You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用asyncio批量下载图片报错:需传入协程却得到None

使用asyncio批量下载图片时触发TypeError错误

目标

想用asyncio实现多张图片的批量下载。

问题

运行代码时出现TypeError,错误提示:TypeError: a coroutine was expected, got None

错误栈

Traceback (most recent call last):
File "C:\Users\Ze\AppData\Roaming\JetBrains\PyCharmCE2022.1\scratches\scratch_1.py", line 175, in
loop.run_until_complete(main())
File "C:\ProgramData\Anaconda3\lib\asyncio\base_events.py", line 647, in run_until_complete
return future.result()
File "C:\Users\Ze\AppData\Roaming\JetBrains\PyCharmCE2022.1\scratches\scratch_1.py", line 163, in main
await get_image_links(all_links)
File "C:\Users\Ze\AppData\Roaming\JetBrains\PyCharmCE2022.1\scratches\scratch_1.py", line 89, in get_image_links
current_task = asyncio.create_task(await fetch_image(img, i))
File "C:\ProgramData\Anaconda3\lib\asyncio\tasks.py", line 361, in create_task
task = loop.create_task(coro)
File "C:\ProgramData\Anaconda3\lib\asyncio\base_events.py", line 438, in create_task
task = tasks.Task(coro, loop=self, name=name)
TypeError: a coroutine was expected, got None

相关代码

import asyncio
import requests
from bs4 import BeautifulSoup
import re
import os

async def get_image_links(all_links):
    for index, link in enumerate(all_links.split()):
        response = requests.get(link, cookies=cookies, headers=headers).content

        soup = BeautifulSoup(response, 'html.parser')
        page_html = str(soup.find("div", {"class": "entry-content"}))

        all_images = list(set([re.sub(r'-[0-9]+x[0-9]+', '', x) for x in re.findall(r'((?:https?://)[^",]+(?:jpe?g|webp|png))', page_html)]))

        entry_data = str(soup.find("footer", {"class": "entry-meta"}))

        agency = re.search('Category: .+?>([^<]+)', entry_data).group(1)
        name = re.search('Tags: .+?>([^<]+)', entry_data).group(1)
        folder_name = name + " - " + agency

        location = os.path.join(base_location, folder_name)

        if not os.path.exists(location):
            os.makedirs(location)

        async def fetch_image(img, img_index):
            image_name = img.split('/')[-1]
            save_path = os.path.join(location, image_name)

            # 如果图片已下载则跳过
            if os.path.exists(save_path):
                print("Skipping")
                return

            print(f"Fetching image {img_index + 1} of {len(all_images)} for link {index + 1} of {len(all_links.split())}")
            save_image(save_path, requests.get(img).content)

        tasks = []

        for i, img in enumerate(all_images):
            current_task = asyncio.create_task(await fetch_image(img, i))

            tasks.append(current_task)

        await asyncio.gather(*tasks)

async def main():
    await get_image_links(all_links)

all_links = [#links]
base_location = "./downloads"
cookies = {}  # 替换为实际cookies
headers = {}  # 替换为实际headers
def save_path(path, content):
    with open(path, 'wb') as f:
        f.write(content)

# 避免多事件循环冲突
if asyncio.get_event_loop().is_running():
    import nest_asyncio
    nest_asyncio.apply()

loop = asyncio.get_event_loop()
loop.run_until_complete(main())
loop.close()

解决方案

1. 修复任务创建的核心错误

报错根源是创建任务时错误使用了await:

current_task = asyncio.create_task(await fetch_image(img, i))

await fetch_image(img, i)会直接执行协程并获取返回值(这里协程无返回值,得到None),而asyncio.create_task()需要传入协程对象,不是执行结果。修改为:

current_task = asyncio.create_task(fetch_image(img, i))

2. 替换同步请求为异步实现

原代码用requests.get()是同步阻塞操作,会彻底卡住asyncio事件循环,无法实现真正的异步并发。需要改用异步HTTP库aiohttp:

  • 先安装依赖:pip install aiohttp
  • 修改后的核心代码:
import aiohttp

async def get_image_links(all_links):
    # 创建aiohttp会话,复用连接提升效率
    async with aiohttp.ClientSession(cookies=cookies, headers=headers) as session:
        for index, link in enumerate(all_links.split()):
            async with session.get(link) as response:
                response_content = await response.read()

            soup = BeautifulSoup(response_content, 'html.parser')
            page_html = str(soup.find("div", {"class": "entry-content"}))

            all_images = list(set([re.sub(r'-[0-9]+x[0-9]+', '', x) for x in re.findall(r'((?:https?://)[^",]+(?:jpe?g|webp|png))', page_html)]))

            entry_data = str(soup.find("footer", {"class": "entry-meta"}))

            agency = re.search('Category: .+?>([^<]+)', entry_data).group(1)
            name = re.search('Tags: .+?>([^<]+)', entry_data).group(1)
            folder_name = name + " - " + agency

            location = os.path.join(base_location, folder_name)

            if not os.path.exists(location):
                os.makedirs(location)

            async def fetch_image(img, img_index):
                image_name = img.split('/')[-1]
                save_path = os.path.join(location, image_name)

                if os.path.exists(save_path):
                    print("Skipping")
                    return

                print(f"Fetching image {img_index + 1} of {len(all_images)} for link {index + 1} of {len(all_links.split())}")
                async with session.get(img) as img_response:
                    img_content = await img_response.read()
                    save_image(save_path, img_content)

            # 用列表推导式批量创建任务,代码更简洁
            tasks = [asyncio.create_task(fetch_image(img, i)) for i, img in enumerate(all_images)]
            await asyncio.gather(*tasks)

3. 额外优化点

  • 使用async with管理aiohttp会话和请求,自动释放资源
  • 字符串拼接改用f-string,可读性更强
  • 批量创建任务用列表推导式,简化代码

内容的提问来源于stack exchange,提问作者José Guedes

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.24 23:24:56