使用asyncio批量下载图片报错:需传入协程却得到None
目标
想用asyncio实现多张图片的批量下载。
问题
运行代码时出现TypeError,错误提示:TypeError: a coroutine was expected, got None
错误栈
Traceback (most recent call last):
File "C:\Users\Ze\AppData\Roaming\JetBrains\PyCharmCE2022.1\scratches\scratch_1.py", line 175, in
loop.run_until_complete(main())
File "C:\ProgramData\Anaconda3\lib\asyncio\base_events.py", line 647, in run_until_complete
return future.result()
File "C:\Users\Ze\AppData\Roaming\JetBrains\PyCharmCE2022.1\scratches\scratch_1.py", line 163, in main
await get_image_links(all_links)
File "C:\Users\Ze\AppData\Roaming\JetBrains\PyCharmCE2022.1\scratches\scratch_1.py", line 89, in get_image_links
current_task = asyncio.create_task(await fetch_image(img, i))
File "C:\ProgramData\Anaconda3\lib\asyncio\tasks.py", line 361, in create_task
task = loop.create_task(coro)
File "C:\ProgramData\Anaconda3\lib\asyncio\base_events.py", line 438, in create_task
task = tasks.Task(coro, loop=self, name=name)
TypeError: a coroutine was expected, got None
相关代码
import asyncio import requests from bs4 import BeautifulSoup import re import os async def get_image_links(all_links): for index, link in enumerate(all_links.split()): response = requests.get(link, cookies=cookies, headers=headers).content soup = BeautifulSoup(response, 'html.parser') page_html = str(soup.find("div", {"class": "entry-content"})) all_images = list(set([re.sub(r'-[0-9]+x[0-9]+', '', x) for x in re.findall(r'((?:https?://)[^",]+(?:jpe?g|webp|png))', page_html)])) entry_data = str(soup.find("footer", {"class": "entry-meta"})) agency = re.search('Category: .+?>([^<]+)', entry_data).group(1) name = re.search('Tags: .+?>([^<]+)', entry_data).group(1) folder_name = name + " - " + agency location = os.path.join(base_location, folder_name) if not os.path.exists(location): os.makedirs(location) async def fetch_image(img, img_index): image_name = img.split('/')[-1] save_path = os.path.join(location, image_name) # 如果图片已下载则跳过 if os.path.exists(save_path): print("Skipping") return print(f"Fetching image {img_index + 1} of {len(all_images)} for link {index + 1} of {len(all_links.split())}") save_image(save_path, requests.get(img).content) tasks = [] for i, img in enumerate(all_images): current_task = asyncio.create_task(await fetch_image(img, i)) tasks.append(current_task) await asyncio.gather(*tasks) async def main(): await get_image_links(all_links) all_links = [#links] base_location = "./downloads" cookies = {} # 替换为实际cookies headers = {} # 替换为实际headers def save_path(path, content): with open(path, 'wb') as f: f.write(content) # 避免多事件循环冲突 if asyncio.get_event_loop().is_running(): import nest_asyncio nest_asyncio.apply() loop = asyncio.get_event_loop() loop.run_until_complete(main()) loop.close()
解决方案
1. 修复任务创建的核心错误
报错根源是创建任务时错误使用了await:
current_task = asyncio.create_task(await fetch_image(img, i))
await fetch_image(img, i)会直接执行协程并获取返回值(这里协程无返回值,得到None),而asyncio.create_task()需要传入协程对象,不是执行结果。修改为:
current_task = asyncio.create_task(fetch_image(img, i))
2. 替换同步请求为异步实现
原代码用requests.get()是同步阻塞操作,会彻底卡住asyncio事件循环,无法实现真正的异步并发。需要改用异步HTTP库aiohttp:
- 先安装依赖:
pip install aiohttp - 修改后的核心代码:
import aiohttp async def get_image_links(all_links): # 创建aiohttp会话,复用连接提升效率 async with aiohttp.ClientSession(cookies=cookies, headers=headers) as session: for index, link in enumerate(all_links.split()): async with session.get(link) as response: response_content = await response.read() soup = BeautifulSoup(response_content, 'html.parser') page_html = str(soup.find("div", {"class": "entry-content"})) all_images = list(set([re.sub(r'-[0-9]+x[0-9]+', '', x) for x in re.findall(r'((?:https?://)[^",]+(?:jpe?g|webp|png))', page_html)])) entry_data = str(soup.find("footer", {"class": "entry-meta"})) agency = re.search('Category: .+?>([^<]+)', entry_data).group(1) name = re.search('Tags: .+?>([^<]+)', entry_data).group(1) folder_name = name + " - " + agency location = os.path.join(base_location, folder_name) if not os.path.exists(location): os.makedirs(location) async def fetch_image(img, img_index): image_name = img.split('/')[-1] save_path = os.path.join(location, image_name) if os.path.exists(save_path): print("Skipping") return print(f"Fetching image {img_index + 1} of {len(all_images)} for link {index + 1} of {len(all_links.split())}") async with session.get(img) as img_response: img_content = await img_response.read() save_image(save_path, img_content) # 用列表推导式批量创建任务,代码更简洁 tasks = [asyncio.create_task(fetch_image(img, i)) for i, img in enumerate(all_images)] await asyncio.gather(*tasks)
3. 额外优化点
- 使用
async with管理aiohttp会话和请求,自动释放资源 - 字符串拼接改用f-string,可读性更强
- 批量创建任务用列表推导式,简化代码
内容的提问来源于stack exchange,提问作者José Guedes

