使用aiohttp请求图片返回403 Forbidden,requests.get却返回200的问题
问题描述
使用aiohttp异步下载特定站点的图片时,返回403 Forbidden错误,但相同URL使用requests.get同步请求可成功获取图片。仅这类站点的URL出现该问题,aiohttp代码在其他站点URL上运行正常。相关代码及响应信息如下:
requests.get 实现代码
import requests from io import BytesIO async def download_image(self, url: str): ## is_url是验证URL有效性的小函数 if not is_url(url): return None VALID_MIME_TYPES = { "image/jpeg": ".jpeg", "image/png": ".png", "image/jpg": ".jpg", "image/gif": ".gif", "image/tiff": ".tiff", "image/webp": ".webp", "image/apng": ".apng", "image/svg+xml": ".svg", "application/octet-stream": get_file_extension_from_url(url=url) # get_file_extension_from_url用于从URL获取图片类型 } response = requests.get(url) # 可成功下载图片 mimetype = response.headers.get("Content-Type", "").lower() if mimetype in VALID_MIME_TYPES: # 生成文件名 file_name = f"cimage.{VALID_MIME_TYPES[mimetype]}" content = response.content # 转换为BytesIO流 return BytesIO(content), file_name, mimetype else: return None image_url = "https://img.evbuc.com/https%3A%2F%2Fcdn.evbuc.com%2Fimages%2F602272019%2F1430182031443%2F1%2Foriginal.20230920-130504?w=940&auto=format%2Ccompress&q=75&sharp=10&rect=0%2C15%2C1200%2C600&s=13645e838fd09f2552c8f8500410abec" image_output = download_image(image_url)
aiohttp 实现代码
import aiohttp, asyncio from io import BytesIO async def download_image(self, url: str, session: aiohttp.ClientSession): """ Args: url (str): image url session (aiohttp.ClientSession): 使用通用会话提升速度 """ if not is_url(url): return None VALID_MIME_TYPES = { "image/jpeg": ".jpeg", "image/png": ".png", "image/jpg": ".jpg", "image/gif": ".gif", "image/tiff": ".tiff", "image/webp": ".webp", "image/apng": ".apng", "image/svg+xml": ".svg", "application/octet-stream": get_file_extension_from_url(url=url) } res = await session.request(method="GET", url=url) mimetype = res.headers.get("Content-Type", "").lower() if mimetype in VALID_MIME_TYPES: file_name = f"cimage.{VALID_MIME_TYPES[mimetype]}" content = await res.read() return BytesIO(content), file_name, mimetype else: return None if __name__ == "__main__": image_url = "https://img.evbuc.com/https%3A%2F%2Fcdn.evbuc.com%2Fimages%2F602272019%2F1430182031443%2F1%2Foriginal.20230920-130504?w=940&auto=format%2Ccompress&q=75&sharp=10&rect=0%2C15%2C1200%2C600&s=13645e838fd09f2552c8f8500410abec" image_urls = [image_url] * 1 # 测试时可增加数量 async def main(): async with aiohttp.ClientSession(trust_env=True) as session: tasks = [] for url in image_urls: task = asyncio.create_task(download_image(url=url, session=session)) tasks.append(task) # 返回图片输出的三元组列表 images = await asyncio.gather(*tasks) asyncio.run(main())
aiohttp 请求响应信息
<ClientResponse(https://img.evbuc.com/https:%2F%2Fcdn.evbuc.com%2Fimages%2F602272019%2F1430182031443%2F1%2Foriginal.20230920-130504?w=940&auto=format,compress&q=75&sharp=10&rect=0,15,1200,600&s=13645e838fd09f2552c8f8500410abec) [403 Forbidden]> <CIMultiDictProxy('Content-Type': 'text/plain', 'Content-Length': '14', 'Connection': 'keep-alive', 'Cache-Control': 'public, max-age=5', 'Server': 'imgix', 'x-imgix-id': 'e3fb1d9c2f4cdf79dc45ca6fa20455560bdc05a5', 'x-imgix-proxy-status': '403', 'x-imgix-proxy-reason': '', 'X-Imgix-Render-Farm': '01.140360', 'Date': 'Wed, 18 Oct 2023 20:11:20 GMT', 'Accept-Ranges': 'bytes', 'Access-Control-Allow-Origin': '*', 'Timing-Allow-Origin': '*', 'Cross-Origin-Resource-Policy': 'cross-origin', 'X-Content-Type-Options': 'nosniff', 'X-Served-By': 'cache-sjc10076-SJC, cache-bom4734-BOM', 'X-Cache': 'Error from cloudfront', 'Via': '1.1 9e8c29342ff6f7610166562f3559cbe4.cloudfront.net (CloudFront)', 'X-Amz-Cf-Pop': 'BOM78-P1', 'X-Amz-Cf-Id': 'cFMR0YKkz5pLrgzH-IkmYd0JTYqgZPT-wDKbTdxDiOr_ZJH_v3xLeg==', 'Age': '0')>
问题原因
目标站点使用的imgix/CloudFront反爬机制识别到了aiohttp与requests的默认请求头差异,核心是User-Agent字段:
- requests默认的User-Agent格式为
python-requests/{version} - aiohttp默认的User-Agent格式为
Python/{python_version} aiohttp/{aiohttp_version}
服务器将aiohttp的默认请求头判定为非合法请求,因此返回403 Forbidden。
解决方案
在aiohttp请求中添加与requests一致的请求头,重点模拟User-Agent字段。修改后的aiohttp代码如下:
import aiohttp, asyncio from io import BytesIO async def download_image(self, url: str, session: aiohttp.ClientSession): if not is_url(url): return None VALID_MIME_TYPES = { "image/jpeg": ".jpeg", "image/png": ".png", "image/jpg": ".jpg", "image/gif": ".gif", "image/tiff": ".tiff", "image/webp": ".webp", "image/apng": ".apng", "image/svg+xml": ".svg", "application/octet-stream": get_file_extension_from_url(url=url) } # 添加模拟requests的请求头 headers = { "User-Agent": "python-requests/2.31.0", # 可替换为你实际使用的requests版本 "Accept": "*/*", "Accept-Encoding": "gzip, deflate", "Connection": "keep-alive" } res = await session.request(method="GET", url=url, headers=headers) mimetype = res.headers.get("Content-Type", "").lower() if mimetype in VALID_MIME_TYPES: file_name = f"cimage.{VALID_MIME_TYPES[mimetype]}" content = await res.read() return BytesIO(content), file_name, mimetype else: return None if __name__ == "__main__": image_url = "https://img.evbuc.com/https%3A%2F%2Fcdn.evbuc.com%2Fimages%2F602272019%2F1430182031443%2F1%2Foriginal.20230920-130504?w=940&auto=format%2Ccompress&q=75&sharp=10&rect=0%2C15%2C1200%2C600&s=13645e838fd09f2552c8f8500410abec" image_urls = [image_url] * 1 async def main(): async with aiohttp.ClientSession(trust_env=True) as session: tasks = [asyncio.create_task(download_image(url=url, session=session)) for url in image_urls] images = await asyncio.gather(*tasks) asyncio.run(main())
补充说明
- 可以通过
requests.utils.default_headers()获取requests的完整默认请求头,直接复制到aiohttp中使用,确保请求头完全一致 - 若仍出现403,可尝试添加
Referer字段,模拟从目标站点页面跳转过来的请求
内容的提问来源于stack exchange,提问作者Harsh Rao
相关产品推荐
相关产品推荐

