You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用aiohttp请求图片返回403 Forbidden,requests.get却返回200的问题

问题描述

使用aiohttp异步下载特定站点的图片时,返回403 Forbidden错误,但相同URL使用requests.get同步请求可成功获取图片。仅这类站点的URL出现该问题,aiohttp代码在其他站点URL上运行正常。相关代码及响应信息如下:

requests.get 实现代码

import requests
from io import BytesIO

async def download_image(self, url: str):
    
    ## is_url是验证URL有效性的小函数
    if not is_url(url):
        return None

    VALID_MIME_TYPES = {
        "image/jpeg": ".jpeg",
        "image/png": ".png",
        "image/jpg": ".jpg",
        "image/gif": ".gif",
        "image/tiff": ".tiff",
        "image/webp": ".webp",
        "image/apng": ".apng",
        "image/svg+xml": ".svg",
        "application/octet-stream": get_file_extension_from_url(url=url)  
        # get_file_extension_from_url用于从URL获取图片类型
    }

    response = requests.get(url)    # 可成功下载图片
    mimetype = response.headers.get("Content-Type", "").lower()
    
    if mimetype in VALID_MIME_TYPES:
            # 生成文件名
            file_name = f"cimage.{VALID_MIME_TYPES[mimetype]}"
            content = response.content
            
            # 转换为BytesIO流
            return BytesIO(content), file_name, mimetype
    else:
        return None

image_url = "https://img.evbuc.com/https%3A%2F%2Fcdn.evbuc.com%2Fimages%2F602272019%2F1430182031443%2F1%2Foriginal.20230920-130504?w=940&auto=format%2Ccompress&q=75&sharp=10&rect=0%2C15%2C1200%2C600&s=13645e838fd09f2552c8f8500410abec"
image_output = download_image(image_url)

aiohttp 实现代码

import aiohttp, asyncio
from io import BytesIO

async def download_image(self, url: str, session: aiohttp.ClientSession):
    """
    Args:
        url (str): image url
        session (aiohttp.ClientSession): 使用通用会话提升速度
    """
    if not is_url(url):
        return None

    VALID_MIME_TYPES = {
        "image/jpeg": ".jpeg",
        "image/png": ".png",
        "image/jpg": ".jpg",
        "image/gif": ".gif",
        "image/tiff": ".tiff",
        "image/webp": ".webp",
        "image/apng": ".apng",
        "image/svg+xml": ".svg",
        "application/octet-stream": get_file_extension_from_url(url=url)
    }
    res = await session.request(method="GET", url=url)

    mimetype = res.headers.get("Content-Type", "").lower()

    if mimetype in VALID_MIME_TYPES:
        file_name = f"cimage.{VALID_MIME_TYPES[mimetype]}"
        content = await res.read()
        return BytesIO(content), file_name, mimetype
    else:
        return None

if __name__ == "__main__":
    image_url = "https://img.evbuc.com/https%3A%2F%2Fcdn.evbuc.com%2Fimages%2F602272019%2F1430182031443%2F1%2Foriginal.20230920-130504?w=940&auto=format%2Ccompress&q=75&sharp=10&rect=0%2C15%2C1200%2C600&s=13645e838fd09f2552c8f8500410abec"
    image_urls = [image_url] * 1 # 测试时可增加数量

    async def main():
        async with aiohttp.ClientSession(trust_env=True) as session:
            tasks = []
            for url in image_urls:
                task = asyncio.create_task(download_image(url=url, session=session))
                tasks.append(task)
                
            # 返回图片输出的三元组列表
            images = await asyncio.gather(*tasks)


    asyncio.run(main())

aiohttp 请求响应信息

<ClientResponse(https://img.evbuc.com/https:%2F%2Fcdn.evbuc.com%2Fimages%2F602272019%2F1430182031443%2F1%2Foriginal.20230920-130504?w=940&amp;auto=format,compress&amp;q=75&amp;sharp=10&amp;rect=0,15,1200,600&amp;s=13645e838fd09f2552c8f8500410abec) [403 Forbidden]>
<CIMultiDictProxy('Content-Type': 'text/plain', 'Content-Length': '14', 'Connection': 'keep-alive', 'Cache-Control': 'public, max-age=5', 'Server': 'imgix', 'x-imgix-id': 'e3fb1d9c2f4cdf79dc45ca6fa20455560bdc05a5', 'x-imgix-proxy-status': '403', 'x-imgix-proxy-reason': '', 'X-Imgix-Render-Farm': '01.140360', 'Date': 'Wed, 18 Oct 2023 20:11:20 GMT', 'Accept-Ranges': 'bytes', 'Access-Control-Allow-Origin': '*', 'Timing-Allow-Origin': '*', 'Cross-Origin-Resource-Policy': 'cross-origin', 'X-Content-Type-Options': 'nosniff', 'X-Served-By': 'cache-sjc10076-SJC, cache-bom4734-BOM', 'X-Cache': 'Error from cloudfront', 'Via': '1.1 9e8c29342ff6f7610166562f3559cbe4.cloudfront.net (CloudFront)', 'X-Amz-Cf-Pop': 'BOM78-P1', 'X-Amz-Cf-Id': 'cFMR0YKkz5pLrgzH-IkmYd0JTYqgZPT-wDKbTdxDiOr_ZJH_v3xLeg==', 'Age': '0')>
问题原因

目标站点使用的imgix/CloudFront反爬机制识别到了aiohttp与requests的默认请求头差异,核心是User-Agent字段:

  • requests默认的User-Agent格式为python-requests/{version}
  • aiohttp默认的User-Agent格式为Python/{python_version} aiohttp/{aiohttp_version}
    服务器将aiohttp的默认请求头判定为非合法请求,因此返回403 Forbidden。
解决方案

在aiohttp请求中添加与requests一致的请求头,重点模拟User-Agent字段。修改后的aiohttp代码如下:

import aiohttp, asyncio
from io import BytesIO

async def download_image(self, url: str, session: aiohttp.ClientSession):
    if not is_url(url):
        return None

    VALID_MIME_TYPES = {
        "image/jpeg": ".jpeg",
        "image/png": ".png",
        "image/jpg": ".jpg",
        "image/gif": ".gif",
        "image/tiff": ".tiff",
        "image/webp": ".webp",
        "image/apng": ".apng",
        "image/svg+xml": ".svg",
        "application/octet-stream": get_file_extension_from_url(url=url)
    }
    # 添加模拟requests的请求头
    headers = {
        "User-Agent": "python-requests/2.31.0",  # 可替换为你实际使用的requests版本
        "Accept": "*/*",
        "Accept-Encoding": "gzip, deflate",
        "Connection": "keep-alive"
    }
    res = await session.request(method="GET", url=url, headers=headers)

    mimetype = res.headers.get("Content-Type", "").lower()

    if mimetype in VALID_MIME_TYPES:
        file_name = f"cimage.{VALID_MIME_TYPES[mimetype]}"
        content = await res.read()
        return BytesIO(content), file_name, mimetype
    else:
        return None

if __name__ == "__main__":
    image_url = "https://img.evbuc.com/https%3A%2F%2Fcdn.evbuc.com%2Fimages%2F602272019%2F1430182031443%2F1%2Foriginal.20230920-130504?w=940&amp;auto=format%2Ccompress&amp;q=75&amp;sharp=10&amp;rect=0%2C15%2C1200%2C600&amp;s=13645e838fd09f2552c8f8500410abec"
    image_urls = [image_url] * 1

    async def main():
        async with aiohttp.ClientSession(trust_env=True) as session:
            tasks = [asyncio.create_task(download_image(url=url, session=session)) for url in image_urls]
            images = await asyncio.gather(*tasks)

    asyncio.run(main())

补充说明

  • 可以通过requests.utils.default_headers()获取requests的完整默认请求头,直接复制到aiohttp中使用,确保请求头完全一致
  • 若仍出现403,可尝试添加Referer字段,模拟从目标站点页面跳转过来的请求

内容的提问来源于stack exchange,提问作者Harsh Rao

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.08 01:32:02