Python Playwright下载PDF出现net::ERR_ABORTED错误求解决方案
解决Playwright下载PDF报错及拦截Blob传输的方案
错误原因分析
你遇到的net::ERR_ABORTED错误,是因为设置always_open_pdf_externally后,调用page.goto访问PDF链接时,浏览器会尝试调用外部程序打开PDF,导致页面导航被强制中止。
解决PDF下载错误的方法
方案1:移除外部打开偏好,模拟内置阅读器下载
删掉自定义PDF外部打开的配置,页面加载PDF后,模拟点击Chromium内置阅读器的下载按钮:
from playwright.async_api import Playwright, async_playwright import asyncio import os downloads_path = os.path.join(os.getcwd(), "pwtest", "downloads") user_dir = os.path.join(os.getcwd(), "pwtest", "user_dir") # 用os.makedirs高效创建多级目录(替代多个try-except) os.makedirs(downloads_path, exist_ok=True) os.makedirs(os.path.join(user_dir, "Default"), exist_ok=True) async def run(playwright: Playwright) -> None: browser = await playwright.chromium.launch_persistent_context( user_dir, accept_downloads=True, headless=False, slow_mo=1000 ) browser.set_default_timeout(10000) page = await browser.new_page() file_name = "test_d.pdf" async with page.expect_download() as download_info: await page.goto("https://www.africau.edu/images/default/sample.pdf", timeout=5000) # 点击内置PDF阅读器的下载按钮(Chromium默认选择器) await page.click("button[aria-label='下载']") download = await download_info.value await download.save_as(os.path.join(downloads_path, file_name)) print(f"文件已保存至: {os.path.join(downloads_path, file_name)}") await browser.close() async def main() -> None: async with async_playwright() as playwright: await run(playwright) asyncio.run(main())
方案2:直接发起网络请求下载(无需页面渲染)
如果不需要渲染页面,直接用Playwright的请求API获取PDF内容,效率更高:
from playwright.async_api import Playwright, async_playwright import asyncio import os downloads_path = os.path.join(os.getcwd(), "pwtest", "downloads") os.makedirs(downloads_path, exist_ok=True) async def run(playwright: Playwright) -> None: browser = await playwright.chromium.launch(headless=False) context = await browser.new_context() page = await context.new_page() # 发起GET请求获取PDF二进制内容 response = await page.request.get("https://www.africau.edu/images/default/sample.pdf") pdf_content = await response.body() # 写入本地文件 file_path = os.path.join(downloads_path, "test_d.pdf") with open(file_path, "wb") as f: f.write(pdf_content) print(f"文件已保存至: {file_path}") await browser.close() async def main() -> None: async with async_playwright() as playwright: await run(playwright) asyncio.run(main())
拦截Blob传输的方法
使用page.route拦截目标请求,获取响应的Blob数据,可直接保存或处理:
from playwright.async_api import Playwright, async_playwright import asyncio import os downloads_path = os.path.join(os.getcwd(), "pwtest", "downloads") os.makedirs(downloads_path, exist_ok=True) async def run(playwright: Playwright) -> None: browser = await playwright.chromium.launch(headless=False) context = await browser.new_context() page = await context.new_page() # 定义拦截处理逻辑 async def intercept_blob(route): # 继续请求获取原始响应 response = await route.fetch() # 获取Blob二进制内容 blob_content = await response.body() # 保存拦截到的内容 with open(os.path.join(downloads_path, "intercepted_blob.pdf"), "wb") as f: f.write(blob_content) print("已拦截并保存Blob内容") # 让页面继续加载原始响应 await route.fulfill(response=response) # 匹配所有PDF格式的请求(可根据需求修改匹配规则) await page.route("**/*.pdf", intercept_blob) # 访问页面触发请求 await page.goto("https://www.africau.edu/images/default/sample.pdf") await browser.close() async def main() -> None: async with async_playwright() as playwright: await run(playwright) asyncio.run(main())
内容的提问来源于stack exchange,提问作者NanoNerd
相关产品推荐
相关产品推荐

