You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

传递pyppeteer.Page至多进程时触发PermissionError,求解决方案

如何在多进程中复用Pyppeteer浏览器页面以降低开销?

背景

原有代码可运行但效率低下:每个进程独立启动浏览器实例,处理任务时开销过大。

原代码:

import asyncio
from pyppeteer import launch
from multiprocessing import Process

async def f(x):
    print("async def f(x,page):",x)    
    browser = await launch(headless=False, autoClose=False)
    page = (await browser.pages())[0]    
    await page.goto('https://example.com')
    h1 = await page.querySelector("body > div > h1")
    await page.evaluate(f'(element) => element.textContent="{x}"', h1)    

def p(x):
    print("def p(x,page):",x)
    asyncio.run(f(x))

async def main():
    pro = Process(target=p, args=("1111",))
    pro.start()    
    pro = Process(target=p, args=("2222",))
    pro.start()    

if __name__ =="__main__":
    asyncio.get_event_loop().run_until_complete(main())

为降低开销,尝试复用浏览器页面,将pyppeteer.Page作为参数传递给multiprocessing.Process,但运行时触发错误:

尝试的代码:

import asyncio
from pyppeteer import launch
from multiprocessing import Process

async def f(x,page):
    print("async def f(x,page):",x)

    await page.goto('https://example.com')
    h1 = await page.querySelector("body > div > h1")
    await page.evaluate(f'(element) => element.textContent="{x}"', h1)

def p(x,page):
    print("def p(x,page):",x)
    asyncio.run(f(x,page))

async def main():
    browser = await launch(headless=False, autoClose=False)
    page = (await browser.pages())[0]

    pro = Process(target=p, args=("1111",page))
    pro.start()    

if __name__ =="__main__":
    asyncio.get_event_loop().run_until_complete(main())

错误信息

c:\Users\mimmi\python\ttttt.py:24: DeprecationWarning: There is no current event loop
  asyncio.get_event_loop().run_until_complete(main())
Traceback (most recent call last):
  File "c:\Users\mimmi\python\ttttt.py", line 24, in <module>
    asyncio.get_event_loop().run_until_complete(main())
  File "C:\python\python311\Lib\asyncio\base_events.py", line 650, in run_until_complete
    return future.result()
           ^^^^^^^^^^^^^^^
  File "c:\Users\mimmi\python\ttttt.py", line 21, in main
    pro.start()    
    ^^^^^^^^^^^
  File "C:\python\python311\Lib\multiprocessing\process.py", line 121, in start
    self._popen = self._Popen(self)
              ^^^^^^^^^^^^^^^^^
  File "C:\python\python311\Lib\multiprocessing\context.py", line 224, in _Popen
    return _default_context.get_context().Process._Popen(process_obj)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "C:\python\python311\Lib\multiprocessing\context.py", line 336, in _Popen
    return Popen(process_obj)
           ^^^^^^^^^^^^^^^^^^
  File "C:\python\python311\Lib\multiprocessing\popen_spawn_win32.py", line 94, in __init__
    reduction.dump(process_obj, to_child)
  File "C:\python\python311\Lib\multiprocessing\reduction.py", line 60, in dump
    ForkingPickler(file, protocol).dump(obj)
TypeError: cannot pickle '_thread.lock' object
Traceback (most recent call last):
  File "<string>", line 1, in <module>
  File "C:\python\python311\Lib\multiprocessing\spawn.py", line 111, in spawn_main
    new_handle = reduction.duplicate(pipe_handle,
                 ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "C:\python\python311\Lib\multiprocessing\reduction.py", line 79, in duplicate
    return _winapi.DuplicateHandle(
           ^^^^^^^^^^^^^^^^^^^^^^^^
PermissionError: [WinError 5] Access is denied

环境信息

  • Windows 11
  • Python 3.11
  • pyppeteer 1.0.2

解决方案

问题根源

multiprocessing跨进程传递参数时需要序列化对象,但pyppeteer.Page和Browser内部包含线程锁、系统进程句柄等无法序列化的资源,直接传递会触发序列化错误,同时Windows的进程模型会引发权限问题。

方案1:异步并发复用单个浏览器实例(推荐)

Pyppeteer基于异步IO设计,单进程内用asyncio.gather实现多任务并发,复用同一个浏览器的多个页面,开销远低于多进程各起一个浏览器。

修改后的代码:

import asyncio
from pyppeteer import launch

async def process_task(x, browser):
    print(f"处理任务: {x}")
    # 为每个任务创建独立页面,避免任务间干扰
    page = await browser.newPage()
    try:
        await page.goto('https://example.com')
        h1 = await page.querySelector("body > div > h1")
        await page.evaluate(f'(element) => element.textContent="{x}"', h1)
        # 可添加截图、数据提取等操作
        # await page.screenshot(path=f"{x}_result.png")
    finally:
        # 任务完成后关闭页面,释放资源
        await page.close()

async def main():
    # 仅启动一个浏览器实例
    browser = await launch(headless=False, autoClose=False)
    # 创建多个异步任务并发执行
    tasks = [process_task("1111", browser), process_task("2222", browser)]
    await asyncio.gather(*tasks)
    # 所有任务完成后关闭浏览器
    await browser.close()

if __name__ == "__main__":
    asyncio.run(main())

方案2:多进程+进程间通信(需分离CPU密集任务时使用)

如果必须用多进程(比如同时处理CPU密集型任务),可让单个进程管理浏览器,其他进程通过队列传递任务请求,避免跨进程传递浏览器对象。

示例代码:

import asyncio
from pyppeteer import launch
from multiprocessing import Process, Queue

# 主进程的浏览器处理逻辑
async def browser_worker(task_queue):
    browser = await launch(headless=False, autoClose=False)
    while True:
        # 从队列获取任务(同步转异步)
        task = await asyncio.get_event_loop().run_in_executor(None, task_queue.get)
        if task is None:  # 收到结束信号
            break
        x = task
        print(f"处理任务: {x}")
        page = await browser.newPage()
        try:
            await page.goto('https://example.com')
            h1 = await page.querySelector("body > div > h1")
            await page.evaluate(f'(element) => element.textContent="{x}"', h1)
        finally:
            await page.close()
    await browser.close()

# 子进程:仅负责发送任务
def task_sender(task_queue):
    # 发送任务
    task_queue.put("1111")
    task_queue.put("2222")
    task_queue.put(None)  # 发送结束信号

if __name__ == "__main__":
    task_queue = Queue()
    # 启动子进程
    sender_process = Process(target=task_sender, args=(task_queue,))
    sender_process.start()
    # 主进程运行浏览器任务
    asyncio.run(browser_worker(task_queue))
    sender_process.join()

内容的提问来源于stack exchange,提问作者금밈미

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.12 23:35:51