Python ThreadPool结合Selenium爬虫完成后不关闭且漏存页面如何解决
多线程Selenium爬虫问题修复方案
问题根因
两份代码的问题完全同源,核心问题如下:
- Selenium未设置超时时间,默认无限等待页面加载,遇到网络波动、网站反爬拦截时会直接卡住线程,导致程序一直无法退出
- ThreadPool使用后未显式关闭、等待线程回收,仅调用
map方法不会自动销毁线程池内的子线程 - 子线程threadlocal存储的webdriver实例无法被主线程的
del threaded_data操作释放,webdriver进程残留会阻止程序退出 - 多线程并发修改全局
count变量未加锁,计数逻辑错乱,同时页面加载、文件写入无重试机制,异常后直接跳过导致页面漏存 - 未对文件名做非法字符处理,页面标题含Windows系统不允许的特殊字符时,文件写入直接失败
修复要点
- 给Selenium配置页面加载、元素获取超时时间,超时后主动抛出异常中断等待
- 显式管理ThreadPool生命周期,任务执行完成后调用
close和join方法回收线程 - 线程内任务执行完成后主动调用
driver.quit()释放webdriver资源,不依赖析构函数自动回收 - 增加页面加载重试机制,单次加载失败后重试2-3次再跳过
- 对文件名做非法字符过滤,避免写入失败
- 全局计数器加锁避免并发修改异常
修复后代码(基于你修改的无Class版本)
# libraries import os import re import time from bs4 import BeautifulSoup from selenium import webdriver from selenium.common.exceptions import TimeoutException from multiprocessing.pool import ThreadPool import threading # variables url = "https://eldorado.ua/" directory = os.path.dirname(os.path.realpath(__file__)) env_path = directory + "\chromedriver" chromedriver_path = env_path + "\chromedriver.exe" UserAgent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/95.0.4638.54 " \ "Safari/537.36 " dict1 = {"Смартфоны и телефоны": "https://eldorado.ua/node/c1038944/", "Телевизоры и аудиотехника": "https://eldorado.ua/node/c1038957/", "Ноутбуки, ПК и Планшеты": "https://eldorado.ua/node/c1038958/", "Техника для кухни": "https://eldorado.ua/node/c1088594/", "Техника для дома": "https://eldorado.ua/node/c1088603/", "Игровая зона": "https://eldorado.ua/node/c1285101/", "Гаджеты и аксесуары": "https://eldorado.ua/node/c1215257/", "Посуда": "https://eldorado.ua/node/c1039055/", "Фото и видео": "https://eldorado.ua/node/c1038960/", "Красота и здоровье": "https://eldorado.ua/node/c1178596/", "Авто и инструменты": "https://eldorado.ua/node/c1284654/", "Спорт и туризм": "https://eldorado.ua/node/c1218544/", "Товары для дома и сада": "https://eldorado.ua/node/c1285161/", "Товары для детей": "https://eldorado.ua/node/c1085100/"} count = 0 count_lock = threading.Lock() threaded_data = threading.local() os.environ['PATH'] += env_path # 过滤文件名非法字符 def sanitize_filename(filename): return re.sub(r'[\\/*?:"<>|]', "", filename) def processing_brand_pages(name): driver = put_driver_in_threaded_data() try: with open(f"{directory}\section_pages\\{name}.html", encoding="utf-8") as file: soup = BeautifulSoup(file.read(), "lxml") links = soup.find_all("div", class_="title") for n in links: ref = url + n.find('a').get('href') with count_lock: global count print(n.text, count) count += 1 # 重试3次加载页面 success = False for _ in range(3): try: driver.get(ref) success = True break except TimeoutException: time.sleep(1) continue if not success: print(f"加载失败,跳过页面:{ref}") continue try: safe_name = sanitize_filename(n.text) os.makedirs(f"{directory}\\brand_pages\\{name}", exist_ok=True) with open(f"{directory}\\brand_pages\\{name}\\{safe_name}.html", "w", encoding="utf-8") as file: file.write(driver.page_source) except Exception as ex: print(f"页面{ref}写入失败:{ex}") finally: # 线程任务完成后主动退出driver driver.quit() def put_driver_in_threaded_data(): threaded_driver = getattr(threaded_data, 'driver_in_threaded_data', None) if threaded_driver is None: options = webdriver.ChromeOptions() options.headless = True options.add_experimental_option("excludeSwitches", ['enable-automation']) options.add_argument(f'--user-agent={UserAgent}') options.add_argument('--disable-blink-features=AutomationControlled') threaded_driver = webdriver.Chrome(executable_path=chromedriver_path, options=options) # 设置超时时间,15秒没加载完直接抛异常 threaded_driver.set_page_load_timeout(15) threaded_driver.set_script_timeout(15) setattr(threaded_data, 'driver_in_threaded_data', threaded_driver) return threaded_driver if __name__ == "__main__": pool = ThreadPool(processes=6) pool.map(processing_brand_pages, dict1.keys()) # 显式关闭线程池,等待所有线程回收 pool.close() pool.join()
内容的提问来源于stack exchange,提问作者Arondy
相关产品推荐
相关产品推荐

