Headless Chrome(Selenium)请求无流量问题排查求助
问题:Headless Chrome批量加载URL无网络流量(Selenium)
我尝试使用Selenium在Headless Chrome环境中批量打开两个CSV文件中的URL,代码运行无报错,但通过Wireshark检测发现完全没有网络流量产生。
用同一数据集测试pycurl时功能正常,能正常产生网络请求。怀疑是Kali Linux平台的问题,也尝试过切换Firefox和Chrome驱动,以及调整代码中注释的Chrome启动选项,但均未解决问题。
以下是我认为功能正常但实际无效的代码:
import pandas as pd from tqdm import tqdm import requests from selenium import webdriver from selenium.webdriver.chrome.options import Options from concurrent.futures import ThreadPoolExecutor # Read the CSV files with encoding top_domains_df = pd.read_csv('Top1MillionDomians.csv', encoding='utf-8', dtype=str) malware_urls_df = pd.read_csv('Malurl.csv', encoding='utf-8', dtype=str) # Get the first column from each DataFrame top_domains = top_domains_df.iloc[:, 0].tolist() malware_urls = malware_urls_df.iloc[:, 0].tolist() # Initialize Chrome options chrome_options = Options() #chrome_options.add_argument('--headless') ###chrome_options.add_argument('--disable-gpu') ##chrome_options.add_argument('--no-sandbox') #chrome_options.add_argument('--disable-dev-shm-usage') chrome_options.headless = True # Function to process a URL def process_url(url): try: driver = webdriver.Chrome(options=chrome_options) driver.set_page_load_timeout(15) # Maximum 15 seconds per URL driver.get(url) # Add your processing logic here driver.quit() return url, True except: driver.quit() return url, False def perform_request(url): try: response = requests.get(url, timeout=5) return url, response.status_code == 200 except: return url, False def process_with_progress(url, pbar): result = perform_request(url) update_progress_bar(result, pbar) return result def update_progress_bar(result, pbar): url, success = result if success: pbar.set_description(f"Processing: {url} (Success)") else: pbar.set_description(f"Processing: {url} (Error)") pbar.update(1) print(f"Processed URL: {url} (Success: {success})") if __name__ == '__main__': batch_size = 500 num_processes = 4 # Number of parallel processes with tqdm(total=len(malware_urls), desc="Processing URLs") as pbar: with ThreadPoolExecutor(max_workers=num_processes) as executor: futures = [] for url in malware_urls: future = executor.submit(process_with_progress, url, pbar) futures.append(future) for future in futures: future.result() print("All URLs processed.")
内容的提问来源于stack exchange,提问作者mhoro
相关产品推荐
相关产品推荐

