Python获取重定向URL时Chrome历史日志未及时更新的问题求助
解决Chrome历史文件未及时更新导致重定向URL重复的问题
我使用Shahin Shirazi的代码批量获取重定向链接的目标URL,但运行时部分场景会重复写入上一个重定向的目标URL,显然Chrome历史文件未及时更新,添加延时处理后问题依旧,相关代码如下:
from bs4 import BeautifulSoup as soup import webbrowser import sqlite3 import pandas as pd import shutil import time import os # ---------------------FILEREADING------------------------------------------ colnames = ['Column1'] Filename_Link = "reflinks.csv" data = pd.read_csv(Filename_Link, names=colnames) links = data.Column1.tolist() # ---------------------Create a new .csv file that we are going to fill with the target urls------------------------------------------ output_data = "output_data.csv" fd = open(output_data, "a",encoding='utf-8-sig') topline = "url, target_url," fd.write(topline) fd.write("\n") # ---------------------Define the history folder of the browser ------------------------------------------ #source file is where the history of your webbroser is saved, I was using chrome, but it should be the same process if you are using different browser source_file = 'C:\\Users\\xxx\\AppData\\Local\\Google\\Chrome\\User Data\\Default\\History' # could not directly connect to history file as it was locked and had to make a copy of it in different location destination_file = 'C:\\Users\\xxx\\Downloads\\History' # ---------------------Run the code to get target urls for all redirect links ------------------------------------------ for link in links: webbrowser.open(link) time.sleep(30) # there is some delay to update the history file, so 30 sec wait give it enough time to make sure your last url get logged os.system("taskkill /im chrome.exe /f") shutil.copy(source_file,destination_file) # copying the file. time.sleep(10) # I added some delay to copy the files con = sqlite3.connect('C:\\Users\\xxx\\Downloads\\History') # connecting to browser history cursor = con.execute("SELECT * FROM urls") names = [description[0] for description in cursor.description] urls = cursor.fetchall() con.close() df_history = pd.DataFrame(urls,columns=names) last_url = df_history.loc[len(df_history)-1,'url'] print(last_url) fd.write(link.replace(",","").replace("\n"," ")) fd.write(",") fd.write(last_url.replace(",","").replace("\n"," ")) fd.write(",") fd.write("\n") fd.close()
解决方案
1. 改用Chrome DevTools协议直接获取最终URL(最可靠)
放弃依赖Chrome历史文件,直接通过Selenium控制Chrome实例,实时捕获页面跳转后的最终URL,彻底避免历史文件更新延迟问题:
from selenium import webdriver from selenium.webdriver.chrome.options import Options import pandas as pd import time # 读取待处理链接 colnames = ['Column1'] Filename_Link = "reflinks.csv" data = pd.read_csv(Filename_Link, names=colnames) links = data.Column1.tolist() # 配置Chrome无头模式(可选,也可以去掉用可视化窗口) chrome_options = Options() chrome_options.add_argument("--headless=new") chrome_options.add_argument("--disable-gpu") chrome_options.add_argument("--no-sandbox") # 初始化浏览器 driver = webdriver.Chrome(options=chrome_options) driver.set_page_load_timeout(30) # 设置页面加载超时时间 # 写入输出文件表头 output_data = "output_data.csv" with open(output_data, "w", encoding='utf-8-sig') as fd: fd.write("url,target_url\n") for link in links: try: driver.get(link) time.sleep(2) # 等待JS跳转完成,可根据实际情况调整时长 final_url = driver.current_url print(f"{link} -> {final_url}") # 写入结果到CSV fd.write(f"{link.replace(',','').replace('\n',' ')},{final_url.replace(',','').replace('\n',' ')}\n") except Exception as e: print(f"处理{link}失败: {str(e)}") fd.write(f"{link.replace(',','').replace('\n',' ')},获取失败\n") # 关闭浏览器 driver.quit()
2. 优化历史文件读取逻辑
如果坚持使用历史文件方案,需解决两个核心问题:确保Chrome完全退出、正确获取最新访问记录:
from bs4 import BeautifulSoup as soup import webbrowser import sqlite3 import pandas as pd import shutil import time import os import psutil # ---------------------FILEREADING------------------------------------------ colnames = ['Column1'] Filename_Link = "reflinks.csv" data = pd.read_csv(Filename_Link, names=colnames) links = data.Column1.tolist() # ---------------------Create output file------------------------------------------ output_data = "output_data.csv" with open(output_data, "w", encoding='utf-8-sig') as fd: fd.write("url,target_url\n") # ---------------------Define history paths------------------------------------------ source_file = 'C:\\Users\\xxx\\AppData\\Local\\Google\\Chrome\\User Data\\Default\\History' destination_file = 'C:\\Users\\xxx\\Downloads\\History' # ---------------------Process each link------------------------------------------ for link in links: webbrowser.open(link) time.sleep(30) # 等待页面跳转完成 # 强制结束所有Chrome进程,确保历史文件写入磁盘 for proc in psutil.process_iter(['name']): if proc.info['name'] == 'chrome.exe': try: proc.kill() except: pass time.sleep(15) # 等待进程完全退出 # 复制历史文件 shutil.copy(source_file, destination_file) time.sleep(2) # 连接数据库,按访问时间取最新记录 con = sqlite3.connect(destination_file) # last_visit_time是Chrome历史表的时间戳字段,降序排序取第一条 cursor = con.execute("SELECT url FROM urls ORDER BY last_visit_time DESC LIMIT 1") last_url = cursor.fetchone()[0] con.close() print(last_url) # 写入结果 with open(output_data, "a", encoding='utf-8-sig') as fd: fd.write(f"{link.replace(',','').replace('\n',' ')},{last_url.replace(',','').replace('\n',' ')}\n")
3. 使用requests库直接追踪重定向(无浏览器,效率最高)
如果目标链接仅涉及HTTP/HTTPS协议重定向(无JavaScript跳转),可以直接用requests库追踪重定向,无需启动浏览器:
import requests import pandas as pd # 读取链接列表 colnames = ['Column1'] Filename_Link = "reflinks.csv" data = pd.read_csv(Filename_Link, names=colnames) links = data.Column1.tolist() # 写入结果 output_data = "output_data.csv" with open(output_data, "w", encoding='utf-8-sig') as fd: fd.write("url,target_url\n") for link in links: try: # allow_redirects=True自动追踪所有重定向 response = requests.get(link, allow_redirects=True, timeout=10) final_url = response.url print(f"{link} -> {final_url}") fd.write(f"{link.replace(',','').replace('\n',' ')},{final_url.replace(',','').replace('\n',' ')}\n") except Exception as e: print(f"处理{link}失败: {str(e)}") fd.write(f"{link.replace(',','').replace('\n',' ')},获取失败\n")
内容的提问来源于stack exchange,提问作者Scijens
相关产品推荐
相关产品推荐

