You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python获取重定向URL时Chrome历史日志未及时更新的问题求助

解决Chrome历史文件未及时更新导致重定向URL重复的问题

我使用Shahin Shirazi的代码批量获取重定向链接的目标URL,但运行时部分场景会重复写入上一个重定向的目标URL,显然Chrome历史文件未及时更新,添加延时处理后问题依旧,相关代码如下:

from bs4 import BeautifulSoup as soup
import webbrowser
import sqlite3
import pandas as pd
import shutil
import time
import os

# ---------------------FILEREADING------------------------------------------

colnames = ['Column1']
Filename_Link = "reflinks.csv"
data = pd.read_csv(Filename_Link, names=colnames)
links = data.Column1.tolist()

# ---------------------Create a new .csv file that we are going to fill with the target urls------------------------------------------

output_data = "output_data.csv"
fd = open(output_data, "a",encoding='utf-8-sig')
topline = "url, target_url,"
fd.write(topline)
fd.write("\n")         

# ---------------------Define the history folder of the browser ------------------------------------------

#source file is where the history of your webbroser is saved, I was using chrome, but it should be the same process if you are using different browser
source_file = 'C:\\Users\\xxx\\AppData\\Local\\Google\\Chrome\\User Data\\Default\\History'
# could not directly connect to history file as it was locked and had to make a copy of it in different location
destination_file = 'C:\\Users\\xxx\\Downloads\\History'
    

# ---------------------Run the code to get target urls for all redirect links ------------------------------------------

for link in links:
    webbrowser.open(link)
    time.sleep(30) # there is some delay to update the history file, so 30 sec wait give it enough time to make sure your last url get logged
    os.system("taskkill /im chrome.exe /f")    
    shutil.copy(source_file,destination_file) # copying the file.
    time.sleep(10) # I added some delay to copy the files
    con = sqlite3.connect('C:\\Users\\xxx\\Downloads\\History') # connecting to browser history
    cursor = con.execute("SELECT * FROM urls")
    names = [description[0] for description in cursor.description]
    urls = cursor.fetchall()
    con.close()
    df_history = pd.DataFrame(urls,columns=names)
    last_url = df_history.loc[len(df_history)-1,'url']
    print(last_url)
    
    fd.write(link.replace(",","").replace("\n"," "))
    fd.write(",")
    
    fd.write(last_url.replace(",","").replace("\n"," "))
    fd.write(",")
   
       
    fd.write("\n")
    
fd.close()

解决方案

1. 改用Chrome DevTools协议直接获取最终URL(最可靠)

放弃依赖Chrome历史文件,直接通过Selenium控制Chrome实例,实时捕获页面跳转后的最终URL,彻底避免历史文件更新延迟问题:

from selenium import webdriver
from selenium.webdriver.chrome.options import Options
import pandas as pd
import time

# 读取待处理链接
colnames = ['Column1']
Filename_Link = "reflinks.csv"
data = pd.read_csv(Filename_Link, names=colnames)
links = data.Column1.tolist()

# 配置Chrome无头模式(可选,也可以去掉用可视化窗口)
chrome_options = Options()
chrome_options.add_argument("--headless=new")
chrome_options.add_argument("--disable-gpu")
chrome_options.add_argument("--no-sandbox")

# 初始化浏览器
driver = webdriver.Chrome(options=chrome_options)
driver.set_page_load_timeout(30)  # 设置页面加载超时时间

# 写入输出文件表头
output_data = "output_data.csv"
with open(output_data, "w", encoding='utf-8-sig') as fd:
    fd.write("url,target_url\n")

    for link in links:
        try:
            driver.get(link)
            time.sleep(2)  # 等待JS跳转完成,可根据实际情况调整时长
            final_url = driver.current_url
            print(f"{link} -> {final_url}")
            
            # 写入结果到CSV
            fd.write(f"{link.replace(',','').replace('\n',' ')},{final_url.replace(',','').replace('\n',' ')}\n")
        except Exception as e:
            print(f"处理{link}失败: {str(e)}")
            fd.write(f"{link.replace(',','').replace('\n',' ')},获取失败\n")

# 关闭浏览器
driver.quit()

2. 优化历史文件读取逻辑

如果坚持使用历史文件方案,需解决两个核心问题:确保Chrome完全退出、正确获取最新访问记录:

from bs4 import BeautifulSoup as soup
import webbrowser
import sqlite3
import pandas as pd
import shutil
import time
import os
import psutil

# ---------------------FILEREADING------------------------------------------

colnames = ['Column1']
Filename_Link = "reflinks.csv"
data = pd.read_csv(Filename_Link, names=colnames)
links = data.Column1.tolist()

# ---------------------Create output file------------------------------------------

output_data = "output_data.csv"
with open(output_data, "w", encoding='utf-8-sig') as fd:
    fd.write("url,target_url\n")         

# ---------------------Define history paths------------------------------------------

source_file = 'C:\\Users\\xxx\\AppData\\Local\\Google\\Chrome\\User Data\\Default\\History'
destination_file = 'C:\\Users\\xxx\\Downloads\\History'
    

# ---------------------Process each link------------------------------------------

for link in links:
    webbrowser.open(link)
    time.sleep(30)  # 等待页面跳转完成
    
    # 强制结束所有Chrome进程,确保历史文件写入磁盘
    for proc in psutil.process_iter(['name']):
        if proc.info['name'] == 'chrome.exe':
            try:
                proc.kill()
            except:
                pass
    time.sleep(15)  # 等待进程完全退出
    
    # 复制历史文件
    shutil.copy(source_file, destination_file)
    time.sleep(2)
    
    # 连接数据库,按访问时间取最新记录
    con = sqlite3.connect(destination_file)
    # last_visit_time是Chrome历史表的时间戳字段,降序排序取第一条
    cursor = con.execute("SELECT url FROM urls ORDER BY last_visit_time DESC LIMIT 1")
    last_url = cursor.fetchone()[0]
    con.close()
    
    print(last_url)
    
    # 写入结果
    with open(output_data, "a", encoding='utf-8-sig') as fd:
        fd.write(f"{link.replace(',','').replace('\n',' ')},{last_url.replace(',','').replace('\n',' ')}\n")

3. 使用requests库直接追踪重定向(无浏览器,效率最高)

如果目标链接仅涉及HTTP/HTTPS协议重定向(无JavaScript跳转),可以直接用requests库追踪重定向,无需启动浏览器:

import requests
import pandas as pd

# 读取链接列表
colnames = ['Column1']
Filename_Link = "reflinks.csv"
data = pd.read_csv(Filename_Link, names=colnames)
links = data.Column1.tolist()

# 写入结果
output_data = "output_data.csv"
with open(output_data, "w", encoding='utf-8-sig') as fd:
    fd.write("url,target_url\n")
    
    for link in links:
        try:
            # allow_redirects=True自动追踪所有重定向
            response = requests.get(link, allow_redirects=True, timeout=10)
            final_url = response.url
            print(f"{link} -> {final_url}")
            fd.write(f"{link.replace(',','').replace('\n',' ')},{final_url.replace(',','').replace('\n',' ')}\n")
        except Exception as e:
            print(f"处理{link}失败: {str(e)}")
            fd.write(f"{link.replace(',','').replace('\n',' ')},获取失败\n")

内容的提问来源于stack exchange,提问作者Scijens

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.01 08:35:41