You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何为网页爬虫程序搭建支持文件导入导出的Tkinter GUI界面?

实现思路
  • 使用Tkinter内置的文件选择对话框替代手动输入路径,自动筛选csv/xlsx格式的输入文件,避免路径拼写错误
  • 支持用户自定义输出文件的保存路径和格式,无需在代码里硬编码输出文件名
  • 增加运行状态提示,用户可以直观看到当前程序运行进度
  • 爬虫逻辑和GUI代码完全分离,后续修改爬虫规则或者新增其他新闻源的爬取函数都不用改GUI部分的代码
可直接运行的完整代码
import tkinter as tk
from tkinter import filedialog, messagebox
import pandas as pd
from urllib.request import Request, urlopen
from bs4 import BeautifulSoup as bs
from newspaper import Article, Config
import os

# ---------------------- 爬虫逻辑(基于原有代码做了适配修改)----------------------
# 注意:你原有代码中的 column_names、headers、url_category 三个变量直接替换下方占位内容即可
column_names = ['URL', 'Article Date', 'Article Title', 'Author', 'Source', 'Type', 'Text', 'Category']
headers = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/92.0.4515.107 Safari/537.36'}
url_category = {} # 保留你原来的分类映射字典

def atime_scrape(asia_times, output_path):
    # create dataframe 
    atime = pd.DataFrame(columns = column_names)
    # pass url list to URL column
    atime['URL'] = asia_times
    # create dictionaries 
    atime_date = {}
    atime_title = {}
    atime_auth = {}
    atime_type = {}
    atime_corpus = {}
    atime_summary = {}
    atime_category = {}

    total = len(atime['URL'])
    for idx, i in enumerate(atime['URL']):
        # 更新运行状态
        status_label.config(text=f"正在爬取第{idx+1}/{total}条:{i[:50]}...")
        root.update()

        # general
        req = Request(i, headers=headers)    # make the request 
        page = urlopen(req).read()           # get the response
        soup = bs(page, 'html.parser')       # parse the response into a bs object

        # date
        for x in soup.findAll('meta', {'property':'article:published_time'}):  
            atime_date[i] = x['content'].split('T',1)[0]

        # title
        for x in soup.findAll('meta', {'property':'og:title'}):
            atime_title[i] = x['content']

        # author 
        for x in soup.findAll('meta',  {'name':'twitter:data1'}):
            atime_auth[i] = x['content']

        # type
        for x in soup.findAll('meta', {'property':'og:type'}):
            atime_type[i] = x['content']

        # text         
        user_agent = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/92.0.4515.107 Safari/537.36'
        config = Config()
        config.browser_user_agent = user_agent
        page = Article(i, config=config)
        page.download()
        page.parse()
        atime_corpus[i] = page.text.replace('\xa0',' ').replace('\n',' ')

        # category   
        for k, v in url_category.items():
            if str(i) == str(k):
                atime_category[i] = v

    # map data by URL to dataframe
    atime['Article Date'] = atime['URL'].map(atime_date)
    atime['Article Title'] = atime['URL'].map(atime_title)
    atime['Author'] = atime['URL'].map(atime_auth)
    atime['Source'] = 'Asia Times'
    atime['Type'] = atime['URL'].map(atime_type)
    atime['Text'] = atime['URL'].map(atime_corpus)
    atime['Category'] = atime['URL'].map(atime_category)

    # 根据输出路径后缀自动判断保存格式
    if output_path.endswith('.csv'):
        atime.to_csv(output_path, index=False, encoding='utf-8-sig')
    elif output_path.endswith('.xlsx'):
        atime.to_excel(output_path, index=False)
    return True

# ---------------------- GUI 逻辑 ----------------------
def select_input_file():
    path = filedialog.askopenfilename(
        title="选择存储URL的文件",
        filetypes=[("CSV文件", "*.csv"), ("Excel文件", "*.xlsx *.xls")]
    )
    input_entry.delete(0, tk.END)
    input_entry.insert(0, path)

def select_output_file():
    path = filedialog.asksaveasfilename(
        title="选择结果保存位置",
        defaultextension=".csv",
        filetypes=[("CSV文件", "*.csv"), ("Excel文件", "*.xlsx")]
    )
    output_entry.delete(0, tk.END)
    output_entry.insert(0, path)

def run_scrape():
    input_path = input_entry.get().strip()
    output_path = output_entry.get().strip()
    if not input_path or not os.path.exists(input_path):
        messagebox.showerror("错误", "请选择有效的输入文件")
        return
    if not output_path:
        messagebox.showerror("错误", "请选择结果保存路径")
        return
    
    try:
        # 读取输入文件的URL列表
        if input_path.endswith('.csv'):
            df = pd.read_csv(input_path)
        else:
            df = pd.read_excel(input_path)
        # 若输入文件的URL列名不是URL,修改下方对应的列名即可
        url_list = df['URL'].dropna().tolist()
        if not url_list:
            messagebox.showerror("错误", "输入文件中没有找到有效URL")
            return
        
        # 执行爬虫
        success = atime_scrape(url_list, output_path)
        if success:
            status_label.config(text="爬取完成")
            messagebox.showinfo("成功", f"爬取完成,结果已保存到:{output_path}")
    except Exception as e:
        messagebox.showerror("运行错误", f"程序运行出错:{str(e)}")

# 初始化主窗口
root = tk.Tk()
root.title("新闻URL爬虫工具")
root.geometry("600x250")
root.resizable(False, False)

# 输入文件选择组件
tk.Label(root, text="输入URL文件:").place(x=20, y=20)
input_entry = tk.Entry(root, width=50)
input_entry.place(x=120, y=20)
tk.Button(root, text="选择文件", command=select_input_file).place(x=480, y=18)

# 输出文件选择组件
tk.Label(root, text="输出结果文件:").place(x=20, y=70)
output_entry = tk.Entry(root, width=50)
output_entry.place(x=120, y=70)
tk.Button(root, text="选择路径", command=select_output_file).place(x=480, y=68)

# 执行按钮
run_btn = tk.Button(root, text="开始爬取", command=run_scrape, bg="#4CAF50", fg="white", font=("微软雅黑", 12), width=15)
run_btn.place(x=230, y=120)

# 状态提示
status_label = tk.Label(root, text="等待操作", fg="gray")
status_label.place(x=20, y=190)

root.mainloop()
注意事项
  • 运行前先安装所有依赖:pip install pandas beautifulsoup4 newspaper3k openpyxl
  • 如果爬取的URL数量较多,可以把爬虫逻辑放到子线程中运行,避免GUI界面暂时无响应
  • 你原有代码中的分类映射url_category、请求头headers、列名column_names直接替换代码中对应的占位部分即可

内容的提问来源于stack exchange,提问作者Quanty

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.10.04 21:18:01