如何为网页爬虫程序搭建支持文件导入导出的Tkinter GUI界面?
实现思路
- 使用Tkinter内置的文件选择对话框替代手动输入路径,自动筛选csv/xlsx格式的输入文件,避免路径拼写错误
- 支持用户自定义输出文件的保存路径和格式,无需在代码里硬编码输出文件名
- 增加运行状态提示,用户可以直观看到当前程序运行进度
- 爬虫逻辑和GUI代码完全分离,后续修改爬虫规则或者新增其他新闻源的爬取函数都不用改GUI部分的代码
可直接运行的完整代码
import tkinter as tk from tkinter import filedialog, messagebox import pandas as pd from urllib.request import Request, urlopen from bs4 import BeautifulSoup as bs from newspaper import Article, Config import os # ---------------------- 爬虫逻辑(基于原有代码做了适配修改)---------------------- # 注意:你原有代码中的 column_names、headers、url_category 三个变量直接替换下方占位内容即可 column_names = ['URL', 'Article Date', 'Article Title', 'Author', 'Source', 'Type', 'Text', 'Category'] headers = {'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/92.0.4515.107 Safari/537.36'} url_category = {} # 保留你原来的分类映射字典 def atime_scrape(asia_times, output_path): # create dataframe atime = pd.DataFrame(columns = column_names) # pass url list to URL column atime['URL'] = asia_times # create dictionaries atime_date = {} atime_title = {} atime_auth = {} atime_type = {} atime_corpus = {} atime_summary = {} atime_category = {} total = len(atime['URL']) for idx, i in enumerate(atime['URL']): # 更新运行状态 status_label.config(text=f"正在爬取第{idx+1}/{total}条:{i[:50]}...") root.update() # general req = Request(i, headers=headers) # make the request page = urlopen(req).read() # get the response soup = bs(page, 'html.parser') # parse the response into a bs object # date for x in soup.findAll('meta', {'property':'article:published_time'}): atime_date[i] = x['content'].split('T',1)[0] # title for x in soup.findAll('meta', {'property':'og:title'}): atime_title[i] = x['content'] # author for x in soup.findAll('meta', {'name':'twitter:data1'}): atime_auth[i] = x['content'] # type for x in soup.findAll('meta', {'property':'og:type'}): atime_type[i] = x['content'] # text user_agent = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/92.0.4515.107 Safari/537.36' config = Config() config.browser_user_agent = user_agent page = Article(i, config=config) page.download() page.parse() atime_corpus[i] = page.text.replace('\xa0',' ').replace('\n',' ') # category for k, v in url_category.items(): if str(i) == str(k): atime_category[i] = v # map data by URL to dataframe atime['Article Date'] = atime['URL'].map(atime_date) atime['Article Title'] = atime['URL'].map(atime_title) atime['Author'] = atime['URL'].map(atime_auth) atime['Source'] = 'Asia Times' atime['Type'] = atime['URL'].map(atime_type) atime['Text'] = atime['URL'].map(atime_corpus) atime['Category'] = atime['URL'].map(atime_category) # 根据输出路径后缀自动判断保存格式 if output_path.endswith('.csv'): atime.to_csv(output_path, index=False, encoding='utf-8-sig') elif output_path.endswith('.xlsx'): atime.to_excel(output_path, index=False) return True # ---------------------- GUI 逻辑 ---------------------- def select_input_file(): path = filedialog.askopenfilename( title="选择存储URL的文件", filetypes=[("CSV文件", "*.csv"), ("Excel文件", "*.xlsx *.xls")] ) input_entry.delete(0, tk.END) input_entry.insert(0, path) def select_output_file(): path = filedialog.asksaveasfilename( title="选择结果保存位置", defaultextension=".csv", filetypes=[("CSV文件", "*.csv"), ("Excel文件", "*.xlsx")] ) output_entry.delete(0, tk.END) output_entry.insert(0, path) def run_scrape(): input_path = input_entry.get().strip() output_path = output_entry.get().strip() if not input_path or not os.path.exists(input_path): messagebox.showerror("错误", "请选择有效的输入文件") return if not output_path: messagebox.showerror("错误", "请选择结果保存路径") return try: # 读取输入文件的URL列表 if input_path.endswith('.csv'): df = pd.read_csv(input_path) else: df = pd.read_excel(input_path) # 若输入文件的URL列名不是URL,修改下方对应的列名即可 url_list = df['URL'].dropna().tolist() if not url_list: messagebox.showerror("错误", "输入文件中没有找到有效URL") return # 执行爬虫 success = atime_scrape(url_list, output_path) if success: status_label.config(text="爬取完成") messagebox.showinfo("成功", f"爬取完成,结果已保存到:{output_path}") except Exception as e: messagebox.showerror("运行错误", f"程序运行出错:{str(e)}") # 初始化主窗口 root = tk.Tk() root.title("新闻URL爬虫工具") root.geometry("600x250") root.resizable(False, False) # 输入文件选择组件 tk.Label(root, text="输入URL文件:").place(x=20, y=20) input_entry = tk.Entry(root, width=50) input_entry.place(x=120, y=20) tk.Button(root, text="选择文件", command=select_input_file).place(x=480, y=18) # 输出文件选择组件 tk.Label(root, text="输出结果文件:").place(x=20, y=70) output_entry = tk.Entry(root, width=50) output_entry.place(x=120, y=70) tk.Button(root, text="选择路径", command=select_output_file).place(x=480, y=68) # 执行按钮 run_btn = tk.Button(root, text="开始爬取", command=run_scrape, bg="#4CAF50", fg="white", font=("微软雅黑", 12), width=15) run_btn.place(x=230, y=120) # 状态提示 status_label = tk.Label(root, text="等待操作", fg="gray") status_label.place(x=20, y=190) root.mainloop()
注意事项
- 运行前先安装所有依赖:
pip install pandas beautifulsoup4 newspaper3k openpyxl - 如果爬取的URL数量较多,可以把爬虫逻辑放到子线程中运行,避免GUI界面暂时无响应
- 你原有代码中的分类映射
url_category、请求头headers、列名column_names直接替换代码中对应的占位部分即可
内容的提问来源于stack exchange,提问作者Quanty
相关产品推荐
相关产品推荐

