求助:实现遍历网页XML链接并批量追加数据至单个CSV文件
批量遍历XML链接并合并数据到CSV
修改后的完整代码
import typing import urllib.request import pandas as pd from bs4 import BeautifulSoup from pandas import DataFrame # 提取目标页面的XML链接并清理数据 theurl = "https://data.food.gov.uk/catalog/datasets/38dd8d6a-5ab1-4f50-b753-ab33288e3200" thepage = urllib.request.urlopen(theurl) soup = BeautifulSoup(thepage) project_href = [i['href'] for i in soup.find_all('a', href=True) if i['href'] != "#"] links = pd.DataFrame(project_href, columns=['Url']) # 保留原有的链接清理逻辑 n = 12 links.drop(index=links.index[:n], inplace=True) b = 15 links.drop(links.tail(b).index, inplace=True) links.drop([409], axis=0, inplace=True) # XML处理相关函数 def get_feed(url): """从指定URL抓取XML源""" try: response = urllib.request.urlopen(urllib.request.Request(url, headers={'User-Agent': 'Mozilla'})) source = BeautifulSoup(response, 'lxml-xml', from_encoding=response.info().get_param('charset')) return source except Exception as e: print(f"获取XML失败 {url}: {str(e)}") return None def get_elements(xml, item='EstablishmentDetail'): """提取XML中的字段名""" try: items = xml.find_all(item) if not items: return None elements = [element.name for element in items[0].find_all()] return elements except Exception as e: print(f"提取字段失败: {str(e)}") return None def feed_to_df(url, item='EstablishmentDetail'): """将单个XML链接转换为DataFrame""" xml = get_feed(url) if not xml: return None elements = get_elements(xml) if not isinstance(elements, typing.List): return None df = pd.DataFrame(columns=elements) items = xml.find_all(item) for item in items: row = {} for element in elements: elem = item.find(element) row[element] = elem.text if elem else '' df.loc[len(df)] = row # 替代已弃用的append方法,提升效率 return df # 遍历所有链接并追加到同一个CSV output_path = 'C:/FDSA_3.csv' first_write = True # 控制表头仅写入一次 for idx, url in enumerate(links['Url'], 1): print(f"处理第 {idx}/{len(links)} 个链接: {url}") df = feed_to_df(url) if df is not None and not df.empty: # 第一次写入含表头,后续追加不含表头 df.to_csv(output_path, mode='a', index=False, header=first_write) first_write = False else: print(f"链接 {url} 未提取到有效数据") print(f"所有链接处理完成,数据已保存至 {output_path}")
关键修改说明
- 替换行添加方式:用
df.loc[len(df)]替代df.append,避免使用已被pandas弃用的方法,同时提升数据写入效率 - 批量遍历逻辑:通过
enumerate遍历links中的所有URL,实时打印处理进度,方便跟踪 - 表头控制:用
first_write变量确保CSV文件仅在第一次写入时添加表头,避免重复表头问题 - 异常增强:在XML抓取和字段提取函数中添加针对性的错误提示,快速定位失败的链接
- 有效性判断:仅当提取到非空的有效DataFrame时才写入CSV,避免无效数据干扰
内容的提问来源于stack exchange,提问作者Jess
相关产品推荐
相关产品推荐

