You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

求助:实现遍历网页XML链接并批量追加数据至单个CSV文件

批量遍历XML链接并合并数据到CSV

修改后的完整代码

import typing
import urllib.request
import pandas as pd
from bs4 import BeautifulSoup
from pandas import DataFrame

# 提取目标页面的XML链接并清理数据
theurl = "https://data.food.gov.uk/catalog/datasets/38dd8d6a-5ab1-4f50-b753-ab33288e3200"
thepage = urllib.request.urlopen(theurl)
soup = BeautifulSoup(thepage)

project_href = [i['href'] for i in soup.find_all('a', href=True) if i['href'] != "#"]
links = pd.DataFrame(project_href, columns=['Url'])

# 保留原有的链接清理逻辑
n = 12
links.drop(index=links.index[:n], inplace=True)
b = 15
links.drop(links.tail(b).index, inplace=True)
links.drop([409], axis=0, inplace=True)

# XML处理相关函数
def get_feed(url):
    """从指定URL抓取XML源"""
    try:
        response = urllib.request.urlopen(urllib.request.Request(url, headers={'User-Agent': 'Mozilla'}))
        source = BeautifulSoup(response, 'lxml-xml', from_encoding=response.info().get_param('charset'))
        return source
    except Exception as e:
        print(f"获取XML失败 {url}: {str(e)}")
        return None

def get_elements(xml, item='EstablishmentDetail'):
    """提取XML中的字段名"""
    try:
        items = xml.find_all(item)
        if not items:
            return None
        elements = [element.name for element in items[0].find_all()]
        return elements
    except Exception as e:
        print(f"提取字段失败: {str(e)}")
        return None

def feed_to_df(url, item='EstablishmentDetail'):
    """将单个XML链接转换为DataFrame"""
    xml = get_feed(url)
    if not xml:
        return None
    elements = get_elements(xml)
    if not isinstance(elements, typing.List):
        return None
    
    df = pd.DataFrame(columns=elements)
    items = xml.find_all(item)
    
    for item in items:
        row = {}
        for element in elements:
            elem = item.find(element)
            row[element] = elem.text if elem else ''
        df.loc[len(df)] = row  # 替代已弃用的append方法,提升效率
    return df

# 遍历所有链接并追加到同一个CSV
output_path = 'C:/FDSA_3.csv'
first_write = True  # 控制表头仅写入一次

for idx, url in enumerate(links['Url'], 1):
    print(f"处理第 {idx}/{len(links)} 个链接: {url}")
    df = feed_to_df(url)
    if df is not None and not df.empty:
        # 第一次写入含表头,后续追加不含表头
        df.to_csv(output_path, mode='a', index=False, header=first_write)
        first_write = False
    else:
        print(f"链接 {url} 未提取到有效数据")

print(f"所有链接处理完成,数据已保存至 {output_path}")

关键修改说明

  • 替换行添加方式:用df.loc[len(df)]替代df.append,避免使用已被pandas弃用的方法,同时提升数据写入效率
  • 批量遍历逻辑:通过enumerate遍历links中的所有URL,实时打印处理进度,方便跟踪
  • 表头控制:用first_write变量确保CSV文件仅在第一次写入时添加表头,避免重复表头问题
  • 异常增强:在XML抓取和字段提取函数中添加针对性的错误提示,快速定位失败的链接
  • 有效性判断:仅当提取到非空的有效DataFrame时才写入CSV,避免无效数据干扰

内容的提问来源于stack exchange,提问作者Jess

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.20 11:31:06