You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python递归将嵌套XML转字典列表时如何处理重复标签

使用Python lxml解析嵌套XML并处理重复标签

问题背景

现有如下字符串格式的XML数据,需要使用Python的lxml包解析并遍历,生成指定格式的输出:
更新:已更新代码和输出内容

<A xmlns="dfjdlfkdjflsd">
  <B>
    <B>
      <B1>B_1</B1>
      <B2>B_2</B2>
      <B3>
        <B31>B3_1</B31>
        <B32>B3_2</B32>
        <B33>
          <B331>
            <B3311></B3311>
          </B331>
          <B332>
            <B3321></B3321>
          </B332>
        </B33>
        <B34>
          <B341>
            <B3411></B3411>
          </B341>
          <B342>
            <B3421></B3421>
          </B342>
        </B34>
        <B35>
          <B351>B35_1</B351>
          <B352>
            <B3521>B352_1</B3521>
            <B3522>B352_2</B3522>
            <B3523>B352_3</B3523>
            <B3524>
              <B35241>
                <B352411></B352411>
                <B352412></B352412>
                <B352413></B352413>
              </B35241>
            </B3524>
          </B352>
          <B352>
            <B3521>B352_4</B3521>
            <B3522>B352_5</B3522>
            <B3523>B352_6</B3523>
            <B3524>
              <B35241>
                <B352411></B352411>
                <B352412></B352412>
                <B352413></B352413>
              </B35241>
            </B3524>
          </B352>
          <B352>
            <B3521>B352_7</B3521>
            <B3522>B352_8</B3522>
            <B3523>B352_9</B3523>
            <B3524>
              <B35241>
                <B352411></B352411>
                <B352412></B352412>
                <B352413></B352413>
              </B35241>
            </B3524>
          </B352>
        </B35>
        <B36>
          <B361>B36_1</B361>
          <B362>B36_2</B362>
        </B36>
      </B3>
    </B>
  </B>
  <C>
    <C1>B_1</C1>
    <C2>B_2</C2>
    <C3>
      <C31>C3_1</C31>
      <C32>C3_2</C32>
      <C33>
        <C331>
          <C3311></C3311>
        </C331>
        <C332>
          <C3321></C3321>
        </C332>
      </C33>
    </C3>
  </C>
</A>

预期输出格式

[{'B1': 'B_1',
    'B2': 'B_2',
    'B3_B31': 'B3_1',
    'B3_B32': 'B3_2',
    'B3_B33_B331_B3311': '-',
    'B3_B33_B332_B3321': '-',
    'B3_B34_B341_B3411': '-',
    'B3_B34_B342_B3421': '-',
    'B3_B35_B352': [
        {
            'B3_B35_B352_B3521': 'B352_1',
            'B3_B35_B352_B3522': 'B352_2',
            'B3_B35_B352_B3523': 'B352_3',
            'B3_B35_B352_B3524_B35241_B352411': '-',
            'B3_B35_B352_B3524_B35241_B352412': '-',
            'B3_B35_B352_B3524_B35241_B352413': '-'
        },
        {
            'B3_B35_B352_B3521': 'B352_4',
            'B3_B35_B352_B3522': 'B352_5',
            'B3_B35_B352_B3523': 'B352_6',
            'B3_B35_B352_B3524_B35241_B352411': '-',
            'B3_B35_B352_B3524_B35241_B352412': '-',
            'B3_B35_B352_B3524_B35241_B352413': '-'
        },
        {
            'B3_B35_B352_B3521': 'B352_7',
            'B3_B35_B352_B3522': 'B352_8',
            'B3_B35_B352_B3523': 'B352_9',
            'B3_B35_B352_B3524_B35241_B352411': '-',
            'B3_B35_B352_B3524_B35241_B352412': '-',
            'B3_B35_B352_B3524_B35241_B352413': '-'
        }
    ],
    'B3_B36_B361': 'B36_1',
    'B3_B36_B362': 'B36_2'},
   {'C1': 'B_1',
    'C2': 'B_2',
    'C3_C31': 'C3_1',
    'C3_C32': 'C3_2',
    'C3_C33_C331_C3311': '-',
    'C3_C33_C332_C3321': '-'}]

现有问题

原有代码已经可以实现普通嵌套XML的遍历输出,但无法正确处理重复XML标签的场景:

  • 重复标签的键名不符合要求,统一生成了duplicate键而不是用重复标签的路径作为键
  • 重复列表中的标签内容存在混淆
    现有代码逻辑使用单独的_handle_duplicates方法处理重复标签,希望可以直接在_flatten方法中完成重复标签处理,无需额外单独实现方法。

原有代码

class ParseXML:
    def __init__(self, xml_input):
        self.main_output = []
        parser = et.XMLParser(recover=True)
        self.tree = et.fromstring(re.sub('\s*xmlns(:\w+)?="[^"]*"', '', xml_input), parser=parser)
    def parse_xml(self):
        for interface in list(self.tree):
            temp_output = {}
            for children in interface:
                temp_list = []
                temp_dict = {}
                for key, value in self._flatten(children):
                    if key in temp_output:
                        if key in temp_dict:
                            temp_list.append(temp_dict)
                            temp_dict = {}
                        temp_dict.update({key: value})
                    else:
                        temp_output.update({key: value})
                temp = self._handle_duplicates(temp_output, temp_dict, temp_list) if temp_dict else temp_output
            self.main_output.append(temp)
        return self.main_output
    def _flatten(self, node, tags=None):
        if tags is None:
            tags = []
        children = list(node)
        if not children:
            if node.text is None:
                yield '_'.join(tags + [node.tag]), '-'
            else:
                yield '_'.join(tags + [node.tag]), node.text
        else:
            for child in children:
                for key_val in self._flatten(child, tags + [node.tag]):
                    yield key_val
    def _handle_duplicates(self, temp_output, temp_dict, temp_list):
        temp_list.append(temp_dict)
        temp = {}
        for dup in temp_dict:
            temp.update({dup: temp_output.pop(dup)})
        temp_list.append(temp)
        temp_output.update({'duplicate': temp_list})
        return temp_output
if __name__ == '__main__':
    parse = ParseXML(data)
    output = parse.parse_xml()
    pprint(output)

解决方案

修改思路

  1. 遍历节点前先统计子标签的出现次数,识别出重复标签
  2. 非重复标签正常递归扁平化,生成路径拼接的键值对直接合并到结果
  3. 重复标签逐个处理,每个标签的扁平化结果作为独立字典存入列表,以「父路径+重复标签名」作为列表的键

完整实现代码

from lxml import etree as et
import re
from pprint import pprint
class ParseXML:
    def __init__(self, xml_input):
        self.main_output = []
        parser = et.XMLParser(recover=True)
        # 移除命名空间
        clean_xml = re.sub(r'\s*xmlns(:\w+)?="[^"]*"', '', xml_input)
        self.tree = et.fromstring(clean_xml, parser=parser)
    def parse_xml(self):
        # 遍历根节点A的直接子节点(B、C节点)
        for root_child in self.tree:
            # 处理B/C节点下的实际内容节点
            for content_node in list(root_child):
                self.main_output.append(self._flatten(content_node))
        return self.main_output
    def _flatten(self, node, parent_tags=None):
        if parent_tags is None:
            parent_tags = []
        children = list(node)
        # 无子节点直接返回键值对
        if not children:
            node_text = node.text.strip() if node.text and node.text.strip() else '-'
            return {'_'.join(parent_tags + [node.tag]): node_text}
        # 统计子标签出现次数,识别重复标签
        tag_count = {}
        for child in children:
            tag_count[child.tag] = tag_count.get(child.tag, 0) + 1
        result = {}
        for child in children:
            child_tag = child.tag
            # 非重复标签直接递归合并结果
            if tag_count[child_tag] == 1:
                child_res = self._flatten(child, parent_tags)
                result.update(child_res)
            # 重复标签存入列表,键为父路径+重复标签名
            else:
                list_key = '_'.join(parent_tags + [child_tag])
                if list_key not in result:
                    result[list_key] = []
                # 递归处理每个重复子节点,路径追加当前重复标签名
                child_res = self._flatten(child, parent_tags + [child_tag])
                result[list_key].append(child_res)
        return result
if __name__ == '__main__':
    # 此处data为上述XML字符串
    parse = ParseXML(data)
    output = parse.parse_xml()
    pprint(output)

运行上述代码即可完全匹配预期输出格式。


内容的提问来源于stack exchange,提问作者Tony Montana

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.10.01 18:45:04