Python递归将嵌套XML转字典列表时如何处理重复标签
使用Python lxml解析嵌套XML并处理重复标签
问题背景
现有如下字符串格式的XML数据,需要使用Python的lxml包解析并遍历,生成指定格式的输出:
更新:已更新代码和输出内容
<A xmlns="dfjdlfkdjflsd"> <B> <B> <B1>B_1</B1> <B2>B_2</B2> <B3> <B31>B3_1</B31> <B32>B3_2</B32> <B33> <B331> <B3311></B3311> </B331> <B332> <B3321></B3321> </B332> </B33> <B34> <B341> <B3411></B3411> </B341> <B342> <B3421></B3421> </B342> </B34> <B35> <B351>B35_1</B351> <B352> <B3521>B352_1</B3521> <B3522>B352_2</B3522> <B3523>B352_3</B3523> <B3524> <B35241> <B352411></B352411> <B352412></B352412> <B352413></B352413> </B35241> </B3524> </B352> <B352> <B3521>B352_4</B3521> <B3522>B352_5</B3522> <B3523>B352_6</B3523> <B3524> <B35241> <B352411></B352411> <B352412></B352412> <B352413></B352413> </B35241> </B3524> </B352> <B352> <B3521>B352_7</B3521> <B3522>B352_8</B3522> <B3523>B352_9</B3523> <B3524> <B35241> <B352411></B352411> <B352412></B352412> <B352413></B352413> </B35241> </B3524> </B352> </B35> <B36> <B361>B36_1</B361> <B362>B36_2</B362> </B36> </B3> </B> </B> <C> <C1>B_1</C1> <C2>B_2</C2> <C3> <C31>C3_1</C31> <C32>C3_2</C32> <C33> <C331> <C3311></C3311> </C331> <C332> <C3321></C3321> </C332> </C33> </C3> </C> </A>
预期输出格式
[{'B1': 'B_1', 'B2': 'B_2', 'B3_B31': 'B3_1', 'B3_B32': 'B3_2', 'B3_B33_B331_B3311': '-', 'B3_B33_B332_B3321': '-', 'B3_B34_B341_B3411': '-', 'B3_B34_B342_B3421': '-', 'B3_B35_B352': [ { 'B3_B35_B352_B3521': 'B352_1', 'B3_B35_B352_B3522': 'B352_2', 'B3_B35_B352_B3523': 'B352_3', 'B3_B35_B352_B3524_B35241_B352411': '-', 'B3_B35_B352_B3524_B35241_B352412': '-', 'B3_B35_B352_B3524_B35241_B352413': '-' }, { 'B3_B35_B352_B3521': 'B352_4', 'B3_B35_B352_B3522': 'B352_5', 'B3_B35_B352_B3523': 'B352_6', 'B3_B35_B352_B3524_B35241_B352411': '-', 'B3_B35_B352_B3524_B35241_B352412': '-', 'B3_B35_B352_B3524_B35241_B352413': '-' }, { 'B3_B35_B352_B3521': 'B352_7', 'B3_B35_B352_B3522': 'B352_8', 'B3_B35_B352_B3523': 'B352_9', 'B3_B35_B352_B3524_B35241_B352411': '-', 'B3_B35_B352_B3524_B35241_B352412': '-', 'B3_B35_B352_B3524_B35241_B352413': '-' } ], 'B3_B36_B361': 'B36_1', 'B3_B36_B362': 'B36_2'}, {'C1': 'B_1', 'C2': 'B_2', 'C3_C31': 'C3_1', 'C3_C32': 'C3_2', 'C3_C33_C331_C3311': '-', 'C3_C33_C332_C3321': '-'}]
现有问题
原有代码已经可以实现普通嵌套XML的遍历输出,但无法正确处理重复XML标签的场景:
- 重复标签的键名不符合要求,统一生成了
duplicate键而不是用重复标签的路径作为键 - 重复列表中的标签内容存在混淆
现有代码逻辑使用单独的_handle_duplicates方法处理重复标签,希望可以直接在_flatten方法中完成重复标签处理,无需额外单独实现方法。
原有代码
class ParseXML: def __init__(self, xml_input): self.main_output = [] parser = et.XMLParser(recover=True) self.tree = et.fromstring(re.sub('\s*xmlns(:\w+)?="[^"]*"', '', xml_input), parser=parser) def parse_xml(self): for interface in list(self.tree): temp_output = {} for children in interface: temp_list = [] temp_dict = {} for key, value in self._flatten(children): if key in temp_output: if key in temp_dict: temp_list.append(temp_dict) temp_dict = {} temp_dict.update({key: value}) else: temp_output.update({key: value}) temp = self._handle_duplicates(temp_output, temp_dict, temp_list) if temp_dict else temp_output self.main_output.append(temp) return self.main_output def _flatten(self, node, tags=None): if tags is None: tags = [] children = list(node) if not children: if node.text is None: yield '_'.join(tags + [node.tag]), '-' else: yield '_'.join(tags + [node.tag]), node.text else: for child in children: for key_val in self._flatten(child, tags + [node.tag]): yield key_val def _handle_duplicates(self, temp_output, temp_dict, temp_list): temp_list.append(temp_dict) temp = {} for dup in temp_dict: temp.update({dup: temp_output.pop(dup)}) temp_list.append(temp) temp_output.update({'duplicate': temp_list}) return temp_output if __name__ == '__main__': parse = ParseXML(data) output = parse.parse_xml() pprint(output)
解决方案
修改思路
- 遍历节点前先统计子标签的出现次数,识别出重复标签
- 非重复标签正常递归扁平化,生成路径拼接的键值对直接合并到结果
- 重复标签逐个处理,每个标签的扁平化结果作为独立字典存入列表,以「父路径+重复标签名」作为列表的键
完整实现代码
from lxml import etree as et import re from pprint import pprint class ParseXML: def __init__(self, xml_input): self.main_output = [] parser = et.XMLParser(recover=True) # 移除命名空间 clean_xml = re.sub(r'\s*xmlns(:\w+)?="[^"]*"', '', xml_input) self.tree = et.fromstring(clean_xml, parser=parser) def parse_xml(self): # 遍历根节点A的直接子节点(B、C节点) for root_child in self.tree: # 处理B/C节点下的实际内容节点 for content_node in list(root_child): self.main_output.append(self._flatten(content_node)) return self.main_output def _flatten(self, node, parent_tags=None): if parent_tags is None: parent_tags = [] children = list(node) # 无子节点直接返回键值对 if not children: node_text = node.text.strip() if node.text and node.text.strip() else '-' return {'_'.join(parent_tags + [node.tag]): node_text} # 统计子标签出现次数,识别重复标签 tag_count = {} for child in children: tag_count[child.tag] = tag_count.get(child.tag, 0) + 1 result = {} for child in children: child_tag = child.tag # 非重复标签直接递归合并结果 if tag_count[child_tag] == 1: child_res = self._flatten(child, parent_tags) result.update(child_res) # 重复标签存入列表,键为父路径+重复标签名 else: list_key = '_'.join(parent_tags + [child_tag]) if list_key not in result: result[list_key] = [] # 递归处理每个重复子节点,路径追加当前重复标签名 child_res = self._flatten(child, parent_tags + [child_tag]) result[list_key].append(child_res) return result if __name__ == '__main__': # 此处data为上述XML字符串 parse = ParseXML(data) output = parse.parse_xml() pprint(output)
运行上述代码即可完全匹配预期输出格式。
内容的提问来源于stack exchange,提问作者Tony Montana
相关产品推荐
相关产品推荐

