You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何解析含列表索引的嵌套键,实现扁平化字典转嵌套结构

扁平化字典转嵌套结构(支持列表索引解析)

问题概述

现有convert函数仅能处理.分隔的简单键解析,无法识别a[0][0]这类带列表索引的键格式,导致索引部分被当作普通字典键保留,无法生成符合预期的嵌套列表结构,同时还需处理e-f这类键转成e的命名转换。

输入示例

input_data = {
    "a.b.c[0][0].d": "i",
    "a.b.c[0][1].e-f": "j",
    "a.b.c[0][2].g-h[0]": "x",
    "a.b.c[0][3].g-h[1]": "y",
    "a.b.c[0][4].g-h[2]": "z",
    "x": [
        {
            "b": {
                "x.y.z": "a"
            },
        },
        {
            "b": {
                "d.f.a": "b"
            },
        },
        {
            "b": {
                "g.h.q": "c"
            },
        },
    ],
}

期望输出

output = {
    "a": {
        "b": {
            "c": [
                [
                    {
                        "d": "i",
                        "e": "j",
                        "g": ["x", "y", "z"]
                    }
                ]
            ]
        }
    },
    "x": [
        {
            "b": {
                "x": {"y": {"z": "a"}}
            }
        },
        {
            "b": {
                "d": {"f": {"a": "b"}}
            }
        },
        {
            "b": {
                "g": {"h": {"q": "c"}}
            }
        }
    ]
}

当前代码

def convert(data: dict):
    if isinstance(data, dict):
        new_dict = {}
        for key, value in data.items():
            if not key or key is None or key == '':
                continue
            parts = key.split(".") if "." in key else key.split("_")
            current_dict = new_dict
            for part in parts[:-1]:
                current_dict.setdefault(part, {})
                current_dict = current_dict[part]
            if isinstance(value, dict):
                current_dict[parts[-1]] = convert(value)
            elif isinstance(value, list):
                new_list = []
                for item in value:
                    if isinstance(item, dict):
                        new_list.append(convert(item))
                    elif isinstance(item, str) and "," in item:
                        new_list.extend(item.split(","))
                    else:
                        new_list.append(item)
                current_dict[parts[-1]] = new_list
            else:
                current_dict[parts[-1]] = value
        return new_dict
    else:
        return data

print(convert(input_data))

当前错误输出

{'a': {'b': {'c[0][0]': {'d': 'i'}, 'c[0][1]': {'e-f': 'j'}, 'c[0][2]': {'g-h[0]': 'x'}, 'c[0][3]': {'g-h[1]': 'y'}, 'c[0][4]': {'g-h[2]': 'z'}}}, 'x': [{'b': {'x': {'y': {'z': 'a'}}}}, {'b': {'d': {'f': {'a': 'b'}}}}, {'b': {'g': {'h': {'q': 'c'}}}}]}

解决方案

修改后的代码增加了键段解析逻辑,能识别列表索引并生成对应层级的列表,同时处理了e-f这类键的命名转换:

import re
from typing import Dict, Any, List, Tuple

def parse_key_part(part: str) -> Tuple[str, List[int]]:
    # 拆分键名和索引部分,比如 c[0][0] -> ('c', [0,0]), g-h[0] -> ('g', [0])
    match = re.match(r'^([^\[]+)(\[(\d+)\])*$', part)
    if not match:
        return part, []
    key_name = match.group(1)
    # 处理e-f转成e,g-h转成g的情况
    key_name = key_name.split('-')[0]
    # 提取所有索引
    index_matches = re.findall(r'\[(\d+)\]', part)
    indices = [int(idx) for idx in index_matches]
    return key_name, indices

def convert(data: Dict[str, Any]) -> Dict[str, Any]:
    if isinstance(data, dict):
        new_dict = {}
        for key, value in data.items():
            if not key or key is None:
                continue
            # 先按.拆分键的各个部分
            parts = key.split('.')
            current = new_dict
            # 处理除最后一个部分外的所有键段
            for part in parts[:-1]:
                key_name, indices = parse_key_part(part)
                # 先进入字典层级
                if key_name not in current:
                    current[key_name] = {}
                current = current[key_name]
                # 处理每个列表索引
                for idx in indices:
                    # 如果当前不是列表,或者列表长度不够,扩展列表并填充空字典
                    if not isinstance(current, list):
                        current = [{}]
                    while len(current) <= idx:
                        current.append({})
                    current = current[idx]
            # 处理最后一个键段
            last_part = parts[-1]
            key_name, indices = parse_key_part(last_part)
            # 递归处理值
            processed_value = convert(value) if isinstance(value, (dict, list)) else value
            if indices:
                # 如果当前是字典,先创建对应的键并初始化为列表
                if isinstance(current, dict):
                    if key_name not in current:
                        current[key_name] = []
                    current = current[key_name]
                # 处理每个索引,确保列表长度足够
                for idx in indices[:-1]:
                    while len(current) <= idx:
                        current.append([])
                    current = current[idx]
                # 处理最后一个索引,设置值
                final_idx = indices[-1]
                while len(current) <= final_idx:
                    current.append(None)
                current[final_idx] = processed_value
            else:
                # 没有索引,直接设置字典键值
                current[key_name] = processed_value
        return new_dict
    elif isinstance(data, list):
        # 处理列表中的每个元素
        return [convert(item) if isinstance(item, (dict, list)) else item for item in data]
    else:
        return data

# 测试
input_data = {
    "a.b.c[0][0].d": "i",
    "a.b.c[0][1].e-f": "j",
    "a.b.c[0][2].g-h[0]": "x",
    "a.b.c[0][3].g-h[1]": "y",
    "a.b.c[0][4].g-h[2]": "z",
    "x": [
        {
            "b": {
                "x.y.z": "a"
            },
        },
        {
            "b": {
                "d.f.a": "b"
            },
        },
        {
            "b": {
                "g.h.q": "c"
            },
        },
    ],
}

print(convert(input_data))

关键改动说明

  1. 键段解析函数parse_key_part:用正则提取键名和列表索引,同时将e-f这类键转换成e,符合期望输出的命名规则。
  2. 列表索引处理逻辑:遍历每个索引时,动态扩展列表长度以确保索引位置存在,自动填充空字典或列表作为占位符。
  3. 递归处理增强:不仅递归处理字典,还对列表中的每个元素进行递归解析,确保嵌套在列表中的扁平化键也能被正确转换。
  4. 动态类型适配:根据当前层级的类型(字典/列表)自动切换处理逻辑,确保生成的结构符合预期。

内容的提问来源于stack exchange,提问作者Pranaya Behera

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.23 16:28:12