You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用Python将特定格式的TXT文件转换为合法JSON?

缩进式键值对TXT转合法JSON的问题

输入TXT格式

我有一个采用4空格缩进的TXT文件,内容如下:

key1=value1
key2
    key2_1=value2_1
    key2_2
        key2_2_1=value2_2_1
    key2_3=value2_3_1,value2_3_2,value2_3_3
key3=value3_1,value3_2,value3_3

期望输出的合法JSON

希望转换为如下结构的合法JSON:

{
    "key1": "value1",
    "key2": {
        "key2_1": "value2_1",
        "key2_2": {
            "key2_2_1": "value2_2_1"
        },
        "key2_3": ["value2_3_1", "value2_3_2", "value2_3_3"]
    },
    "key3": ["value3_1", "value3_2", "value3_3"]
}

尝试的Python代码

我用了以下Python代码尝试转换:

import json

# helper method to convert equals sign to indentation for easier parsing
def convertIndentation(inputString):
    indentCount = 0
    indentVal = "    "
    for position, eachLine in enumerate(inputString):
        if "=" not in eachLine:
            continue
        else:
            strSplit = eachLine.split("=", 1)
            #get previous indentation
            prevIndent = inputString[position].count(indentVal)
            newVal = (indentVal * (prevIndent + 1)) + strSplit[1]
            inputString[position] = strSplit[0] + '\n'
            inputString.insert(position+1, newVal)
    flatList = "".join(inputString)
    return flatList

# helper class for node usage
class Node:
    def __init__(self, indented_line):
        self.children = []
        self.level = len(indented_line) - len(indented_line.lstrip())
        self.text = indented_line.strip()

    def add_children(self, nodes):
        childlevel = nodes[0].level

        while nodes:
            node = nodes.pop(0)
            if node.level == childlevel: # add node as a child
                self.children.append(node)
            elif node.level > childlevel: # add nodes as grandchildren of the last child
                nodes.insert(0,node)
                self.children[-1].add_children(nodes)
            elif node.level <= self.level: # this node is a sibling, no more children
                nodes.insert(0,node)
                return

    def as_dict(self):
        if len(self.children) > 1:
            return {self.text: [node.as_dict() for node in self.children]}
        elif len(self.children) == 1:
            return {self.text: self.children[0].as_dict()}
        else:
            return self.text

# process our file here
filename = "your_input_file.txt"  # 替换为实际文件名
with open(filename, 'r') as fh:
    fileContent = fh.readlines()
    fileParse = convertIndentation(fileContent)
    # convert equals signs to indentation
    root = Node('root')
    root.add_children([Node(line) for line in fileParse.splitlines() if line.strip()])
    d = root.as_dict()['root']
    # this variable is storing the json output
    jsonOutput = json.dumps(d, indent = 4, sort_keys = False)
    print(jsonOutput)

当前运行结果及报错

运行后得到的输出是:

[
    {
        "key1": "value1"
    },
    {
        "key2": [
            {
                "key2_1": "value2_1"
            },
            {
                "key2_2": {
                    "key2_2_1": "value2_2_1"
                }
            },
            {
                "key2_3": "value2_3_1,value2_3_2,value2_3_3"
            },
        ]
    },
    {
        "key3": "value3_1,value3_2,value3_3"
    }
]

用json.load读取输出文件时,出现以下错误:

with open(r'C:\Users\nigel\OneDrive\Documents\LAB\lean\sample_01.02_R00.json', 'r', encoding='utf-8') as read_file:
    data = json.load(read_file)

报错信息:

JSONDecodeError                           Traceback (most recent call last)
Input In [2], in <cell line: 1>()
      1 with open(r'C:\Users\nigel\OneDrive\Documents\LAB\lean\sample_01.02_R00.json', 'r', encoding='utf-8') as read_file:
----> 2     data = json.load(read_file)

File ~\Anaconda3\lib\json\__init__.py:293, in load(fp, cls, object_hook, parse_float, parse_int, parse_constant, object_pairs_hook, **kw)
    274 def load(fp, *, cls=None, object_hook=None, parse_float=None,
    275         parse_int=None, parse_constant=None, object_pairs_hook=None, **kw):
    276     """Deserialize ``fp`` (a ``.read()``-supporting file-like object containing
    277     a JSON document) to a Python object.
    278 
   (...)
    291     kwarg; otherwise ``JSONDecoder`` is used.
    292     """
---> 293     return loads(fp.read(),
    294         cls=cls, object_hook=object_hook,
    295         parse_float=parse_float, parse_int=parse_int,
    296         parse_constant=parse_constant, object_pairs_hook=object_pairs_hook, **kw)

File ~\Anaconda3\lib\json\__init__.py:346, in loads(s, cls, object_hook, parse_float, parse_int, parse_constant, object_pairs_hook, **kw)
    341     s = s.decode(detect_encoding(s), 'surrogatepass')
    343 if (cls is None and object_hook is None and
    344         parse_int is None and parse_float is None and
    345         parse_constant is None and object_pairs_hook is None and not kw):
---> 346     return _default_decoder.decode(s)
    347 if cls is None:
    348     cls = JSONDecoder

File ~\Anaconda3\lib\json\decoder.py:337, in JSONDecoder.decode(self, s, _w)
    332 def decode(self, s, _w=WHITESPACE.match):
    333     """Return the Python representation of ``s`` (a ``str`` instance
    334     containing a JSON document).
    335 
    336     """
---> 337     obj, end = self.raw_decode(s, idx=_w(s, 0).end())
    338     end = _w(s, end).end()
    339     if end != len(s):

File ~\Anaconda3\lib\json\decoder.py:353, in JSONDecoder.raw_decode(self, s, idx)
    344 """Decode a JSON document from ``s`` (a ``str`` beginning with
    345 a JSON document) and return a 2-tuple of the Python
    346 representation and the index in ``s`` where the document ended.
   (...)
    350 
    351 """
    352 try:
---> 353     obj, end = self.scan_once(s, idx)
    354 except StopIteration as err:
    355     raise JSONDecodeError("Expecting value", s, err.value) from None

JSONDecodeError: Expecting property name enclosed in double quotes: line 10 column 5 (char 165)

问题在于生成的输出是数组包裹多个单键字典,而非一个顶层字典,同时逗号分隔的值没有转换为数组,且存在JSON不允许的尾随逗号,导致解析失败。

解决方案

以下是修复后的代码,能正确处理缩进结构、键值对和数组转换:

import json

INDENT = "    "  # 4空格缩进

def parse_line(line):
    """解析单行,返回(缩进级别, 键, 值)"""
    stripped = line.strip()
    if not stripped:
        return None
    indent_level = (len(line) - len(line.lstrip())) // len(INDENT)
    if "=" in stripped:
        key, value = stripped.split("=", 1)
        # 处理数组值
        if "," in value:
            value = [v.strip() for v in value.split(",")]
        return indent_level, key.strip(), value
    else:
        # 没有=的是父键,值为字典
        return indent_level, stripped.strip(), {}

def build_hierarchy(lines):
    """构建层级结构"""
    root = {}
    stack = [(root, -1)]  # (当前字典, 当前缩进级别)
    
    for line in lines:
        parsed = parse_line(line)
        if not parsed:
            continue
        level, key, value = parsed
        
        # 找到当前层级的父节点
        while stack[-1][1] >= level:
            stack.pop()
        parent, parent_level = stack[-1]
        
        # 添加当前键到父节点
        parent[key] = value
        # 如果值是字典,加入栈中处理子节点
        if isinstance(value, dict):
            stack.append((value, level))
    
    return root

# 处理文件
filename = "your_input_file.txt"  # 替换为实际文件名
with open(filename, 'r') as f:
    lines = f.readlines()

result = build_hierarchy(lines)
# 生成合法JSON,确保没有尾随逗号
json_output = json.dumps(result, indent=4, ensure_ascii=False)

# 保存或打印
print(json_output)
with open("output.json", 'w', encoding='utf-8') as f:
    f.write(json_output)

代码说明

  1. parse_line函数:解析每一行,计算缩进级别,拆分键值对,将逗号分隔的值转换为数组,没有=的行标记为字典类型的父键。
  2. build_hierarchy函数:使用栈结构维护当前层级的字典,根据缩进级别将键值对添加到对应的父节点中,构建正确的嵌套结构。
  3. 最终用json.dumps生成合法JSON,自动处理引号和逗号,避免尾随逗号问题。

运行该代码后,将生成符合期望的合法JSON,可直接被json.load解析。

内容的提问来源于stack exchange,提问作者NigelBlainey

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.10 13:15:21