Python实现嵌套字典列表格式TXT文件转表格及表头提取
解决Twitter API嵌套字典数据解析与格式化问题
问题说明
正在学习数据解析,目标是创建可复用代码模板,仅需修改循环、函数及参数即可复用。通过Twitter API爬取话题标签相关推文,得到嵌套字典组成的列表并保存为TXT文件,当前需要完成两个任务:
- 清洗文本并转换为表格,但因数据是嵌套字典结构(每条数据含
domain和entity两个键,对应值均为嵌套字典),无法正确识别表头 - 实现每个字典条目以
domain开头换行的格式化输出
数据样例
[{'domain': {'id': '46', 'name': 'Business Taxonomy', 'description': 'Categories within Brand Verticals that narrow down the scope of Brands'}, 'entity': {'id': '1557696848252391426', 'name': 'Financial Services Business', 'description': 'Brands, companies, advertisers and every non-person handle with the profit intent related to Banks, Credit cards, Insurance, Investments, Stocks '}}, {'domain': {'id': '46', 'name': 'Business Taxonomy', 'description': 'Categories within Brand Verticals that narrow down the scope of Brands'}, 'entity': {'id': '1557697333571112960', 'name': 'Technology Business', 'description': 'Brands, companies, advertisers and every non-person handle with the profit intent related to softwares, apps, communication equipments, hardwares'}}, {'domain': {'id': '30', 'name': 'Entities [Entity Service]', 'description': 'Entity Service top level domain, every item that is in Entity Service should be in this domain'}, 'entity': {'id': '1007360414114435072', 'name': 'Bitcoin cryptocurrency', 'description': 'Bitcoin Cryptocurrency'}}, {'domain': {'id': '30', 'name': 'Entities [Entity Service]', 'description': 'Entity Service top level domain, every item that is in Entity Service should be in this domain'}, 'entity': {'id': '1007361429752594432', 'name': 'Ethereum cryptocurrency', 'description': 'Ethereum Cryptocurrency'}}, {'domain': {'id': '47', 'name': 'Brand', 'description': 'Brands and Companies'}, 'entity': {'id': '1372588659346612225', 'name': 'Binance'}}, {'domain': {'id': '30', 'name': 'Entities [Entity Service]', 'description': 'Entity Service top level domain, every item that is in Entity Service should be in this domain'}, 'entity': {'id': '857879456773357569', 'name': 'Technology', 'description': 'Technology'}}, {'domain': {'id': '66', 'name': 'Interests and Hobbies Category', 'description': 'A grouping of interests and hobbies entities, like Novelty Food or Destinations'}, 'entity': {'id': '913142676819648512', 'name': 'Cryptocurrencies', 'description': 'Cryptocurrency'}}, {'domain': {'id': '30', 'name': 'Entities [Entity Service]', 'description': 'Entity Service top level domain, every item that is in Entity Service should be in this domain'}, 'entity': {'id': '1001503516555337728', 'name': 'Blockchain', 'description': 'Blockchain'}}, {'domain': {'id': '66', 'name': 'Interests and Hobbies Category', 'description': 'A grouping of interests and hobbies entities, like Novelty Food or Destinations'}, 'entity': {'id': '1369311988040355840', 'name': 'NFTs', 'description': 'Non-fungible tokens'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '781974596148793345', 'name': 'Business & finance'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '781974596794716162', 'name': 'Financial services'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '847894353708068864', 'name': 'Investing', 'description': 'Investing'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '848920371311001600', 'name': 'Technology', 'description': 'Technology and computing'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '913142676819648512', 'name': 'Cryptocurrencies', 'description': 'Cryptocurrency'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1007360414114435072', 'name': 'Bitcoin cryptocurrency', 'description': 'Bitcoin Cryptocurrency'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1007361429752594432', 'name': 'Ethereum cryptocurrency', 'description': 'Ethereum Cryptocurrency'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1369311988040355840', 'name': 'NFTs', 'description': 'Non-fungible tokens'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1390680741206368263', 'name': 'Cryptocurrency exchanges'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1478776259068907541', 'name': 'Cryptotokens'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1484181943616884743', 'name': 'Cryptocoins'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1486271512655003652', 'name': 'Web3'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1491481998862348291', 'name': 'Digital asset industry'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1492162686204854274', 'name': 'Digital assets & cryptocurrency', 'description': 'Cryptocurrency'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1521397643909365760', 'name': 'NFT development'}}, {'domain': {'id': '131', 'name': 'Unified Twitter Taxonomy', 'description': 'A taxonomy of user interests. '}, 'entity': {'id': '1536439027636678656', 'name': 'Decentralized finance'}}, {'domain': {'id': '174', 'name': 'Digital Assets & Crypto', 'description': 'For cryptocurrency entities'}, 'entity': {'id': '1007360414114435072', 'name': 'Bitcoin cryptocurrency', 'description': 'Bitcoin Cryptocurrency'}}, {'domain': {'id': '174', 'name': 'Digital Assets & Crypto', 'description': 'For cryptocurrency entities'}, 'entity': {'id': '1007361429752594432', 'name': 'Ethereum cryptocurrency', 'description': 'Ethereum Cryptocurrency'}}, {'domain': {'id': '174', 'name': 'Digital Assets & Crypto', 'description': 'For cryptocurrency entities'}, 'entity': {'id': '1478776259068907541', 'name': 'Cryptotokens'}}]
尝试的代码
import json import re import os from tabulate import tabulate file = open('binance_hash_tweets_micro.txt', 'r+') read = file.readlines() file.close() modified = [] #this modified variable is a empty list that can be parsed into using loops that call modified for row in read: modified.append(row) print(modified) header = modified.pop(0) def fixed_length(text,length): if len(text) > length: text = text[:length] elif len(text) < length: text = (text + " " * length) [:length] return text for column in header: print(fixed_length(column,20), end = " ") print()
解决方案
1. 解析嵌套字典生成表格
原代码的核心问题是将文件按行读取,而实际文件中是一个完整的Python列表字符串,无法通过pop(0)获取表头。需要先将字符串转换为Python对象,再展开嵌套字典为平级字段,最后生成表格。
import ast from tabulate import tabulate # 读取文件并转换为Python列表 with open('binance_hash_tweets_micro.txt', 'r', encoding='utf-8') as f: data_str = f.read() data = ast.literal_eval(data_str) # 处理单引号的字典字符串 # 提取所有可能的表头(覆盖所有嵌套字段) headers = ['domain_id', 'domain_name', 'domain_description', 'entity_id', 'entity_name', 'entity_description'] # 构造表格行数据 rows = [] for item in data: domain = item['domain'] entity = item['entity'] row = [ domain.get('id', ''), domain.get('name', ''), domain.get('description', ''), entity.get('id', ''), entity.get('name', ''), entity.get('description', '') ] rows.append(row) # 打印表格 print(tabulate(rows, headers=headers, tablefmt='grid'))
2. 实现每个条目以domain开头换行的格式化
遍历每个字典条目,先输出domain的内容,再输出entity的内容,每个条目之间添加空行分隔:
import ast with open('binance_hash_tweets_micro.txt', 'r', encoding='utf-8') as f: data_str = f.read() data = ast.literal_eval(data_str) # 格式化输出每个条目 for idx, item in enumerate(data, 1): domain = item['domain'] entity = item['entity'] print(f"条目 {idx}:") print(f"domain:") for k, v in domain.items(): print(f" {k}: {v}") print(f"entity:") for k, v in entity.items(): print(f" {k}: {v}") print() # 条目间换行
内容的提问来源于stack exchange,提问作者throothewire
相关产品推荐
相关产品推荐

