如何用Python编辑复杂JSON数据并计算营养成分值
需求与实现方案
需求说明
我生成了一个包含units数组的JSON对象,数组内对象数量不固定。其中units第一个对象的键和值已分别存储在first_keys = []和first_values = []中。需要实现:
- 计算第2个及以后对象的营养值(如carb、protein等),公式为:
当前对象amount × 第一个对象对应营养值 ÷ 第一个对象amount - 保留原有的单位格式(比如"g"、"kcal")
示例JSON结构
{ "food_id": 0, "food_name": "NAME", "food_image": "IMAGE", "food_kcal": "KCAL", "food_url": "FOOD_URL", "food_description": "DESC", "meal_time": "null", "food_category": "", "food_first_unit": "Yemek Kaşığı", "carb_percent": "72", "protein_percent": "23", "fat_percent": "4", "units": [ { "unit": "100 Gram", "amount": "100", "kcal": "505 kcal", "carb": "65 g", "fiber": "3 g", "protein": "5 g", "fat": "24 g", "saturated_fat": "10 g", "salt": "0.7 g", "sugar": "34 g" }, { "unit": "1 Adet", "amount": "5", "kcal": "", "carb": "", "fiber": "", "protein": "", "fat": "", "saturated_fat": "", "salt": "", "sugar": "" }, { "unit": "1 Porsiyon", "amount": "30", "kcal": "", "carb": "", "fiber": "", "protein": "", "fat": "", "saturated_fat": "", "salt": "", "sugar": "" }, { "unit": "1 Paket", "amount": "90", "kcal": "", "carb": "", "fiber": "", "protein": "", "fat": "", "saturated_fat": "", "salt": "", "sugar": "" } ] }
现有代码
import requests from bs4 import BeautifulSoup import json url = 'www.example.com' response = requests.get(url) response.encoding = 'utf-8' html = response.text soup = BeautifulSoup(html, 'html.parser') items = soup.find_all('li') units = [] first_key = [] first_values = [] names = [] grams = [] for item in items: name = item.get('data-ntr-srvname-param') if(str(name) not in "None"): names.append(str(name).replace("Gramda", "Gram")) items = soup.find_all("li", {"data-ntr-gram-param": True}) for item in items: grams.append(float(item["data-ntr-gram-param"])) for row in soup.select('table[data-ntr-target="facts"] tr'): name = row.select_one('td:nth-of-type(1)') value = row.select_one('td:nth-of-type(2)') if name is not None and value is not None: first_key.append(name.text.strip()) first_values.append(value.text.strip()) for index, gram in enumerate(grams): replacedGram=str(gram).replace(".0", "") units.append({ "unit": f"{names[index]}", "amount": f"{replacedGram}" }) data_dict = dict(zip(first_key, first_values)) for index, gram in enumerate(grams): units[index].update(data_dict) json_obj = { "food_id": 0, "food_name": "NAME", "food_image": "IMAGE", "food_kcal": "KCAL", "food_url": "FOOD_URL", "food_description": "DESC", "meal_time": "null", "food_category":"", "food_first_unit": "FIRST", "carb_percent": "72", "protein_percent": "23", "fat_percent": "4", "units": units } print("***************") print(json.dumps(json_obj, indent=4, ensure_ascii=False)) print("***************")
修改后的实现代码
要完成营养值计算,需先解析第一个单位的数值与单位,再遍历后续单位批量计算:
import requests from bs4 import BeautifulSoup import json url = 'www.example.com' response = requests.get(url) response.encoding = 'utf-8' html = response.text soup = BeautifulSoup(html, 'html.parser') items = soup.find_all('li') units = [] first_keys = [] first_values = [] names = [] grams = [] # 提取单位名称 for item in items: name = item.get('data-ntr-srvname-param') if name is not None: names.append(str(name).replace("Gramda", "Gram")) # 提取每个单位的amount值 items = soup.find_all("li", {"data-ntr-gram-param": True}) for item in items: grams.append(float(item["data-ntr-gram-param"])) # 提取第一个单位的营养键值对 for row in soup.select('table[data-ntr-target="facts"] tr'): name = row.select_one('td:nth-of-type(1)') value = row.select_one('td:nth-of-type(2)') if name is not None and value is not None: first_keys.append(name.text.strip()) first_values.append(value.text.strip()) # 构建基础units数组 for index, gram in enumerate(grams): replaced_gram = str(gram).replace(".0", "") units.append({ "unit": names[index], "amount": replaced_gram }) # 解析第一个单位的数值和单位,生成计算模板 first_amount = float(units[0]["amount"]) nutrient_template = {} for key, value in zip(first_keys, first_values): # 拆分数值和单位(兼容"505 kcal"、"65 g"、"0.7 g"等格式) parts = value.split() if len(parts) == 2: num, unit = parts nutrient_template[key] = (float(num), unit) elif len(parts) == 1: # 处理无单位的特殊情况(可根据实际需求调整) nutrient_template[key] = (float(parts[0]), "") # 给第一个单位填充完整营养值 for key, (num, unit) in nutrient_template.items(): units[0][key] = f"{num} {unit}" if unit else str(num) # 计算后续单位的营养值 for unit in units[1:]: current_amount = float(unit["amount"]) ratio = current_amount / first_amount for key, (base_num, unit_str) in nutrient_template.items(): calculated_num = base_num * ratio # 优化数值显示格式:整数显示为整数,小数保留1位 if calculated_num.is_integer(): calculated_num = int(calculated_num) else: calculated_num = round(calculated_num, 1) unit[key] = f"{calculated_num} {unit_str}" if unit_str else str(calculated_num) # 构建最终JSON对象 json_obj = { "food_id": 0, "food_name": "NAME", "food_image": "IMAGE", "food_kcal": "KCAL", "food_url": "FOOD_URL", "food_description": "DESC", "meal_time": "null", "food_category": "", "food_first_unit": "FIRST", "carb_percent": "72", "protein_percent": "23", "fat_percent": "4", "units": units } print("***************") print(json.dumps(json_obj, indent=4, ensure_ascii=False)) print("***************")
关键实现说明
- 营养模板解析:将第一个单位的营养值拆分为数值和单位,确保计算后保留原格式
- 比例计算:通过当前单位amount与第一个单位amount的比值,批量计算所有营养项
- 数值格式优化:自动判断整数/小数,提升输出可读性
- 批量填充:遍历后续单位,一次性完成所有营养字段的计算与填充
内容的提问来源于stack exchange,提问作者Deniz
相关产品推荐
相关产品推荐

