You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用Python的BeautifulSoup提取Open Food Facts多页食品数据?

Open Food Facts分页爬取无输出问题解决

我正在从Open Food Facts网站提取食品数据,单页数据能正常提取并输出到CSV,但分页循环提取时终端完全没有输出。我是第一次用Python和BeautifulSoup编程,尝试过移除baseurl拼接商品链接,但问题依旧,希望能得到代码改进的具体建议。

原代码如下:

import requests
import openpyxl
from bs4 import BeautifulSoup

# excel = openpyxl.Workbook()
# sheet = excel.active
# sheet.title = "Open food facts"

# sheet.append(['Name', 'Barcode', 'Common Name', 'Quantity', 'Packaging', 'Categories', 'Labels, certifications, awards',
#               'Stores', 'Countries where sold', 'Nutri_Score_Description', 'NOVA_Description', 'ECO_Score_Description',
#               'ingredient', 'allergen', 'fat_in_quantity', 'saturated_fat_in_quantity', 'sugar_in_quantity', 'salt_in_quantity'])

baseurl = "https://world.openfoodfacts.org/"

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36"
}

productlinks = []

for x in range(2, 3):
    r = requests.get(f'https://world.openfoodfacts.org/{x}')
    soup = BeautifulSoup(r.content, 'lxml')
    productlist = soup.find_all('li', class_='list_product_a')
    for item in productlist:
        for link in item.find_all('a', href=True):
            productlinks.append(baseurl + link['href'])

food_data = []
for link in productlinks:
    r = requests.get(link, headers=headers)
    soup = BeautifulSoup(r.content, 'lxml')

    try:
        item_name = soup.find('h2', class_='title-1').text.strip()
    except:
        item_name = "No names"

    try:
        common_name = soup.find('span', class_='field_value').text.strip()
    except:
        common_name = "No common names"

    try:
        barcode = soup.find('span', id='barcode').text.strip()
    except:
        barcode = "No barcodes"

    try:
        qty = soup.find('span', id='field_quantity_value').text.strip()
    except:
        qty = "No Quantitys"

    try:
        list_packaging = soup.find_all('span', id='field_packaging_value')
        packaging_list = [package.text.strip() for package in list_packaging]
        packaging = ', '.join(packaging_list) if packaging_list else 'No Packages'
    except:
        packaging = 'No Packages'

    try:
        list_categories = soup.find_all('span', id='field_categories_value')
        category_list = [category.text.strip() for category in list_categories]
        categories = ', '.join(category_list) if category_list else 'No Categories'
    except:
        categories = 'No Categories'

    try:
        list_labels = soup.find_all('span', id='field_labels_value')
        label_list = [label.text.strip() for label in list_labels]
        labels = ', '.join(label_list) if label_list else 'No Labels, certifications, awards'
    except:
        labels = 'No Labels, certifications, awards'

    try:
        list_stores = soup.find_all('span', id='field_stores_value')
        store_list = [store.text.strip() for store in list_stores]
        stores = ', '.join(store_list) if store_list else 'No Stores'
    except:
        stores = 'No Stores'

    try:
        list_countries = soup.find_all('span', id='field_countries_value')
        country_list = [country.text.strip() for country in list_countries]
        countries = ', '.join(country_list) if country_list else 'No Countries'
    except:
        countries = 'No Countries'

    try:
        a_element = soup.find('a', href="#panel_nutriscore_content")
        Nutri_Score_Description = a_element.find('h4').text.strip()
    except:
        Nutri_Score_Description = "NO Nutri Score Description Available"

    try:
        a_element = soup.find('a', href="#panel_nova_content")
        NOVA_Description = a_element.find('h4').text.strip()
    except:
        NOVA_Description = "NO NOVA Description Available"

    try:
        a_element = soup.find('a', href="#panel_ecoscore_content")
        ECO_Score_Description = a_element.find('h4').text.strip()
    except:
        ECO_Score_Description = "NO ECO Score Description Available"

    try:
        ingredient = soup.find('div', id='panel_ingredients_content').find('div', class_='panel_text').get_text(strip=True)
    except:
        ingredient = 'No Ingredients'

    try:
        allergen = soup.find('div', id='panel_ingredients_content').find('strong', text='Allergens:').next_sibling.strip()
    except:
        allergen = 'No Allergen'

    try:
        a_element = soup.find('a', href="#panel_nutrient_level_fat_content")
        fat_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip()
    except:
        fat_in_quantity = "No Fat In Quantitys"

    try:
        a_element = soup.find('a', href="#panel_nutrient_level_saturated-fat_content")
        saturated_fat_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip()
    except:
        saturated_fat_in_quantity = "No Saturated Fat In Quantitys"

    try:
        a_element = soup.find('a', href="#panel_nutrient_level_sugars_content")
        sugar_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip()
    except:
        sugar_in_quantity = "No Sugar In Quantitys"

    try:
        a_element = soup.find('a', href="#panel_nutrient_level_salt_content")
        salt_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip()
    except:
        salt_in_quantity = "No Salt In Quantitys"

    food = {
        'Name': item_name,
        'Barcode': barcode,
        'Common Name': common_name,
        'Quantity': qty,
        'Packaging': packaging,
        'Categories': categories,
        'Labels, certifications, awards':labels,
        'Stores': stores,
        'Countries where sold': countries,
        'Nutri_Score_Description': Nutri_Score_Description,
        'NOVA_Description': NOVA_Description,
        'ECO_Score_Description':ECO_Score_Description,
        'ingredient': ingredient,
        'allergen': allergen,
        'fat_in_quantity': fat_in_quantity,
        'saturated_fat_in_quantity': saturated_fat_in_quantity,
        'sugar_in_quantity': sugar_in_quantity,
        'salt_in_quantity':salt_in_quantity
    }

    print(item_name, barcode, common_name, qty, packaging, categories, labels,
              stores, countries, Nutri_Score_Description, NOVA_Description, ECO_Score_Description,
              ingredient, allergen, fat_in_quantity, saturated_fat_in_quantity, sugar_in_quantity, salt_in_quantity)

    # sheet.append([item_name, barcode, common_name, qty, packaging, categories, labels,
    #           stores, countries, Nutri_Score_Description, NOVA_Description, ECO_Score_Description,
    #           ingredient, allergen, fat_in_quantity, saturated_fat_in_quantity, sugar_in_quantity, salt_in_quantity])
# excel.save('openfoodfact.xlsx')

我曾尝试修改链接拼接部分代码,从:

productlinks = []

for x in range(2, 3):
    r = requests.get(f'https://world.openfoodfacts.org/{x}')
    soup = BeautifulSoup(r.content, 'lxml')
    productlist = soup.find_all('li', class_='list_product_a')
    for item in productlist:
        for link in item.find_all('a', href=True):
            productlinks.append(baseurl + link['href'])

改为:

productlinks = []

for x in range(2, 3):
    r = requests.get(f'https://world.openfoodfacts.org/{x}')
    soup = BeautifulSoup(r.content, 'lxml')
    productlist = soup.find_all('li', class_='list_product_a')
    for item in productlist:
        for link in item.find_all('a', href=True):
            productlinks.append(link['href'])

但终端还是没有输出。


问题排查与改进方案

1. 修复分页URL与商品链接拼接

  • 分页URL错误:原代码用https://world.openfoodfacts.org/{x}访问第二页,实际网站的分页地址是https://world.openfoodfacts.org/page/{x},用原URL会返回404错误,导致无法获取商品列表。
  • 商品链接必须拼接baseurl:页面内的商品链接是相对路径(比如/product/3017620422003),直接使用会导致请求地址无效,必须和baseurl拼接成完整绝对地址。

2. 修正商品列表选择器

原代码用li.list_product_a查找商品项,但当前网站的商品列表项类名是product-item,这个错误会导致productlist为空,后续循环无法执行。

3. 增加请求验证与调试信息

在请求后添加状态码打印,确认请求是否成功;在商品列表为空时打印提示,方便排查问题。

4. 优化请求逻辑

  • 分页请求也使用headers,避免被网站识别为爬虫拦截。
  • 每个商品项只有一个链接,用find代替find_all,减少不必要的循环。
  • 添加延时,避免短时间内请求过多被限制。

修改后的完整代码

import requests
import openpyxl
from bs4 import BeautifulSoup
import time

baseurl = "https://world.openfoodfacts.org/"

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36"
}

productlinks = []

# 爬取第2页到第2页(测试用,可修改范围如range(2,5)爬2-4页)
for x in range(2, 3):
    # 修正分页URL
    page_url = f"{baseurl}page/{x}"
    r = requests.get(page_url, headers=headers)
    # 打印请求状态码,确认是否成功
    print(f"请求第{x}页,状态码:{r.status_code}")
    if r.status_code != 200:
        print(f"第{x}页请求失败")
        continue
    soup = BeautifulSoup(r.content, 'lxml')
    # 修正商品列表选择器
    productlist = soup.find_all('li', class_='product-item')
    print(f"第{x}页找到{len(productlist)}个商品")
    if not productlist:
        continue
    for item in productlist:
        # 每个商品只取第一个有效链接
        link = item.find('a', href=True)
        if link:
            full_link = baseurl + link['href'].lstrip('/')  # 移除可能的重复斜杠
            productlinks.append(full_link)
            print(f"添加商品链接:{full_link}")

print(f"共获取{len(productlinks)}个商品链接")

food_data = []
for link in productlinks:
    r = requests.get(link, headers=headers)
    print(f"请求商品页面,状态码:{r.status_code}")
    if r.status_code != 200:
        print(f"商品页面{link}请求失败")
        continue
    soup = BeautifulSoup(r.content, 'lxml')

    try:
        item_name = soup.find('h2', class_='title-1').text.strip()
    except:
        item_name = "No names"

    try:
        # 修正common_name选择器:第一个field_value是品牌,实际通用名在id为field_generic_name_value的元素中
        common_name = soup.find('span', id='field_generic_name_value').text.strip()
    except:
        common_name = "No common names"

    try:
        barcode = soup.find('span', id='barcode').text.strip()
    except:
        barcode = "No barcodes"

    try:
        qty = soup.find('span', id='field_quantity_value').text.strip()
    except:
        qty = "No Quantitys"

    try:
        list_packaging = soup.find_all('span', id='field_packaging_value')
        packaging_list = [package.text.strip() for package in list_packaging]
        packaging = ', '.join(packaging_list) if packaging_list else 'No Packages'
    except:
        packaging = 'No Packages'

    try:
        list_categories = soup.find_all('span', id='field_categories_value')
        category_list = [category.text.strip() for category in list_categories]
        categories = ', '.join(category_list) if category_list else 'No Categories'
    except:
        categories = 'No Categories'

    try:
        list_labels = soup.find_all('span', id='field_labels_value')
        label_list = [label.text.strip() for label in list_labels]
        labels = ', '.join(label_list) if label_list else 'No Labels, certifications, awards'
    except:
        labels = 'No Labels, certifications, awards'

    try:
        list_stores = soup.find_all('span', id='field_stores_value')
        store_list = [store.text.strip() for store in list_stores]
        stores = ', '.join(store_list) if store_list else 'No Stores'
    except:
        stores = 'No Stores'

    try:
        list_countries = soup.find_all('span', id='field_countries_value')
        country_list = [country.text.strip() for country in list_countries]
        countries = ', '.join(country_list) if country_list else 'No Countries'
    except:
        countries = 'No Countries'

    try:
        a_element = soup.find('a', href="#panel_nutriscore_content")
        Nutri_Score_Description = a_element.find('h4').text.strip()
    except:
        Nutri_Score_Description = "NO Nutri Score Description Available"

    try:
        a_element = soup.find('a', href="#panel_nova_content")
        NOVA_Description = a_element.find('h4').text.strip()
    except:
        NOVA_Description = "NO NOVA Description Available"

    try:
        a_element = soup.find('a', href="#panel_ecoscore_content")
        ECO_Score_Description = a_element.find('h4').text.strip()
    except:
        ECO_Score_Description = "NO ECO Score Description Available"

    try:
        ingredient = soup.find('div', id='panel_ingredients_content').find('div', class_='panel_text').get_text(strip=True)
    except:
        ingredient = 'No Ingredients'

    try:
        allergen = soup.find('div', id='panel_ingredients_content').find('strong', text='Allergens:').next_sibling.strip()
    except:
        allergen = 'No Allergen'

    try:
        a_element = soup.find('a', href="#panel_nutrient_level_fat_content")
        fat_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip()
    except:
        fat_in_quantity = "No Fat In Quantitys"

    try:
        a_element = soup.find('a', href="#panel_nutrient_level_saturated-fat_content")
        saturated_fat_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip()
    except:
        saturated_fat_in_quantity = "No Saturated Fat In Quantitys"

    try:
        a_element = soup.find('a', href="#panel_nutrient_level_sugars_content")
        sugar_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip()
    except:
        sugar_in_quantity = "No Sugar In Quantitys"

    try:
        a_element = soup.find('a', href="#panel_nutrient_level_salt_content")
        salt_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip()
    except:
        salt_in_quantity = "No Salt In Quantitys"

    food = {
        'Name': item_name,
        'Barcode': barcode,
        'Common Name': common_name,
        'Quantity': qty,
        'Packaging': packaging,
        'Categories
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.15 13:02:57