如何用Python的BeautifulSoup提取Open Food Facts多页食品数据?
Open Food Facts分页爬取无输出问题解决
我正在从Open Food Facts网站提取食品数据,单页数据能正常提取并输出到CSV,但分页循环提取时终端完全没有输出。我是第一次用Python和BeautifulSoup编程,尝试过移除baseurl拼接商品链接,但问题依旧,希望能得到代码改进的具体建议。
原代码如下:
import requests import openpyxl from bs4 import BeautifulSoup # excel = openpyxl.Workbook() # sheet = excel.active # sheet.title = "Open food facts" # sheet.append(['Name', 'Barcode', 'Common Name', 'Quantity', 'Packaging', 'Categories', 'Labels, certifications, awards', # 'Stores', 'Countries where sold', 'Nutri_Score_Description', 'NOVA_Description', 'ECO_Score_Description', # 'ingredient', 'allergen', 'fat_in_quantity', 'saturated_fat_in_quantity', 'sugar_in_quantity', 'salt_in_quantity']) baseurl = "https://world.openfoodfacts.org/" headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36" } productlinks = [] for x in range(2, 3): r = requests.get(f'https://world.openfoodfacts.org/{x}') soup = BeautifulSoup(r.content, 'lxml') productlist = soup.find_all('li', class_='list_product_a') for item in productlist: for link in item.find_all('a', href=True): productlinks.append(baseurl + link['href']) food_data = [] for link in productlinks: r = requests.get(link, headers=headers) soup = BeautifulSoup(r.content, 'lxml') try: item_name = soup.find('h2', class_='title-1').text.strip() except: item_name = "No names" try: common_name = soup.find('span', class_='field_value').text.strip() except: common_name = "No common names" try: barcode = soup.find('span', id='barcode').text.strip() except: barcode = "No barcodes" try: qty = soup.find('span', id='field_quantity_value').text.strip() except: qty = "No Quantitys" try: list_packaging = soup.find_all('span', id='field_packaging_value') packaging_list = [package.text.strip() for package in list_packaging] packaging = ', '.join(packaging_list) if packaging_list else 'No Packages' except: packaging = 'No Packages' try: list_categories = soup.find_all('span', id='field_categories_value') category_list = [category.text.strip() for category in list_categories] categories = ', '.join(category_list) if category_list else 'No Categories' except: categories = 'No Categories' try: list_labels = soup.find_all('span', id='field_labels_value') label_list = [label.text.strip() for label in list_labels] labels = ', '.join(label_list) if label_list else 'No Labels, certifications, awards' except: labels = 'No Labels, certifications, awards' try: list_stores = soup.find_all('span', id='field_stores_value') store_list = [store.text.strip() for store in list_stores] stores = ', '.join(store_list) if store_list else 'No Stores' except: stores = 'No Stores' try: list_countries = soup.find_all('span', id='field_countries_value') country_list = [country.text.strip() for country in list_countries] countries = ', '.join(country_list) if country_list else 'No Countries' except: countries = 'No Countries' try: a_element = soup.find('a', href="#panel_nutriscore_content") Nutri_Score_Description = a_element.find('h4').text.strip() except: Nutri_Score_Description = "NO Nutri Score Description Available" try: a_element = soup.find('a', href="#panel_nova_content") NOVA_Description = a_element.find('h4').text.strip() except: NOVA_Description = "NO NOVA Description Available" try: a_element = soup.find('a', href="#panel_ecoscore_content") ECO_Score_Description = a_element.find('h4').text.strip() except: ECO_Score_Description = "NO ECO Score Description Available" try: ingredient = soup.find('div', id='panel_ingredients_content').find('div', class_='panel_text').get_text(strip=True) except: ingredient = 'No Ingredients' try: allergen = soup.find('div', id='panel_ingredients_content').find('strong', text='Allergens:').next_sibling.strip() except: allergen = 'No Allergen' try: a_element = soup.find('a', href="#panel_nutrient_level_fat_content") fat_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip() except: fat_in_quantity = "No Fat In Quantitys" try: a_element = soup.find('a', href="#panel_nutrient_level_saturated-fat_content") saturated_fat_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip() except: saturated_fat_in_quantity = "No Saturated Fat In Quantitys" try: a_element = soup.find('a', href="#panel_nutrient_level_sugars_content") sugar_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip() except: sugar_in_quantity = "No Sugar In Quantitys" try: a_element = soup.find('a', href="#panel_nutrient_level_salt_content") salt_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip() except: salt_in_quantity = "No Salt In Quantitys" food = { 'Name': item_name, 'Barcode': barcode, 'Common Name': common_name, 'Quantity': qty, 'Packaging': packaging, 'Categories': categories, 'Labels, certifications, awards':labels, 'Stores': stores, 'Countries where sold': countries, 'Nutri_Score_Description': Nutri_Score_Description, 'NOVA_Description': NOVA_Description, 'ECO_Score_Description':ECO_Score_Description, 'ingredient': ingredient, 'allergen': allergen, 'fat_in_quantity': fat_in_quantity, 'saturated_fat_in_quantity': saturated_fat_in_quantity, 'sugar_in_quantity': sugar_in_quantity, 'salt_in_quantity':salt_in_quantity } print(item_name, barcode, common_name, qty, packaging, categories, labels, stores, countries, Nutri_Score_Description, NOVA_Description, ECO_Score_Description, ingredient, allergen, fat_in_quantity, saturated_fat_in_quantity, sugar_in_quantity, salt_in_quantity) # sheet.append([item_name, barcode, common_name, qty, packaging, categories, labels, # stores, countries, Nutri_Score_Description, NOVA_Description, ECO_Score_Description, # ingredient, allergen, fat_in_quantity, saturated_fat_in_quantity, sugar_in_quantity, salt_in_quantity]) # excel.save('openfoodfact.xlsx')
我曾尝试修改链接拼接部分代码,从:
productlinks = [] for x in range(2, 3): r = requests.get(f'https://world.openfoodfacts.org/{x}') soup = BeautifulSoup(r.content, 'lxml') productlist = soup.find_all('li', class_='list_product_a') for item in productlist: for link in item.find_all('a', href=True): productlinks.append(baseurl + link['href'])
改为:
productlinks = [] for x in range(2, 3): r = requests.get(f'https://world.openfoodfacts.org/{x}') soup = BeautifulSoup(r.content, 'lxml') productlist = soup.find_all('li', class_='list_product_a') for item in productlist: for link in item.find_all('a', href=True): productlinks.append(link['href'])
但终端还是没有输出。
问题排查与改进方案
1. 修复分页URL与商品链接拼接
- 分页URL错误:原代码用
https://world.openfoodfacts.org/{x}访问第二页,实际网站的分页地址是https://world.openfoodfacts.org/page/{x},用原URL会返回404错误,导致无法获取商品列表。 - 商品链接必须拼接baseurl:页面内的商品链接是相对路径(比如
/product/3017620422003),直接使用会导致请求地址无效,必须和baseurl拼接成完整绝对地址。
2. 修正商品列表选择器
原代码用li.list_product_a查找商品项,但当前网站的商品列表项类名是product-item,这个错误会导致productlist为空,后续循环无法执行。
3. 增加请求验证与调试信息
在请求后添加状态码打印,确认请求是否成功;在商品列表为空时打印提示,方便排查问题。
4. 优化请求逻辑
- 分页请求也使用headers,避免被网站识别为爬虫拦截。
- 每个商品项只有一个链接,用
find代替find_all,减少不必要的循环。 - 添加延时,避免短时间内请求过多被限制。
修改后的完整代码
import requests import openpyxl from bs4 import BeautifulSoup import time baseurl = "https://world.openfoodfacts.org/" headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36" } productlinks = [] # 爬取第2页到第2页(测试用,可修改范围如range(2,5)爬2-4页) for x in range(2, 3): # 修正分页URL page_url = f"{baseurl}page/{x}" r = requests.get(page_url, headers=headers) # 打印请求状态码,确认是否成功 print(f"请求第{x}页,状态码:{r.status_code}") if r.status_code != 200: print(f"第{x}页请求失败") continue soup = BeautifulSoup(r.content, 'lxml') # 修正商品列表选择器 productlist = soup.find_all('li', class_='product-item') print(f"第{x}页找到{len(productlist)}个商品") if not productlist: continue for item in productlist: # 每个商品只取第一个有效链接 link = item.find('a', href=True) if link: full_link = baseurl + link['href'].lstrip('/') # 移除可能的重复斜杠 productlinks.append(full_link) print(f"添加商品链接:{full_link}") print(f"共获取{len(productlinks)}个商品链接") food_data = [] for link in productlinks: r = requests.get(link, headers=headers) print(f"请求商品页面,状态码:{r.status_code}") if r.status_code != 200: print(f"商品页面{link}请求失败") continue soup = BeautifulSoup(r.content, 'lxml') try: item_name = soup.find('h2', class_='title-1').text.strip() except: item_name = "No names" try: # 修正common_name选择器:第一个field_value是品牌,实际通用名在id为field_generic_name_value的元素中 common_name = soup.find('span', id='field_generic_name_value').text.strip() except: common_name = "No common names" try: barcode = soup.find('span', id='barcode').text.strip() except: barcode = "No barcodes" try: qty = soup.find('span', id='field_quantity_value').text.strip() except: qty = "No Quantitys" try: list_packaging = soup.find_all('span', id='field_packaging_value') packaging_list = [package.text.strip() for package in list_packaging] packaging = ', '.join(packaging_list) if packaging_list else 'No Packages' except: packaging = 'No Packages' try: list_categories = soup.find_all('span', id='field_categories_value') category_list = [category.text.strip() for category in list_categories] categories = ', '.join(category_list) if category_list else 'No Categories' except: categories = 'No Categories' try: list_labels = soup.find_all('span', id='field_labels_value') label_list = [label.text.strip() for label in list_labels] labels = ', '.join(label_list) if label_list else 'No Labels, certifications, awards' except: labels = 'No Labels, certifications, awards' try: list_stores = soup.find_all('span', id='field_stores_value') store_list = [store.text.strip() for store in list_stores] stores = ', '.join(store_list) if store_list else 'No Stores' except: stores = 'No Stores' try: list_countries = soup.find_all('span', id='field_countries_value') country_list = [country.text.strip() for country in list_countries] countries = ', '.join(country_list) if country_list else 'No Countries' except: countries = 'No Countries' try: a_element = soup.find('a', href="#panel_nutriscore_content") Nutri_Score_Description = a_element.find('h4').text.strip() except: Nutri_Score_Description = "NO Nutri Score Description Available" try: a_element = soup.find('a', href="#panel_nova_content") NOVA_Description = a_element.find('h4').text.strip() except: NOVA_Description = "NO NOVA Description Available" try: a_element = soup.find('a', href="#panel_ecoscore_content") ECO_Score_Description = a_element.find('h4').text.strip() except: ECO_Score_Description = "NO ECO Score Description Available" try: ingredient = soup.find('div', id='panel_ingredients_content').find('div', class_='panel_text').get_text(strip=True) except: ingredient = 'No Ingredients' try: allergen = soup.find('div', id='panel_ingredients_content').find('strong', text='Allergens:').next_sibling.strip() except: allergen = 'No Allergen' try: a_element = soup.find('a', href="#panel_nutrient_level_fat_content") fat_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip() except: fat_in_quantity = "No Fat In Quantitys" try: a_element = soup.find('a', href="#panel_nutrient_level_saturated-fat_content") saturated_fat_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip() except: saturated_fat_in_quantity = "No Saturated Fat In Quantitys" try: a_element = soup.find('a', href="#panel_nutrient_level_sugars_content") sugar_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip() except: sugar_in_quantity = "No Sugar In Quantitys" try: a_element = soup.find('a', href="#panel_nutrient_level_salt_content") salt_in_quantity = a_element.find('h4', class_='evaluation__title').text.strip() except: salt_in_quantity = "No Salt In Quantitys" food = { 'Name': item_name, 'Barcode': barcode, 'Common Name': common_name, 'Quantity': qty, 'Packaging': packaging, 'Categories
相关产品推荐
相关产品推荐

