Python网页爬虫报错:AttributeError: 'NoneType' object has no attribute 'find'
亚马逊埃及站爬虫翻页AttributeError问题解决
我是Python编程新手,编写亚马逊埃及站机械键盘爬虫时,第一页能正常运行,但加了翻页功能后触发AttributeError,提示'NoneType'对象没有'find'属性。怀疑是find()方法写法有问题,尤其是处理a标签的部分,相关代码和报错堆栈如下:
原代码
from bs4 import BeautifulSoup import requests, datetime, time # adjusting the link and the header (only change the user agent) url = "https://www.amazon.eg/s?k=mechinical+keyboard&language=en_AE&crid=1UBHQ81JZMC5G&sprefix=mechinical+keyboard%2Caps%2C206&ref=nb_sb_noss_1" headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/78.0.3904.108 Safari/537.36", "Accept-Encoding":"gzip, deflate", "Accept":"text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", "DNT":"1","Connection":"close", "Upgrade-Insecure-Requests":"1"} def get_data(soup): page = requests.get(url, headers= headers) soup = BeautifulSoup(page.content , "html.parser") # getting the html code of the webpage title = soup.find_all("span", "a-size-base-plus a-color-base a-text-normal") price = soup.find_all("span", {"class": "a-price-whole"}) # making a list for the data to be saved in title_list=[] price_list=[] # for loops to extract the titles and prices for titlex in title: titles= titlex.text title_list.append(titles) for pricex in price: prices=pricex.text.strip() prices = prices[:-1] price_list.append(prices) ## def get_nextpage(soup): pages = soup.find('span', {'class': "s-pagination-strip"}) if not pages.find({'class':'s-pagination-item s-pagination-next s-pagination-disabled'}): url = 'https://www.amazon.eg/-' + str(pages.find({'class': 's-pagination-item s-pagination-next s-pagination-button s-pagination-separator'}).find('a')['href']) return url else: return while True: data = get_data(url) url = get_nextpage(data) if not url: break print(url)
报错堆栈
AttributeError Traceback (most recent call last) Cell In[85], line 43 41 while True: 42 data = get_data(url) ---> 43 url = get_nextpage(data) 44 if not url: 45 break Cell In[85], line 33, in get_nextpage(soup) 32 def get_nextpage(soup): ---> 33 pages = soup.find('span', {'class': "s-pagination-strip"}) 34 if not pages.find({'class':'s-pagination-item s-pagination-next s-pagination-disabled'}): 35 url = 'https://www.amazon.eg/-' + str(pages.find({'class': 's-pagination-item s-pagination-next s-pagination-button s-pagination-separator'}).find('a')['href']) AttributeError: 'NoneType' object has no attribute 'find'
问题根源及解决方案
核心问题
- get_data函数无返回值:
get_data函数接收soup参数但未使用,内部请求固定URL后创建soup却不返回,导致主循环中data = get_data(url)得到None,传给get_nextpage后触发None.find()错误。 - find方法参数错误:
pages.find({'class':'xxx'})缺少标签类型(如span、a),BeautifulSoup无法正确定位元素。 - 空值判断缺失:链式调用
find前未检查结果是否为None,直接调用会触发错误。 - 翻页URL拼接错误:亚马逊下一页是相对路径,不需要额外添加
-,直接用基础域名拼接即可。
修正后的完整代码
from bs4 import BeautifulSoup import requests, datetime, time # 基础域名,用于拼接翻页链接 base_url = "https://www.amazon.eg" # 初始请求URL url = "https://www.amazon.eg/s?k=mechinical+keyboard&language=en_AE&crid=1UBHQ81JZMC5G&sprefix=mechinical+keyboard%2Caps%2C206&ref=nb_sb_noss_1" headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/78.0.3904.108 Safari/537.36", "Accept-Encoding":"gzip, deflate", "Accept":"text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", "DNT":"1","Connection":"close", "Upgrade-Insecure-Requests":"1"} def get_data(url): # 根据传入的URL请求页面 page = requests.get(url, headers=headers) soup = BeautifulSoup(page.content, "html.parser") # 提取标题和价格 title_list = [] price_list = [] titles = soup.find_all("span", class_="a-size-base-plus a-color-base a-text-normal") for titlex in titles: title_list.append(titlex.text.strip()) prices = soup.find_all("span", class_="a-price-whole") for pricex in prices: price_text = pricex.text.strip() # 处理空价格情况,避免索引错误 price_clean = price_text[:-1] if price_text else "N/A" price_list.append(price_clean) # 打印当前页数据统计,方便验证 print(f"当前页抓取到 {len(title_list)} 个商品,{len(price_list)} 个价格") # 返回解析后的soup对象,供翻页函数使用 return soup def get_nextpage(soup): # 定位分页条容器 pagination_strip = soup.find('span', class_="s-pagination-strip") if not pagination_strip: # 无分页条,说明已到最后一页 return None # 检查下一页按钮是否禁用 next_disabled = pagination_strip.find('span', class_="s-pagination-item s-pagination-next s-pagination-disabled") if next_disabled: return None # 找到下一页按钮的a标签 next_button = pagination_strip.find('a', class_="s-pagination-item s-pagination-next s-pagination-button") if not next_button: return None # 拼接完整的下一页URL next_url = base_url + next_button['href'] return next_url # 主循环:循环抓取直到无下一页 while True: soup = get_data(url) url = get_nextpage(soup) if not url: print("已抓取完所有页面,爬虫结束") break print(f"准备抓取下一页:{url}") # 添加延迟,避免触发反爬机制 time.sleep(2)
关键修改说明
- get_data函数重构:接收url参数,请求对应页面后返回soup对象,确保翻页函数能拿到正确的解析结果。
- 强化空值判断:每一步
find操作后都检查是否为None,避免链式调用引发错误。 - 修正find参数:统一使用
class_参数指定类名,同时明确标签类型,确保元素定位准确。 - 正确拼接URL:使用基础域名拼接相对路径,保证翻页链接有效。
- 添加请求延迟:避免短时间内大量请求触发亚马逊反爬限制。
内容的提问来源于stack exchange,提问作者Cordeva
相关产品推荐
相关产品推荐

