Python Ebay价格追踪器爬虫报错:'NoneType' object has no attribute 'text'
问题描述
我跟着教程用Python开发Ebay价格追踪器,爬取Ebay搜索结果页商品标题时,遇到AttributeError: 'NoneType' object has no attribute 'text'错误。出错的代码行是:
'title': item.find('h3', {'class': 's-item__title s-item__title--has-tags'}).text,
完整代码如下:
import requests from bs4 import BeautifulSoup import pandas as pd searchterm = 'screen' def get_data(searchterm): url = f'https://www.ebay.com/sch/i.html?_from=R40&_trksid=p2380057.m570.l1313&_nkw={searchterm}&_sacat=0&LH_PrefLoc=1&LH_Auction=1&rt=nc&LH_Sold=1&LH_Complete=1' r = requests.get(url) soup = BeautifulSoup(r.text, 'html.parser') return soup def parse(soup): productslist = [] results = soup.find_all('div',{'class': 's-item__info clearfix'}) for item in results: product = { 'title': item.find('h3', {'class': 's-item__title s-item__title--has-tags'}).text, 'soldprice': float(item.find('span', {'class': 's-item__price'}).text.replace('$','').replace(',','').strip()), 'solddate': item.find('span', {'class': 's-item__title--tagblock__COMPLETED'}).find('span',{'class': 'POSITIVE'}.text), 'bids': item.find('span', {'class': 's-item__bids'}).text, 'link': item.find('a', {'class': 's-item__link'})['href'], } productslist.append(product) return productslist def output (productslist, searchterm): productsdf = pd.DataFrame(productslist) productsdf.to_csv(searchterm + 'ebaytrackeroutput.csv', index=False) print('Saved to CSV') return soup = get_data(searchterm) productslist = parse(soup) output(productslist, searchterm)
错误原因
- 元素匹配失败:
item.find()返回None,说明当前遍历的item里没有找到指定class的h3标签。部分商品的标题class可能不带s-item__title--has-tags后缀,导致匹配失败。 - 页面结构变更:Ebay的页面结构可能和教程发布时不同,原有的class选择器不再适配所有元素。
- 多处存在风险:除了标题,
solddate、bids等字段都直接调用.text或取['href'],只要find()返回None就会触发类似错误;另外solddate的代码还有语法错误:find('span',{'class': 'POSITIVE'}.text)应该是find('span',{'class': 'POSITIVE'}).text。
解决方法
核心逻辑是先判断元素是否存在,再访问其属性,避免直接对None调用方法或属性:
- 放宽标题的class匹配,只用基础class
s-item__title覆盖所有商品标题。 - 对每个字段添加空值判断,不存在时赋值为
None或跳过该商品。 - 修复
solddate的语法错误。 - 跳过搜索结果顶部可能被误匹配的占位元素(比如"Shop by category"模块)。
修改后的完整代码
import requests from bs4 import BeautifulSoup import pandas as pd searchterm = 'screen' def get_data(searchterm): url = f'https://www.ebay.com/sch/i.html?_from=R40&_trksid=p2380057.m570.l1313&_nkw={searchterm}&_sacat=0&LH_PrefLoc=1&LH_Auction=1&rt=nc&LH_Sold=1&LH_Complete=1' r = requests.get(url) soup = BeautifulSoup(r.text, 'html.parser') return soup def parse(soup): productslist = [] results = soup.find_all('div',{'class': 's-item__info clearfix'}) # 跳过第一个可能的占位元素 for item in results[1:]: # 处理标题 title_elem = item.find('h3', {'class': 's-item__title'}) title = title_elem.text.strip() if title_elem else None # 处理售价 price_elem = item.find('span', {'class': 's-item__price'}) soldprice = float(price_elem.text.replace('$','').replace(',','').strip()) if price_elem else None # 处理售出日期 tag_block = item.find('span', {'class': 's-item__title--tagblock__COMPLETED'}) date_elem = tag_block.find('span', {'class': 'POSITIVE'}) if tag_block else None solddate = date_elem.text.strip() if date_elem else None # 处理竞价数 bids_elem = item.find('span', {'class': 's-item__bids'}) bids = bids_elem.text.strip() if bids_elem else None # 处理商品链接 link_elem = item.find('a', {'class': 's-item__link'}) link = link_elem['href'] if link_elem else None # 只保留有核心信息的商品 if title and soldprice and link: product = { 'title': title, 'soldprice': soldprice, 'solddate': solddate, 'bids': bids, 'link': link, } productslist.append(product) return productslist def output(productslist, searchterm): productsdf = pd.DataFrame(productslist) productsdf.to_csv(f'{searchterm}_ebaytrackeroutput.csv', index=False) print('Saved to CSV') return soup = get_data(searchterm) productslist = parse(soup) output(productslist, searchterm)
内容的提问来源于stack exchange,提问作者EXB
相关产品推荐
相关产品推荐

