Python Selenium爬取KSL:首次调用成功二次触发NoSuchElementException排查
KSL家具分类页面二次爬取CSS选择器失效问题排查与解决
问题描述
爬取KSL分类信息的家具板块时,首次调用get_first_listing函数能通过CSS选择器成功获取首个列表项的链接和标题,但在while循环中第二次调用该函数时,触发selenium.common.exceptions.NoSuchElementException,调整sleep时长后问题依旧。
用户代码如下:
from selenium import webdriver from selenium.webdriver.chrome.options import Options from selenium.webdriver.chrome.service import Service from selenium.webdriver.common.by import By from webdriver_manager.chrome import ChromeDriverManager import time import os chrome_options = Options() chrome_options.add_argument('--headless') chrome_options.add_experimental_option('excludeSwitches', ['enable-logging']) os.environ['WDM_LOG_LEVEL'] = '0' s = Service(ChromeDriverManager().install()) driver = webdriver.Chrome(service=s, options=chrome_options) # driver = webdriver.Chrome(s=path, options=chrome_options) # if you have problems with line 15 # Setting classified_link = 'https://classifieds.ksl.com/search/Furniture' time_to_wait_between_checking = 15 def get_first_listing(): driver.get(classified_link) time.sleep(15) link = driver.find_element(By.CSS_SELECTOR, '#search-results > div > section > div > div:nth-child(1) > section:nth-child(4) > div.listing-item-info > h2 > div > a').get_attribute('href') title = driver.find_element(By.CSS_SELECTOR, '#search-results > div > section > div > div:nth-child(1) > section:nth-child(4) > div.listing-item-info > h2 > div > a').text return (link, title) listing_info = get_first_listing() first_listing_link_temp = listing_info[0] listing_title = listing_info[1] print(f"First Listing Title: {listing_title}, Link: {first_listing_link_temp}") check_count = 0 active = True while active: check_count += 1 time.sleep(time_to_wait_between_checking) print(f"Checking to see if new listing, this is attempt number {check_count}") new_listing_info = get_first_listing() first_listing_link = new_listing_info[0] title = new_listing_info[1] if first_listing_link_temp != first_listing_link: print(f"There is a new ad. Title {title}, Link: {first_listing_link}") active = False break
报错信息:
Traceback (most recent call last): File "C:PATH.py", line 46, in <module> new_listing_info = get_first_listing() File "C:PATH.py", line 26, in get_first_listing link = driver.find_element(By.CSS_SELECTOR, '#search-results > div > section > div > div:nth-child(1) >' File "C:PATH\anaconda3\lib\site-packages\selenium\webdriver\remote\webdriver.py", line 856, in find_element return self.execute(Command.FIND_ELEMENT, { File "C:PATH\anaconda3\lib\site-packages\selenium\webdriver\remote\webdriver.py", line 429, in execute self.error_handler.check_response(response) File "C:PATH\anaconda3\lib\site-packages\selenium\webdriver\remote\errorhandler.py", line 243, in check_response raise exception_class(message, screen, stacktrace) selenium.common.exceptions.NoSuchElementException: Message: no such element: Unable to locate element: {"method":"css selector","selector":"#search-results > div > section > div > div:nth-child(1) > section:nth-child(4) > div.listing-item-info > h2 > div > a"} (Session info: headless chrome=106.0.5249.119) Stacktrace: Backtrace: ... Process finished with exit code 1
原因分析
- CSS选择器过于依赖DOM结构位置:你使用的
div:nth-child(1) > section:nth-child(4)是基于元素在DOM中的固定位置定位的,页面二次加载时可能因为动态插入的广告、置顶推广或弹窗改变元素层级,导致选择器匹配失败。 - 固定sleep不可靠:
time.sleep(15)是固定等待时长,页面加载速度受网络波动、JS渲染影响,第二次加载时可能元素还未完全渲染,就执行了查找操作。 - 无头模式渲染差异:无头Chrome的窗口默认尺寸较小,页面可能触发响应式布局,导致元素结构或位置发生变化。
解决方案
1. 使用健壮的CSS选择器
抛弃依赖位置的选择器,改用元素的类名属性定位,比如KSL的列表项统一带有listing-item类,标题链接在listing-item-info下的h2标签内:
#search-results .listing-item:first-child .listing-item-info h2 a
2. 用显式等待替代固定sleep
使用Selenium的WebDriverWait等待元素加载完成,比固定sleep更灵活可靠,只在元素出现后才继续执行。
3. 优化无头模式配置
设置无头Chrome的窗口尺寸,避免响应式布局影响元素渲染:
chrome_options.add_argument('--window-size=1920,1080')
修改后的完整代码
from selenium import webdriver from selenium.webdriver.chrome.options import Options from selenium.webdriver.chrome.service import Service from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from webdriver_manager.chrome import ChromeDriverManager import time import os chrome_options = Options() chrome_options.add_argument('--headless') chrome_options.add_argument('--window-size=1920,1080') chrome_options.add_experimental_option('excludeSwitches', ['enable-logging']) os.environ['WDM_LOG_LEVEL'] = '0' s = Service(ChromeDriverManager().install()) driver = webdriver.Chrome(service=s, options=chrome_options) # Setting classified_link = 'https://classifieds.ksl.com/search/Furniture' time_to_wait_between_checking = 15 wait = WebDriverWait(driver, 20) # 最长等待20秒 def get_first_listing(): driver.get(classified_link) # 等待第一个列表项的链接加载完成 listing_link = wait.until(EC.presence_of_element_located( (By.CSS_SELECTOR, '#search-results .listing-item:first-child .listing-item-info h2 a') )) link = listing_link.get_attribute('href') title = listing_link.text return (link, title) listing_info = get_first_listing() first_listing_link_temp = listing_info[0] listing_title = listing_info[1] print(f"First Listing Title: {listing_title}, Link: {first_listing_link_temp}") check_count = 0 active = True while active: check_count += 1 time.sleep(time_to_wait_between_checking) print(f"Checking to see if new listing, this is attempt number {check_count}") try: new_listing_info = get_first_listing() first_listing_link = new_listing_info[0] title = new_listing_info[1] if first_listing_link_temp != first_listing_link: print(f"There is a new ad. Title {title}, Link: {first_listing_link}") active = False break except Exception as e: print(f"Check failed: {str(e)}") continue driver.quit() # 程序结束后关闭浏览器
额外建议
- 添加异常捕获:循环中捕获异常,避免单次失败导致程序终止;
- 改用页面刷新:用
driver.refresh()代替driver.get(classified_link),减少服务器请求,提升效率; - 防范反爬:KSL可能对频繁请求的IP进行限制,建议适当延长请求间隔,或使用代理IP。
内容的提问来源于stack exchange,提问作者trainfortendietown
相关产品推荐
相关产品推荐

