动态URL页面爬虫故障求助:hotelcareer.com数据爬取失败
问题分析与代码修复
先梳理你代码里的几个致命问题:
- 缺了
from selenium import webdriver核心导入,直接导致webdriver.Chrome()调用失败 - 分页按钮定位逻辑太粗糙,原XPATH
//a[@class='weiter']容易定位到无效元素 - 循环遍历列表项时反复用
driver.find_element,极易触发**元素过期(stale element)**错误 - 每次爬一条数据就存一次Excel,严重拖慢效率
- 详情页等待的元素不够稳定,容易出现超时
下面是修复后的完整代码:
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.wait import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import pandas as pd import time # 初始化浏览器 driver = webdriver.Chrome() driver.maximize_window() job_links = [] try: driver.get("https://www.hotelcareer.com/jobs/job-offers") except Exception as e: print(f"页面访问失败: {str(e)}") driver.quit() exit() # 处理各类弹窗(Cookie提示+广告弹窗) try: # 处理Cookie同意弹窗 cookie_btn = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.XPATH, "//button[contains(text(), 'Accept')]")) ) cookie_btn.click() time.sleep(1) except: pass try: # 处理右上角关闭弹窗 close_btn = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.XPATH, "//button[@title='Close']")) ) close_btn.click() except: pass # 设置每页显示50条结果 try: # 滚动到分页选项区域再操作 pagination_area = WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.CLASS_NAME, "pagination")) ) driver.execute_script("arguments[0].scrollIntoView();", pagination_area) show_50_btn = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.XPATH, "//a[text()='50']")) ) show_50_btn.click() time.sleep(3) except Exception as e: print(f"设置每页50条失败: {str(e)}") # 批量爬取所有职位链接 while True: try: # 等待当前页职位列表完全加载 WebDriverWait(driver, 15).until( EC.presence_of_all_elements_located((By.XPATH, "//ul[@class='resultlist']/li//a[@data-js-action]")) ) # 一次性获取当前页所有链接元素,避免循环中重复定位 current_link_elements = driver.find_elements(By.XPATH, "//ul[@class='resultlist']/li//a[@data-js-action]") for elem in current_link_elements: link = elem.get_attribute('href') if link: job_links.append(link) print(f"已收集 {len(job_links)} 个职位链接") # 定位并点击下一页 next_page_btn = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.XPATH, "//a[@class='weiter' and contains(text(), 'Next')]")) ) driver.execute_script("arguments[0].scrollIntoView();", next_page_btn) next_page_btn.click() time.sleep(3) except Exception as e: print(f"分页结束或出错: {str(e)}") break # 爬取职位详情信息 job_data = [] for idx, link in enumerate(job_links, 1): print(f"处理第 {idx} 个职位: {link}") detail = {} try: driver.get(link) # 等待详情页核心内容加载完成 WebDriverWait(driver, 15).until( EC.presence_of_element_located((By.TAG_NAME, "h1")) ) # 提取各字段(用find_elements判断元素是否存在,避免直接报错) detail['Job Title'] = driver.find_element(By.TAG_NAME, "h1").text.strip() if driver.find_elements(By.TAG_NAME, "h1") else "" detail['Address'] = driver.find_element(By.XPATH, "//span[@class='location']/a").text.strip() if driver.find_elements(By.XPATH, "//span[@class='location']/a") else "" detail['Email'] = driver.find_element(By.XPATH, "//a[@id='email']").text.strip() if driver.find_elements(By.XPATH, "//a[@id='email']") else "" detail['Website'] = driver.find_element(By.XPATH, "//div[@id='contact_fields']/a[@target='_blank']").text.strip() if driver.find_elements(By.XPATH, "//div[@id='contact_fields']/a[@target='_blank']") else "" job_data.append(detail) except Exception as e: print(f"职位 {link} 处理失败: {str(e)}") continue # 统一保存数据到Excel if job_data: df = pd.DataFrame(job_data) df.to_excel("HotelJobs.xlsx", index=False) print(f"数据已保存到 HotelJobs.xlsx,共 {len(job_data)} 条有效记录") else: print("未获取到有效职位数据") driver.quit()
关键修复说明:
- 补全核心导入:添加
from selenium import webdriver,解决最基础的模块缺失问题 - 优化弹窗处理:新增Cookie弹窗处理逻辑,覆盖更多页面初始化时的弹窗场景
- 批量获取链接:一次性抓取当前页所有链接元素,避免循环中重复定位导致的元素过期错误
- 精准分页定位:给下一页按钮增加文本判断,同时滚动到按钮位置再点击,避免点击失效
- 稳定详情页等待:改用页面标题标签
<h1>作为等待条件,比特定元素更通用稳定 - 优化数据保存:所有数据爬取完成后统一写入Excel,大幅提升效率
- 增强错误日志:每个步骤添加异常打印,方便快速排查问题
内容的提问来源于stack exchange,提问作者Istvan Kalanyos
相关产品推荐
相关产品推荐

