使用Selenium爬取UN职位仅提取首个元素数据的问题求助
问题描述
- 使用Python+Selenium爬取联合国职位门户网站的实习岗位,目前仅能输出首个职位的信息
- 需获取职位原始申请链接:需点击职位条目进入详情页,再点击“Apply now”按钮跳转至对应联合国官网获取链接
- 附原代码及输出:
#import required packages from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.common.keys import Keys import time # 原代码遗漏导入,补充后可正常运行 #define the driver variable driver = webdriver.Chrome() #navigate to url given driver.get("https://www.unjobnet.org/jobs?orgtypes%5B0%5D=United+Nations+System&apptypes%5B0%5D=Internship&keywords=&orderby=closing") #wait 5 seconds for elements to load time.sleep(5) #locate elements based on specified css path divs = driver.find_elements(By.XPATH,'//*[@id="main"]/div[2]') #get the text attribute of each element and print it for div in divs: title = div.find_element(By.XPATH,'//*[@id="main"]/div[2]/div/div[1]/div/div[2]/div[1]').text area = div.find_element(By.XPATH,'//*[@id="main"]/div[2]/div/div[1]/div/div[2]/div[2]').text place = div.find_element(By.XPATH,'//*[@id="main"]/div[2]/div/div[1]/div/div[2]/div[4]').text deadline = div.find_element(By.XPATH,'//*[@id="main"]/div[2]/div/div[1]/div/div[2]/div[7]/span[2]').text print(title, area, place, deadline)
打印结果:
Communications - Intern UNDP - United Nations Development Programme New Delhi (India) Closing soon: 19 Jul 2023
一、修复仅爬取首个职位的问题
问题根源
- 原代码定位的
divs是整个职位列表的容器,而非单个职位条目,导致循环仅执行一次 - 内部元素使用绝对XPATH(从根节点开始匹配),每次都会定位到页面第一个职位的元素,而非当前职位条目下的元素
修改后代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time driver = webdriver.Chrome() driver.get("https://www.unjobnet.org/jobs?orgtypes%5B0%5D=United+Nations+System&apptypes%5B0%5D=Internship&keywords=&orderby=closing") # 用显式等待替代固定sleep,提升稳定性 wait = WebDriverWait(driver, 10) # 定位所有单个职位条目 job_items = wait.until(EC.presence_of_all_elements_located((By.XPATH, '//*[@id="main"]/div[2]/div'))) for item in job_items: # 使用相对XPATH(以.开头),限定在当前职位条目内查找元素 title = item.find_element(By.XPATH, './/div/div[2]/div[1]').text area = item.find_element(By.XPATH, './/div/div[2]/div[2]').text place = item.find_element(By.XPATH, './/div/div[2]/div[4]').text deadline = item.find_element(By.XPATH, './/div/div[2]/div[7]/span[2]').text print(f"职位标题:{title}\n所属机构:{area}\n工作地点:{place}\n截止日期:{deadline}\n---")
二、获取原始申请链接
实现逻辑
- 循环每个职位条目时,点击进入详情页
- 等待“Apply now”按钮加载完成并点击,触发跳转
- 切换到新打开的标签页,获取当前URL即为原始申请链接
- 关闭新标签页,切回列表页继续处理下一个职位
整合代码示例
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time driver = webdriver.Chrome() driver.get("https://www.unjobnet.org/jobs?orgtypes%5B0%5D=United+Nations+System&apptypes%5B0%5D=Internship&keywords=&orderby=closing") wait = WebDriverWait(driver, 10) # 记录主窗口句柄,用于后续切换 main_window = driver.current_window_handle while True: # 定位当前页所有职位条目 job_items = wait.until(EC.presence_of_all_elements_located((By.XPATH, '//*[@id="main"]/div[2]/div'))) for index, item in enumerate(job_items): try: # 点击职位条目进入详情页 item.click() # 等待并点击Apply now按钮 apply_btn = wait.until(EC.element_to_be_clickable((By.XPATH, '//a[contains(text(), "Apply now")]'))) apply_btn.click() # 切换到新标签页 new_window = [win for win in driver.window_handles if win != main_window][0] driver.switch_to.window(new_window) # 获取申请链接 apply_url = driver.current_url # 打印职位信息与链接 title = job_items[index].find_element(By.XPATH, './/div/div[2]/div[1]').text print(f"职位标题:{title}\n申请链接:{apply_url}\n---") # 关闭新标签页,切回主窗口 driver.close() driver.switch_to.window(main_window) except Exception as e: print(f"处理职位出错:{str(e)}") # 出错后强制切回主窗口,避免后续流程阻塞 driver.switch_to.window(main_window) continue # 尝试点击下一页,实现多页爬取 try: next_btn = wait.until(EC.element_to_be_clickable((By.XPATH, '//a[contains(text(), "Next")]'))) next_btn.click() # 等待页面刷新完成 wait.until(EC.staleness_of(job_items[0])) except: # 无下一页时退出循环 break driver.quit()
内容的提问来源于stack exchange,提问作者HAIQI WAN
相关产品推荐
相关产品推荐

