Indeed爬虫列表长度不一致求助:jobTitles/links长于companyName等
Indeed爬虫列表长度不一致问题修复
问题
运行Indeed爬虫时,jobTitles和links列表长度始终大于companyName和jobLocation,各列表长度不统一,无法正常写入CSV。原代码如下:
import hrequests from bs4 import BeautifulSoup import pandas as pd import csv #################### ## INDEED SCRAPER ## #################### # Creates the CSV in Write Mode with open('Indeed_Jobs.csv', 'w', newline='') as file: writer = csv.writer(file) #Creating Lists for info jobTitles = [] companyName = [] jobLocation = [] links = [] #Creates list of URLs URLs = ["https://www.indeed.com/jobs?q=IT+Entry+Level&l=United+States&from=searchOnHP&vjk=55e3c7e5e7a919c9", "https://www.indeed.com/jobs?q=development+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=996b270bd119f225", "https://www.indeed.com/jobs?q=user+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=4fc6a443fd7c11df", "https://www.indeed.com/jobs?q=ux%2Fui+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=254dcd8c33926527", "https://www.indeed.com/jobs?q=data+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=bd4fd45f0feb91c2"] for url in URLs: #Connecting to Indeed and reading HTML target_url = url head= {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/62.0.3202.94 Safari/537.36", "Accept-Encoding": "gzip, deflate, br", "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8", "Connection": "keep-alive", "Accept-Language": "en-US,en;q=0.9,lt;q=0.8,et;q=0.7,de;q=0.6", } resp = hrequests.get(target_url, headers=head) soup = BeautifulSoup(resp.text, 'html.parser') #Finds all items in list for div in soup.find_all('div', {'class': 'css-dekpa e37uo190'}): #Pulls job title for span in div.find_all('span'): jobTitles.append(span.text.strip()) #Pulls link to job for a in div.find_all('a'): links.append("indeed.com" + a.get('href')) #Pulls company name for div in soup.find_all('div', {'class': 'company_location css-17fky0v e37uo190'}): for span in div.find_all('span',{'data-testid': 'company-name'}): companyName.append(span.text.strip()) #Pulls job location for divLocation in div.find_all('div', {'data-testid': 'text-location'}): jobLocation.append(divLocation.text.strip()) # Changing the heading on the CSV depending on which URL was searched if url == "https://www.indeed.com/jobs?q=development+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=996b270bd119f225": additional_row = ["Development - Entry Level"] elif url =="https://www.indeed.com/jobs?q=it+entry+level&l=United+States&from=searchOnHP&vjk=bf9a1055e66b240c": additional_row = ["IT - Entry Level"] elif url == "https://www.indeed.com/jobs?q=user+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=4fc6a443fd7c11df": additional_row = ["User Experience - Entry Level"] elif url == "https://www.indeed.com/jobs?q=ux%2Fui+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=254dcd8c33926527": additional_row = ["UX/UI - Entry Level"] elif url == "https://www.indeed.com/jobs?q=data+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=bd4fd45f0feb91c2": additional_row = ["Data - Entry Level"] else: additional_row = ["Title Not Found"] # Blank Row to make formatting nice blank_row = [''] print(len(jobTitles)) print(len(companyName)) print(len(jobLocation)) print(len(links)) # # # Open the CSV file in append mode # with open('Indeed_Jobs.csv', 'a', newline='') as file: # writer = csv.writer(file) # # Write the header and blank rows for formatting # writer.writerow(blank_row) # writer.writerow(additional_row) # writer.writerow(blank_row) # # # Write jobs to the CSV # df = pd.DataFrame({'Job Title': jobTitles, 'Company Name': companyName, 'Location': jobLocation, 'Link': links}) # df.to_csv(file, mode='a', header=True, index=False)
问题根源
原代码分开遍历不同的DOM块,没有按单个职位条目关联信息:
- 先遍历职位标题和链接的容器,再单独遍历公司和地点的容器
- 若某个职位缺失公司/地点信息,对应的列表不会添加空值,导致所有列表长度错位
- 职位标题的遍历逻辑错误:
div.find_all('span')会抓取多个span标签,导致jobTitles重复添加内容
修复方案
按单个职位条目遍历,在每个职位容器内抓取所有字段,缺失时填充空值,确保每个列表的元素一一对应:
import hrequests from bs4 import BeautifulSoup import pandas as pd import csv #################### ## INDEED SCRAPER ## #################### # 初始化CSV文件(仅执行一次) with open('Indeed_Jobs.csv', 'w', newline='', encoding='utf-8') as file: writer = csv.writer(file) # 写入表头 writer.writerow(["Category", "Job Title", "Company Name", "Location", "Link"]) # 创建列表存储信息 jobTitles = [] companyName = [] jobLocation = [] links = [] categories = [] # 新增:存储职位分类 # 目标URL列表(URL与分类绑定,避免冗余判断) URLs = [ ("https://www.indeed.com/jobs?q=IT+Entry+Level&l=United+States&from=searchOnHP&vjk=55e3c7e5e7a919c9", "IT - Entry Level"), ("https://www.indeed.com/jobs?q=development+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=996b270bd119f225", "Development - Entry Level"), ("https://www.indeed.com/jobs?q=user+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=4fc6a443fd7c11df", "User Experience - Entry Level"), ("https://www.indeed.com/jobs?q=ux%2Fui+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=254dcd8c33926527", "UX/UI - Entry Level"), ("https://www.indeed.com/jobs?q=data+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=bd4fd45f0feb91c2", "Data - Entry Level") ] headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36", "Accept-Encoding": "gzip, deflate, br", "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8", "Connection": "keep-alive", "Accept-Language": "en-US,en;q=0.9" } for url, category in URLs: # 请求页面 resp = hrequests.get(url, headers=headers) soup = BeautifulSoup(resp.text, 'html.parser') # 遍历每个职位条目(使用更稳定的职位容器类) job_cards = soup.find_all('div', class_='job_seen_beacon') for card in job_cards: # 抓取职位标题(精准定位带title属性的span) title_elem = card.find('span', title=True) job_title = title_elem.text.strip() if title_elem else "" jobTitles.append(job_title) # 抓取职位链接(补全完整URL) link_elem = card.find('a', href=True) job_link = f"https://www.indeed.com{link_elem['href']}" if link_elem else "" links.append(job_link) # 抓取公司名称 company_elem = card.find('span', {'data-testid': 'company-name'}) company_name = company_elem.text.strip() if company_elem else "" companyName.append(company_name) # 抓取职位地点 location_elem = card.find('div', {'data-testid': 'text-location'}) job_location = location_elem.text.strip() if location_elem else "" jobLocation.append(job_location) # 添加分类信息 categories.append(category) # 验证列表长度 print(f"职位标题数: {len(jobTitles)}") print(f"公司名称数: {len(companyName)}") print(f"职位地点数: {len(jobLocation)}") print(f"链接数: {len(links)}") print(f"分类数: {len(categories)}") # 写入CSV(避免重复表头) df = pd.DataFrame({ "Category": categories, "Job Title": jobTitles, "Company Name": companyName, "Location": jobLocation, "Link": links }) with open('Indeed_Jobs.csv', 'a', newline='', encoding='utf-8') as file: df.to_csv(file, mode='a', header=False, index=False)
关键修改点
- 按职位条目关联信息:遍历单个职位卡片容器,在每个卡片内抓取所有字段,确保信息一一对应
- 缺失字段填充空值:使用
if-else判断元素是否存在,不存在时添加空字符串"" - 优化URL与分类映射:将URL和对应分类组成元组,避免冗余的
if-elif判断,修复原代码中URL判断不匹配的问题 - 更换稳定的容器类:使用
job_seen_beacon作为职位卡片的主容器,比原代码中的类更稳定 - 修复标题抓取逻辑:通过
title=True的span标签精准定位职位标题,避免重复抓取 - 优化CSV写入:统一使用Pandas写入,避免重复表头,添加UTF-8编码防止乱码
内容的提问来源于stack exchange,提问作者Gannan307
相关产品推荐
相关产品推荐

