You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Indeed爬虫列表长度不一致求助:jobTitles/links长于companyName等

Indeed爬虫列表长度不一致问题修复

问题

运行Indeed爬虫时,jobTitles和links列表长度始终大于companyName和jobLocation,各列表长度不统一,无法正常写入CSV。原代码如下:

import hrequests
from bs4 import BeautifulSoup
import pandas as pd
import csv

####################
## INDEED SCRAPER ##
####################

# Creates the CSV in Write Mode
with open('Indeed_Jobs.csv', 'w', newline='') as file:
    writer = csv.writer(file)

#Creating Lists for info
jobTitles = []
companyName = []
jobLocation = []
links = []

#Creates list of URLs
URLs = ["https://www.indeed.com/jobs?q=IT+Entry+Level&l=United+States&from=searchOnHP&vjk=55e3c7e5e7a919c9", 
        "https://www.indeed.com/jobs?q=development+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=996b270bd119f225",
        "https://www.indeed.com/jobs?q=user+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=4fc6a443fd7c11df",
        "https://www.indeed.com/jobs?q=ux%2Fui+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=254dcd8c33926527",
        "https://www.indeed.com/jobs?q=data+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=bd4fd45f0feb91c2"]

for url in URLs:

    #Connecting to Indeed and reading HTML
    target_url = url
    head= {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/62.0.3202.94 Safari/537.36",
        "Accept-Encoding": "gzip, deflate, br",
        "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8",
        "Connection": "keep-alive",
        "Accept-Language": "en-US,en;q=0.9,lt;q=0.8,et;q=0.7,de;q=0.6",
    }
    resp = hrequests.get(target_url, headers=head)
    soup = BeautifulSoup(resp.text, 'html.parser')

    #Finds all items in list
    for div in soup.find_all('div', {'class': 'css-dekpa e37uo190'}):
        #Pulls job title
        for span in div.find_all('span'):
            jobTitles.append(span.text.strip())
        #Pulls link to job
        for a in div.find_all('a'):
            links.append("indeed.com" + a.get('href'))
        #Pulls company name
    for div in soup.find_all('div', {'class': 'company_location css-17fky0v e37uo190'}):
        for span in div.find_all('span',{'data-testid': 'company-name'}):
            companyName.append(span.text.strip())
        #Pulls job location
        for divLocation in div.find_all('div', {'data-testid': 'text-location'}):
            jobLocation.append(divLocation.text.strip())


    # Changing the heading on the CSV depending on which URL was searched
    if url == "https://www.indeed.com/jobs?q=development+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=996b270bd119f225":
        additional_row = ["Development - Entry Level"]
    elif url =="https://www.indeed.com/jobs?q=it+entry+level&l=United+States&from=searchOnHP&vjk=bf9a1055e66b240c":
        additional_row = ["IT - Entry Level"]
    elif url == "https://www.indeed.com/jobs?q=user+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=4fc6a443fd7c11df":
        additional_row = ["User Experience - Entry Level"]
    elif url == "https://www.indeed.com/jobs?q=ux%2Fui+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=254dcd8c33926527":
        additional_row = ["UX/UI - Entry Level"]
    elif url == "https://www.indeed.com/jobs?q=data+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=bd4fd45f0feb91c2":
        additional_row = ["Data - Entry Level"]
    else:
        additional_row = ["Title Not Found"]

    # Blank Row to make formatting nice
    blank_row = ['']



print(len(jobTitles))
print(len(companyName))
print(len(jobLocation))
print(len(links))

#
#   # Open the CSV file in append mode
#    with open('Indeed_Jobs.csv', 'a', newline='') as file:
#        writer = csv.writer(file)
#        # Write the header and blank rows for formatting
#        writer.writerow(blank_row)
#        writer.writerow(additional_row)
#        writer.writerow(blank_row)
#
#        # Write jobs to the CSV
#        df = pd.DataFrame({'Job Title': jobTitles, 'Company Name': companyName, 'Location': jobLocation, 'Link': links})
#        df.to_csv(file, mode='a', header=True, index=False)

问题根源

原代码分开遍历不同的DOM块,没有按单个职位条目关联信息:

  • 先遍历职位标题和链接的容器,再单独遍历公司和地点的容器
  • 若某个职位缺失公司/地点信息,对应的列表不会添加空值,导致所有列表长度错位
  • 职位标题的遍历逻辑错误:div.find_all('span')会抓取多个span标签,导致jobTitles重复添加内容

修复方案

按单个职位条目遍历,在每个职位容器内抓取所有字段,缺失时填充空值,确保每个列表的元素一一对应:

import hrequests
from bs4 import BeautifulSoup
import pandas as pd
import csv

####################
## INDEED SCRAPER ##
####################

# 初始化CSV文件(仅执行一次)
with open('Indeed_Jobs.csv', 'w', newline='', encoding='utf-8') as file:
    writer = csv.writer(file)
    # 写入表头
    writer.writerow(["Category", "Job Title", "Company Name", "Location", "Link"])

# 创建列表存储信息
jobTitles = []
companyName = []
jobLocation = []
links = []
categories = []  # 新增:存储职位分类

# 目标URL列表(URL与分类绑定,避免冗余判断)
URLs = [
    ("https://www.indeed.com/jobs?q=IT+Entry+Level&l=United+States&from=searchOnHP&vjk=55e3c7e5e7a919c9", "IT - Entry Level"),
    ("https://www.indeed.com/jobs?q=development+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=996b270bd119f225", "Development - Entry Level"),
    ("https://www.indeed.com/jobs?q=user+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=4fc6a443fd7c11df", "User Experience - Entry Level"),
    ("https://www.indeed.com/jobs?q=ux%2Fui+experience+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=254dcd8c33926527", "UX/UI - Entry Level"),
    ("https://www.indeed.com/jobs?q=data+entry+level&l=United+States&from=searchOnDesktopSerp&vjk=bd4fd45f0feb91c2", "Data - Entry Level")
]

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36",
    "Accept-Encoding": "gzip, deflate, br",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8",
    "Connection": "keep-alive",
    "Accept-Language": "en-US,en;q=0.9"
}

for url, category in URLs:
    # 请求页面
    resp = hrequests.get(url, headers=headers)
    soup = BeautifulSoup(resp.text, 'html.parser')
    
    # 遍历每个职位条目(使用更稳定的职位容器类)
    job_cards = soup.find_all('div', class_='job_seen_beacon')
    for card in job_cards:
        # 抓取职位标题(精准定位带title属性的span)
        title_elem = card.find('span', title=True)
        job_title = title_elem.text.strip() if title_elem else ""
        jobTitles.append(job_title)
        
        # 抓取职位链接(补全完整URL)
        link_elem = card.find('a', href=True)
        job_link = f"https://www.indeed.com{link_elem['href']}" if link_elem else ""
        links.append(job_link)
        
        # 抓取公司名称
        company_elem = card.find('span', {'data-testid': 'company-name'})
        company_name = company_elem.text.strip() if company_elem else ""
        companyName.append(company_name)
        
        # 抓取职位地点
        location_elem = card.find('div', {'data-testid': 'text-location'})
        job_location = location_elem.text.strip() if location_elem else ""
        jobLocation.append(job_location)
        
        # 添加分类信息
        categories.append(category)

# 验证列表长度
print(f"职位标题数: {len(jobTitles)}")
print(f"公司名称数: {len(companyName)}")
print(f"职位地点数: {len(jobLocation)}")
print(f"链接数: {len(links)}")
print(f"分类数: {len(categories)}")

# 写入CSV(避免重复表头)
df = pd.DataFrame({
    "Category": categories,
    "Job Title": jobTitles,
    "Company Name": companyName,
    "Location": jobLocation,
    "Link": links
})

with open('Indeed_Jobs.csv', 'a', newline='', encoding='utf-8') as file:
    df.to_csv(file, mode='a', header=False, index=False)

关键修改点

  1. 按职位条目关联信息:遍历单个职位卡片容器,在每个卡片内抓取所有字段,确保信息一一对应
  2. 缺失字段填充空值:使用if-else判断元素是否存在,不存在时添加空字符串""
  3. 优化URL与分类映射:将URL和对应分类组成元组,避免冗余的if-elif判断,修复原代码中URL判断不匹配的问题
  4. 更换稳定的容器类:使用job_seen_beacon作为职位卡片的主容器,比原代码中的类更稳定
  5. 修复标题抓取逻辑:通过title=True的span标签精准定位职位标题,避免重复抓取
  6. 优化CSV写入:统一使用Pandas写入,避免重复表头,添加UTF-8编码防止乱码

内容的提问来源于stack exchange,提问作者Gannan307

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.23 19:03:10