You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python爬取Indeed远程Data Analyst职位写入CSV无数据排查

问题:Indeed爬虫仅生成CSV表头无数据排查

我是Python新手,尝试爬取Indeed平台上的远程Data Analyst职位并写入CSV文件。为解决SSL证书问题设置了verify=False,终端提示爬取完成,但CSV仅显示表头无数据。相关代码及终端输出如下:

import requests
import csv
from bs4 import BeautifulSoup

# Define the dataanalyst variable.
dataanalyst = "data analyst"

def get_job_postings(dataanalyst):
  """Gets the job postings from Indeed for the given keyword."""

  # Get the Indeed search URL for the given keyword.
  search_url = "https://www.indeed.com/jobs?q=data+analyst&l=remote&vjk=30f58c7471301c42".format(keyword)

  # Make a request to the Indeed search URL.
  response = requests.get(search_url, verify=False)

  # Parse the response and get the job postings.
  soup = BeautifulSoup(response.content, "html.parser")
  job_postings = soup.find_all("div", class_="jobsearch-result")

  return job_postings

def write_job_postings_to_csv(job_postings, filename):
  """Writes the job postings to a CSV file."""

  # Create a CSV file to store the job postings.
  with open(filename, "w", newline="") as csvfile:

    # Create a CSV writer object.
    writer = csv.writer(csvfile)

    # Write the header row to the CSV file.
    writer.writerow(["Title", "Company", "Location", "Description"])

    # Write the job postings to the CSV file.
    for job_posting in job_postings:
      title = job_posting.find("h2", class_="jobtitle").text
      company = job_posting.find("span", class_="company").text
      location = job_posting.find("span", class_="location").text
      description = job_posting.find("div", class_="job-snippet").text

      writer.writerow([title, company, location, description])

if __name__ == "__main__":

  # Define the dataanalyst variable.
  dataanalyst = "data+analyst"

  # Get the keyword from the user.
  keyword = "data analyst"

  # Get the job postings from Indeed.
  job_postings = get_job_postings(dataanalyst)

  # Write the job postings to a CSV file.
  write_job_postings_to_csv(job_postings, "remote_data_analyst_positions.csv")

  print("The job postings have been successfully scraped and written to a CSV file.")

终端输出:

PS C:\Users\chlor\OneDrive\Documents\Python> & C:/Users/chlor/AppData/Local/Programs/Python/Python311/python.exe c:/Users/chlor/OneDrive/Documents/Python/Indeed_DataAnalyst_Remote/DataAnalyst_Remote.py
C:\Users\chlor\AppData\Local\Programs\Python\Python311\Lib\site-packages\urllib3\connectionpool.py:1045: InsecureRequestWarning: Unverified HTTPS request is being made to host 'localhost'. Adding certificate verification is strongly advised. See: https://urllib3.readthedocs.io/en/1.26.x/advanced-usage.html#ssl-warnings
warnings.warn(
The job postings have been successfully scraped and written to a CSV file.
PS C:\Users\chlor\OneDrive\Documents\Python>

我期望CSV文件能写入职位信息,但实际只有表头。


错误排查及修复方案

  • URL逻辑错误:

    1. search_url使用.format(keyword)但字符串无占位符,同时硬编码data+analyst导致函数参数dataanalyst完全未被使用,冗余的vjk是会话专属参数,可能引发请求异常。
    2. 终端提示访问localhost,需确认本地无代理/hosts篡改,确保请求目标为Indeed官网。修复后URL生成代码:
      search_url = f"https://www.indeed.com/jobs?q={dataanalyst.replace(' ', '+')}&l=remote"
      
  • HTML元素定位错误:
    Indeed当前职位卡片类名并非jobsearch-result,实际为job_seen_beacon,原代码无法匹配任何节点,导致job_postings为空列表。修复定位代码:

    job_postings = soup.find_all("div", class_="job_seen_beacon")
    
  • 子元素定位无容错处理:
    直接调用.text会因元素不存在触发异常,且Indeed子元素类名已更新(如公司名类为companyName,位置类为companyLocation),修复后代码:

    title = job_posting.find("h2").get_text(strip=True) if job_posting.find("h2") else "N/A"
    company = job_posting.find("span", class_="companyName").get_text(strip=True) if job_posting.find("span", class_="companyName") else "N/A"
    location = job_posting.find("div", class_="companyLocation").get_text(strip=True) if job_posting.find("div", class_="companyLocation") else "N/A"
    description = job_posting.find("div", class_="job-snippet").get_text(strip=True) if job_posting.find("div", class_="job-snippet") else "N/A"
    
  • 冗余变量清理:
    代码重复定义dataanalyst且keyword未被使用,清理后主函数逻辑更清晰:

    if __name__ == "__main__":
        keyword = "data analyst"
        job_postings = get_job_postings(keyword)
        write_job_postings_to_csv(job_postings, "remote_data_analyst_positions.csv")
        print("The job postings have been successfully scraped and written to a CSV file.")
    

修复后完整代码

import requests
import csv
from bs4 import BeautifulSoup

def get_job_postings(keyword):
    """Gets the job postings from Indeed for the given keyword."""
    search_url = f"https://www.indeed.com/jobs?q={keyword.replace(' ', '+')}&l=remote"
    # 添加请求头模拟浏览器,避免被反爬拦截
    headers = {
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36"
    }
    response = requests.get(search_url, headers=headers, verify=True)  # 建议开启证书验证,若仍有问题再关闭
    soup = BeautifulSoup(response.content, "html.parser")
    job_postings = soup.find_all("div", class_="job_seen_beacon")
    return job_postings

def write_job_postings_to_csv(job_postings, filename):
    """Writes the job postings to a CSV file."""
    with open(filename, "w", newline="", encoding="utf-8") as csvfile:
        writer = csv.writer(csvfile)
        writer.writerow(["Title", "Company", "Location", "Description"])
        for job_posting in job_postings:
            title = job_posting.find("h2").get_text(strip=True) if job_posting.find("h2") else "N/A"
            company = job_posting.find("span", class_="companyName").get_text(strip=True) if job_posting.find("span", class_="companyName") else "N/A"
            location = job_posting.find("div", class_="companyLocation").get_text(strip=True) if job_posting.find("div", class_="companyLocation") else "N/A"
            description = job_posting.find("div", class_="job-snippet").get_text(strip=True) if job_posting.find("div", class_="job-snippet") else "N/A"
            writer.writerow([title, company, location, description])

if __name__ == "__main__":
    keyword = "data analyst"
    job_postings = get_job_postings(keyword)
    write_job_postings_to_csv(job_postings, "remote_data_analyst_positions.csv")
    print(f"Successfully scraped {len(job_postings)} job postings and saved to CSV.")

内容的提问来源于stack exchange,提问作者EpicMe

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.21 15:17:56