You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python爬虫作业求助:Indeed/CareerJunction爬取遇403或无结果

作业爬虫问题:Indeed/CareerJunction爬取失败(403或无结果)

我有一项作业任务,要求编写Python应用爬取CareerJunction(careerjunction.co.za)的招聘数据:允许用户输入职位名称,提取首页结果的职位名称、招聘方名称、薪资、职位类型、工作地点、发布日期。之后获准改用Indeed.com,但尝试了以下4段代码后,要么出现403 Forbidden错误,要么无返回结果,即使已经修正HTML类选择器,问题依然存在。


第一段代码(CareerJunction爬取)

def scrape_careerjunction(job_title):
    url = f"https://www.careerjunction.co.za/jobs/results/?keyword={job_title}"
    headers = {
        'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML,  like Gecko) Chrome/91.0.4472.124 Safari/537.36'
    }

    # Send HTTP request
    response = requests.get(url, headers=headers)

    if response.status_code == 200:
        soup = BeautifulSoup(response.content, 'html.parser')
    
        jobs = []
     
        # Find all job listings
        job_listings = soup.find_all('div', class_='listing-item')
    
        for job in job_listings:
            job_title = job.find('h2', class_='title').text.strip()
            recruiter = job.find('span', class_='company').text.strip()
            salary = job.find('span', class_='salary').text.strip()
            position = job.find('span', class_='location').text.strip()
            location = job.find('span', class_='area').text.strip()
            date_posted = job.find('span', class_='time').text.strip()
        
            job_info = {
                'Job Title': job_title,
                'Recruiter': recruiter,
                'Salary': salary,
                'Position': position,
                'Location': location,
                'Date Posted': date_posted
            }
        
            jobs.append(job_info)
        
        return jobs
    else:
        print(f"Failed to retrieve data, status code: {response.status_code}")
        return []

第二段代码(Indeed爬取)

import requests
from bs4 import BeautifulSoup

# Function to scrape job data from Indeed
def scrape_jobs(job_title):
    # Replace spaces in job_title with '+'
    job_title = job_title.replace(' ', '+')

    # The base URL of Indeed
    base_url = 'https://www.indeed.com/jobs?q='

    # Complete URL with the job title
    search_url = f'{base_url}{job_title}'

    # Send a request to the website
    response = requests.get(search_url)

    # Check if the request was successful
    if response.status_code == 200:
        # Parse the content with BeautifulSoup
        soup = BeautifulSoup(response.content, 'html.parser')
    
        # Find all job entries - this will depend on the website's structure
        job_entries = soup.find_all('div', class_='jobsearch-SerpJobCard') # Replace with  actual class
    
        for job in job_entries:
            # Extract the required information
            job_title = job.find('h2', class_='title').text.strip() # Replace with actual class
            company_name = job.find('span', class_='company').text.strip() # Replace with actual class
            job_location = job.find('span', class_='location').text.strip() # Replace with actual class
            job_summary = job.find('div', class_='summary').text.strip() # Replace with actual class
        
            # Print the extracted information
            print(f'Job Title: {job_title}')
            print(f'Company Name: {company_name}')
            print(f'Location: {job_location}')
            print(f'Summary: {job_summary}')
            print('-----------------------------------')
    else:
        print('Failed to retrieve the webpage')

# Example usage
job_to_search = input('Enter a job title to search for: ')
scrape_jobs(job_to_search)

第三段代码(Indeed URL生成)

import csv
from datetime import datetime
import bs4 as BeautifulSoup
import requests

def get_url(position, location):
    """Generate a url from position and location"""
    template = 'https://za.indeed.com/jobs?q={}&l={}'
    url = template.format(position, location)
    return url

url = get_url('senior accountant', 'charlotte nc')

response = requests.get(url)
response

第四段代码(Indeed数据提取)

# import module 
import requests 
from bs4 import BeautifulSoup 

# user define function 
# Scrape the data 
# and get in string 
def getdata(url): 
    r = requests.get(url) 
    return r.text 

# Get Html code using parse 
def html_code(url): 
    # pass the url 
    # into getdata function 
    htmldata = getdata(url) 
    soup = BeautifulSoup(htmldata, 'html.parser') 
    # return html code 
    return(soup) 

# filter job data using 
# find_all function 
def job_data(soup): 
    # find the Html tag 
    # with find() 
    # and convert into string 
    data_str = "" 
    for item in soup.find_all("a", class_="jobtitle turnstileLink"): 
        data_str = data_str + item.get_text() 
    result_1 = data_str.split("\n") 
    return(result_1) 

# filter company_data using 
# find_all function 
def company_data(soup): 
    # find the Html tag 
    # with find() 
    # and convert into string 
    data_str = "" 
    result = "" 
    for item in soup.find_all("div", class_="sjcl"): 
        data_str = data_str + item.get_text() 
    result_1 = data_str.split("\n") 
    res = [] 
    for i in range(1, len(result_1)): 
        if len(result_1[i]) > 1: 
            res.append(result_1[i]) 
    return(res) 

# driver nodes/main function 
if __name__ == "__main__": 
    # Data for URL 
    job = "data+science+internship"
    Location = "Noida%2C+Uttar+Pradesh"
    url = "https://in.indeed.com/jobs?q="+job+"&l="+Location 
    # Pass this URL into the soup 
    # which will return 
    # html string 
    soup = html_code(url) 
    # call job and company data 
    # and store into it var 
    job_res = job_data(soup) 
    com_res = company_data(soup) 
    # Traverse the both data 
    temp = 0
    for i in range(1, len(job_res)): 
        j = temp 
        for j in range(temp, 2+temp): 
            print("Company Name and Address : " + com_res[j]) 
        temp = j 
        print("Job : " + job_res[i]) 
        print("-----------------------------") 

问题原因及解决办法

1. 403错误的核心原因

网站反爬机制识别出脚本请求,而非真实浏览器访问,主要因为:

  • 请求头不完整(仅User-Agent不足以模拟真实浏览器)
  • 未处理会话Cookie
  • 部分站点会验证请求来源

2. 可运行的修复代码(适配Indeed南非站点)

以下代码补全请求头、适配当前页面结构,可解决403和无结果问题:

import requests
from bs4 import BeautifulSoup

def scrape_indeed_jobs(job_title):
    # 格式化搜索关键词
    formatted_job = job_title.replace(" ", "+")
    # 用南非Indeed站点匹配原任务地区
    url = f"https://za.indeed.com/jobs?q={formatted_job}"
    
    # 模拟真实Chrome浏览器的完整请求头
    headers = {
        'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36',
        'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
        'Accept-Language': 'en-US,en;q=0.5',
        'Accept-Encoding': 'gzip, deflate, br',
        'Connection': 'keep-alive',
        'Upgrade-Insecure-Requests': '1'
    }
    
    # 发送请求并允许重定向
    response = requests.get(url, headers=headers, allow_redirects=True)
    
    if response.status_code == 200:
        soup = BeautifulSoup(response.text, 'html.parser')
        jobs = []
        
        # 匹配当前Indeed南非站点的职位卡片类名
        job_cards = soup.find_all('div', class_='job_seen_beacon')
        
        for card in job_cards:
            # 提取职位名称
            job_title_elem = card.find('h2', class_='jobTitle')
            job_title = job_title_elem.find('span').text.strip() if job_title_elem else "未标注"
            
            # 提取公司名称
            company_elem = card.find('span', class_='companyName')
            company = company_elem.text.strip() if company_elem else "未标注"
            
            # 提取工作地点
            location_elem = card.find('div', class_='companyLocation')
            location = location_elem.text.strip() if location_elem else "未标注"
            
            # 提取薪资(部分职位无公开薪资)
            salary_elem = card.find('div', class_='salary-snippet-container')
            salary = salary_elem.text.strip() if salary_elem else "未标注"
            
            # 提取职位类型
            job_type = "未标注"
            type_elem = card.find('div', class_='metadata')
            if type_elem:
                job_type = type_elem.text.strip()
            
            # 提取发布日期
            date_elem = card.find('span', class_='date')
            date_posted = date_elem.text.strip() if date_elem else "未标注"
            
            job_info = {
                '职位名称': job_title,
                '招聘方': company,
                '薪资': salary,
                '职位类型': job_type,
                '工作地点': location,
                '发布日期': date_posted
            }
            jobs.append(job_info)
        
        # 打印结果
        for idx, job in enumerate(jobs, 1):
            print(f"职位 {idx}:")
            for key, value in job.items():
                print(f"{key}: {value}")
            print("-" * 50)
        return jobs
    else:
        print(f"请求失败,状态码: {response.status_code}")
        return []

# 用户交互入口
if __name__ == "__main__":
    job_input = input("请输入要搜索的职位名称: ")
    scrape_indeed_jobs(job_input)

3. 后续注意事项

  • 类名更新:Indeed页面类名会不定期调整,若再次无结果,用浏览器F12开发者工具重新定位元素类名
  • 反爬升级:若仍触发403,可添加time.sleep(1)控制请求间隔,或用requests.Session()维持会话
  • CareerJunction修复:同样需要补全请求头,若页面有动态加载,需用Selenium模拟浏览器操作(作业若允许静态爬取,优先补全请求头)

内容的提问来源于stack exchange,提问作者Xotic Flash24

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 02:39:51