Python爬虫作业求助:Indeed/CareerJunction爬取遇403或无结果
作业爬虫问题:Indeed/CareerJunction爬取失败(403或无结果)
我有一项作业任务,要求编写Python应用爬取CareerJunction(careerjunction.co.za)的招聘数据:允许用户输入职位名称,提取首页结果的职位名称、招聘方名称、薪资、职位类型、工作地点、发布日期。之后获准改用Indeed.com,但尝试了以下4段代码后,要么出现403 Forbidden错误,要么无返回结果,即使已经修正HTML类选择器,问题依然存在。
第一段代码(CareerJunction爬取)
def scrape_careerjunction(job_title): url = f"https://www.careerjunction.co.za/jobs/results/?keyword={job_title}" headers = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36' } # Send HTTP request response = requests.get(url, headers=headers) if response.status_code == 200: soup = BeautifulSoup(response.content, 'html.parser') jobs = [] # Find all job listings job_listings = soup.find_all('div', class_='listing-item') for job in job_listings: job_title = job.find('h2', class_='title').text.strip() recruiter = job.find('span', class_='company').text.strip() salary = job.find('span', class_='salary').text.strip() position = job.find('span', class_='location').text.strip() location = job.find('span', class_='area').text.strip() date_posted = job.find('span', class_='time').text.strip() job_info = { 'Job Title': job_title, 'Recruiter': recruiter, 'Salary': salary, 'Position': position, 'Location': location, 'Date Posted': date_posted } jobs.append(job_info) return jobs else: print(f"Failed to retrieve data, status code: {response.status_code}") return []
第二段代码(Indeed爬取)
import requests from bs4 import BeautifulSoup # Function to scrape job data from Indeed def scrape_jobs(job_title): # Replace spaces in job_title with '+' job_title = job_title.replace(' ', '+') # The base URL of Indeed base_url = 'https://www.indeed.com/jobs?q=' # Complete URL with the job title search_url = f'{base_url}{job_title}' # Send a request to the website response = requests.get(search_url) # Check if the request was successful if response.status_code == 200: # Parse the content with BeautifulSoup soup = BeautifulSoup(response.content, 'html.parser') # Find all job entries - this will depend on the website's structure job_entries = soup.find_all('div', class_='jobsearch-SerpJobCard') # Replace with actual class for job in job_entries: # Extract the required information job_title = job.find('h2', class_='title').text.strip() # Replace with actual class company_name = job.find('span', class_='company').text.strip() # Replace with actual class job_location = job.find('span', class_='location').text.strip() # Replace with actual class job_summary = job.find('div', class_='summary').text.strip() # Replace with actual class # Print the extracted information print(f'Job Title: {job_title}') print(f'Company Name: {company_name}') print(f'Location: {job_location}') print(f'Summary: {job_summary}') print('-----------------------------------') else: print('Failed to retrieve the webpage') # Example usage job_to_search = input('Enter a job title to search for: ') scrape_jobs(job_to_search)
第三段代码(Indeed URL生成)
import csv from datetime import datetime import bs4 as BeautifulSoup import requests def get_url(position, location): """Generate a url from position and location""" template = 'https://za.indeed.com/jobs?q={}&l={}' url = template.format(position, location) return url url = get_url('senior accountant', 'charlotte nc') response = requests.get(url) response
第四段代码(Indeed数据提取)
# import module import requests from bs4 import BeautifulSoup # user define function # Scrape the data # and get in string def getdata(url): r = requests.get(url) return r.text # Get Html code using parse def html_code(url): # pass the url # into getdata function htmldata = getdata(url) soup = BeautifulSoup(htmldata, 'html.parser') # return html code return(soup) # filter job data using # find_all function def job_data(soup): # find the Html tag # with find() # and convert into string data_str = "" for item in soup.find_all("a", class_="jobtitle turnstileLink"): data_str = data_str + item.get_text() result_1 = data_str.split("\n") return(result_1) # filter company_data using # find_all function def company_data(soup): # find the Html tag # with find() # and convert into string data_str = "" result = "" for item in soup.find_all("div", class_="sjcl"): data_str = data_str + item.get_text() result_1 = data_str.split("\n") res = [] for i in range(1, len(result_1)): if len(result_1[i]) > 1: res.append(result_1[i]) return(res) # driver nodes/main function if __name__ == "__main__": # Data for URL job = "data+science+internship" Location = "Noida%2C+Uttar+Pradesh" url = "https://in.indeed.com/jobs?q="+job+"&l="+Location # Pass this URL into the soup # which will return # html string soup = html_code(url) # call job and company data # and store into it var job_res = job_data(soup) com_res = company_data(soup) # Traverse the both data temp = 0 for i in range(1, len(job_res)): j = temp for j in range(temp, 2+temp): print("Company Name and Address : " + com_res[j]) temp = j print("Job : " + job_res[i]) print("-----------------------------")
问题原因及解决办法
1. 403错误的核心原因
网站反爬机制识别出脚本请求,而非真实浏览器访问,主要因为:
- 请求头不完整(仅User-Agent不足以模拟真实浏览器)
- 未处理会话Cookie
- 部分站点会验证请求来源
2. 可运行的修复代码(适配Indeed南非站点)
以下代码补全请求头、适配当前页面结构,可解决403和无结果问题:
import requests from bs4 import BeautifulSoup def scrape_indeed_jobs(job_title): # 格式化搜索关键词 formatted_job = job_title.replace(" ", "+") # 用南非Indeed站点匹配原任务地区 url = f"https://za.indeed.com/jobs?q={formatted_job}" # 模拟真实Chrome浏览器的完整请求头 headers = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36', 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8', 'Accept-Language': 'en-US,en;q=0.5', 'Accept-Encoding': 'gzip, deflate, br', 'Connection': 'keep-alive', 'Upgrade-Insecure-Requests': '1' } # 发送请求并允许重定向 response = requests.get(url, headers=headers, allow_redirects=True) if response.status_code == 200: soup = BeautifulSoup(response.text, 'html.parser') jobs = [] # 匹配当前Indeed南非站点的职位卡片类名 job_cards = soup.find_all('div', class_='job_seen_beacon') for card in job_cards: # 提取职位名称 job_title_elem = card.find('h2', class_='jobTitle') job_title = job_title_elem.find('span').text.strip() if job_title_elem else "未标注" # 提取公司名称 company_elem = card.find('span', class_='companyName') company = company_elem.text.strip() if company_elem else "未标注" # 提取工作地点 location_elem = card.find('div', class_='companyLocation') location = location_elem.text.strip() if location_elem else "未标注" # 提取薪资(部分职位无公开薪资) salary_elem = card.find('div', class_='salary-snippet-container') salary = salary_elem.text.strip() if salary_elem else "未标注" # 提取职位类型 job_type = "未标注" type_elem = card.find('div', class_='metadata') if type_elem: job_type = type_elem.text.strip() # 提取发布日期 date_elem = card.find('span', class_='date') date_posted = date_elem.text.strip() if date_elem else "未标注" job_info = { '职位名称': job_title, '招聘方': company, '薪资': salary, '职位类型': job_type, '工作地点': location, '发布日期': date_posted } jobs.append(job_info) # 打印结果 for idx, job in enumerate(jobs, 1): print(f"职位 {idx}:") for key, value in job.items(): print(f"{key}: {value}") print("-" * 50) return jobs else: print(f"请求失败,状态码: {response.status_code}") return [] # 用户交互入口 if __name__ == "__main__": job_input = input("请输入要搜索的职位名称: ") scrape_indeed_jobs(job_input)
3. 后续注意事项
- 类名更新:Indeed页面类名会不定期调整,若再次无结果,用浏览器F12开发者工具重新定位元素类名
- 反爬升级:若仍触发403,可添加
time.sleep(1)控制请求间隔,或用requests.Session()维持会话 - CareerJunction修复:同样需要补全请求头,若页面有动态加载,需用Selenium模拟浏览器操作(作业若允许静态爬取,优先补全请求头)
内容的提问来源于stack exchange,提问作者Xotic Flash24
相关产品推荐
相关产品推荐

