Python爬取Indeed远程Data Analyst职位写入CSV无数据排查
我是Python新手,尝试爬取Indeed平台上的远程Data Analyst职位并写入CSV文件。为解决SSL证书问题设置了verify=False,终端提示爬取完成,但CSV仅显示表头无数据。相关代码及终端输出如下:
import requests import csv from bs4 import BeautifulSoup # Define the dataanalyst variable. dataanalyst = "data analyst" def get_job_postings(dataanalyst): """Gets the job postings from Indeed for the given keyword.""" # Get the Indeed search URL for the given keyword. search_url = "https://www.indeed.com/jobs?q=data+analyst&l=remote&vjk=30f58c7471301c42".format(keyword) # Make a request to the Indeed search URL. response = requests.get(search_url, verify=False) # Parse the response and get the job postings. soup = BeautifulSoup(response.content, "html.parser") job_postings = soup.find_all("div", class_="jobsearch-result") return job_postings def write_job_postings_to_csv(job_postings, filename): """Writes the job postings to a CSV file.""" # Create a CSV file to store the job postings. with open(filename, "w", newline="") as csvfile: # Create a CSV writer object. writer = csv.writer(csvfile) # Write the header row to the CSV file. writer.writerow(["Title", "Company", "Location", "Description"]) # Write the job postings to the CSV file. for job_posting in job_postings: title = job_posting.find("h2", class_="jobtitle").text company = job_posting.find("span", class_="company").text location = job_posting.find("span", class_="location").text description = job_posting.find("div", class_="job-snippet").text writer.writerow([title, company, location, description]) if __name__ == "__main__": # Define the dataanalyst variable. dataanalyst = "data+analyst" # Get the keyword from the user. keyword = "data analyst" # Get the job postings from Indeed. job_postings = get_job_postings(dataanalyst) # Write the job postings to a CSV file. write_job_postings_to_csv(job_postings, "remote_data_analyst_positions.csv") print("The job postings have been successfully scraped and written to a CSV file.")
终端输出:
PS C:\Users\chlor\OneDrive\Documents\Python> & C:/Users/chlor/AppData/Local/Programs/Python/Python311/python.exe c:/Users/chlor/OneDrive/Documents/Python/Indeed_DataAnalyst_Remote/DataAnalyst_Remote.py
C:\Users\chlor\AppData\Local\Programs\Python\Python311\Lib\site-packages\urllib3\connectionpool.py:1045: InsecureRequestWarning: Unverified HTTPS request is being made to host 'localhost'. Adding certificate verification is strongly advised. See: https://urllib3.readthedocs.io/en/1.26.x/advanced-usage.html#ssl-warnings
warnings.warn(
The job postings have been successfully scraped and written to a CSV file.
PS C:\Users\chlor\OneDrive\Documents\Python>
我期望CSV文件能写入职位信息,但实际只有表头。
错误排查及修复方案
URL逻辑错误:
search_url使用.format(keyword)但字符串无占位符,同时硬编码data+analyst导致函数参数dataanalyst完全未被使用,冗余的vjk是会话专属参数,可能引发请求异常。- 终端提示访问
localhost,需确认本地无代理/hosts篡改,确保请求目标为Indeed官网。修复后URL生成代码:search_url = f"https://www.indeed.com/jobs?q={dataanalyst.replace(' ', '+')}&l=remote"
HTML元素定位错误:
Indeed当前职位卡片类名并非jobsearch-result,实际为job_seen_beacon,原代码无法匹配任何节点,导致job_postings为空列表。修复定位代码:job_postings = soup.find_all("div", class_="job_seen_beacon")子元素定位无容错处理:
直接调用.text会因元素不存在触发异常,且Indeed子元素类名已更新(如公司名类为companyName,位置类为companyLocation),修复后代码:title = job_posting.find("h2").get_text(strip=True) if job_posting.find("h2") else "N/A" company = job_posting.find("span", class_="companyName").get_text(strip=True) if job_posting.find("span", class_="companyName") else "N/A" location = job_posting.find("div", class_="companyLocation").get_text(strip=True) if job_posting.find("div", class_="companyLocation") else "N/A" description = job_posting.find("div", class_="job-snippet").get_text(strip=True) if job_posting.find("div", class_="job-snippet") else "N/A"冗余变量清理:
代码重复定义dataanalyst且keyword未被使用,清理后主函数逻辑更清晰:if __name__ == "__main__": keyword = "data analyst" job_postings = get_job_postings(keyword) write_job_postings_to_csv(job_postings, "remote_data_analyst_positions.csv") print("The job postings have been successfully scraped and written to a CSV file.")
修复后完整代码
import requests import csv from bs4 import BeautifulSoup def get_job_postings(keyword): """Gets the job postings from Indeed for the given keyword.""" search_url = f"https://www.indeed.com/jobs?q={keyword.replace(' ', '+')}&l=remote" # 添加请求头模拟浏览器,避免被反爬拦截 headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36" } response = requests.get(search_url, headers=headers, verify=True) # 建议开启证书验证,若仍有问题再关闭 soup = BeautifulSoup(response.content, "html.parser") job_postings = soup.find_all("div", class_="job_seen_beacon") return job_postings def write_job_postings_to_csv(job_postings, filename): """Writes the job postings to a CSV file.""" with open(filename, "w", newline="", encoding="utf-8") as csvfile: writer = csv.writer(csvfile) writer.writerow(["Title", "Company", "Location", "Description"]) for job_posting in job_postings: title = job_posting.find("h2").get_text(strip=True) if job_posting.find("h2") else "N/A" company = job_posting.find("span", class_="companyName").get_text(strip=True) if job_posting.find("span", class_="companyName") else "N/A" location = job_posting.find("div", class_="companyLocation").get_text(strip=True) if job_posting.find("div", class_="companyLocation") else "N/A" description = job_posting.find("div", class_="job-snippet").get_text(strip=True) if job_posting.find("div", class_="job-snippet") else "N/A" writer.writerow([title, company, location, description]) if __name__ == "__main__": keyword = "data analyst" job_postings = get_job_postings(keyword) write_job_postings_to_csv(job_postings, "remote_data_analyst_positions.csv") print(f"Successfully scraped {len(job_postings)} job postings and saved to CSV.")
内容的提问来源于stack exchange,提问作者EpicMe

