IOL网站新闻爬取求助:DataFrame报错及正文获取失败
爬取IOL网站新闻的问题解决
问题描述
爬取IOL网站(https://www.iol.co.za/news/south-africa/eastern-cape)的新闻标题、日期、链接及正文内容时,遇到两个核心问题:
- 存储数据到pandas DataFrame时触发
ValueError: All arrays must be of the same length - 无法通过文章链接获取正文内容
用户尝试用h标签匹配标题,但因页面元素类名和标签不统一导致提取异常,原代码如下:
import sys, time from bs4 import BeautifulSoup import requests import pandas as pd from selenium import webdriver from webdriver_manager.chrome import ChromeDriverManager from datetime import timedelta from selenium.common.exceptions import TimeoutException from selenium.webdriver.common.by import By from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.support.wait import WebDriverWait import re art_title = [] # to store the titles of all news article art_date = [] # to store the dates of all news article art_link = [] # to store the links of all news article pagesToGet = ['south-africa/eastern-cape'] for i in range(0, len(pagesToGet)): print('processing page : \n') url = 'https://www.iol.co.za' + str(pagesToGet[i]) print(url) driver = webdriver.Chrome(ChromeDriverManager().install()) driver.maximize_window() #time.sleep(5) # allow you to sleep your code before your retrieve the elements from the webpage. Additionally, to # prevent the chrome driver opening a new instance for every url, open the browser outside of the loop. # an exception might be thrown, so the code should be in a try-except block try: # use the browser to get the url. This is suspicious command that might blow up. driver.get("https://www.iol.co.za/news/" +str(pagesToGet[i])) except Exception as e: # this describes what to do if an exception is thrown error_type, error_obj, error_info = sys.exc_info() # get the exception information print('ERROR FOR LINK:', url) # print the link that cause the problem print(error_type, 'Line:', error_info.tb_lineno) # print error info and line that threw the exception continue # ignore this page. Abandon this and go back. time.sleep(3) # Allow 3 seconds for the web page to open # Code to scroll the screen to the end and click on more news till the 15th page before scraping all the news k = 1 while k<=2: scroll_pause_time = 1 # You can set your own pause time. My laptop is a bit slow so I use 1 sec screen_height = driver.execute_script("return window.screen.height;") # get the screen height of the web i = 1 while True: # scroll one screen height each time driver.execute_script("window.scrollTo(0, {screen_height}*{i});".format(screen_height=screen_height, i=i)) i += 1 time.sleep(scroll_pause_time) # update scroll height each time after scrolled, as the scroll height can change after we scrolled the page scroll_height = driver.execute_script("return document.body.scrollHeight;") # Break the loop when the height we need to scroll to is larger than the total scroll height if (screen_height) * i > scroll_height: break driver.find_element(By.CSS_SELECTOR, '.Articles__MoreFromButton-sc-1mrfc98-0').click() k += 1 time.sleep(1) soup = BeautifulSoup(driver.page_source, 'html.parser') news = soup.find_all('article', attrs={'class': 'sc-ifAKCX'}) print(len(news)) # Getting titles, dates, and links for j in news: # Article title title = j.findAll(re.compile('^h[1-6]')) for news_title in title: art_title.append(news_title.text) # Article dates dates = j.find('p', attrs={'class': 'sc-cIShpX'}) if dates is not None: date = dates.text split_date = date.rsplit('|', 1)[1][10:].rsplit('<', 1)[0] art_date.append(split_date) # Article links address = j.find('a').get('href') news_link = 'https://www.iol.co.za' + address art_link.append(news_link) df = pd.DataFrame({'Article_Title': art_title, 'Date': art_date, 'Source': art_link}) # Getting contents new_articles = ...struggling to write the code df['Content'] = news_articles df.to_csv('data.csv') driver.quit()
问题原因分析
- DataFrame长度不匹配:
- 部分article包含多个h标签,导致
art_title被多次追加内容,长度远大于art_date和art_link - 部分article找不到日期元素,
art_date未添加对应占位值,长度不足
- 部分article包含多个h标签,导致
- 正文爬取失败:未实现遍历文章链接、解析正文的逻辑,且未处理请求异常
修复后的完整代码
import sys, time from bs4 import BeautifulSoup import requests import pandas as pd from selenium import webdriver from webdriver_manager.chrome import ChromeDriverManager from selenium.webdriver.common.by import By import re # 初始化存储列表 art_title = [] art_date = [] art_link = [] art_content = [] pagesToGet = ['south-africa/eastern-cape'] # 只初始化一次浏览器,避免重复启动 driver = webdriver.Chrome(ChromeDriverManager().install()) driver.maximize_window() for page in pagesToGet: print(f'processing page: https://www.iol.co.za/news/{page}') try: driver.get(f"https://www.iol.co.za/news/{page}") except Exception as e: error_type, _, error_info = sys.exc_info() print(f'ERROR FOR LINK: https://www.iol.co.za/news/{page}') print(f'{error_type}, Line: {error_info.tb_lineno}') continue time.sleep(3) # 滚动加载更多内容 k = 1 while k <= 2: scroll_pause_time = 1 screen_height = driver.execute_script("return window.screen.height;") i = 1 while True: driver.execute_script(f"window.scrollTo(0, {screen_height}*{i});") i += 1 time.sleep(scroll_pause_time) scroll_height = driver.execute_script("return document.body.scrollHeight;") if screen_height * i > scroll_height: break # 点击加载更多,添加异常处理防止按钮未加载 try: load_more_btn = driver.find_element(By.CSS_SELECTOR, '.Articles__MoreFromButton-sc-1mrfc98-0') load_more_btn.click() except: print('Load more button not found, stopping pagination') break k += 1 time.sleep(1) # 解析页面内容 soup = BeautifulSoup(driver.page_source, 'html.parser') news = soup.find_all('article', attrs={'class': 'sc-ifAKCX'}) print(f'Found {len(news)} articles') for article in news: # 提取标题:只取第一个h标签,避免重复 title_tag = article.find(re.compile('^h[1-6]')) art_title.append(title_tag.text.strip() if title_tag else 'No Title') # 提取日期:处理无日期的情况,保证列表长度一致 date_tag = article.find('p', attrs={'class': 'sc-cIShpX'}) if date_tag: date_text = date_tag.text.strip() # 简化日期提取逻辑,避免索引越界 split_date = date_text.split('|')[-1].strip() if '|' in date_text else date_text art_date.append(split_date) else: art_date.append('No Date') # 提取链接 link_tag = article.find('a') if link_tag and link_tag.get('href'): news_link = 'https://www.iol.co.za' + link_tag.get('href') art_link.append(news_link) else: art_link.append('No Link') # 提取正文内容 for link in art_link: if link == 'No Link': art_content.append('No Content') continue try: # 使用requests获取页面,比selenium更高效 response = requests.get(link, headers={'User-Agent': 'Mozilla/5.0'}) response.raise_for_status() content_soup = BeautifulSoup(response.text, 'html.parser') # 正文通常在特定类的p标签中,添加备用选择器兼容不同页面结构 content_paragraphs = content_soup.find_all('p', attrs={'class': 'sc-12bzhsi-0'}) if not content_paragraphs: content_paragraphs = content_soup.find_all('div', class_='article-body') content = '\n'.join([p.text.strip() for p in content_paragraphs]) art_content.append(content if content else 'No Content') except Exception as e: print(f'Failed to fetch content from {link}: {str(e)}') art_content.append('Failed to fetch content') time.sleep(1) # 添加上限,避免请求过快被拦截 # 创建DataFrame df = pd.DataFrame({ 'Article_Title': art_title, 'Date': art_date, 'Source': art_link, 'Content': art_content }) # 保存到CSV df.to_csv('iol_news.csv', index=False, encoding='utf-8') driver.quit()
修改点说明
- 列表长度统一:每个article对应一个标题、日期、链接,即使元素缺失也添加占位值,确保三个列表长度完全一致
- 标题提取优化:只取每个article下的第一个h标签,避免因多个h标签导致
art_title长度超标 - 正文爬取实现:遍历所有文章链接,用requests高效获取页面,结合备用选择器解析正文,同时处理请求异常
- 浏览器实例优化:将driver初始化移到循环外,避免每次处理页面都重启浏览器,提升效率
- 日期处理简化:调整日期提取逻辑,避免原代码中可能出现的索引越界问题
- 异常处理增强:给加载更多按钮、正文请求添加异常捕获,避免程序中途崩溃
内容的提问来源于stack exchange,提问作者TG_
相关产品推荐
相关产品推荐

