You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

IOL网站新闻爬取求助:DataFrame报错及正文获取失败

爬取IOL网站新闻的问题解决

问题描述

爬取IOL网站(https://www.iol.co.za/news/south-africa/eastern-cape)的新闻标题、日期、链接及正文内容时,遇到两个核心问题:

  • 存储数据到pandas DataFrame时触发ValueError: All arrays must be of the same length
  • 无法通过文章链接获取正文内容

用户尝试用h标签匹配标题,但因页面元素类名和标签不统一导致提取异常,原代码如下:

import sys, time
from bs4 import BeautifulSoup
import requests
import pandas as pd
from selenium import webdriver
from webdriver_manager.chrome import ChromeDriverManager
from datetime import timedelta
from selenium.common.exceptions import TimeoutException
from selenium.webdriver.common.by import By
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
import re


art_title = [] # to store the titles of all news article
art_date = [] # to store the dates of all news article
art_link = [] # to store the links of all news article


pagesToGet = ['south-africa/eastern-cape']


for i in range(0, len(pagesToGet)):
    print('processing page : \n')
    url = 'https://www.iol.co.za' + str(pagesToGet[i])
    print(url)

    driver = webdriver.Chrome(ChromeDriverManager().install())
    driver.maximize_window()

    #time.sleep(5)  # allow you to sleep your code before your retrieve the elements from the webpage. Additionally, to
    # prevent the chrome driver opening a new instance for every url, open the browser outside of the loop.

    # an exception might be thrown, so the code should be in a try-except block
    try:
        # use the browser to get the url. This is suspicious command that might blow up.
        driver.get("https://www.iol.co.za/news/" +str(pagesToGet[i]))

    except Exception as e:  # this describes what to do if an exception is thrown
        error_type, error_obj, error_info = sys.exc_info()  # get the exception information
        print('ERROR FOR LINK:', url)  # print the link that cause the problem
        print(error_type, 'Line:', error_info.tb_lineno)  # print error info and line that threw the exception
        continue  # ignore this page. Abandon this and go back.
    time.sleep(3) # Allow 3 seconds for the web page to open

    # Code to scroll the screen to the end and click on more news till the 15th page before scraping all the news
    k = 1
    while k<=2:
        scroll_pause_time = 1  # You can set your own pause time. My laptop is a bit slow so I use 1 sec
        screen_height = driver.execute_script("return window.screen.height;")  # get the screen height of the web
        i = 1
        while True:
            # scroll one screen height each time
            driver.execute_script("window.scrollTo(0, {screen_height}*{i});".format(screen_height=screen_height, i=i))
            i += 1
            time.sleep(scroll_pause_time)
            # update scroll height each time after scrolled, as the scroll height can change after we scrolled the page
            scroll_height = driver.execute_script("return document.body.scrollHeight;")
            # Break the loop when the height we need to scroll to is larger than the total scroll height
            if (screen_height) * i > scroll_height:
                break
        driver.find_element(By.CSS_SELECTOR, '.Articles__MoreFromButton-sc-1mrfc98-0').click()
        k += 1
        time.sleep(1)

    soup = BeautifulSoup(driver.page_source, 'html.parser')

    news = soup.find_all('article', attrs={'class': 'sc-ifAKCX'})
    print(len(news))

    # Getting titles, dates, and links
    for j in news:
        # Article title
        title = j.findAll(re.compile('^h[1-6]'))
        for news_title in title:
            art_title.append(news_title.text)

        # Article dates
        dates = j.find('p', attrs={'class': 'sc-cIShpX'})
        if dates is not None:
            date = dates.text
            split_date = date.rsplit('|', 1)[1][10:].rsplit('<', 1)[0]
            art_date.append(split_date)

        # Article links
        address = j.find('a').get('href')
        news_link = 'https://www.iol.co.za' + address
        art_link.append(news_link)

    df = pd.DataFrame({'Article_Title': art_title, 'Date': art_date, 'Source': art_link})

    # Getting contents
    new_articles = ...struggling to write the code

    df['Content'] = news_articles


df.to_csv('data.csv')


driver.quit()

问题原因分析

  1. DataFrame长度不匹配:
    • 部分article包含多个h标签,导致art_title被多次追加内容,长度远大于art_date和art_link
    • 部分article找不到日期元素,art_date未添加对应占位值,长度不足
  2. 正文爬取失败:未实现遍历文章链接、解析正文的逻辑,且未处理请求异常

修复后的完整代码

import sys, time
from bs4 import BeautifulSoup
import requests
import pandas as pd
from selenium import webdriver
from webdriver_manager.chrome import ChromeDriverManager
from selenium.webdriver.common.by import By
import re

# 初始化存储列表
art_title = []
art_date = []
art_link = []
art_content = []

pagesToGet = ['south-africa/eastern-cape']

# 只初始化一次浏览器,避免重复启动
driver = webdriver.Chrome(ChromeDriverManager().install())
driver.maximize_window()

for page in pagesToGet:
    print(f'processing page: https://www.iol.co.za/news/{page}')
    try:
        driver.get(f"https://www.iol.co.za/news/{page}")
    except Exception as e:
        error_type, _, error_info = sys.exc_info()
        print(f'ERROR FOR LINK: https://www.iol.co.za/news/{page}')
        print(f'{error_type}, Line: {error_info.tb_lineno}')
        continue
    time.sleep(3)

    # 滚动加载更多内容
    k = 1
    while k <= 2:
        scroll_pause_time = 1
        screen_height = driver.execute_script("return window.screen.height;")
        i = 1
        while True:
            driver.execute_script(f"window.scrollTo(0, {screen_height}*{i});")
            i += 1
            time.sleep(scroll_pause_time)
            scroll_height = driver.execute_script("return document.body.scrollHeight;")
            if screen_height * i > scroll_height:
                break
        # 点击加载更多,添加异常处理防止按钮未加载
        try:
            load_more_btn = driver.find_element(By.CSS_SELECTOR, '.Articles__MoreFromButton-sc-1mrfc98-0')
            load_more_btn.click()
        except:
            print('Load more button not found, stopping pagination')
            break
        k += 1
        time.sleep(1)

    # 解析页面内容
    soup = BeautifulSoup(driver.page_source, 'html.parser')
    news = soup.find_all('article', attrs={'class': 'sc-ifAKCX'})
    print(f'Found {len(news)} articles')

    for article in news:
        # 提取标题:只取第一个h标签,避免重复
        title_tag = article.find(re.compile('^h[1-6]'))
        art_title.append(title_tag.text.strip() if title_tag else 'No Title')

        # 提取日期:处理无日期的情况,保证列表长度一致
        date_tag = article.find('p', attrs={'class': 'sc-cIShpX'})
        if date_tag:
            date_text = date_tag.text.strip()
            # 简化日期提取逻辑,避免索引越界
            split_date = date_text.split('|')[-1].strip() if '|' in date_text else date_text
            art_date.append(split_date)
        else:
            art_date.append('No Date')

        # 提取链接
        link_tag = article.find('a')
        if link_tag and link_tag.get('href'):
            news_link = 'https://www.iol.co.za' + link_tag.get('href')
            art_link.append(news_link)
        else:
            art_link.append('No Link')

# 提取正文内容
for link in art_link:
    if link == 'No Link':
        art_content.append('No Content')
        continue
    try:
        # 使用requests获取页面,比selenium更高效
        response = requests.get(link, headers={'User-Agent': 'Mozilla/5.0'})
        response.raise_for_status()
        content_soup = BeautifulSoup(response.text, 'html.parser')
        # 正文通常在特定类的p标签中,添加备用选择器兼容不同页面结构
        content_paragraphs = content_soup.find_all('p', attrs={'class': 'sc-12bzhsi-0'})
        if not content_paragraphs:
            content_paragraphs = content_soup.find_all('div', class_='article-body')
        content = '\n'.join([p.text.strip() for p in content_paragraphs])
        art_content.append(content if content else 'No Content')
    except Exception as e:
        print(f'Failed to fetch content from {link}: {str(e)}')
        art_content.append('Failed to fetch content')
    time.sleep(1)  # 添加上限,避免请求过快被拦截

# 创建DataFrame
df = pd.DataFrame({
    'Article_Title': art_title,
    'Date': art_date,
    'Source': art_link,
    'Content': art_content
})

# 保存到CSV
df.to_csv('iol_news.csv', index=False, encoding='utf-8')

driver.quit()

修改点说明

  1. 列表长度统一:每个article对应一个标题、日期、链接,即使元素缺失也添加占位值,确保三个列表长度完全一致
  2. 标题提取优化:只取每个article下的第一个h标签,避免因多个h标签导致art_title长度超标
  3. 正文爬取实现:遍历所有文章链接,用requests高效获取页面,结合备用选择器解析正文,同时处理请求异常
  4. 浏览器实例优化:将driver初始化移到循环外,避免每次处理页面都重启浏览器,提升效率
  5. 日期处理简化:调整日期提取逻辑,避免原代码中可能出现的索引越界问题
  6. 异常处理增强:给加载更多按钮、正文请求添加异常捕获,避免程序中途崩溃

内容的提问来源于stack exchange,提问作者TG_

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.07 14:50:28