You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

无法从韩国中央日报商业板块爬取数据并生成目标字典求助

问题描述

我需要从网站https://koreajoongangdaily.joins.com/section/business的多个页面爬取新闻文章,最终整理成包含date、UTC_date、title、authors_name、news_content、url的字典。我编写了代码,但无法生成预期的字典,请求解决。


原代码模块

导入必要函数

from bs4 import BeautifulSoup as soup
import requests
import numpy as np
from pymongo import MongoClient
from bs4 import BeautifulSoup
from selenium import webdriver
import pandas as pd
from time import sleep
import uuid
import datetime
import time
from fake_useragent import UserAgent
import os
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.common.action_chains import ActionChains
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.by import By
from selenium.webdriver.common.desired_capabilities import DesiredCapabilities
import sys
from fake_useragent import UserAgent

import warnings
warnings.filterwarnings('ignore')
import re
from tqdm import tqdm
import pandas as pd

日期处理函数

import datetime

def string_to_date(x):
    return datetime.datetime.strptime(x, '%Y/%m/%d')

def datee(pp):
    return str(pp.date())

获取链接函数

def get_link(res):
    href_list = []
    for res in res_list:  # h3
        link_list = res.select('a')
        for link in link_list:  # a
            href = link.get('href')
            href_list.append(href)
    return href_list

获取文章内容函数

def get_article(url):
    news_list = []
    title_list= []
    page = requests.get(url)
    bsobj = soup(page.content)
    for title in bsobj.findAll('h1',{'class':'view-article-title serif'}):
        title_list.append(title.text.strip())
        
    for news in bsobj.findAll('div',{'class':'article-content-left pb-30'}):
        news = news_list.append(news.text.strip())
        
    author_list = []
    for f in news:
        author = ""
        pattern = r"BY\b(.+)(?=\[.+\])"
        resultsss = re.search(pattern, f)
        if resultsss != None:
            author = resultsss.group(0).strip()[3:]
        authors = author_list.append(author)
        
    #there is date given in every links of the articles hence we can use that    
    date_list_1 = []
    separator = '/business'
    for link in href_list:
        new_set1 = link.replace('https://koreajoongangdaily.joins.com/', '')
        new_set2 = new_set1.split(separator, 1)[0]
        new_set3 = date_list_1.append(new_set2)
        new_set4 = list(map(datee, new_set_4))
   #no separate time so add 00:00:00 for UTC    
    p=[]
    for x in new_set4:
        utc_date = p.append(str(x) + " 00:00:00")
        
    #print(news_list)   
    return news_list, title_list, authors, new_set4, utc_date

主爬取函数

def scrape_the_article(n):
    options = webdriver.ChromeOptions()
    
    lists = ['disable-popup-blocking']

    caps = DesiredCapabilities().CHROME
    caps["pageLoadStrategy"] = "normal"

    options.add_argument("--window-size=1920,1080")
    options.add_argument("--disable-extensions")
    options.add_argument("--disable-notifications")
    options.add_argument("--disable-Advertisement")
    options.add_argument("--disable-popup-blocking")

    driver = webdriver.Chrome(executable_path= r"E:\chromedriver\chromedriver.exe", options=options) #paste your own choromedriver path
    url = "https://koreajoongangdaily.joins.com/section/business"
    driver.get(url)
    
    page = 0
    for step in tqdm(range(n)):          # set the page range here, how many page you want to scrape
        page += 1
        time.sleep(2)
        try:
            button = driver.find_element_by_class_name("service-more-btn")
            button.click()
        except Exception as e:
            print("trying to scroll")
            driver.execute_script("window.scrollBy(0, 100);")
        print("Page: ", page)
        
        
    html = driver.page_source
    bs = BeautifulSoup(html, 'html.parser')
    res_list = bs.select('div[class="mid-article3"]')
    
    for res in res_list:
        links = get_article_links(res)
        article = get_article(links) 

调用函数

scrape_the_article(4)

预期字典结构

data = {'date': new_set4, 'utc_date_time': utc_date, 'title': title_list,'author': authors,
            'content': news_list,'link': href_list}

问题分析与修复方案

你的代码存在多个逻辑错误,导致无法生成预期字典,以下是具体问题和修正后的完整代码:

核心问题点

  1. 函数命名与调用不匹配:主函数中调用get_article_links,但实际定义的函数是get_link。
  2. 变量作用域错误:get_article函数中使用全局变量href_list,但该变量未在函数内定义或传入。
  3. 日期处理逻辑混乱:new_set_4未定义就使用,日期转换流程断裂。
  4. 作者提取逻辑错误:循环遍历news(实际是append返回的None),无法正确提取作者。
  5. 数据收集方式错误:未将单篇文章的字段关联,而是收集成独立列表,无法组合成目标字典。
  6. 主函数未保存结果:调用get_article后未收集返回值,也未最终生成字典。

修正后的完整代码

from bs4 import BeautifulSoup as soup
import requests
from selenium import webdriver
import datetime
import time
import re
from tqdm import tqdm
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.common.desired_capabilities import DesiredCapabilities

import warnings
warnings.filterwarnings('ignore')

# 日期处理函数
def string_to_date(x):
    return datetime.datetime.strptime(x, '%Y/%m/%d')

def date_to_str(date_obj):
    return str(date_obj.date())

# 获取所有文章链接
def get_all_links(driver):
    html = driver.page_source
    bs = BeautifulSoup(html, 'html.parser')
    res_list = bs.select('div.mid-article3')
    href_list = []
    for res in res_list:
        link_list = res.select('a')
        for link in link_list:
            href = link.get('href')
            if href not in href_list:  # 去重
                href_list.append(href)
    return href_list

# 处理单篇文章
def parse_article(url):
    # 从URL提取日期
    date_str = url.replace('https://koreajoongangdaily.joins.com/', '').split('/business')[0]
    date_obj = string_to_date(date_str)
    date = date_to_str(date_obj)
    utc_date = f"{date} 00:00:00"
    
    # 请求文章页面
    page = requests.get(url)
    bsobj = soup(page.content, 'html.parser')
    
    # 提取标题
    title = bsobj.find('h1', {'class':'view-article-title serif'}).text.strip() if bsobj.find('h1', {'class':'view-article-title serif'}) else "No Title"
    
    # 提取内容
    content_div = bsobj.find('div', {'class':'article-content-left pb-30'})
    news_content = content_div.text.strip() if content_div else "No Content"
    
    # 提取作者
    author = "Unknown"
    pattern = r"BY\s+(.+?)\s+\["
    result = re.search(pattern, news_content)
    if result:
        author = result.group(1).strip()
    
    # 返回单篇文章的字典
    return {
        'date': date,
        'UTC_date': utc_date,
        'title': title,
        'authors_name': author,
        'news_content': news_content,
        'url': url
    }

# 主爬取函数
def scrape_the_article(n):
    options = webdriver.ChromeOptions()
    caps = DesiredCapabilities().CHROME
    caps["pageLoadStrategy"] = "normal"

    options.add_argument("--window-size=1920,1080")
    options.add_argument("--disable-extensions")
    options.add_argument("--disable-notifications")
    options.add_argument("--disable-popup-blocking")

    driver = webdriver.Chrome(executable_path=r"E:\chromedriver\chromedriver.exe", options=options)
    url = "https://koreajoongangdaily.joins.com/section/business"
    driver.get(url)
    
    # 加载更多页面
    for step in tqdm(range(n)):
        time.sleep(2)
        try:
            button = driver.find_element_by_class_name("service-more-btn")
            button.click()
        except Exception:
            driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
        print(f"已加载第 {step+1} 页")
    
    # 获取所有链接
    all_links = get_all_links(driver)
    driver.quit()
    
    # 遍历链接,解析每篇文章
    final_data = []
    for link in tqdm(all_links):
        try:
            article_dict = parse_article(link)
            final_data.append(article_dict)
        except Exception as e:
            print(f"解析链接 {link} 失败: {e}")
    
    return final_data

# 调用并验证结果
if __name__ == "__main__":
    scraped_data = scrape_the_article(4)
    # 打印前2条数据验证
    for item in scraped_data[:2]:
        print(item)

修正说明

  1. 重构函数职责:拆分get_all_links和parse_article,分别负责链接提取和单篇文章解析,逻辑更清晰。
  2. 修复变量作用域:所有变量在函数内部定义或通过参数传入,消除全局变量依赖。
  3. 优化日期处理:从URL直接提取日期并转换,流程连贯无错误。
  4. 修正作者提取:调整正则表达式匹配逻辑,避免无效遍历。
  5. 匹配预期数据结构:每篇文章生成独立字典,最终收集为列表,完全符合需求字段。
  6. 增加错误处理:捕获解析异常,避免程序崩溃。

内容的提问来源于stack exchange,提问作者Starlord22

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.15 08:40:31