You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

运行Python新闻发布日期提取脚本时遭遇WinError10053连接中止错误

问题描述

我编写了一个Python脚本,用于从新闻文章中提取发布日期,将存储在文本文件中的新闻URL(每行一个)按日期分组,每天的新闻存入单独的文件。脚本可以运行,但处理30万条URL耗时极长(有时需要数周),最终还会抛出如下连接错误:

Traceback (most recent call last):
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\connectionpool.py", line 703, in urlopen
    httplib_response = self._make_request(
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\connectionpool.py", line 386, in _make_request
    self._validate_conn(conn)
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\connectionpool.py", line 1042, in _validate_conn
    conn.connect()
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\connection.py", line 414, in connect
    self.sock = ssl_wrap_socket(
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\util\ssl_.py", line 449, in ssl_wrap_socket
    ssl_sock = _ssl_wrap_socket_impl(
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\util\ssl_.py", line 493, in _ssl_wrap_socket_impl
    return ssl_context.wrap_socket(sock, server_hostname=server_hostname)
  File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.10_3.10.2288.0_x64__qbz5n2kfra8p0\lib\ssl.py", line 513, in wrap_socket
    return self.sslsocket_class._create(
  File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.10_3.10.2288.0_x64__qbz5n2kfra8p0\lib\ssl.py", line 1071, in _create
    self.do_handshake()
  File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.10_3.10.2288.0_x64__qbz5n2kfra8p0\lib\ssl.py", line 1342, in do_handshake
    self._sslobj.do_handshake()
ConnectionAbortedError: [WinError 10053] An established connection was aborted by the software in your host machine

During handling of the above exception, another exception occurred:

Traceback (most recent call last):
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\requests\adapters.py", line 489, in send
    resp = conn.urlopen(
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\connectionpool.py", line 787, in urlopen
    retries = retries.increment(
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\util\retry.py", line 550, in increment
    raise six.reraise(type(error), error, _stacktrace)
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\packages\six.py", line 769, in reraise
    raise value.with_traceback(tb)
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\connectionpool.py", line 703, in urlopen
    httplib_response = self._make_request(
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\connectionpool.py", line 386, in _make_request
    self._validate_conn(conn)
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\connectionpool.py", line 1042, in _validate_conn
    conn.connect()
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\connection.py", line 414, in connect
    self.sock = ssl_wrap_socket(
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\util\ssl_.py", line 449, in ssl_wrap_socket
    ssl_sock = _ssl_wrap_socket_impl(
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\urllib3\util\ssl_.py", line 493, in _ssl_wrap_socket_impl
    return ssl_context.wrap_socket(sock, server_hostname=server_hostname)
  File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.10_3.10.2288.0_x64__qbz5n2kfra8p0\lib\ssl.py", line 513, in wrap_socket
    return self.sslsocket_class._create(
  File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.10_3.10.2288.0_x64__qbz5n2kfra8p0\lib\ssl.py", line 1071, in _create
    self.do_handshake()
  File "C:\Program Files\WindowsApps\PythonSoftwareFoundation.Python.3.10_3.10.2288.0_x64__qbz5n2kfra8p0\lib\ssl.py", line 1342, in do_handshake
    self._sslobj.do_handshake()
urllib3.exceptions.ProtocolError: ('Connection aborted.', ConnectionAbortedError(10053, 'An established connection was aborted by the software in your host machine', None, 10053, None))

During handling of the above exception, another exception occurred:

Traceback (most recent call last):
  File "C:\Users\hhallak\Desktop\split_by_date.py", line 65, in <module>
    split(links)
  File "C:\Users\hhallak\Desktop\split_by_date.py", line 25, in split
    with requests.get(link, stream=True) as response:
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\requests\api.py", line 73, in get
    return request("get", url, params=params, **kwargs)
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\requests\api.py", line 59, in request
    return session.request(method=method, url=url, **kwargs)
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\requests\sessions.py", line 587, in request
    resp = self.send(prep, **send_kwargs)
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\requests\sessions.py", line 701, in send
    r = adapter.send(request, **kwargs)
  File "C:\Users\hhallak\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\requests\adapters.py", line 547, in send
    raise ConnectionError(err, request=request)
requests.exceptions.ConnectionError: ('Connection aborted.', ConnectionAbortedError(10053, 'An established connection was aborted by the software in your host machine', None, 10053, None))

原运行代码

import os
import sys
from newspaper import Article   
import requests
import json

def split(links):
    exists = os.path.exists("output")
    if not exists:
        # Create a new directory because it does not exist
        os.makedirs("output")

    exists = os.path.exists("output_redo")
    if not exists:
        # Create a new directory because it does not exist
        os.makedirs("output_redo")

    for link in links:
        with requests.get(link, stream=True) as response:
            response = requests.get(link)
            if response.status_code == 200:
                final_link = link
            else:
                final_link = response.url 

            try:
                story = Article(final_link)
                story.download()
                story.parse()
                date_time = str(story.publish_date)
                split_date = date_time.split()  
                date = split_date[0]
                with open("output/" + date + ".txt", "a", encoding = 'utf-8') as output_file:
                    output_file.write(link + "\n")
            except:
                print("The script was not able to extract published date. Moving the url to be crawled later.")
                print("link: ", link)
                print("status code: ", response.status_code)
                print("final link: ", final_link)
                with open("output_redo/" + "links_to_redo" + ".txt", "a", encoding = 'utf-8') as output_redo:
                        output_redo.write(link + "\n")            
            continue

if __name__ == "__main__":
    if len(sys.argv) != 2:
        print ("Usage: Python split_by_date.py <file_name>")
        print ("e.g: python split_by_date.py input_file.txt")
        sys.exit()
    else:
        file_name = sys.argv[1]
        with open(file_name, "r", encoding = 'utf-8') as input_file:
            input_data = input_file.read()
            links = input_data.split("\n")
            del links[-1]
        split(links)

问题分析与优化方案

1. 解决WinError 10053连接错误

这个错误是因为频繁请求被本地安全软件(防火墙/杀毒)或目标服务器中断,且原脚本没有重试机制。

  • 移除重复请求:原代码中先执行requests.get(link, stream=True)又调用requests.get(link),完全重复,直接删除第一个请求
  • 添加重试机制:使用requests.Session配合HTTPAdapter实现自动重试,处理连接中断问题
  • 设置超时时间:给请求添加超时,避免长时间挂起占用资源

2. 大幅提升运行速度

原脚本是单线程串行处理,30万条URL必然耗时极长,改成多线程并发处理:

  • 使用线程池:用concurrent.futures.ThreadPoolExecutor实现并发,控制并发数(建议10-20,避免被目标服务器封禁)
  • 复用连接池:用requests.Session保持TCP连接,减少握手开销
  • 优化文件IO:批量写入结果,减少文件打开/关闭的次数

3. 代码细节优化

  • 精准异常捕获:替换全局except:为特定异常捕获,避免误吞键盘中断等关键错误
  • 过滤无效URL:处理空URL和无效链接,避免无效请求
  • 日期处理简化:直接用story.publish_date.date().isoformat()获取日期字符串,无需转成字符串再分割

优化后的代码

import os
import sys
from newspaper import Article, ArticleException
import requests
from urllib3.util.retry import Retry
from requests.adapters import HTTPAdapter
from concurrent.futures import ThreadPoolExecutor, as_completed
from collections import defaultdict

# 初始化带重试的requests会话
def init_session():
    session = requests.Session()
    retry_strategy = Retry(
        total=3,
        backoff_factor=1,
        status_forcelist=[429, 500, 502, 503, 504],
        allowed_methods=["GET"]
    )
    adapter = HTTPAdapter(max_retries=retry_strategy)
    session.mount("https://", adapter)
    session.mount("http://", adapter)
    return session

# 处理单个链接
def process_link(link, session):
    try:
        # 检查链接有效性
        if not link.strip():
            return None, link
        
        # 获取最终链接(处理重定向)
        response = session.get(link, timeout=10)
        response.raise_for_status()
        final_link = response.url if response.history else link

        # 提取发布日期
        story = Article(final_link)
        story.download()
        story.parse()
        
        if not story.publish_date:
            return None, link
        
        date_str = story.publish_date.date().isoformat()
        return date_str, link
    except (requests.exceptions.RequestException, ArticleException) as e:
        print(f"处理链接失败 {link}: {str(e)}")
        return None, link
    except Exception as e:
        print(f"未知错误 {link}: {str(e)}")
        return None, link

def split(links):
    # 创建输出目录
    os.makedirs("output", exist_ok=True)
    os.makedirs("output_redo", exist_ok=True)

    # 初始化会话
    session = init_session()
    date_links = defaultdict(list)
    failed_links = []

    # 多线程处理
    with ThreadPoolExecutor(max_workers=15) as executor:
        futures = {executor.submit(process_link, link, session): link for link in links}
        
        for future in as_completed(futures):
            date_str, link = future.result()
            if date_str:
                date_links[date_str].append(link)
            else:
                failed_links.append(link)

    # 批量写入日期分组文件
    for date_str, links_list in date_links.items():
        with open(f"output/{date_str}.txt", "a", encoding='utf-8') as f:
            f.write("\n".join(links_list) + "\n")

    # 写入失败链接
    if failed_links:
        with open("output_redo/links_to_redo.txt", "a", encoding='utf-8') as f:
            f.write("\n".join(failed_links) + "\n")

if __name__ == "__main__":
    if len(sys.argv) != 2:
        print("用法: Python split_by_date.py <文件名>")
        print("示例: python split_by_date.py input_file.txt")
        sys.exit()
    
    file_name = sys.argv[1]
    with open(file_name, "r",
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.03 00:15:40