You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python爬虫脚本扩展:从CSV读取多URL爬取Justia案件链接

问题描述

我是Python新手,现有异步爬虫脚本可爬取单个支持分页的Justia搜索URL,提取页面中的案件编号与案件链接,将其打印到控制台并保存至文件。现在想扩展功能,让脚本从单列CSV文件读取多个搜索URL,对每个URL执行爬取操作。

修改脚本时先出现SyntaxError,调整后无语法错误但运行完全没输出,以下是各阶段代码:

初始可用脚本

from aiohttp import ClientSession
from pyuseragents import random
from bs4 import BeautifulSoup
from asyncio import run

class DocketsJustia:

    def __init__(self):
        self.headers = {
            'authority': 'dockets.justia.com',
            'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
            'accept-language': 'en-US,en;q=0.5',
            'cache-control': 'max-age=0',
            'referer': 'https://dockets.justia.com/search?parties=Agfa&cases=between&sort-by-last-update=false&after=2015-1-1&before=2023-3-27',
            'user-agent': random(),
        }

        self.PatchFile = "nametxt.txt"

    async def Parser(self, session):
        count = 1

        while True:

            params = {
                'parties': 'Agfa',
                'page': f'{count}',
            }

            async with session.get(f'https://dockets.justia.com/search?parties=Agfa&cases=between&sort-by-last-update=false&after=2015-1-1&before=2023-3-27&page={count}',
                                   params=params) as response:
                links = BeautifulSoup(await response.text(), "lxml").find_all("div", {
                    "class": "has-padding-content-block-30 -zb"})

                for link in links:
                    try:
                        case_link = link.find("a", {"class": "case-name"}).get("href")
                        case_number = link.find("span", {"class": "citation"}).text
                        print(case_number + "\t" + case_link + "\n")

                        with open(self.PatchFile, "a", encoding='utf-8') as file:
                            file.write(case_number + "\t" + case_link + "\n")
                    except:
                        pass
            count += 1

    async def LoggerParser(self):
        async with ClientSession(headers=self.headers) as session:
            await self.Parser(session)

def StartDocketsJustia():
    run(DocketsJustia().LoggerParser())

if __name__ == '__main__':
    StartDocketsJustia()

出现SyntaxError的脚本

from aiohttp import ClientSession
from pyuseragents import random
from bs4 import BeautifulSoup
from asyncio import run

class DocketsJustia:

    def __init__(self):
        self.headers = {
            'authority': 'dockets.justia.com',
            'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
            'accept-language': 'en-US,en;q=0.5',
            'cache-control': 'max-age=0',
            'referer': 'https://dockets.justia.com/search?parties=Agfa&cases=between&sort-by-last-update=false&after=2015-1-1&before=2023-3-27',
            'user-agent': random(),
        }

        self.PatchFile = "nametxt.txt"

#    old line: async def Parser(self, session):
    async def Parser(selfself, session, searchUrl):
        count = 1

        while True:

            params = {
                'parties': 'Agfa',
                'page': f'{count}',
            }

            async with session.get(searchUrl, params=params as response:
                links = BeautifulSoup(await response.text(), "lxml").find_all("div", {
                    "class": "has-padding-content-block-30 -zb"})

                for link in links:
                    try:
                        case_link = link.find("a", {"class": "case-name"}).get("href")
                        case_number = link.find("span", {"class": "citation"}).text
                        print(case_number + "\t" + case_link + "\n")

                        with open(self.PatchFile, "a", encoding='utf-8') as file:
                            file.write(case_number + "\t" + case_link + "\n")
                    except:
                        pass
            count += 1

    async def LoggerParser(self):
#       old line: async with ClientSession(headers=self.headers) as session:
        searchUrls=set(pd.read_csv('input_file.csv', header=None)[0])
            for url in searchUrls: await self.Parser(session, url)

def StartDocketsJustia():
    run(DocketsJustia().LoggerParser())

if __name__ == '__main__':
    StartDocketsJustia()

当前无输出的脚本

from aiohttp import ClientSession
from pyuseragents import random
from bs4 import BeautifulSoup
from asyncio import run
import pandas as pd

class DocketsJustia:

    def __init__(self):
        self.headers = {
            'authority': 'dockets.justia.com',
            'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
            'accept-language': 'en-US,en;q=0.5',
            'cache-control': 'max-age=0',
            'user-agent': random(),
        }

        self.PatchFile = "nametxt.txt"

#    old line: async def Parser(self, session):
    async def Parser(selfself, session, searchUrl):
        count = 1

        while True:
            async with session.get(f"{searchUrl}&page={count}") as response:
                links = BeautifulSoup(await response.text(), "lxml").find_all("div", {
                    "class": "has-padding-content-block-30 -zb"})

                for link in links:
                    try:
                        case_link = link.find("a", {"class": "case-name"}).get("href")
                        case_number = link.find("span", {"class": "citation"}).text
                        print(case_number + "\t" + case_link + "\n")

                        with open(self.PatchFile, "a", encoding='utf-8') as file:
                            file.write(case_number + "\t" + case_link + "\n")
                    except:
                        pass
            count += 1

    async def LoggerParser(self):
        async with ClientSession(headers=self.headers) as session:
            searchUrls=set(pd.read_csv('input_file.csv', header=None)[0])
            for url in searchUrls:
                await self.Parser(session, url)

def StartDocketsJustia():
    run(DocketsJustia().LoggerParser())

if __name__ == '__main__':
    StartDocketsJustia()
错误分析与修正方案

1. 语法错误阶段的核心问题

  • 方法参数笔误:Parser(selfself, session, searchUrl) 中的selfself应为self
  • 调用session.get时缺少右括号:params=params as response 应改为params=params) as response
  • LoggerParser方法中未导入pandas就调用pd.read_csv,且for循环缩进错误
  • 未创建ClientSession实例就调用self.Parser(session, url),session变量未定义

2. 无输出阶段的核心问题

  • 仍存在selfself的笔误,导致无法访问类的self.PatchFile等属性,异常被宽泛的except:掩盖
  • 无限循环while True无终止条件,空页面时会持续空跑
  • 分页URL拼接逻辑不严谨,若原URL无参数会出现格式错误
  • 未处理CSV文件不存在、URL为空等异常情况
修正后的完整脚本
from aiohttp import ClientSession
from pyuseragents import random
from bs4 import BeautifulSoup
from asyncio import run
import pandas as pd

class DocketsJustia:
    def __init__(self):
        self.headers = {
            'authority': 'dockets.justia.com',
            'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
            'accept-language': 'en-US,en;q=0.5',
            'cache-control': 'max-age=0',
            'user-agent': random(),
        }
        self.PatchFile = "nametxt.txt"

    async def Parser(self, session, searchUrl):
        count = 1
        while True:
            # 兼容带/不带参数的URL,正确拼接分页参数
            page_url = f"{searchUrl}&page={count}" if '?' in searchUrl else f"{searchUrl}?page={count}"
            
            async with session.get(page_url) as response:
                # 请求失败则终止当前URL爬取
                if response.status != 200:
                    print(f"[{response.status}] 爬取失败: {page_url}")
                    break
                
                soup = BeautifulSoup(await response.text(), "lxml")
                links = soup.find_all("div", {"class": "has-padding-content-block-30 -zb"})
                
                # 无结果则判定为最后一页,终止循环
                if not links:
                    print(f"URL {searchUrl} 爬取完成,共 {count-1} 页")
                    break
                
                for link in links:
                    try:
                        case_link = link.find("a", {"class": "case-name"}).get("href")
                        case_number = link.find("span", {"class": "citation"}).text
                        output_line = f"{case_number}\t{case_link}\n"
                        print(output_line.strip())
                        
                        with open(self.PatchFile, "a", encoding='utf-8') as file:
                            file.write(output_line)
                    except AttributeError:
                        # 仅捕获元素查找失败的异常,不掩盖其他问题
                        print("解析案件信息时跳过无效条目")
                        continue
            count += 1

    async def LoggerParser(self):
        async with ClientSession(headers=self.headers) as session:
            try:
                # 读取CSV并过滤空值、去重
                search_urls = pd.read_csv('input_file.csv', header=None)[0].dropna().unique().tolist()
                if not search_urls:
                    print("CSV文件中无有效URL")
                    return
                
                for url in search_urls:
                    print(f"开始爬取: {url}")
                    await self.Parser(session, url)
            except FileNotFoundError:
                print("未找到input_file.csv文件")
            except Exception as e:
                print(f"读取CSV出错: {str(e)}")

def StartDocketsJustia():
    run(DocketsJustia().LoggerParser())

if __name__ == '__main__':
    StartDocketsJustia()

修正要点说明

  • 修复selfself笔误,确保类属性正常访问
  • 优化分页URL拼接逻辑,兼容不同格式的输入URL
  • 添加响应状态码检查与空页面判断,避免无限循环
  • 缩小异常捕获范围,便于排查问题
  • 增加CSV读取的异常处理与空值过滤
  • 添加爬取进度提示,方便跟踪执行状态

内容的提问来源于stack exchange,提问作者PressMeister

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.26 05:55:09