Python爬虫脚本扩展:从CSV读取多URL爬取Justia案件链接
问题描述
我是Python新手,现有异步爬虫脚本可爬取单个支持分页的Justia搜索URL,提取页面中的案件编号与案件链接,将其打印到控制台并保存至文件。现在想扩展功能,让脚本从单列CSV文件读取多个搜索URL,对每个URL执行爬取操作。
修改脚本时先出现SyntaxError,调整后无语法错误但运行完全没输出,以下是各阶段代码:
初始可用脚本
from aiohttp import ClientSession from pyuseragents import random from bs4 import BeautifulSoup from asyncio import run class DocketsJustia: def __init__(self): self.headers = { 'authority': 'dockets.justia.com', 'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7', 'accept-language': 'en-US,en;q=0.5', 'cache-control': 'max-age=0', 'referer': 'https://dockets.justia.com/search?parties=Agfa&cases=between&sort-by-last-update=false&after=2015-1-1&before=2023-3-27', 'user-agent': random(), } self.PatchFile = "nametxt.txt" async def Parser(self, session): count = 1 while True: params = { 'parties': 'Agfa', 'page': f'{count}', } async with session.get(f'https://dockets.justia.com/search?parties=Agfa&cases=between&sort-by-last-update=false&after=2015-1-1&before=2023-3-27&page={count}', params=params) as response: links = BeautifulSoup(await response.text(), "lxml").find_all("div", { "class": "has-padding-content-block-30 -zb"}) for link in links: try: case_link = link.find("a", {"class": "case-name"}).get("href") case_number = link.find("span", {"class": "citation"}).text print(case_number + "\t" + case_link + "\n") with open(self.PatchFile, "a", encoding='utf-8') as file: file.write(case_number + "\t" + case_link + "\n") except: pass count += 1 async def LoggerParser(self): async with ClientSession(headers=self.headers) as session: await self.Parser(session) def StartDocketsJustia(): run(DocketsJustia().LoggerParser()) if __name__ == '__main__': StartDocketsJustia()
出现SyntaxError的脚本
from aiohttp import ClientSession from pyuseragents import random from bs4 import BeautifulSoup from asyncio import run class DocketsJustia: def __init__(self): self.headers = { 'authority': 'dockets.justia.com', 'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7', 'accept-language': 'en-US,en;q=0.5', 'cache-control': 'max-age=0', 'referer': 'https://dockets.justia.com/search?parties=Agfa&cases=between&sort-by-last-update=false&after=2015-1-1&before=2023-3-27', 'user-agent': random(), } self.PatchFile = "nametxt.txt" # old line: async def Parser(self, session): async def Parser(selfself, session, searchUrl): count = 1 while True: params = { 'parties': 'Agfa', 'page': f'{count}', } async with session.get(searchUrl, params=params as response: links = BeautifulSoup(await response.text(), "lxml").find_all("div", { "class": "has-padding-content-block-30 -zb"}) for link in links: try: case_link = link.find("a", {"class": "case-name"}).get("href") case_number = link.find("span", {"class": "citation"}).text print(case_number + "\t" + case_link + "\n") with open(self.PatchFile, "a", encoding='utf-8') as file: file.write(case_number + "\t" + case_link + "\n") except: pass count += 1 async def LoggerParser(self): # old line: async with ClientSession(headers=self.headers) as session: searchUrls=set(pd.read_csv('input_file.csv', header=None)[0]) for url in searchUrls: await self.Parser(session, url) def StartDocketsJustia(): run(DocketsJustia().LoggerParser()) if __name__ == '__main__': StartDocketsJustia()
当前无输出的脚本
from aiohttp import ClientSession from pyuseragents import random from bs4 import BeautifulSoup from asyncio import run import pandas as pd class DocketsJustia: def __init__(self): self.headers = { 'authority': 'dockets.justia.com', 'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7', 'accept-language': 'en-US,en;q=0.5', 'cache-control': 'max-age=0', 'user-agent': random(), } self.PatchFile = "nametxt.txt" # old line: async def Parser(self, session): async def Parser(selfself, session, searchUrl): count = 1 while True: async with session.get(f"{searchUrl}&page={count}") as response: links = BeautifulSoup(await response.text(), "lxml").find_all("div", { "class": "has-padding-content-block-30 -zb"}) for link in links: try: case_link = link.find("a", {"class": "case-name"}).get("href") case_number = link.find("span", {"class": "citation"}).text print(case_number + "\t" + case_link + "\n") with open(self.PatchFile, "a", encoding='utf-8') as file: file.write(case_number + "\t" + case_link + "\n") except: pass count += 1 async def LoggerParser(self): async with ClientSession(headers=self.headers) as session: searchUrls=set(pd.read_csv('input_file.csv', header=None)[0]) for url in searchUrls: await self.Parser(session, url) def StartDocketsJustia(): run(DocketsJustia().LoggerParser()) if __name__ == '__main__': StartDocketsJustia()
错误分析与修正方案
1. 语法错误阶段的核心问题
- 方法参数笔误:
Parser(selfself, session, searchUrl)中的selfself应为self - 调用
session.get时缺少右括号:params=params as response应改为params=params) as response LoggerParser方法中未导入pandas就调用pd.read_csv,且for循环缩进错误- 未创建
ClientSession实例就调用self.Parser(session, url),session变量未定义
2. 无输出阶段的核心问题
- 仍存在
selfself的笔误,导致无法访问类的self.PatchFile等属性,异常被宽泛的except:掩盖 - 无限循环
while True无终止条件,空页面时会持续空跑 - 分页URL拼接逻辑不严谨,若原URL无参数会出现格式错误
- 未处理CSV文件不存在、URL为空等异常情况
修正后的完整脚本
from aiohttp import ClientSession from pyuseragents import random from bs4 import BeautifulSoup from asyncio import run import pandas as pd class DocketsJustia: def __init__(self): self.headers = { 'authority': 'dockets.justia.com', 'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7', 'accept-language': 'en-US,en;q=0.5', 'cache-control': 'max-age=0', 'user-agent': random(), } self.PatchFile = "nametxt.txt" async def Parser(self, session, searchUrl): count = 1 while True: # 兼容带/不带参数的URL,正确拼接分页参数 page_url = f"{searchUrl}&page={count}" if '?' in searchUrl else f"{searchUrl}?page={count}" async with session.get(page_url) as response: # 请求失败则终止当前URL爬取 if response.status != 200: print(f"[{response.status}] 爬取失败: {page_url}") break soup = BeautifulSoup(await response.text(), "lxml") links = soup.find_all("div", {"class": "has-padding-content-block-30 -zb"}) # 无结果则判定为最后一页,终止循环 if not links: print(f"URL {searchUrl} 爬取完成,共 {count-1} 页") break for link in links: try: case_link = link.find("a", {"class": "case-name"}).get("href") case_number = link.find("span", {"class": "citation"}).text output_line = f"{case_number}\t{case_link}\n" print(output_line.strip()) with open(self.PatchFile, "a", encoding='utf-8') as file: file.write(output_line) except AttributeError: # 仅捕获元素查找失败的异常,不掩盖其他问题 print("解析案件信息时跳过无效条目") continue count += 1 async def LoggerParser(self): async with ClientSession(headers=self.headers) as session: try: # 读取CSV并过滤空值、去重 search_urls = pd.read_csv('input_file.csv', header=None)[0].dropna().unique().tolist() if not search_urls: print("CSV文件中无有效URL") return for url in search_urls: print(f"开始爬取: {url}") await self.Parser(session, url) except FileNotFoundError: print("未找到input_file.csv文件") except Exception as e: print(f"读取CSV出错: {str(e)}") def StartDocketsJustia(): run(DocketsJustia().LoggerParser()) if __name__ == '__main__': StartDocketsJustia()
修正要点说明
- 修复
selfself笔误,确保类属性正常访问 - 优化分页URL拼接逻辑,兼容不同格式的输入URL
- 添加响应状态码检查与空页面判断,避免无限循环
- 缩小异常捕获范围,便于排查问题
- 增加CSV读取的异常处理与空值过滤
- 添加爬取进度提示,方便跟踪执行状态
内容的提问来源于stack exchange,提问作者PressMeister
相关产品推荐
相关产品推荐

