Python实现自动识别内部Web页面最新报告下载链接方案问询
自动检测并下载最新报告的实现方案
核心思路
- 先访问报告列表页面,解析页面内容获取所有报告的日期和对应下载链接
- 解析日期字符串并排序,筛选出日期最新的报告链接
- 复用重试逻辑完成最新报告的下载
修改后的完整代码
import urllib.request from urllib.error import HTTPError import sys import time from html.parser import HTMLParser from datetime import datetime # 替换为实际的内部报告列表页面URL REPORT_LIST_URL = 'https://api.test/v1/reports/list' # 最大重试次数 MAX_ATTEMPTS = 3 class ReportParser(HTMLParser): def __init__(self): super().__init__() self.in_report_row = False self.current_row_data = [] self.reports = [] self.current_download_link = None def handle_starttag(self, tag, attrs): # 识别报告行(根据实际页面HTML结构调整标签,比如<tr>) if tag == 'tr': self.in_report_row = True self.current_row_data = [] # 捕获下载链接 if tag == 'a' and self.in_report_row: for attr_name, attr_value in attrs: if attr_name == 'href': self.current_download_link = attr_value def handle_data(self, data): if self.in_report_row: cleaned_data = data.strip() if cleaned_data: self.current_row_data.append(cleaned_data) def handle_endtag(self, tag): if tag == 'tr' and self.in_report_row: self.in_report_row = False # 确保行数据完整且存在下载链接 if len(self.current_row_data) >= 4 and self.current_download_link: self.reports.append({ 'date_str': self.current_row_data[0], 'download_url': self.current_download_link }) self.current_download_link = None def fetch_url_with_retry(target_url, max_attempts=MAX_ATTEMPTS): attempts_left = max_attempts attempt_count = 0 while attempts_left > 0: try: response = urllib.request.urlopen(target_url) if attempt_count > 0: print(f"成功访问 {target_url},重试次数:{attempt_count}") return response.read().decode('utf-8') except HTTPError: attempts_left -= 1 attempt_count += 1 print(f"无法访问 {target_url},正在重试...第 {attempt_count} 次") time.sleep(10) if attempts_left == 0: print(f"访问 {target_url} 的所有重试次数已耗尽,退出程序") sys.exit(1) def get_latest_report_download_url(): # 获取列表页面内容 page_content = fetch_url_with_retry(REPORT_LIST_URL) # 解析页面提取报告信息 parser = ReportParser() parser.feed(page_content) if not parser.reports: print("未在列表页面找到任何报告") sys.exit(1) # 解析日期并排序(适配"Jul 4, 2023"格式,可根据实际调整) def parse_report_date(report_item): try: return datetime.strptime(report_item['date_str'], '%b %d, %Y') except ValueError: # 兼容带时间的日期格式,如"Jul 4, 2023 14:30:00" return datetime.strptime(report_item['date_str'], '%b %d, %Y %H:%M:%S') # 按日期降序排序,取第一条为最新报告 parser.reports.sort(key=lambda x: parse_report_date(x), reverse=True) latest_report = parser.reports[0] print(f"找到最新报告:{latest_report['date_str']},下载链接:{latest_report['download_url']}") return latest_report['download_url'] def fetch_source_data(): # 获取最新报告的下载链接 download_url = get_latest_report_download_url() # 下载报告内容 report_content = fetch_url_with_retry(download_url) # 保存到本地(可调整路径和文件名) with open('latest_report.csv', 'w', encoding='utf-8') as f: f.write(report_content) print("最新报告已成功下载并保存") if __name__ == "__main__": fetch_source_data()
关键说明
HTML解析适配:
- 代码中的
ReportParser基于标准库HTMLParser实现,需根据实际页面的HTML结构调整标签判断逻辑(比如报告行不是<tr>则修改对应判断) - 若列表页面返回纯文本表格,可替换解析逻辑:跳过表头行,按空格拆分每行字段,提取日期和对应下载链接(需确认文本中链接的存储格式)
- 代码中的
日期格式兼容:
- 默认处理
Jul 4, 2023格式的日期,若实际日期带时间,需调整parse_report_date中的格式字符串
- 默认处理
重试逻辑优化:
- 将原有重试逻辑封装为通用函数
fetch_url_with_retry,复用在列表页访问和报告下载环节,修复了原代码中attempts未初始化的问题
- 将原有重试逻辑封装为通用函数
版本兼容性:
- 仅使用Python标准库,完全兼容3.6.8和3.10.11版本
内容的提问来源于stack exchange,提问作者rantish
相关产品推荐
相关产品推荐

