使用BeautifulSoup爬取MDPI遥感期刊PDF时仅能获取30篇的问题排查
解决MDPI期刊每期仅能爬取30篇PDF的问题
问题描述
我使用Python和BeautifulSoup爬取MDPI旗下《Remote Sensing》期刊的PDF文件,目标是将各卷、各期的文章PDF下载至本地。该期刊每卷包含多个期,每期又有多篇文章,文章PDF对应的元素class为UD_Listings_ArticlePDF。但目前代码每期最多仅能下载30篇文章,而多数期的文章数量远超过30篇(例如第8卷第2期)。已尝试扩展选择器,但仍只能获取30个PDF链接,请求协助排查问题。
原代码:
import requests from bs4 import BeautifulSoup import os import time # Base URL for the journal (change if the base URL pattern changes) base_url = "https://www.mdpi.com/2072-4292" # Directory to save the PDFs os.makedirs("mdpi_pdfs", exist_ok=True) # Define the range of volumes and issues to scrape start_volume = 1 end_volume = 16 # Change this number based on the latest volume available # Time delay between requests in seconds request_delay = 4 # Time delay between requests to avoid 429 errors # Maximum number of retries after 429 errors max_retries = 5 # Iterate over each volume for volume_num in range(start_volume, end_volume + 1): print(f"\nProcessing Volume {volume_num}...") # Assume a reasonable number of issues per volume start_issue = 1 end_issue = 30 # You may need to adjust this based on the number of issues per volume for issue_num in range(start_issue, end_issue + 1): issue_url = f"{base_url}/{volume_num}/{issue_num}" print(f" Processing Issue URL: {issue_url}") retries = 0 while retries < max_retries: try: # Get the content of the issue webpage response = requests.get(issue_url) # If issue URL doesn't exist, break the loop for this volume if response.status_code == 404: print(f" Issue {issue_num} in Volume {volume_num} does not exist. Moving to next volume.") time.sleep(request_delay * 5) break # Handle 429 errors gracefully if response.status_code == 429: print(f" Received 429 error. Too many requests. Retrying in {request_delay * 5} seconds...") retries += 1 time.sleep(request_delay * 5) # Exponential backoff strategy continue response.raise_for_status() # Check for other request errors # Parse the page content soup = BeautifulSoup(response.content, "html.parser") # Find all links that lead to PDFs pdf_links = soup.find_all("a", class_="UD_Listings_ArticlePDF") # Adjust class if needed if not pdf_links: print(f" No PDF links found for Issue {issue_num} in Volume {volume_num}.") break # Download each PDF for the current issue for index, link in enumerate(pdf_links, start=1): try: # Construct the full URL for the PDF pdf_url = f"https://www.mdpi.com{link['href']}" # Create a unique file name with volume and issue information pdf_name = f"mdpi_volume_{volume_num}_issue_{issue_num}_article_{index}.pdf" pdf_path = os.path.join("mdpi_pdfs", pdf_name) print(f" Downloading: {pdf_url}") # Download the PDF pdf_response = requests.get(pdf_url) pdf_response.raise_for_status() # Check for request errors # Save the PDF file with open(pdf_path, "wb") as file: file.write(pdf_response.content) print(f" Successfully downloaded: {pdf_name}") # Sleep after each successful download time.sleep(request_delay) except Exception as e: print(f" Failed to download {pdf_url}. Error: {e}") # Exit the retry loop since request was successful break except Exception as e: print(f" Failed to process Issue {issue_num} in Volume {volume_num}. Error: {e}") retries += 1 if retries < max_retries: print(f" Retrying in {request_delay * 2} seconds... (Retry {retries}/{max_retries})") time.sleep(request_delay * 2) else: print(f" Maximum retries reached. Skipping Issue {issue_num} in Volume {volume_num}.") print("\nDownload process completed for all specified volumes and issues!")
问题原因
MDPI期刊的期页面采用分页加载机制:默认仅渲染第一页的30篇文章,后续文章需要通过URL中的page参数来加载(例如https://www.mdpi.com/2072-4292/8/2?page=2对应第二页的30篇)。原代码仅请求了基础URL,未处理分页,因此只能获取前30篇文章的PDF链接。
解决方案
针对每期页面,循环请求带有不同page参数的URL,直到某一页返回的PDF链接为空(说明已加载完所有文章),再整合所有分页的链接进行下载。
修改后的代码:
import requests from bs4 import BeautifulSoup import os import time # Base URL for the journal (change if the base URL pattern changes) base_url = "https://www.mdpi.com/2072-4292" # Directory to save the PDFs os.makedirs("mdpi_pdfs", exist_ok=True) # Define the range of volumes and issues to scrape start_volume = 1 end_volume = 16 # Change this number based on the latest volume available # Time delay between requests in seconds request_delay = 4 # Time delay between requests to avoid 429 errors # Maximum number of retries after 429 errors max_retries = 5 # Iterate over each volume for volume_num in range(start_volume, end_volume + 1): print(f"\nProcessing Volume {volume_num}...") # Assume a reasonable number of issues per volume start_issue = 1 end_issue = 30 # You may need to adjust this based on the number of issues per volume for issue_num in range(start_issue, end_issue + 1): issue_base_url = f"{base_url}/{volume_num}/{issue_num}" print(f" Processing Issue: Volume {volume_num}, Issue {issue_num}") all_pdf_links = [] page = 1 while True: issue_url = f"{issue_base_url}?page={page}" print(f" Requesting page {page}: {issue_url}") retries = 0 success = False while retries < max_retries: try: response = requests.get(issue_url) if response.status_code == 404: if page == 1: print(f" Issue {issue_num} in Volume {volume_num} does not exist. Moving to next volume.") else: print(f" Page {page} does not exist. Reached end of issue.") success = False break if response.status_code == 429: print(f" Received 429 error. Retrying in {request_delay * 5} seconds...") retries += 1 time.sleep(request_delay * 5) continue response.raise_for_status() soup = BeautifulSoup(response.content, "html.parser") pdf_links = soup.find_all("a", class_="UD_Listings_ArticlePDF") if not pdf_links: print(f" No more PDF links found on page {page}. Reached end of issue.") success = False break all_pdf_links.extend(pdf_links) print(f" Found {len(pdf_links)} PDF links on page {page}") success = True break except Exception as e: print(f" Failed to process page {page}. Error: {e}") retries += 1 if retries < max_retries: print(f" Retrying in {request_delay * 2} seconds... (Retry {retries}/{max_retries})") time.sleep(request_delay * 2) else: print(f" Maximum retries reached for page {page}. Skipping this page.") success = False break if not success: break page += 1 time.sleep(request_delay) if not all_pdf_links: print(f" No PDF links found for Issue {issue_num} in Volume {volume_num}.") continue # Download all collected PDF links print(f" Starting download of {len(all_pdf_links)} articles for Issue {issue_num}") for index, link in enumerate(all_pdf_links, start=1): try: pdf_url = f"https://www.mdpi.com{link['href']}" pdf_name = f"mdpi_volume_{volume_num}_issue_{issue_num}_article_{index}.pdf" pdf_path = os.path.join("mdpi_pdfs", pdf_name) print(f" Downloading: {pdf_url}") pdf_response = requests.get(pdf_url) pdf_response.raise_for_status() with open(pdf_path, "wb") as file: file.write(pdf_response.content) print(f" Successfully downloaded: {pdf_name}") time.sleep(request_delay) except Exception as e: print(f" Failed to download {pdf_url}. Error: {e}") print("\nDownload process completed for all specified volumes and issues!")
代码修改说明
- 添加了分页循环逻辑:从
page=1开始,逐页请求直到没有更多PDF链接或页面返回404 - 新增
all_pdf_links列表,用于收集所有分页的PDF链接 - 拆分了页面请求和下载逻辑:先收集所有链接,再统一下载(也可保持边请求边下载,根据需求调整)
- 优化了错误处理,针对分页请求单独处理重试逻辑
内容的提问来源于stack exchange,提问作者Christopher Phillips
相关产品推荐
相关产品推荐

