You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Python+Selenium爬取时生成空白PDF的问题求助

问题:使用Selenium从Jamabandi网站打印PDF时生成空白文件

在网页https://jamabandi.nic.in/land%20records/NakalRecord上,通过Selenium选择各下拉框首个选项后触发生成PDF操作,网页上明明显示有表格内容,但手动和自动化打印出的PDF始终为空白。相关代码如下:

from selenium import webdriver
from selenium.webdriver.chrome.service import Service

service = Service()
options = webdriver.ChromeOptions()
# Set up preferences for printing to PDF
settings = {
    "recentDestinations": [{"id": "Save as PDF", "origin": "local", "account": ""}],
    "selectedDestinationId": "Save as PDF",
    "version": 2
}
prefs = {
    'printing.print_preview_sticky_settings.appState': json.dumps(settings),
    'printing.print_to_file': True,
    'printing.print_to_file.path': '/Users/jatin/Downloads/output.pdf'  # Specify the desired output path
}
chrome_options.add_experimental_option('prefs', prefs)

import urllib.request
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.chrome.options import Options
from webdriver_manager.chrome import ChromeDriverManager
# Set up Chrome options
chrome_options = Options()
# chrome_options.add_argument('--headless')  # Optional: Run Chrome in headless mode
chrome_options.add_argument('--kiosk-printing')
try:
    service = Service(ChromeDriverManager().install())
except ValueError:
    latest_chromedriver_version_url = "https://chromedriver.storage.googleapis.com/LATEST_RELEASE"
    latest_chromedriver_version = urllib.request.urlopen(latest_chromedriver_version_url).read().decode('utf-8')
    service = Service(ChromeDriverManager(version=latest_chromedriver_version).install())

    
options = Options()
url='https://jamabandi.nic.in/land%20records/NakalRecord'
# options.add_argument('--headless') #optional.
driver = webdriver.Chrome(service=service, options=options)
driver.get(url)


dropdown_district = Select(driver.find_element(By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddldname"]'))
dropdown_district.select_by_index(1)
# Select the tehsil dropdown element and choose the first option,we will loop here for multiple anchals
drop_down_tehsil = Select(driver.find_element(By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddltname"]'))
drop_down_tehsil.select_by_index(1)
drop_down_vill = Select(driver.find_element(By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddlvname"]'))
drop_down_vill.select_by_index(1)
drop_down_year = Select(driver.find_element(By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddlPeriod"]'))
drop_down_year.select_by_index(1)
owner_names=Select(driver.find_element(By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ListBox1"]'))
dropdown_locator = (By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ListBox1"]')
drop_down_owner = Select(driver.find_element(By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddlOwner"]'))
drop_down_owner.select_by_index(1)
owner_names =Select(driver.find_element(By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ListBox1"]'))
owner_names.select_by_index(2)
page_source = BeautifulSoup(driver.page_source, 'html.parser')
table = page_source.find_all('table')
div_col_lg_12 = page_source.find('div', class_='col-lg-12')

# Find links within the selected div
links_within_div = div_col_lg_12.find_all('td')
links_within_div
# Perform actions on the links or retrieve their attributes
for link in links_within_div:
    k=link.find_all('a')
    if len(k)>0:
        new_link=(k[0]['href'])
        
javascript_code = str(new_link)

# Execute the JavaScript code
driver.execute_script(javascript_code)

window_handles=driver.window_handles
driver.switch_to.window(window_handles[-1])

# Open the print dialog using JavaScript
driver.execute_script('window.print();')

问题分析及解决办法

核心问题点

  1. 代码中重复定义options和chrome_options变量,导致打印偏好配置未被正确应用
  2. 未等待新窗口的PDF内容完全渲染就触发打印
  3. 通过执行href脚本打开新窗口的方式稳定性不足
  4. 未启用打印背景图形的配置,可能导致部分内容无法被捕获

修正后的代码

import json
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import Select, WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from webdriver_manager.chrome import ChromeDriverManager
import urllib.request

# 统一Chrome配置,避免变量冲突
chrome_options = Options()
# 配置PDF打印参数,启用背景图形确保内容完整
print_settings = {
    "recentDestinations": [{"id": "Save as PDF", "origin": "local", "account": ""}],
    "selectedDestinationId": "Save as PDF",
    "version": 2,
    "isCssBackgroundEnabled": True
}
prefs = {
    'printing.print_preview_sticky_settings.appState': json.dumps(print_settings),
    'printing.print_to_file': True,
    'printing.print_to_file.path': '/Users/jatin/Downloads/output.pdf'
}
chrome_options.add_experimental_option('prefs', prefs)
chrome_options.add_argument('--kiosk-printing')
# chrome_options.add_argument('--headless=new')  # 如需无头模式,建议使用新版无头参数

# 初始化Chrome驱动
try:
    service = Service(ChromeDriverManager().install())
except ValueError:
    latest_version = urllib.request.urlopen("https://chromedriver.storage.googleapis.com/LATEST_RELEASE").read().decode('utf-8')
    service = Service(ChromeDriverManager(version=latest_version).install())

driver = webdriver.Chrome(service=service, options=chrome_options)
driver.get('https://jamabandi.nic.in/land%20records/NakalRecord')

# 显式等待所有元素可交互,避免页面未加载完成就操作
wait = WebDriverWait(driver, 10)

# 依次选择下拉框选项
district_dropdown = Select(wait.until(EC.element_to_be_clickable((By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddldname"]'))))
district_dropdown.select_by_index(1)

tehsil_dropdown = Select(wait.until(EC.element_to_be_clickable((By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddltname"]'))))
tehsil_dropdown.select_by_index(1)

vill_dropdown = Select(wait.until(EC.element_to_be_clickable((By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddlvname"]'))))
vill_dropdown.select_by_index(1)

year_dropdown = Select(wait.until(EC.element_to_be_clickable((By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddlPeriod"]'))))
year_dropdown.select_by_index(1)

owner_type_dropdown = Select(wait.until(EC.element_to_be_clickable((By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ddlOwner"]'))))
owner_type_dropdown.select_by_index(1)

owner_name_dropdown = Select(wait.until(EC.element_to_be_clickable((By.XPATH, '//*[@id="ctl00_ContentPlaceHolder1_ListBox1"]'))))
owner_name_dropdown.select_by_index(2)

# 等待Nakal链接加载完成,直接点击而非执行脚本
nakal_link = wait.until(EC.element_to_be_clickable((By.XPATH, '//div[@class="col-lg-12"]//td//a')))
original_window = driver.current_window_handle
nakal_link.click()

# 等待新窗口打开并切换
wait.until(EC.number_of_windows_to_be(2))
for window_handle in driver.window_handles:
    if window_handle != original_window:
        driver.switch_to.window(window_handle)
        break

# 等待PDF页面完全渲染,增加额外等待确保内容加载完成
wait.until(EC.presence_of_element_located((By.TAG_NAME, 'body')))
driver.implicitly_wait(5)

# 触发打印
driver.execute_script('window.print();')

# 结束操作后关闭驱动
# driver.quit()

关键优化说明

  • 合并了重复的Chrome配置变量,确保打印偏好生效
  • 全程使用显式等待,避免因页面加载延迟导致的元素未就绪问题
  • 直接点击链接元素替代执行href脚本,提升操作稳定性
  • 启用了isCssBackgroundEnabled配置,确保网页样式和背景内容能被正确打印
  • 增加了PDF页面渲染后的等待时间,避免内容未加载完成就触发打印

内容的提问来源于stack exchange,提问作者jatin rajani

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.02 12:45:32