You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Selenium爬取EPFO网站时验证码裁剪功能失效求助

解决EPFO网站验证码裁剪异常问题

我在用Selenium编写Python脚本爬取EPFO网站数据时,遇到验证码裁剪结果不符合预期的问题,裁剪出的验证码不完整。以下是我的代码:

import os
import time
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.options import Options
from selenium.common.exceptions import NoSuchElementException
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
import pytesseract
import re
import pandas as pd
from io import BytesIO
from PIL import Image

DOWNLOAD_DIR = "C:/Users/acer/OneDrive/Desktop/python_dev_coding_challenge/data"
pytesseract.pytesseract.tesseract_cmd = r"C:/Program Files/Tesseract-OCR/tesseract.exe"
def get_captcha_text(driver):
    captcha_element = driver.find_element(By.ID, "capImg")

    # Get the location and size of the captcha image element
    location = captcha_element.location
    size = captcha_element.size

    # Take a screenshot of the captcha image
    captcha_screenshot = driver.get_screenshot_as_png()

    # Crop the screenshot to get only the captcha image
    captcha_image = Image.open(BytesIO(captcha_screenshot)).crop(
        (location['x'], location['y'], location['x'] + size['width'],
         location['y'] + size['height'])
    )

    # Save the captcha image to a file (optional)
    captcha_image.save("captcha.png")

    img = captcha_image.convert('L')  # Convert to Grayscale
    contrast_image = img.point(lambda p: p * 1.5)

    # Process the captcha image using pytesseract
    captcha_text = pytesseract.image_to_string(contrast_image)
    remove_spaces = re.sub(r'\s', '', captcha_text)
    remove_char = re.sub(r'[^\w]', '', remove_spaces)
    return remove_char.upper()

def scrape_data(company_name: str):
    '''
    Scrape data from the EPFO website
    '''

    # Create Selenium driver
    options = Options()
    # TODO: Add whatever options you might think are helpful
    prefs = {"download.default_directory": os.path.join(os.getcwd(), DOWNLOAD_DIR)}  # Set the download directory to the data folder
    options.add_experimental_option("prefs", prefs)
    driver = webdriver.Chrome()

    # Open the EPFO website
    driver.get('https://unifiedportal-epfo.epfindia.gov.in/publicPortal/no-auth/misReport/home/loadEstSearchHome')

    # Use time library to visualize the browser else tab will close
    time.sleep(5)

    # TODO: Fill out the code for the following steps
    # Step 1
    # Enter the company name in the search box
    try:
        search_box = driver.find_element(By.ID, "estName")
        search_box.send_keys(company_name)
    except NoSuchElementException as e:
        print(f"Error finding search box: {e}")
    time.sleep(2)

    captcha_value = get_captcha_text(driver)
    print("captcha text:", captcha_value)

    # Enter the captcha in the captcha box
    captcha_input = driver.find_element(By.ID, 'captchaImg')
    captcha_input.send_keys(captcha_value)

    # Click on the search button
    search_button = WebDriverWait(driver, 10).until(
        EC.presence_of_element_located((By.ID, "searchEmployer"))
    )
    search_button.click()

    # Wait for the search results to load
    time.sleep(2)

    # Step 2
    # Click on the "View Details" button
    view_details_button = driver.find_element(By.LINK_TEXT, "View Details")
    view_details_button.click()

    # Wait for the page to load
    time.sleep(2)

    # Click on the "View Payment Details" button in the new tab
    view_payment_details_button = driver.find_element(By.LINK_TEXT, "View Payment Details")
    view_payment_details_button.click()

    # Switch to the new tab
    driver.switch_to.window(driver.window_handles[1])

    # Step 3
    # Click on the "Excel" button to download the Excel file
    excel_button = driver.find_element(By.LINK_TEXT, "Excel")
    excel_button.click()

    # Wait for the download to complete
    time.sleep(5)

    # Close the browser
    driver.quit()

    print("Driver closed successfully...!!!")


def test_scrape_data():
    '''
    Test the scraped data
    '''
    # Convert xlsx file to csv due to some issues with pandas
    from xlsx2csv import Xlsx2csv
    Xlsx2csv("C:/Users/acer/OneDrive/Desktop/python_dev_coding_challenge/data/Payment Details.xlsx",
             outputencoding="utf-8").convert("payment_details.csv")

    df = pd.read_csv("payment_details.csv")

    assert set(df.columns) == set(['TRRN', 'Date Of Credit', 'Amount', 'Wage Month', 'No. of Employee', 'ECR'])
    assert df['TRRN'].loc[0] == 3171702000767
    assert df['Date Of Credit'].loc[0] == '03-FEB-2017 14:35:15'
    assert df['Amount'].loc[0] == 334901
    assert df['Wage Month'].loc[0] == 'DEC-16'
    assert df['No. of Employee'].loc[0] == 83
    assert df['ECR'].loc[0] == 'YES'
    print("All tests passed!")


def main():
    print("Hello World!")

    # Uncomment the following line when you are ready to test the scraping function
    scrape_data("MGH LOGISTICS PVT LTD")

    # Uncomment the following tests whenever scraping is completed.
    test_scrape_data()
    # TODO: Feel free to add any edge cases which you might think are helpful


if __name__ == "__main__":
    main()

解决方案

1. 修正坐标缩放问题

Selenium返回的location是相对视口的坐标,若浏览器有系统缩放(比如Windows的125%显示缩放),会导致裁剪偏移。改用元素的rect属性配合设备像素比校正:

def get_captcha_text(driver):
    captcha_element = driver.find_element(By.ID, "capImg")
    
    # 获取设备像素比,校正坐标
    scale = driver.execute_script("return window.devicePixelRatio;")
    rect = captcha_element.rect
    x = rect['x'] * scale
    y = rect['y'] * scale
    width = rect['width'] * scale
    height = rect['height'] * scale
    
    captcha_screenshot = driver.get_screenshot_as_png()
    captcha_image = Image.open(BytesIO(captcha_screenshot)).crop(
        (x, y, x + width, y + height)
    )
    
    captcha_image.save("captcha.png")
    
    img = captcha_image.convert('L')
    contrast_image = img.point(lambda p: p * 1.5)
    captcha_text = pytesseract.image_to_string(contrast_image)
    remove_spaces = re.sub(r'\s', '', captcha_text)
    remove_char = re.sub(r'[^\w]', '', remove_spaces)
    return remove_char.upper()

2. 直接下载验证码图片

绕过页面截图的坐标问题,直接获取验证码图片URL下载:

def get_captcha_text(driver):
    captcha_element = driver.find_element(By.ID, "capImg")
    captcha_url = captcha_element.get_attribute('src')
    
    # 需先安装requests库:pip install requests
    import requests
    response = requests.get(captcha_url)
    captcha_image = Image.open(BytesIO(response.content))
    
    captcha_image.save("captcha.png")
    
    img = captcha_image.convert('L')
    contrast_image = img.point(lambda p: p * 1.5)
    captcha_text = pytesseract.image_to_string(contrast_image)
    remove_spaces = re.sub(r'\s', '', captcha_text)
    remove_char = re.sub(r'[^\w]', '', remove_spaces)
    return remove_char.upper()

3. 强制浏览器100%缩放

启动浏览器时添加参数,避免自动缩放导致的坐标偏差:

options = Options()
options.add_argument("--force-device-scale-factor=1")  # 强制100%缩放
prefs = {"download.default_directory": os.path.join(os.getcwd(), DOWNLOAD_DIR)}
options.add_experimental_option("prefs", prefs)
driver = webdriver.Chrome(options=options)

4. 修正验证码输入框ID错误

代码中验证码输入框的ID应为captcha,而非captchaImg,需修改:

# 原错误代码
captcha_input = driver.find_element(By.ID, 'captchaImg')
# 修正后
captcha_input = driver.find_element(By.ID, 'captcha')

内容的提问来源于stack exchange,提问作者Shubham Dikshit

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.02 03:05:06