使用Selenium爬取EPFO网站时验证码裁剪功能失效求助
解决EPFO网站验证码裁剪异常问题
我在用Selenium编写Python脚本爬取EPFO网站数据时,遇到验证码裁剪结果不符合预期的问题,裁剪出的验证码不完整。以下是我的代码:
import os import time from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.chrome.options import Options from selenium.common.exceptions import NoSuchElementException from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import pytesseract import re import pandas as pd from io import BytesIO from PIL import Image DOWNLOAD_DIR = "C:/Users/acer/OneDrive/Desktop/python_dev_coding_challenge/data" pytesseract.pytesseract.tesseract_cmd = r"C:/Program Files/Tesseract-OCR/tesseract.exe" def get_captcha_text(driver): captcha_element = driver.find_element(By.ID, "capImg") # Get the location and size of the captcha image element location = captcha_element.location size = captcha_element.size # Take a screenshot of the captcha image captcha_screenshot = driver.get_screenshot_as_png() # Crop the screenshot to get only the captcha image captcha_image = Image.open(BytesIO(captcha_screenshot)).crop( (location['x'], location['y'], location['x'] + size['width'], location['y'] + size['height']) ) # Save the captcha image to a file (optional) captcha_image.save("captcha.png") img = captcha_image.convert('L') # Convert to Grayscale contrast_image = img.point(lambda p: p * 1.5) # Process the captcha image using pytesseract captcha_text = pytesseract.image_to_string(contrast_image) remove_spaces = re.sub(r'\s', '', captcha_text) remove_char = re.sub(r'[^\w]', '', remove_spaces) return remove_char.upper() def scrape_data(company_name: str): ''' Scrape data from the EPFO website ''' # Create Selenium driver options = Options() # TODO: Add whatever options you might think are helpful prefs = {"download.default_directory": os.path.join(os.getcwd(), DOWNLOAD_DIR)} # Set the download directory to the data folder options.add_experimental_option("prefs", prefs) driver = webdriver.Chrome() # Open the EPFO website driver.get('https://unifiedportal-epfo.epfindia.gov.in/publicPortal/no-auth/misReport/home/loadEstSearchHome') # Use time library to visualize the browser else tab will close time.sleep(5) # TODO: Fill out the code for the following steps # Step 1 # Enter the company name in the search box try: search_box = driver.find_element(By.ID, "estName") search_box.send_keys(company_name) except NoSuchElementException as e: print(f"Error finding search box: {e}") time.sleep(2) captcha_value = get_captcha_text(driver) print("captcha text:", captcha_value) # Enter the captcha in the captcha box captcha_input = driver.find_element(By.ID, 'captchaImg') captcha_input.send_keys(captcha_value) # Click on the search button search_button = WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.ID, "searchEmployer")) ) search_button.click() # Wait for the search results to load time.sleep(2) # Step 2 # Click on the "View Details" button view_details_button = driver.find_element(By.LINK_TEXT, "View Details") view_details_button.click() # Wait for the page to load time.sleep(2) # Click on the "View Payment Details" button in the new tab view_payment_details_button = driver.find_element(By.LINK_TEXT, "View Payment Details") view_payment_details_button.click() # Switch to the new tab driver.switch_to.window(driver.window_handles[1]) # Step 3 # Click on the "Excel" button to download the Excel file excel_button = driver.find_element(By.LINK_TEXT, "Excel") excel_button.click() # Wait for the download to complete time.sleep(5) # Close the browser driver.quit() print("Driver closed successfully...!!!") def test_scrape_data(): ''' Test the scraped data ''' # Convert xlsx file to csv due to some issues with pandas from xlsx2csv import Xlsx2csv Xlsx2csv("C:/Users/acer/OneDrive/Desktop/python_dev_coding_challenge/data/Payment Details.xlsx", outputencoding="utf-8").convert("payment_details.csv") df = pd.read_csv("payment_details.csv") assert set(df.columns) == set(['TRRN', 'Date Of Credit', 'Amount', 'Wage Month', 'No. of Employee', 'ECR']) assert df['TRRN'].loc[0] == 3171702000767 assert df['Date Of Credit'].loc[0] == '03-FEB-2017 14:35:15' assert df['Amount'].loc[0] == 334901 assert df['Wage Month'].loc[0] == 'DEC-16' assert df['No. of Employee'].loc[0] == 83 assert df['ECR'].loc[0] == 'YES' print("All tests passed!") def main(): print("Hello World!") # Uncomment the following line when you are ready to test the scraping function scrape_data("MGH LOGISTICS PVT LTD") # Uncomment the following tests whenever scraping is completed. test_scrape_data() # TODO: Feel free to add any edge cases which you might think are helpful if __name__ == "__main__": main()
解决方案
1. 修正坐标缩放问题
Selenium返回的location是相对视口的坐标,若浏览器有系统缩放(比如Windows的125%显示缩放),会导致裁剪偏移。改用元素的rect属性配合设备像素比校正:
def get_captcha_text(driver): captcha_element = driver.find_element(By.ID, "capImg") # 获取设备像素比,校正坐标 scale = driver.execute_script("return window.devicePixelRatio;") rect = captcha_element.rect x = rect['x'] * scale y = rect['y'] * scale width = rect['width'] * scale height = rect['height'] * scale captcha_screenshot = driver.get_screenshot_as_png() captcha_image = Image.open(BytesIO(captcha_screenshot)).crop( (x, y, x + width, y + height) ) captcha_image.save("captcha.png") img = captcha_image.convert('L') contrast_image = img.point(lambda p: p * 1.5) captcha_text = pytesseract.image_to_string(contrast_image) remove_spaces = re.sub(r'\s', '', captcha_text) remove_char = re.sub(r'[^\w]', '', remove_spaces) return remove_char.upper()
2. 直接下载验证码图片
绕过页面截图的坐标问题,直接获取验证码图片URL下载:
def get_captcha_text(driver): captcha_element = driver.find_element(By.ID, "capImg") captcha_url = captcha_element.get_attribute('src') # 需先安装requests库:pip install requests import requests response = requests.get(captcha_url) captcha_image = Image.open(BytesIO(response.content)) captcha_image.save("captcha.png") img = captcha_image.convert('L') contrast_image = img.point(lambda p: p * 1.5) captcha_text = pytesseract.image_to_string(contrast_image) remove_spaces = re.sub(r'\s', '', captcha_text) remove_char = re.sub(r'[^\w]', '', remove_spaces) return remove_char.upper()
3. 强制浏览器100%缩放
启动浏览器时添加参数,避免自动缩放导致的坐标偏差:
options = Options() options.add_argument("--force-device-scale-factor=1") # 强制100%缩放 prefs = {"download.default_directory": os.path.join(os.getcwd(), DOWNLOAD_DIR)} options.add_experimental_option("prefs", prefs) driver = webdriver.Chrome(options=options)
4. 修正验证码输入框ID错误
代码中验证码输入框的ID应为captcha,而非captchaImg,需修改:
# 原错误代码 captcha_input = driver.find_element(By.ID, 'captchaImg') # 修正后 captcha_input = driver.find_element(By.ID, 'captcha')
内容的提问来源于stack exchange,提问作者Shubham Dikshit
相关产品推荐
相关产品推荐

