You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Python从网站弹窗提取邮箱地址遇阻,求解决方案

问题:提取https://www.hotelleriesuisse.ch网站弹窗中的邮箱地址

尝试从该网站提取邮箱地址时,点击邮箱图标会弹出新弹窗。使用Selenium的get_attribute方法获取data-mailto-token和data-mailto-vector属性失败,试过Selenium及其他跨平台库均无效果,求Python实现提取这类弹窗中邮箱地址的方法。

用户提供的代码:

from selenium import webdriver
from webdriver_manager.chrome import ChromeDriverManager
import time
from selenium.webdriver.support.ui import WebDriverWait

from selenium.webdriver.common.by import By
from selenium.common.exceptions import TimeoutException, NoSuchElementException
from selenium.webdriver.support import expected_conditions as ec
from selenium.webdriver.common.action_chains import ActionChains
from selenium.webdriver.common.keys import Keys
from selenium.common.exceptions import ElementClickInterceptedException
from selenium.webdriver.support.select import Select
from bs4 import BeautifulSoup
import requests
import re


#card_small = driver.find_elements_by_class_name("Card small")

i_num = 1

list_links = []

list_links_all = []

num_inc = 1

for i_p in range(0,14):

    url = "https://www.hotelleriesuisse.ch/de/branche-und-politik/branchenverzeichnis/hotel-page-"+str(num_inc)+"?filterValues=QWN0aXZlLEluYWN0aXZlOzs7OzQsMzs7Ozs7OzQ5LDEzLDUsNDU7&cHash=30901b0e3080a928cd0ad32522e81b3f"
    headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/102.0.0.0 Safari/537.36'}

    driver = webdriver.Chrome(ChromeDriverManager().install())
    driver.get(url)

    time.sleep(5)


    driver.find_element_by_css_selector("body > div.cc-window.cc-banner.cc-type-info.cc-theme-block.cc-bottom.cc-visible > div > div.cc-actions > a.cc-btn.cc-allow").click()

    try:
        driver.execute_script("window.scrollTo(0,2150)")

        target = driver.find_elements_by_tag_name("a")

        for i in target:
            list_links.append(i.get_attribute("href"))

        for i in range(10,22):
            url_new = list_links[i]
            print(url_new)
            headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/102.0.0.0 Safari/537.36'}
            page = requests.get(url_new, headers=headers)
            soup = BeautifulSoup(page.text, 'html.parser')

            name = soup.find('span',class_="Avatar--name")
            address = soup.find_all('span', class_="Button--label")
            phone = soup.find_all('span', class_="Button--label")


            if name != None:
                name_text = soup.find('span', class_="Avatar--name").text
                #print(name_text)

            if address != None:
                for i in address:
                    search=i.select("span p")
                    if search != []:
                        print(search[0].text)
            if phone != None:
                for i in phone:
                    match = re.search("[+]\d{2} \d{2} \d{3} \d{2} \d{2}",i.text)
                    if match !=None:
                        print(match.group())
            time.sleep(5)

            driver.get(url_new)

            try:

                driver.execute_script("window.scrollTo(0,900)")

                time.sleep(5)

                element=driver.find_element_by_link_text("E-Mail")

                info = element.get_attribute("data-mailto-token")

                print(info)

                element.click()



            except NoSuchElementException:
                pass




        list_links = []
        num_inc = num_inc + 1
        i_num = i_num + 1

        driver.close()

        """
        driver.find_element_by_css_selector("#main-content > section.CardGrid > nav > a.Button.nolabel.primary.Pagination--button.Pagination--next").click()
        time.sleep(5)
        print("This is the end of page: "+str(i_num))
        i_num = i_num + 1
        time.sleep(5)
        """
    except ElementClickInterceptedException:
        break

解决方案

1. 核心逻辑分析

该网站的邮箱采用AES加密存储,data-mailto-token和data-mailto-vector是解密所需的参数,网站前端内置了解密函数HotellerieSuisse.decryptMailto(token, vector),无需点击弹窗即可直接调用获取真实邮箱。

2. 优化后的代码实现

from selenium import webdriver
from webdriver_manager.chrome import ChromeDriverManager
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.common.by import By
from selenium.common.exceptions import TimeoutException, NoSuchElementException, ElementClickInterceptedException
from selenium.webdriver.support import expected_conditions as ec
import re

# 初始化浏览器(仅执行一次,避免重复创建资源)
driver = webdriver.Chrome(ChromeDriverManager().install())
wait = WebDriverWait(driver, 10)

i_num = 1
list_links = []
num_inc = 1

for i_p in range(0,14):
    url = f"https://www.hotelleriesuisse.ch/de/branche-und-politik/branchenverzeichnis/hotel-page-{num_inc}?filterValues=QWN0aXZlLEluYWN0aXZlOzs7OzQsMzs7Ozs7OzQ5LDEzLDUsNDU7&cHash=30901b0e3080a928cd0ad32522e81b3f"
    driver.get(url)
    
    # 处理Cookie弹窗(显式等待,替代固定sleep)
    try:
        cookie_btn = wait.until(ec.element_to_be_clickable((By.CSS_SELECTOR, "a.cc-btn.cc-allow")))
        cookie_btn.click()
    except TimeoutException:
        pass
    
    # 滚动页面并筛选有效酒店链接
    driver.execute_script("window.scrollTo(0,2150)")
    target_links = wait.until(ec.presence_of_all_elements_located((By.TAG_NAME, "a")))
    list_links = [link.get_attribute("href") for link in target_links if link.get_attribute("href") and "hotel-detail" in link.get_attribute("href")]
    
    # 遍历酒店详情页
    for url_new in list_links[10:22]:
        print(f"\n当前页面:{url_new}")
        driver.get(url_new)
        
        # 提取酒店名称
        try:
            name_text = wait.until(ec.presence_of_element_located((By.CLASS_NAME, "Avatar--name"))).text
            print(f"酒店名称:{name_text}")
        except TimeoutException:
            name_text = None
        
        # 提取地址
        address_elements = driver.find_elements(By.CLASS_NAME, "Button--label")
        for elem in address_elements:
            address_p = elem.find_elements(By.TAG_NAME, "p")
            if address_p:
                print(f"地址:{address_p[0].text}")
        
        # 提取电话
        phone_elements = driver.find_elements(By.CLASS_NAME, "Button--label")
        for elem in phone_elements:
            match = re.search(r"[+]\d{2} \d{2} \d{3} \d{2} \d{2}", elem.text)
            if match:
                print(f"电话:{match.group()}")
        
        # 提取邮箱(调用前端解密函数)
        try:
            driver.execute_script("window.scrollTo(0,900)")
            email_btn = wait.until(ec.element_to_be_clickable((By.LINK_TEXT, "E-Mail")))
            
            # 获取加密参数
            token = email_btn.get_attribute("data-mailto-token")
            vector = email_btn.get_attribute("data-mailto-vector")
            
            # 执行前端解密JS,直接获取邮箱
            decrypted_email = driver.execute_script(f"""
                return HotellerieSuisse.decryptMailto('{token}', '{vector}');
            """)
            print(f"邮箱:{decrypted_email}")
            
        except (TimeoutException, NoSuchElementException):
            print("未找到邮箱按钮")
            pass
    
    list_links = []
    num_inc += 1
    i_num += 1

# 关闭浏览器
driver.quit()

关键优化点

  • 复用浏览器实例:将driver初始化移到循环外,避免重复创建销毁浏览器,提升效率。
  • 显式等待替代固定sleep:用WebDriverWait等待元素加载,避免因网络延迟导致的元素查找失败。
  • 直接调用解密函数:无需点击弹窗,通过execute_script调用网站内置的解密函数,直接获取真实邮箱。
  • 筛选有效链接:只保留包含hotel-detail的链接,减少无效遍历。

内容的提问来源于stack exchange,提问作者linus otte

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.12 23:40:27