You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何使用Selenium检测网站真实状态?解决urllib检测偏差问题

问题描述

我正在处理一批URL,希望使用Selenium检测网站状态。此前用urllib检测显示状态正常,但有时在Selenium或浏览器中打开这些URL却无法正常访问。请问有没有办法通过Selenium检测网站状态,确保其能正常访问?

用户提供的原始代码

import time
from urllib.request import urlopen
from urllib.error import URLError
from urllib.error import HTTPError
from http import HTTPStatus
from selenium import webdriver
from Base_Class import *


# get the status of a website
def get_website_status(url):
    # handle connection errors
    try:
        # open a connection to the server with a timeout
        with urlopen(url, timeout=3) as connection:
            # get the response code, e.g. 200
            code = connection.getcode()
            return code
    except HTTPError as e:
        return e.code
    except URLError as e:
        return e.reason
    except:
        return e


# interpret an HTTP response code into a status
def get_status(code):
    if code == HTTPStatus.OK:
        return 'OK'
    return 'ERROR'


# check status of a list of websites
def check_status_urls():
    http = 0
    https = 0
    db_conn = base_class.table_selected_urls()
    db_conn.execute("SELECT url FROM SELECTED_URLS LIMIT 50")
    urls = db_conn.fetchall()
    url_protocols = ['http://','https://']

    #driver = base_class.web_driver()
    #driver.current_url()
    for url in urls:
        for url_protocol in url_protocols:
           full_https_url = url_protocol + url[0]
           Http_Https_status = get_website_status(full_https_url)
        # interpret the status
           status = get_status(Http_Https_status)
    # report status
           #print(f'{status:5s}')

           if full_https_url.split(':')[0] == 'https' and status == 'OK':
               print('https:: '+ full_https_url + ' ' + status)
               https += 1

           if full_https_url.split(':')[0] == 'http' and status == 'OK':
               print('http:: ' + full_https_url + ' ' + status)
               http += 1

    print('Number of https :: ' + str(https))
    print('Number of http :: ' + str(http))


# list of urls to check

# check all urls
check_status_urls()

解决方案

核心差异原因

urllib仅发起基础HTTP请求获取状态码,不会处理JS渲染、反爬验证(如UA检测、验证码)、页面资源加载失败等浏览器层面的问题,而Selenium模拟真实浏览器环境,这些差异会导致两者检测结果不一致。

基于Selenium的可靠检测方案

要确保网站能在浏览器中正常访问,需从页面加载完整性和内容可用性两方面验证,而非仅依赖HTTP状态码。以下是实现步骤和改进代码:

关键实现要点

  • 使用无头浏览器提升检测效率
  • 配置页面加载超时避免无限等待
  • 通过显式等待验证核心元素加载,确保页面正常渲染
  • 分类捕获不同类型的浏览器异常,精准定位问题

改进后的代码

import time
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import (TimeoutException, WebDriverException)
from Base_Class import *

def check_website_with_selenium(url):
    driver = None
    try:
        # 配置Chrome无头模式,减少资源占用
        options = webdriver.ChromeOptions()
        options.add_argument('--headless=new')
        options.add_argument('--disable-gpu')
        options.add_argument('--no-sandbox')
        options.add_argument('--user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36')
        
        driver = webdriver.Chrome(options=options)
        # 设置页面加载和脚本执行超时
        driver.set_page_load_timeout(10)
        driver.set_script_timeout(8)
        
        driver.get(url)
        
        # 等待页面核心元素加载(可替换为目标网站的关键标识元素,如登录按钮、logo等)
        WebDriverWait(driver, 8).until(
            EC.presence_of_element_located((By.TAG_NAME, 'body'))
        )
        
        # 额外验证:检查页面标题是否有效,排除空白或错误页面
        if driver.title and len(driver.title.strip()) > 0:
            return "OK"
        else:
            return "BLANK_PAGE"
            
    except TimeoutException:
        return "LOAD_TIMEOUT"
    except WebDriverException as e:
        err_msg = str(e)
        if "Failed to connect" in err_msg:
            return "CONNECTION_FAILED"
        elif "404" in err_msg:
            return "NOT_FOUND"
        elif "SSL_ERROR" in err_msg:
            return "SSL_ERROR"
        else:
            return f"ERROR: {err_msg[:60]}"
    finally:
        # 确保浏览器进程关闭
        if driver:
            driver.quit()

def check_status_urls():
    http_ok = 0
    https_ok = 0
    db_conn = base_class.table_selected_urls()
    db_conn.execute("SELECT url FROM SELECTED_URLS LIMIT 50")
    urls = db_conn.fetchall()
    url_protocols = ['http://','https://']

    for url in urls:
        for protocol in url_protocols:
            full_url = protocol + url[0]
            status = check_website_with_selenium(full_url)
            
            if status == "OK":
                print(f"{protocol[:-3]}:: {full_url} {status}")
                if protocol == 'https://':
                    https_ok += 1
                else:
                    http_ok += 1
            else:
                print(f"{protocol[:-3]}:: {full_url} {status}")

    print(f'Number of https OK :: {https_ok}')
    print(f'Number of http OK :: {http_ok}')

check_status_urls()

代码说明

  • User-Agent配置:模拟真实浏览器UA,避免被反爬机制拦截
  • 显式等待:替换time.sleep(),仅在元素加载完成后继续执行,提升稳定性
  • 异常细化:区分连接失败、超时、SSL错误等不同问题,便于后续排查
  • 资源清理:finally块确保浏览器进程被关闭,避免资源泄漏

内容的提问来源于stack exchange,提问作者stack overflow

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.15 00:40:41