Python网页解析:如何避免被重定向到网站本地版本
解决站点重定向到本地版本的问题
要强制访问美国版站点,需从Selenium浏览器配置和Requests请求头两方面修改,模拟美国地区的访问环境:
1. 核心修改思路
站点会根据请求的语言、地域Cookie、IP等信息自动重定向,我们需要通过以下方式规避:
- 给Chrome浏览器设置美式英语语言偏好
- 添加站点专属的地区锁定Cookie(以Nike为例)
- 给Requests请求添加美式英语的Accept-Language头
2. 修改后的完整代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.chrome.options import Options import time from urllib.parse import urlparse from bs4 import BeautifulSoup import requests def get_navigation_links(url, limit=500, wait_time=5): def validate_url(url_string): try: result = urlparse(url_string) if not result.scheme: url_string = "https://" + url_string result = urlparse(url_string) return url_string if result.netloc else None except: return None validated_url = validate_url(url) if not validated_url: raise ValueError("Invalid URL") base_netloc = urlparse(validated_url).netloc.split(':')[0] # Try JavaScript-rendered version first (Selenium) try: chrome_options = Options() chrome_options.add_argument("--headless") chrome_options.add_argument("--disable-gpu") chrome_options.add_argument("--no-sandbox") chrome_options.add_argument("--window-size=1920,1080") # 强制设置浏览器语言为美式英语 chrome_options.add_argument("--lang=en-US") chrome_options.add_experimental_option("prefs", { "intl.accept_languages": "en-US,en" }) driver = webdriver.Chrome(options=chrome_options) driver.get(validated_url) # 添加Nike地区锁定Cookie,强制访问美国版 driver.add_cookie({ "name": "country", "value": "US", "domain": ".nike.com" }) driver.add_cookie({ "name": "locale", "value": "en_US", "domain": ".nike.com" }) driver.refresh() # 刷新页面应用Cookie time.sleep(wait_time) # 等待JS渲染完成 # 检查是否仍有重定向 current_url = driver.current_url if base_netloc in current_url and current_url != validated_url: print(f"Redirect detected: {current_url}. Scraping original URL.") # 提取站内导航链接 a_tags = driver.find_elements(By.TAG_NAME, "a") seen = set() nav_links = [] for a in a_tags: try: href = a.get_attribute("href") text = a.text.strip() if href and text and urlparse(href).netloc.split(':')[0] == base_netloc: if href not in seen: seen.add(href) nav_links.append((text, href)) except: continue driver.quit() # Selenium未找到链接时,用BeautifulSoup降级处理 if not nav_links: print("No navigation links found via Selenium. Falling back to BeautifulSoup.") # 设置请求头模拟美国地区访问 headers = { "Accept-Language": "en-US,en;q=0.9", "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36" } soup = BeautifulSoup(requests.get(validated_url, headers=headers).text, 'html.parser') a_tags = soup.find_all('a') for a in a_tags: href = a.get('href') text = a.get_text(strip=True) if href and text and urlparse(href).netloc.split(':')[0] == base_netloc: if href not in seen: seen.add(href) nav_links.append((text, href)) return nav_links[:limit] except Exception as e: print(f"[Selenium failed: {e}] Falling back to BeautifulSoup.") # 异常时的降级处理,同样添加请求头 headers = { "Accept-Language": "en-US,en;q=0.9", "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36" } soup = BeautifulSoup(requests.get(validated_url, headers=headers).text, 'html.parser') a_tags = soup.find_all('a') seen = set() nav_links = [] for a in a_tags: href = a.get('href') text = a.get_text(strip=True) if href and text and urlparse(href).netloc.split(':')[0] == base_netloc: if href not in seen: seen.add(href) nav_links.append((text, href)) return nav_links[:limit]
额外补充
- 如果上述方法仍被重定向,可直接将
validated_url替换为美国版站点的直链:https://www.nike.com/us/en_us,跳过自动重定向逻辑 - 针对其他站点时,需手动查看该站点的地区锁定Cookie名称和值,再对应修改代码中的Cookie参数
内容的提问来源于stack exchange,提问作者adrCoder
相关产品推荐
相关产品推荐

