为何Twitter登录UI脚本仅在开发环境生效,服务器端无法运行?
Twitter无头爬虫服务器端登录故障排查
问题背景
近期Twitter实施重大变更:未登录用户无法查看用户时间线,导致原有爬虫全部失效。现有登录脚本在本地无头模式环境下可正常运行,但部署到服务器端后无法工作,核心故障为点击Next按钮后无法找到密码输入框。已尝试添加sleep语句排查是否因执行过快导致问题,未解决。
环境限制
服务器为IaaS架构,因技术债务使用旧版本Selenium;本地开发环境使用最新版Firefox。
故障代码
TWITTER_URL_BASE = "https://twitter.com/" SELENIUM_INSTANCE_WAIT = 1 @classmethod def _get_driver(cls): driver = None firefox_options = FirefoxOptions() firefox_options.headless = True firefox_options.add_argument("width=1920") firefox_options.add_argument("height=1080") firefox_options.add_argument("window-size=1920,1080") firefox_options.add_argument("disable-gpu") # https://stackoverflow.com/questions/24653127/selenium-error-no-display-specified # export MOZ_HEADLESS=1 firefox_options.binary_location = "/usr/bin/firefox" # firefox_options.set_preference("extensions.enabledScopes", 0) # firefox_options.set_preference("gfx.webrender.all", False) # firefox_options.set_preference("layers.acceleration.disabled", True) firefox_binary = FirefoxBinary("/usr/bin/firefox") firefox_profile = FirefoxProfile() firefox_options.binary = "/usr/bin/firefox" # firefox_binary firefox_options.profile = firefox_profile capabilities = DesiredCapabilities.FIREFOX.copy() capabilities["pageLoadStrategy"] = "normal" firefox_options._caps = capabilities try: driver = webdriver.Firefox( firefox_profile=firefox_profile, firefox_binary=firefox_binary, options=firefox_options, desired_capabilities=capabilities, ) except Exception as e: cls.log_response("_get_driver", 500, "Crash: {}".format(e)) cls.log_response("_get_driver", 500, traceback.format_exc()) return driver def _login_scraper_user(cls, driver, scraper_account): driver.implicitly_wait(5) driver.get(TWITTER_URL_BASE) WebDriverWait(driver, 10).until( lambda dr: dr.execute_script("return document.readyState") == "complete" ) time.sleep(SELENIUM_INSTANCE_WAIT) username_inputs = driver.find_elements_by_css_selector("input[name='text']") if not username_inputs: return False username_input_parent = ( username_inputs[0].find_element_by_xpath("..").find_element_by_xpath("..") ) username_input_parent.click() time.sleep(SELENIUM_INSTANCE_WAIT) username_inputs[0].click() time.sleep(SELENIUM_INSTANCE_WAIT) username_inputs[0].send_keys(scraper_account["username"]) time.sleep(SELENIUM_INSTANCE_WAIT) next_buttons = driver.find_elements_by_xpath('//span[text()="Next"]') if not next_buttons: return False next_buttons[0].click() time.sleep(SELENIUM_INSTANCE_WAIT) password_inputs = driver.find_elements_by_css_selector("input[name='password']") if not password_inputs: return False password_input_parent = ( password_inputs[0].find_element_by_xpath("..").find_element_by_xpath("..") ) password_input_parent.click() time.sleep(SELENIUM_INSTANCE_WAIT) password_inputs[0].click() time.sleep(SELENIUM_INSTANCE_WAIT) password_inputs[0].send_keys(scraper_account["password"]) time.sleep(SELENIUM_INSTANCE_WAIT) login_buttons = driver.find_elements_by_xpath('//span[text()="Log in"]') if not login_buttons: return False login_buttons[0].click() time.sleep(SELENIUM_INSTANCE_WAIT) if driver.find_elements_by_xpath( '//span[text()="Boost your account security"]' ): close_buttons = driver.find_elements_by_css_selector( "div[data-testid='app-bar-close']" ) if not close_buttons: return False close_buttons[0].click() driver.implicitly_wait(0) return True
补充说明
API定价调整引发爬虫对抗,反爬虫措施已波及普通用户(此趋势已于5月1日预测)。
内容的提问来源于stack exchange,提问作者Csaba Toth
相关产品推荐
相关产品推荐

