使用Python Requests和Selenium提交LinkedIn表单数据遇403错误
问题
尝试用Python自动化更新LinkedIn个人资料,已通过Selenium成功登录并进入个人主页,提取了会话Cookies和CSRF Token,用requests提交「关于」板块表单时始终返回403 Forbidden错误。
操作步骤:
- Selenium登录:成功登录LinkedIn并进入个人主页
- 提取Cookies与CSRF Token:从Selenium会话提取Cookies,从主页HTML获取CSRF Token
- Requests提交表单:携带Cookies和CSRF Token提交「关于」板块表单数据
用户代码
from selenium import webdriver from selenium.webdriver.chrome.service import Service from webdriver_manager.chrome import ChromeDriverManager from selenium.webdriver.common.by import By from selenium.webdriver.common.keys import Keys from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import requests import json import re username = 'my_username' password = 'my_password' options = webdriver.ChromeOptions() # options.add_argument('--headless') # for headless mode driver = webdriver.Chrome(service=Service(ChromeDriverManager().install())) def get_cookies(driver): return {cookie['name']: cookie['value'] for cookie in driver.get_cookies()} def extract_csrf_token(html): match = re.search(r'name="csrfToken" content="(.*?)"', html) return match.group(1) if match else None def about(session, link, val, csrf_token, headers): summary_field_name = 'gai-text-form-component-profileEditFormElement-SUMMARY-profile-ACoAAE9YpSsBDmNZhmxG3ss87KP-i8rqiD6k-gM-summary' data = { summary_field_name: val, 'save_button_name': 'Save', 'csrfToken': csrf_token } response = session.post(link, data=data, headers=headers) if response.status_code == 200: print("Form submitted successfully!") else: print(f"Failed to submit form: {response.status_code}") print(response.text) try: driver.get('https://www.linkedin.com/login') username_field = driver.find_element(By.ID, 'username') username_field.send_keys(username) password_field = driver.find_element(By.ID, 'password') password_field.send_keys(password) password_field.send_keys(Keys.RETURN) WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.XPATH, '//a[@class="ember-view block"]')) ).click() profile_page = driver.current_url driver.get(profile_page) csrf_token = extract_csrf_token(driver.page_source) base = "https://www.linkedin.com/in/timmy-scrown-a6399b311/edit/forms/{}/new/?profileFormEntryPoint=PROFILE_COMPLETION_HUB" cookies = driver.get_cookies() session = requests.Session() for cookie in cookies: session.cookies.set(cookie['name'], cookie['value']) headers = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36', 'Content-Type': 'application/x-www-form-urlencoded', 'Referer': profile_page, 'X-CSRF-Token': csrf_token, 'X-Requested-With': 'XMLHttpRequest' } sections = { "about": "summary", "education": "education", "positions": "position", "career_break": "career-break", "skills": "skills", "certifications": "certification", "projects": "project", "courses": "course", "volunteer_experience": "volunteer-experience", "publications": "publication", "patents": "patent", "honors_awards": "honor", "languages": "language", "organizations": "organization", "test_scores": "test-score", "causes": "causes" } functions = { "about": about, "education": education, "positions": positions, "career_break": career_break, "skills": skills, "certifications": certifications, "projects": projects, "courses": courses, "volunteer_experience": volunteer_experience, "publications": publications, "patents": patents, "honors_awards": honors_awards, "languages": languages, "organizations": organizations, "test_scores": test_scores, "causes": causes } with open('core.json', 'r') as file: data = json.load(file) for key, value in data.items(): if key in functions: link = base.format(sections[key]) functions[key](session, link, value, csrf_token, headers) print(link, key, value) finally: driver.quit()
core.json内容
{ "about": "Experienced software engineer with a passion for developing innovative programs.", "education": [ { "school": "University of Technology", "degree": "Bachelor of Science in Computer Science", "field_of_study": "Computer Science", "start_year": "2016", "end_year": "2020" } ], "positions": [ { "title": "Senior Developer", "company": "TechCorp", "location": "New York, NY", "start_month": "January", "start_year": "2021", "end_month": "Present", "description": "Leading a team of developers to create innovative software solutions." } ], "skills": ["Python", "Java", "JavaScript"], "certifications": [ { "name": "Certified Java Developer", "issuing_organization": "Oracle", "issue_date": "2021-06" } ] }
错误信息
Failed to submit form: 403 <!DOCTYPE html> <html lang="en"> <head> <title>Unauthorized</title> <style> html, body, pre { margin: 0; padding: 0; font-family: Monaco, 'Lucida Console', monospace; background: #ECECEC; } h1 { margin: 0; background: #333; padding: 20px 45px; color: #fff; text-shadow: 1px 1px 1px rgba(0,0,0,.3); border-bottom: 1px solid #111; font-size: 28px; } p#detail { margin: 0; padding: 15px 45px; background: #888; border-top: 4px solid #666; color: #111; text-shadow: 1px 1px 1px rgba(255,255,255,.3); font-size: 14px; border-bottom: 1px solid #333; } </style> </head> <body> <h1>Unauthorized</h1> <p id="detail"> You must be authenticated to access this page. </p> </body> </html>
解决方案
1. 修复Cookie传递逻辑
Selenium提取的Cookie需要完整同步到requests会话,尤其是核心认证Cookie li_at。替换原Cookie设置代码,用字典直接更新更可靠:
# 替换原Cookie循环设置代码 cookies_dict = {cookie['name']: cookie['value'] for cookie in driver.get_cookies()} session.cookies.update(cookies_dict)
2. 准确提取CSRF Token
避免用正则匹配HTML源码,直接通过Selenium定位meta标签获取Token:
def extract_csrf_token(driver): meta_tag = WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.XPATH, '//meta[@name="csrfToken"]')) ) return meta_tag.get_attribute('content')
3. 补充关键请求头字段
LinkedIn对请求头校验严格,需补充Origin、匹配真实浏览器的User-Agent和Accept字段:
# 获取浏览器真实User-Agent user_agent = driver.execute_script('return navigator.userAgent') headers = { 'User-Agent': user_agent, 'Content-Type': 'application/x-www-form-urlencoded', 'Referer': profile_page, 'X-CSRF-Token': csrf_token, 'X-Requested-With': 'XMLHttpRequest', 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8', 'Origin': 'https://www.linkedin.com' }
4. 动态获取表单字段名
代码中固定的summary_field_name包含用户ID,会动态变化。先通过Selenium进入编辑页面提取真实字段名:
# 进入「关于」编辑页面 driver.get(base.format("summary")) summary_field = WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.NAME, lambda x: x.startswith('gai-text-form-component-profileEditFormElement-SUMMARY'))) ) summary_field_name = summary_field.get_attribute('name')
5. 验证请求URL正确性
硬编码的URL模板可能失效,建议先通过Selenium进入编辑页面,获取实际的表单提交地址:
edit_page_url = base.format("summary") driver.get(edit_page_url) # 获取表单的action属性 form_action = driver.find_element(By.TAG_NAME, 'form').get_attribute('action')
用该地址作为requests的提交URL。
内容的提问来源于stack exchange,提问作者Achilles
相关产品推荐
相关产品推荐

