Selenium爬取CCI网站多页数据及多进程实现问题求助
CCI网站房地产中介数据爬取问题
爬取需求
- 点击列表中每条带有
Attestation de ...前缀的链接,爬取对应详情页全部字段后返回列表页 - 处理完当前页所有链接后自动翻页,直到没有下一页分页按钮
- 最终完成全区域所有数据的爬取
待解决问题
- 已实现单条链接爬取后返回列表的功能,无法实现整页所有链接遍历、自动翻页到最后一页的逻辑
- 全区域爬取预估耗时数天到数周,需要用multiprocessing提升效率,但无相关实现经验
现有代码
from selenium import webdriver from selenium.common.exceptions import NoSuchElementException from selenium.webdriver.support.wait import WebDriverWait from webdriver_manager.chrome import ChromeDriverManager from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.support.select import Select import time with open('output_test_auvergne.csv', 'w') as file: file.write("business_names; town_pc; region \n") driver = webdriver.Chrome(ChromeDriverManager().install()) # initialise chrome driver driver.get( 'https://www.cci.fr/agent-immobilier?company_name=agences%20immobili%C3%A8res%20&brand_name=&siren=&numero_carte=&code_region=84&city=&code_postal=&person_name=&state_recherche=1&name_region=AUVERGNE-RHONE-ALPES&__cf_chl_captcha_tk__=pmd_74hrnIdUsNgz2TJJCM33kpVYFY4hRG420hx18Sk1ITA-1634596843-0-gqNtZGzNBBCjcnBszQil') driver.maximize_window() time.sleep(3) cookie = driver.find_element_by_xpath("//*[@id='tarteaucitronPersonalize2']") try: cookie.click() finally: pass visited_pages = ['1'] with open('output_test_auvergne.csv', 'w') as file: while True: table_rows = driver.find_elements_by_css_selector('table tr') business_name = None for row in table_rows: #loop for identifying elements try: business_name = row.find_element_by_css_selector('.titre_entreprise').text continue except: pass try: if row.get_attribute('class') == 'lien-fiche': for index, td in enumerate(row.find_elements_by_css_selector('td')): if index == 0: attestation_name = td.text if index == 1: city = td.text if index == 2: region = td.text except: pass link2 = row.find_element_by_xpath('//*[@id="main-content"]/div[5]/table/tbody/tr/td/a') for links2 in link2: link2.click() num_attestation = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[1]/strong').text date_delivrance = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[2]/div[1]/strong').text delivre_par = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[2]/div[1]/div/strong').text date_disponibilite = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[2]/div[2]/strong').text president = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[2]/div[3]/strong').text fonction = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[3]/div[3]/strong').text etendue_pouvoir = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[3]/div[4]/strong').text num_carte_pro = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[1]/div[1]/strong[1]').text dispo_carte_pro = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[1]/div[1]/strong[2]').text date_delivrance_carte_pro = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[1]/div[2]/strong').text organisme_delivrance = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[1]/div[3]/strong').text titulaire_carte = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[2]/div[2]/strong[1]').text forme_juridique = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[2]/div[2]/strong[2]').text adresse = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[2]/div[4]/strong/div[1]').text nom_commercial = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[2]/div[5]/strong[1]').text num_identification = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[2]/div[5]/strong[2]').text representant_legal_nom = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[5]/div/strong[1]').text representant_legal_prnom = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[5]/div/strong[2]').text garantie_fonciere = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[6]/div[2]/div[3]/strong').text detention_fonds = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[6]/div[2]/div[2]').text assurance_nom = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[7]/div[3]/strong').text adresse_assurance = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[7]/div[4]/strong').text cp_ville_assurance = driver.find_element_by_xpath('//*[@id="agent-immobilier__document"]/div[2]/div[7]/div[6]/strong').text driver.back(); file.write(business_name + ";" + attestation_name + ';' + city + ";" + region + ";" + num_attestation + ";" + date_delivrance + ";" + delivre_par + ";" + date_disponibilite + ";" + (president) + ";" + fonction + ";" + etendue_pouvoir + ";" + num_carte_pro + ";" + dispo_carte_pro + ";" + date_delivrance_carte_pro + ";" + organisme_delivrance + ";" + titulaire_carte + ";" + forme_juridique + ";" + adresse + ";" + nom_commercial + ";" + num_identification + ";" + representant_legal_nom + ";" + representant_legal_prnom + ";" + garantie_fonciere + ";" + detention_fonds + ";" + assurance_nom + ";" + adresse_assurance + ";" + cp_ville_assurance + "\n") pass number_of_pages = driver.find_element_by_css_selector("a[rel='next']").click() current_page = driver.find_element_by_css_selector('a[title="Page courante"]').text if current_page in visited_pages: break visited_pages =+ current_page file.close() driver.close()
解决方案
现有代码问题修复
原有代码核心问题如下:
- 链接查找逻辑错误:使用绝对xpath查找
//*[@id="main-content"]/div[5]/table/tbody/tr/td/a会每次匹配到页面第一个链接,不会对应到当前遍历的行,需改为从当前row元素下查找相对路径 - 翻页逻辑位置错误:翻页代码写在单条链接处理的循环内部,处理完第一条链接就会触发翻页,无法遍历当前页所有链接
- 变量操作错误:
visited_pages =+ current_page写法错误,应该用列表append方法添加已访问页码 - 资源提前释放:
file.close()和driver.close()写在循环内部,第一次循环结束就会关闭文件和浏览器,无法继续后续爬取 - 无异常兜底:详情页字段爬取没有异常捕获,任意字段缺失就会导致程序崩溃
以下是修复后的单进程可运行代码:
from selenium import webdriver from selenium.common.exceptions import NoSuchElementException, ElementClickInterceptedException from webdriver_manager.chrome import ChromeDriverManager import time import csv # 初始化csv文件 csv_header = ["business_names", "attestation_name", "city", "region", "num_attestation", "date_delivrance", "delivre_par", "date_disponibilite", "president", "fonction", "etendue_pouvoir", "num_carte_pro", "dispo_carte_pro", "date_delivrance_carte_pro", "organisme_delivrance", "titulaire_carte", "forme_juridique", "adresse", "nom_commercial", "num_identification", "representant_legal_nom", "representant_legal_prnom", "garantie_fonciere", "detention_fonds", "assurance_nom", "adresse_assurance", "cp_ville_assurance"] with open('output_test_auvergne.csv', 'w', encoding='utf-8', newline='') as f: writer = csv.writer(f, delimiter=';') writer.writerow(csv_header) # 初始化浏览器 options = webdriver.ChromeOptions() options.add_argument("--start-maximized") # 可选无头模式提升效率 # options.add_argument("--headless=new") driver = webdriver.Chrome(ChromeDriverManager().install(), options=options) base_url = 'https://www.cci.fr/agent-immobilier?company_name=agences%20immobili%C3%A8res%20&brand_name=&siren=&numero_carte=&code_region=84&city=&code_postal=&person_name=&state_recherche=1&name_region=AUVERGNE-RHONE-ALPES&page=0' driver.get(base_url) time.sleep(3) # 处理cookie try: driver.find_element_by_xpath("//*[@id='tarteaucitronPersonalize2']").click() time.sleep(1) except: pass visited_pages = set() def safe_get_text(xpath): """字段爬取兜底方法,不存在返回空字符串""" try: return driver.find_element_by_xpath(xpath).text.strip() except: return "" while True: # 获取当前页码,判断是否已访问过 try: current_page = driver.find_element_by_css_selector('a[title="Page courante"]').text except: current_page = str(len(visited_pages)+1) if current_page in visited_pages: break visited_pages.add(current_page) print(f"正在处理第{current_page}页") # 先提取当前页所有链接和基础信息,避免返回后元素失效 page_items = [] table_rows = driver.find_elements_by_css_selector('table tr') business_name = None for row in table_rows: # 提取企业名称 try: business_name = row.find_element_by_css_selector('.titre_entreprise').text.strip() continue except: pass # 提取链接行信息 try: if row.get_attribute('class') == 'lien-fiche': tds = row.find_elements_by_css_selector('td') attestation_name = tds[0].text.strip() if len(tds)>0 else "" city = tds[1].text.strip() if len(tds)>1 else "" region = tds[2].text.strip() if len(tds)>2 else "" detail_url = tds[0].find_element_by_tag_name('a').get_attribute('href') page_items.append({ "business_name": business_name, "attestation_name": attestation_name, "city": city, "region": region, "detail_url": detail_url }) except Exception as e: continue # 遍历处理当前页所有详情 for item in page_items: driver.get(item['detail_url']) time.sleep(2) # 爬取详情字段 item['num_attestation'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[1]/strong') item['date_delivrance'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[2]/div[1]/strong') item['delivre_par'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[2]/div[1]/div/strong') item['date_disponibilite'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[2]/div[2]/strong') item['president'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[2]/div[3]/strong') item['fonction'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[3]/div[3]/strong') item['etendue_pouvoir'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[3]/div[4]/strong') item['num_carte_pro'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[1]/div[1]/strong[1]') item['dispo_carte_pro'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[1]/div[1]/strong[2]') item['date_delivrance_carte_pro'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[1]/div[2]/strong') item['organisme_delivrance'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[1]/div[3]/strong') item['titulaire_carte'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[2]/div[2]/strong[1]') item['forme_juridique'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[4]/div[2]/div[2]/strong[2]') item['adresse'] = safe_get_text('//*[@id="agent-immobilier__document"]/div[2]/div[4
相关产品推荐
相关产品推荐

