You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于BeautifulSoup实现多URL多页面网络爬虫的适配问题求助

多URL批量爬取修改方案

核心调整逻辑

只需要在原有单URL爬取逻辑外层套一层URL列表遍历逻辑,每次处理新的分类URL时重置页码变量即可,原有页面解析、分页判断的逻辑无需改动。本次修改还优化了URL列表的生成方式,不用手动写5条重复的URL规则,后续要调整分类范围直接修改range参数即可。

完整可运行代码

import requests
from bs4 import BeautifulSoup as BS
from datetime import datetime
import pandas as pd
import re
import csv

today = datetime.today().strftime('%y%m%d ')

def main():
    # 所有分类的数据都存在同一个列表里,所以列表定义放在URL循环外层
    name = []
    date = []
    address = []
    district = []
    city = []
    price = []
    area_sqm = []
    rooms = []
    floor = []
    commission_year = []
    building_floors = []
    garage = []
    balcony = []
    windows = []
    window_type = []
    floor_type = []
    door_type = []
    leasing = []
    description = []
    link = []

    BASE = 'https://www.unegui.mn'
    # 列表推导式生成1-5对应的分类URL,替代硬编码的5条规则
    URLS = [f'{BASE}/l-hdlh/l-hdlh-zarna/oron-suuts-zarna/{i}-r/?page=' for i in range(1,6)]
    COLUMNS=['Name','Date','Address','District','City','Price','Area_sqm','Rooms','Floor','Commission_year',
             'Building_floors','Garage', 'Balcony','Windows','Window_type','Floor_type','door_type','Leasing','Description','Link']
    
    with requests.Session() as session:
        # 外层循环遍历所有分类URL
        for current_url in URLS:
            print(f'开始处理分类URL: {current_url}')
            # 每个分类的页码从0开始重置,避免继承上一个分类的页码数值
            page = 0
            while True:
                (r := session.get(f'{current_url}{page+1}')).raise_for_status()
                m = re.search('.*page=(\d+)$', r.url)
                if m and int(m.group(1)) == page:
                    break
                page += 1
                print(f'正在爬取第 {page} 页')
                soup = BS(r.text, 'lxml')
                for tag in soup.findAll('div', class_='list-announcement-block'):
                    _name = tag.find('a', attrs={'itemprop': 'name'})
                    name.append(_name.get('content', 'N/A'))
                    if (_link := _name.get('href', None)):
                        link.append(f'{BASE}{_link}')
                        (_r := session.get(link[-1])).raise_for_status()
                        _spanlist = BS(_r.text, 'lxml').find_all('span', class_='value-chars')
                        floor_type.append(_spanlist[0].get_text().strip())
                        balcony.append(_spanlist[1].get_text().strip())
                        garage.append(_spanlist[2].get_text().strip())
                        window_type.append(_spanlist[3].get_text().strip())
                        door_type.append(_spanlist[4].get_text().strip())   
                        windows.append(_spanlist[5].get_text().strip())
                        
                        _alist = BS(_r.text, 'lxml').find_all('a', class_='value-chars')
                        commission_year.append(_alist[0].get_text().strip())
                        building_floors.append(_alist[1].get_text().strip())
                        area_sqm.append(_alist[2].get_text().strip())
                        floor.append(_alist[3].get_text().strip())
                        leasing.append(_alist[4].get_text().strip())
                        district.append(_alist[5].get_text().strip())
                        address.append(_alist[6].get_text().strip())
                        
                    rooms.append(tag.find('div', attrs={'announcement-block__breadcrumbs'}).get_text().split('»')[1].strip())
                    description.append(tag.find('div', class_='announcement-block__description').get_text().strip())
                    date.append(tag.find('div', class_='announcement-block__date').get_text().split(',')[0].strip())
                    city.append((tag.find('meta', attrs={'itemprop': 'areaServed'})).get('content'))
                    if (_price := tag.find('meta', attrs={'itemprop': 'price'})) is None:
                        _price = tag.find('div', class_='announcement-block__price _premium')
                    price.append(_price.get_text().strip() if _price else 'N/A')
    df = pd.DataFrame(zip(name, date, address, district, city, 
                              price, area_sqm, rooms, floor, commission_year,
                              building_floors, garage, balcony, windows, window_type,
                              floor_type, door_type, leasing, description, link), columns=COLUMNS)
    return(df)

if __name__ == '__main__':
    df = main()
    df.to_csv(f'{today}HPD.csv', encoding='cp1251', errors='ignore', index=False)

内容的提问来源于stack exchange,提问作者WX1505

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.28 16:45:04