You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

从prisonstudies.org爬取监狱人口数据并写入CSV的技术求助

爬取监狱人口数据并导出CSV的解决方案

我想要从指定网站爬取各国历年监狱人口及监狱人口率数据,数据在各国专属页面中呈现,最终要导出包含Country name、Year、Total Prison Population、Prison Population Date四列的CSV文件。预期输出示例如下:

Country name,Year,Total Prison Population,Prison Population Date
Algeria,2000,33.992,108
Algeria,2003,39.806,122
Algeria,2004,44.231,134

我已经完成了部分代码,但对后续操作存在困惑,现有代码如下:

import requests
import elementpath
from xml.etree import ElementTree as ET
from bs4 import BeautifulSoup
from os.path import basename, dirname,abspath

url = "https://www.prisonstudies.org/world-prison-brief-data"

def parseCountries(url):
    r = requests.get(url)
    soup = ET.parse(r.text, 'lxml')
    regions = soup.findAll('div', {'class' : 'item-list'})
    out = {}
    for reg in regions:
        items = reg.findAll('a', href=True)
        for i in items:
            if i.text.strip() != '':
                out[i.text.strip()] = i['href']
    return(out)

def yearTableParser(countryUrl, countryName):
    r = requests.get(countryUrl)
    soup = BeautifulSoup(r.text, 'lxml')
    yearTab = soup.find('table', {'id':'views-aggregator-datatable'})
    out = []
    if yearTab is not None:
        rows = yearTab.findAll('tr')
        for r in rows:
            dat = r.findAll('td')
            if dat != []:
                out.append([countryName, dat[0].text.strip(),dat[1].text.replace('c','').replace(',','.').strip(),dat[2].text.replace('c','').replace(',','.').strip()])
    return(out) 

问题修正与后续实现步骤

1. 修复国家列表解析函数

原代码里用处理XML的ElementTree解析HTML是错误的,换成BeautifulSoup才能正确解析网页内容,同时要拼接完整的国家页面URL:

def parseCountries(url):
    r = requests.get(url)
    soup = BeautifulSoup(r.text, 'lxml')
    regions = soup.findAll('div', {'class': 'item-list'})
    out = {}
    for reg in regions:
        items = reg.findAll('a', href=True)
        for i in items:
            country_name = i.text.strip()
            if country_name:
                out[country_name] = f"https://www.prisonstudies.org{i['href']}"
    return out

2. 添加主逻辑:遍历爬取并写入CSV

用csv模块处理文件写入,确保表头和数据格式符合要求:

import csv

def main():
    # 获取所有国家的名称和对应页面URL
    countries = parseCountries(url)
    csv_file = "prison_population_data.csv"
    
    # 打开CSV文件准备写入
    with open(csv_file, 'w', newline='', encoding='utf-8') as f:
        # 定义表头
        headers = ['Country name', 'Year', 'Total Prison Population', 'Prison Population Date']
        writer = csv.DictWriter(f, fieldnames=headers)
        
        # 写入表头
        writer.writeheader()
        
        # 逐个爬取国家数据
        for name, url in countries.items():
            print(f"正在爬取:{name}")
            country_data = yearTableParser(url, name)
            # 逐条写入CSV
            for row in country_data:
                writer.writerow({
                    'Country name': row[0],
                    'Year': row[1],
                    'Total Prison Population': row[2],
                    'Prison Population Date': row[3]
                })
    print(f"所有数据已保存到 {csv_file}")

3. 完整可运行代码

整合所有部分,还可以添加延迟避免给服务器造成压力:

import requests
from bs4 import BeautifulSoup
import csv
import time

url = "https://www.prisonstudies.org/world-prison-brief-data"

def parseCountries(url):
    r = requests.get(url)
    soup = BeautifulSoup(r.text, 'lxml')
    regions = soup.findAll('div', {'class': 'item-list'})
    out = {}
    for reg in regions:
        items = reg.findAll('a', href=True)
        for i in items:
            country_name = i.text.strip()
            if country_name:
                out[country_name] = f"https://www.prisonstudies.org{i['href']}"
    return out

def yearTableParser(countryUrl, countryName):
    r = requests.get(countryUrl)
    soup = BeautifulSoup(r.text, 'lxml')
    yearTab = soup.find('table', {'id':'views-aggregator-datatable'})
    out = []
    if yearTab is not None:
        rows = yearTab.findAll('tr')
        for r in rows:
            dat = r.findAll('td')
            if dat != []:
                year = dat[0].text.strip()
                total_pop = dat[1].text.replace('c','').replace(',','.').strip()
                rate = dat[2].text.replace('c','').replace(',','.').strip()
                out.append([countryName, year, total_pop, rate])
    return out 

def main():
    countries = parseCountries(url)
    csv_file = "prison_population_data.csv"
    
    with open(csv_file, 'w', newline='', encoding='utf-8') as f:
        headers = ['Country name', 'Year', 'Total Prison Population', 'Prison Population Date']
        writer = csv.DictWriter(f, fieldnames=headers)
        
        writer.writeheader()
        
        for name, url in countries.items():
            print(f"爬取中:{name}")
            country_data = yearTableParser(url, name)
            for row in country_data:
                writer.writerow({
                    'Country name': row[0],
                    'Year': row[1],
                    'Total Prison Population': row[2],
                    'Prison Population Date': row[3]
                })
            # 每次爬取后延迟1秒,避免请求过于频繁
            time.sleep(1)
    print(f"数据已成功保存至 {csv_file}")

if __name__ == "__main__":
    main()

额外提示

  • 可以添加异常捕获代码,比如用try-except包裹请求逻辑,避免单个国家爬取失败导致程序崩溃
  • 若遇到反爬限制,可以考虑添加请求头(比如User-Agent)模拟浏览器访问

内容的提问来源于stack exchange,提问作者Pelin Kasap

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.20 16:42:36