技术协助请求:实现GDELT数据自动处理及解决gdeltr2包问题
解决GDELT事件数据自动下载、解压与合并的方案
直接操作GDELT原始文件可规避gdeltr2包的变量缺失问题,以下是基于Python的完整实现方案:
核心流程
- 抓取目标时段的GDELT事件数据压缩包链接
- 批量下载文件至本地目录
- 自动解压并读取CSV(手动补全官方完整变量表头)
- 合并所有数据并导出最终文件
完整代码实现
import os import requests from bs4 import BeautifulSoup import zipfile import pandas as pd from datetime import datetime, timedelta # 配置参数 start_date = "2024-01-01" # 起始日期(按需修改) end_date = "2024-01-03" # 结束日期(按需修改) save_dir = "./gdelt_data" # 本地文件保存目录 final_output = "./gdelt_merged_data.csv" # 合并后数据输出路径 # GDELT事件数据索引页地址 gdelt_events_index = "http://data.gdeltproject.org/events/index.html" # GDELT事件数据完整变量表头(官方定义全量字段) gdelt_full_headers = [ "GLOBALEVENTID", "SQLDATE", "MonthYear", "Year", "FractionDate", "Actor1Code", "Actor1Name", "Actor1CountryCode", "Actor1KnownGroupCode", "Actor1EthnicCode", "Actor1Religion1Code", "Actor1Religion2Code", "Actor1Type1Code", "Actor1Type2Code", "Actor1Type3Code", "Actor2Code", "Actor2Name", "Actor2CountryCode", "Actor2KnownGroupCode", "Actor2EthnicCode", "Actor2Religion1Code", "Actor2Religion2Code", "Actor2Type1Code", "Actor2Type2Code", "Actor2Type3Code", "IsRootEvent", "EventCode", "EventBaseCode", "EventRootCode", "QuadClass", "GoldsteinScale", "NumMentions", "NumSources", "NumArticles", "AvgTone", "Actor1Geo_Type", "Actor1Geo_FullName", "Actor1Geo_CountryCode", "Actor1Geo_ADM1Code", "Actor1Geo_ADM2Code", "Actor1Geo_Lat", "Actor1Geo_Long", "Actor1Geo_FeatureID", "Actor2Geo_Type", "Actor2Geo_FullName", "Actor2Geo_CountryCode", "Actor2Geo_ADM1Code", "Actor2Geo_ADM2Code", "Actor2Geo_Lat", "Actor2Geo_Long", "Actor2Geo_FeatureID", "ActionGeo_Type", "ActionGeo_FullName", "ActionGeo_CountryCode", "ActionGeo_ADM1Code", "ActionGeo_ADM2Code", "ActionGeo_Lat", "ActionGeo_Long", "ActionGeo_FeatureID", "DATEADDED", "SOURCEURL" ] # 创建本地保存目录 os.makedirs(save_dir, exist_ok=True) # 1. 抓取目标时段的压缩包链接 response = requests.get(gdelt_events_index) soup = BeautifulSoup(response.text, "html.parser") all_links = soup.find_all("a", href=True) # 生成目标日期范围的字符串格式(YYYYMMDD) date_format = "%Y%m%d" start_dt = datetime.strptime(start_date, "%Y-%m-%d") end_dt = datetime.strptime(end_date, "%Y-%m-%d") date_range = [start_dt + timedelta(days=x) for x in range((end_dt - start_dt).days + 1)] target_dates = [dt.strftime(date_format) for dt in date_range] # 筛选符合日期的zip文件链接 target_links = [] for link in all_links: href = link["href"] if href.endswith(".export.CSV.zip"): file_date = href.split(".")[0] if file_date in target_dates: target_links.append(f"http://data.gdeltproject.org/events/{href}") # 2. 批量下载文件 for link in target_links: filename = os.path.join(save_dir, link.split("/")[-1]) if not os.path.exists(filename): print(f"正在下载: {filename}") r = requests.get(link, stream=True) with open(filename, "wb") as f: for chunk in r.iter_content(chunk_size=1024*1024): f.write(chunk) # 3. 解压并合并数据 merged_df = pd.DataFrame() for zip_filename in os.listdir(save_dir): if zip_filename.endswith(".zip"): zip_path = os.path.join(save_dir, zip_filename) with zipfile.ZipFile(zip_path, "r") as zf: csv_filename = zf.namelist()[0] with zf.open(csv_filename) as f: # 读取CSV并绑定全量表头 df = pd.read_csv(f, sep="\t", names=gdelt_full_headers, encoding="utf-8") merged_df = pd.concat([merged_df, df], ignore_index=True) print(f"处理完成: {zip_filename}") # 4. 导出合并后的数据 merged_df.to_csv(final_output, index=False, encoding="utf-8") print(f"合并完成,结果已保存至: {final_output}")
关键提示
- 全量变量保障:手动定义的
gdelt_full_headers包含GDELT事件数据的所有官方字段,彻底避免第三方包的变量遗漏问题 - 大文件优化:若处理跨月/跨年的大时段数据,建议在
pd.read_csv中添加chunksize=100000参数分块读取,防止内存溢出 - 日期匹配逻辑:通过生成目标日期字符串匹配文件名,确保仅下载指定时段的有效数据
- 编码兼容性:强制指定UTF-8编码读取文件,避免中文或特殊字符乱码
内容的提问来源于stack exchange,提问作者Injae Jeon
相关产品推荐
相关产品推荐

