使用rvest包进行网页抓取时如何循环遍历多个县FIPS代码
遍历抓取科罗拉多州各县农业补贴数据实现方案
首先将单县抓取逻辑封装为可复用函数,再遍历目标FIPS列表执行,支持两种输出形式。
1. 导入依赖并定义抓取函数
library(rvest) library(dplyr) library(tidyr) # 定义单县补贴数据抓取函数,入参为县FIPS代码 get_county_subsidy <- function(fips_code) { # 构造请求链接 link <- paste0("https://farm.ewg.org/regionsummary.php?fips=", fips_code) page <- read_html(link) # 提取各字段 year <- page %>% html_nodes("tr~ tr+ tr td:nth-child(1)") %>% html_text() subs <- page %>% html_nodes("td:nth-child(3)") %>% html_text() subsidy_data <- data.frame(subs) subs <- data.frame(do.call("rbind", strsplit(as.character(subsidy_data$subs), "$", fixed = TRUE))) sub_data <- cbind(year, subs) sub_data <- sub_data[-c(28),] # 剔除无效行,若不同县结构不同可调整该行逻辑 cons_sub_rec <- page %>% html_nodes("td~ td+ td small:nth-child(1) em") %>% html_text() cons_sub_rec <- cons_sub_rec[-c(28)] dis_sub_rec <- page %>% html_nodes("small:nth-child(3) em") %>% html_text() comm_sub_rec <- page %>% html_nodes("small:nth-child(5) em") %>% html_text() ins_sub_rec <- page %>% html_nodes("small:nth-child(7) em") %>% html_text() # 合并字段并补充FIPS标识 sub_data <- cbind(year, subs, cons_sub_rec, dis_sub_rec, comm_sub_rec, ins_sub_rec) sub_data$fips <- fips_code # 每次请求后暂停2秒,避免触发反爬 Sys.sleep(2) return(sub_data) }
2. 输出方案A:每个县生成独立CSV文件
# 定义目标FIPS列表,后续可自行补充其他县代码 fips_list <- c("08003", "08005", "08007", "08009", "08011") # 定义文件保存路径,替换为你本地的实际路径 save_path <- "your/local/filepath/" # 遍历FIPS列表执行抓取并保存 for (fips in fips_list) { county_data <- get_county_subsidy(fips) write.csv(county_data, paste0(save_path, "ewg_sub_", fips, ".csv"), row.names = TRUE) # 可取消下一行注释打印进度:print(paste0("已完成FIPS=", fips, "的抓取")) }
3. 输出方案B:所有数据汇总后统一保存为CSV
# 定义目标FIPS列表,后续可自行补充其他县代码 fips_list <- c("08003", "08005", "08007", "08009", "08011") # 定义文件保存路径,替换为你本地的实际路径 save_path <- "your/local/filepath/" # 遍历抓取所有县数据后合并为总数据框 all_data <- lapply(fips_list, get_county_subsidy) %>% bind_rows() # 保存汇总数据 write.csv(all_data, paste0(save_path, "ewg_sub_all_colorado.csv"), row.names = TRUE)
可选优化建议
- 可调整
Sys.sleep()的等待秒数,平衡抓取速度和反爬风险 - 可在函数中加入
tryCatch错误捕获逻辑,遇到单个县抓取失败时不会中断整体任务 - 若后续发现不同县的有效行数不一致,可调整剔除无效行的逻辑,改为按字段长度匹配筛选
内容的提问来源于stack exchange,提问作者JohnnyJohnson
相关产品推荐
相关产品推荐

