You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用rvest从加拿大法律网页提取物种时的节点定位问题

问题

我编写了一段R代码,试图从加拿大法律网页提取物种信息,但无法通过html_nodes定位到各section。执行以下代码时返回{xml_nodeset (0)}:

section <- div_content %>% html_nodes(xpath = paste0("//h2[contains(text(), '", header, "')]/following-sibling::div[contains(@class, 'ProvisionList')]"))

尝试添加<br>标签匹配文本仍未成功。核心需求:

  • 提取id为425426的div中的数据
  • 获取scheduleLabel下的scheduleTitleText文本
  • 添加SchedHeadL1(物种所在章节标题)、BilingualGroupTitleText(动植物分组)列
  • 生成包含法语名、英文名、拉丁名的物种嵌套列表

完整代码如下:

library(rvest)
library(dplyr)
library(stringr)

# URL of the webpage
url <- "https://laws.justice.gc.ca/fra/lois/S-15.3/TexteComplet.html"

# Read the webpage content
webpage <- read_html(url)

# Extract the div with id "425426"
div_content <- webpage %>% html_node("#425426")

# Extract the header h2 with class "scheduleTitleText" from the class "scheduleLabel" and id "h-425427"
schedule_label <- div_content %>% html_node("h2.scheduleLabel#h-425427") %>% html_text()

# Extract all h2 headers with class "SchedHeadL1"
headers <- div_content %>% html_nodes("h2.SchedHeadL1") %>% html_text()


# Use str_extract to extract the "PARTIE #" part
partie_numbers <- str_extract(headers, "PARTIE \\d+")

# Use str_remove to remove the "PARTIE #" part from the original strings
descriptions <- str_remove(headers, "PARTIE \\d+")

# Combine into a data frame
result <- data.frame(Partie = partie_numbers, Description = descriptions, stringsAsFactors = FALSE)

headers_prep = result |> 
  unite(pd, Partie, Description, sep = "<br>") |> pull(pd)

# Initialize lists to store the extracted data
group_titles <- list()
item_first <- list()
item_second <- list()
scientific_names <- list()
latin_names <- list()

# Loop through each header to extract the associated content
for (header in headers) {
  # Extract the section associated with the current header
  section <- div_content %>% html_nodes(xpath = paste0("//h2[contains(text(), '", header, "')]/following-sibling::div[contains(@class, 'ProvisionList')]"))
  
  # Extract BilingualGroupTitleText within the section
  group_title <- section %>% html_nodes(".BilingualGroupTitleText") %>% html_text()
  group_titles <- c(group_titles, group_title)
  
  # Extract BilingualItemFirst within the section
  item_first_section <- section %>% html_nodes(".BilingualItemFirst") %>% html_text()
  item_first <- c(item_first, item_first_section)
  
  # Extract BilingualItemSecond within the section
  item_second_section <- section %>% html_nodes(".BilingualItemSecond") %>% html_text()
  item_second <- c(item_second, item_second_section)
  
  # Extract otherLang (scientific names) within the section
  scientific_name_section <- section %>% html_nodes(".otherLang") %>% html_text()
  scientific_names <- c(scientific_names, scientific_name_section)
  
  # Extract scientific Latin names from BilingualItemFirst
  latin_name_section <- str_extract(item_first_section, "\\(([^)]+)\\)") %>% str_replace_all("[()]", "")
  latin_names <- c(latin_names, latin_name_section)
}

# Ensure all columns have the same length by repeating the last element if necessary
max_length <- max(length(headers), length(group_titles), length(item_first), length(item_second), length(scientific_names), length(latin_names))

schedule_label <- rep(schedule_label, length.out = max_length)
headers <- rep(headers, length.out = max_length)
group_titles <- rep(group_titles, length.out = max_length)
item_first <- rep(item_first, length.out = max_length)
item_second <- rep(item_second, length.out = max_length)
scientific_names <- rep(scientific_names, length.out = max_length)
latin_names <- rep(latin_names, length.out = max_length)

# Create a data frame
data <- data.frame(
  ScheduleLabel = schedule_label,
  Header = headers,
  GroupTitle = group_titles,
  ItemFirst = item_first,
  ItemSecond = item_second,
  ScientificName = scientific_names,
  LatinName = latin_names,
  stringsAsFactors = FALSE
)

解决方法

问题根源

  1. XPath上下文错误:使用绝对路径//h2会从整个文档根节点查找,而非限定的div_content内部,应该用相对路径.//h2。
  2. 文本匹配失效:headers中的文本包含换行和多余空格,直接用contains(text(), ...)无法精准匹配,需要清理文本或用normalize-space()处理。
  3. 列表长度不匹配:循环中直接拼接列表的方式容易导致数据错位,需按层级分组提取数据。

修正后的代码

library(rvest)
library(dplyr)
library(stringr)

# 读取网页内容
url <- "https://laws.justice.gc.ca/fra/lois/S-15.3/TexteComplet.html"
webpage <- read_html(url)

# 定位目标div
div_content <- webpage %>% html_node("#425426")

# 获取schedule标题
schedule_label <- div_content %>% html_node("h2#h-425427") %>% html_text(trim = TRUE)

# 按章节分组提取数据
sections <- div_content %>% html_nodes(".SchedHeadL1") %>%
  map(function(header_node) {
    # 获取清理后的章节标题
    header_text <- header_node %>% html_text(trim = TRUE)
    
    # 定位当前标题后的第一个ProvisionList(相对路径)
    provision_list <- header_node %>% html_node(xpath = "./following-sibling::div[contains(@class, 'ProvisionList')][1]")
    
    # 提取分组标题
    group_titles <- provision_list %>% html_nodes(".BilingualGroupTitleText") %>% html_text(trim = TRUE)
    
    # 提取所有物种条目
    species_wrappers <- provision_list %>% html_nodes(".BilingualItemWrapper")
    
    # 处理分组与物种的对应关系
    if(length(group_titles) > 0) {
      # 找到每个分组对应的物种起始位置
      group_positions <- map_int(group_titles, function(g) {
        provision_list %>% html_node(paste0(".BilingualGroupTitleText[normalize-space(text())='", g, "']/following-sibling::div[1]")) %>%
          xml_attr("id") %>%
          which(map_chr(species_wrappers, ~.x %>% xml_attr("id")) == .)
      })
      # 拆分物种列表为对应分组
      split_indices <- c(group_positions, length(species_wrappers)+1)
      
      map2(group_titles, seq_along(group_titles), function(title, idx) {
        start <- split_indices[idx]
        end <- split_indices[idx+1]-1
        species_wrappers[start:end] %>%
          map_dfr(function(wrapper) {
            tibble(
              Header = header_text,
              GroupTitle = title,
              FrenchName = wrapper %>% html_node(".BilingualItemFirst") %>% html_text(trim = TRUE),
              EnglishName = wrapper %>% html_node(".BilingualItemSecond") %>% html_text(trim = TRUE),
              LatinName = str_extract(wrapper %>% html_node(".BilingualItemFirst") %>% html_text(trim = TRUE), "\\(([^)]+)\\)") %>% str_remove_all("[()]") %>% str_trim()
            )
          })
      }) %>% bind_rows()
    } else {
      # 无分组时直接提取所有物种
      species_wrappers %>%
        map_dfr(function(wrapper) {
          tibble(
            Header = header_text,
            GroupTitle = NA_character_,
            FrenchName = wrapper %>% html_node(".BilingualItemFirst") %>% html_text(trim = TRUE),
            EnglishName = wrapper %>% html_node(".BilingualItemSecond") %>% html_text(trim = TRUE),
            LatinName = str_extract(wrapper %>% html_node(".BilingualItemFirst") %>% html_text(trim = TRUE), "\\(([^)]+)\\)") %>% str_remove_all("[()]") %>% str_trim()
          )
        })
    }
  }) %>% bind_rows()

# 组装最终数据框
final_data <- sections %>% mutate(ScheduleLabel = schedule_label) %>%
  select(ScheduleLabel, Header, GroupTitle, FrenchName, EnglishName, LatinName)

# 查看结果示例
head(final_data)

关键改进

  • 相对XPath:用./following-sibling::div确保在当前标题的同级元素中查找,限定在目标div内部。
  • 文本清理:html_text(trim = TRUE)去除多余空格和换行,解决文本匹配问题。
  • 层级分组提取:通过map系列函数按章节、分组层级处理数据,避免列表长度错位。
  • 自动关联分组:通过节点位置匹配,自动将分组标题与对应物种绑定,无需手动重复填充。

内容的提问来源于stack exchange,提问作者M. Beausoleil

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.15 13:34:51