You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

处理academictwitteR返回的嵌套Dataframe时do.call执行报错求助

问题

使用academictwitteR包的get_all_tweets函数提取推文后,得到一个嵌套数据框(部分列本身是数据框),示例数据集如下:

df <- structure(list(author_id = c("18166614", "590204178", "394175206", 
"2581787594", "835979767"), entities = structure(list(hashtags = list(
    structure(list(start = c(91L, 170L, 185L, 194L, 205L), end = c(100L, 
    184L, 193L, 203L, 216L), tag = c("phishing", "cybersecurity", 
    "Vishing", "SMiShing", "PhishProof")), class = "data.frame", row.names = c(NA, 
    5L)), structure(list(start = c(72L, 151L, 166L, 175L, 186L
    ), end = c(81L, 165L, 174L, 184L, 197L), tag = c("phishing", 
    "cybersecurity", "Vishing", "SMiShing", "PhishProof")), class = "data.frame", row.names = c(NA, 
    5L)), structure(list(start = 14L, end = 22L, tag = "Vishing"), class = "data.frame", row.names = 1L), 
    structure(list(start = 0L, end = 8L, tag = "Vishing"), class = "data.frame", row.names = 1L), 
    NULL), urls = list(structure(list(start = 146L, end = 169L, 
    url = "", expanded_url = "", 
    display_url = ""), class = "data.frame", row.names = 1L), 
    structure(list(start = 127L, end = 150L, url = "", 
        expanded_url = " ", display_url = ""), class = "data.frame", row.names = 1L), 
    structure(list(start = 103L, end = 126L, url = "", 
        expanded_url = "https://www.biocatch.com/resources/data-sheets/using-behavioural-biometrics-to-combat-vishing-and-app-fraud", 
        display_url = "biocatch.com/resources/data…"), class = "data.frame", row.names = 1L), 
    structure(list(start = c(89L, 113L), end = c(112L, 136L), 
        url = c("", ""
        ),expanded_url = c("app", 
        "ht/photo/1"
        ), display_url = c("biocatch.com/resources/data…", "pic.twitter.com/LZcmikqlwA"
        ), media_key = c(NA, "3_1090735909463621632")), class = "data.frame", row.names = 1:2), 
    NULL), annotations = list(structure(list(start = c(3L, 206L
), end = c(16L, 215L), probability = c(0.5826, 0.555), type = c("Other", 
"Other"), normalized_text = c("InspiredeLearn", "PhishProof")), class = "data.frame", row.names = 1:2), 
    structure(list(start = 187L, end = 196L, probability = 0.5316, 
        type = "Other", normalized_text = "PhishProof"), class = "data.frame", row.names = 1L), 
    NULL, NULL, NULL), mentions = list(NULL, NULL, structure(list(
    start = 3L, end = 12L, username = "BioCatch", id = "2581787594"), class = "data.frame", row.names = 1L), 
    NULL, NULL)), class = "data.frame", row.names = c(NA, -5L
)), source = c("IFTTT", "Hootsuite Inc.", "Twitter for iPhone", 
"HubSpot", "Twitter for iPhone"), id = c("1090739397803343872", 
"1090738965160824832", "1090736184614113281", "1090735911036506112", 
"1090735511914901504"), edit_history_tweet_ids = list("1090739397803343872", 
    "1090738965160824832", "1090736184614113281", "1090735911036506112", 
    "1090735511914901504"), created_at = c("2019-01-30T22:31:52.000Z", 
"2019-01-30T22:30:08.000Z", "2019-01-30T22:19:05.000Z", "2019-01-30T22:18:00.000Z", 
"2019-01-30T22:16:25.000Z"), possibly_sensitive = c(FALSE, FALSE, 
FALSE, FALSE, FALSE), lang = c("en", "en", "en", "en", "en"), 
    public_metrics = structure(list(retweet_count = c(0L, 1L, 
    1L, 1L, 0L), reply_count = c(0L, 0L, 0L, 0L, 0L), like_count = c(0L,7L, 0L, 2L, 0L), quote_count = c(0L, 0L, 0L, 0L, 0L)), class = "data.frame", row.names = c(NA, 
    -5L)), text = c("RT InspiredeLearn: If 2017 was the year of ransomware, 2018 was the year of the phish. And #phishing attacks are only growing more sophisticated.  #cybersecurity #Vishing #SMiShing  #PhishProof", 
    "If 2017 was the year of ransomware, 2018 was the year of the phish. And #phishing attacks are only growing more sophisticated.  #cybersecurity #Vishing #SMiShing  #PhishProof", 
    "RT @BioCatch: #Vishing scams are on the rise around the globe. Here's how banks can combat the threat. ", 
    "#Vishing scams are on the rise around the globe. Here's how banks can combat the threat.", 
    "Man we gotta get a grip on these “Vishing” calls I get everyday all day"
    ), conversation_id = c("1090739397803343872", "1090738965160824832", 
    "1090736184614113281", "1090735911036506112", "1090735511914901504"
    ), referenced_tweets = list(NULL, NULL, structure(list(type = "retweeted", 
        id = "1090735911036506112"), class = "data.frame", row.names = 1L), 
        NULL, NULL), attachments = structure(list(media_keys = list(
        NULL, NULL, NULL, "3_1090735909463621632", NULL)), class = "data.frame", row.names = c(NA, 
    -5L)), in_reply_to_user_id = c(NA_character_, NA_character_, 
    NA_character_, NA_character_, NA_character_), geo = structure(list(
        place_id = c(NA_character_, NA_character_, NA_character_, 
        NA_character_, NA_character_)), class = "data.frame", row.names = c(NA, 
    -5L))), row.names = c(NA, -5L), class = c("tbl_df", "tbl", 
"data.frame"))

执行do.call(data.frame, df)命令尝试将嵌套列转换为常规列时,触发错误:

Error in (function (..., row.names = NULL, check.rows = FALSE, check.names = TRUE,  : 
  arguments imply differing number of rows: 0, 1

解决方案

错误根源是嵌套列内的子数据框行数不一致(部分有5行、1行,部分为NULL),do.call(data.frame, ...)无法处理这种异构结构。推荐用tidyverse工具集处理嵌套数据框,以下是几种实用方案:

方案1:分步展开并聚合内层列表

先展开顶层嵌套数据框列,再将内层列表列聚合为字符串(保留原推文行数):

library(tidyverse)

# 展开顶层嵌套数据框,用下划线分隔列名
df_step1 <- df %>%
  unnest_wider(public_metrics, names_sep = "_") %>%
  unnest_wider(entities, names_sep = "_") %>%
  unnest_wider(attachments, names_sep = "_") %>%
  unnest_wider(geo, names_sep = "_")

# 处理内层列表列,将多值合并为字符串,空值设为NA
df_final <- df_step1 %>%
  mutate(
    hashtags = map_chr(hashtags, ~ifelse(is.null(.x), NA_character_, paste(.x$tag, collapse = ", "))),
    urls = map_chr(urls, ~ifelse(is.null(.x), NA_character_, paste(.x$expanded_url, collapse = ", "))),
    mentions = map_chr(mentions, ~ifelse(is.null(.x), NA_character_, paste(.x$username, collapse = ", ")))
  ) %>%
  # 移除不需要的原始嵌套列
  select(-annotations, -edit_history_tweet_ids, -referenced_tweets, -media_keys)

方案2:完全展开所有嵌套结构(行数会增加)

如果需要保留子数据框的每一行细节,用unnest_longer完全展开:

library(tidyverse)

df_full_expand <- df %>%
  # 展开顶层数据框列
  unnest_wider(public_metrics, names_sep = "_") %>%
  unnest_wider(entities, names_sep = "_") %>%
  # 展开内层列表列,保留空行
  unnest_longer(hashtags, keep_empty = TRUE) %>%
  unnest_wider(hashtags, names_sep = "_") %>%
  unnest_longer(urls, keep_empty = TRUE) %>%
  unnest_wider(urls, names_sep = "_") %>%
  unnest_longer(mentions, keep_empty = TRUE) %>%
  unnest_wider(mentions, names_sep = "_") %>%
  unnest_wider(attachments, names_sep = "_") %>%
  unnest_wider(geo, names_sep = "_")

注意:完全展开后,单条推文会根据子数据框行数拆分,总行数会增加。

方案3:简化列名(可选)

展开后列名可能冗长,用janitor包简化:

library(janitor)

df_final <- df_final %>%
  clean_names()

内容的提问来源于stack exchange,提问作者AneesBaqir

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.15 03:20:31