You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于R语言分析PubMed文献:年度词频与词云构建技术求助

具体实现方法

一、用tm包清洗文本

先定义通用的文本清洗函数,再批量处理各年份的DataFrame:

library(tm)
library(dplyr)

# 定义文本清洗函数
clean_abstract <- function(text_col) {
  # 创建语料库
  corpus <- VCorpus(VectorSource(text_col))
  # 转小写
  corpus <- tm_map(corpus, content_transformer(tolower))
  # 去除标点
  corpus <- tm_map(corpus, removePunctuation)
  # 去除数字
  corpus <- tm_map(corpus, removeNumbers)
  # 去除英文停用词(可自定义停用词列表)
  corpus <- tm_map(corpus, removeWords, stopwords("english"))
  # 去除自定义无用词(比如和研究无关的通用词)
  custom_stopwords <- c("study", "patient", "patients", "result", "results")
  corpus <- tm_map(corpus, removeWords, custom_stopwords)
  # 去除多余空格
  corpus <- tm_map(corpus, stripWhitespace)
  # 可选:词干提取(还原词根)
  # corpus <- tm_map(corpus, stemDocument)
  
  # 将清洗后的语料转回字符向量
  cleaned_text <- sapply(corpus, as.character)
  return(cleaned_text)
}

# 批量处理各年份DataFrame(假设year_dfs是按年份拆分的DataFrame列表)
for (i in seq_along(year_dfs)) {
  year_dfs[[i]] <- year_dfs[[i]] %>%
    mutate(cleaned_abstract = clean_abstract(abstract))
}

二、统计年度词频

对每个年份的清洗后文本,生成词频统计表:

# 定义词频统计函数
get_word_freq <- function(cleaned_text) {
  # 创建文档-词项矩阵
  dtm <- DocumentTermMatrix(VCorpus(VectorSource(cleaned_text)))
  # 转换为词频数据框
  word_freq <- colSums(as.matrix(dtm))
  word_freq <- sort(word_freq, decreasing = TRUE)
  return(data.frame(word = names(word_freq), freq = word_freq, stringsAsFactors = FALSE))
}

# 生成各年份词频表,存储为列表
year_word_freqs <- lapply(year_dfs, function(df) {
  get_word_freq(df$cleaned_abstract)
})
# 给词频列表命名对应年份(假设year_dfs的名称是年份)
names(year_word_freqs) <- names(year_dfs)

三、绘制年度词频变化图

以Top10高频词为例,用ggplot2绘制跨年度词频变化:

library(ggplot2)

# 整理成可视化所需的长格式数据
freq_long <- bind_rows(lapply(names(year_word_freqs), function(year) {
  year_word_freqs[[year]] %>%
    head(10) %>%
    mutate(year = as.integer(year))
}))

# 绘制折线图
ggplot(freq_long, aes(x = year, y = freq, color = word, group = word)) +
  geom_line(size = 1) +
  geom_point(size = 2) +
  labs(title = "年度Top10高频词变化趋势", x = "年份", y = "词频") +
  theme_minimal() +
  theme(legend.position = "bottom")

# 或者绘制年度柱状对比图(按年份分面)
ggplot(freq_long, aes(x = reorder(word, -freq), y = freq, fill = word)) +
  geom_col() +
  facet_wrap(~year, scales = "free_x") +
  labs(title = "各年度Top10高频词", x = "词汇", y = "词频") +
  theme_minimal() +
  theme(axis.text.x = element_text(angle = 45, hjust = 1))

四、用wordcloud2生成年度词云

循环生成各年份的词云,可保存为图片:

library(wordcloud2)
library(htmlwidgets)

# 循环处理每个年份的词频表
for (year in names(year_word_freqs)) {
  freq_df <- year_word_freqs[[year]]
  # 生成词云(可调整size、color等参数)
  wc <- wordcloud2(freq_df, size = 1.5, color = "random-dark", backgroundColor = "white")
  # 保存词云为HTML文件(可后续转为图片)
  saveWidget(wc, paste0("platinum_resistant_cancer_wordcloud_", year, ".html"))
}

内容的提问来源于stack exchange,提问作者Aidi

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.14 22:21:38