基于R语言分析PubMed文献:年度词频与词云构建技术求助
具体实现方法
一、用tm包清洗文本
先定义通用的文本清洗函数,再批量处理各年份的DataFrame:
library(tm) library(dplyr) # 定义文本清洗函数 clean_abstract <- function(text_col) { # 创建语料库 corpus <- VCorpus(VectorSource(text_col)) # 转小写 corpus <- tm_map(corpus, content_transformer(tolower)) # 去除标点 corpus <- tm_map(corpus, removePunctuation) # 去除数字 corpus <- tm_map(corpus, removeNumbers) # 去除英文停用词(可自定义停用词列表) corpus <- tm_map(corpus, removeWords, stopwords("english")) # 去除自定义无用词(比如和研究无关的通用词) custom_stopwords <- c("study", "patient", "patients", "result", "results") corpus <- tm_map(corpus, removeWords, custom_stopwords) # 去除多余空格 corpus <- tm_map(corpus, stripWhitespace) # 可选:词干提取(还原词根) # corpus <- tm_map(corpus, stemDocument) # 将清洗后的语料转回字符向量 cleaned_text <- sapply(corpus, as.character) return(cleaned_text) } # 批量处理各年份DataFrame(假设year_dfs是按年份拆分的DataFrame列表) for (i in seq_along(year_dfs)) { year_dfs[[i]] <- year_dfs[[i]] %>% mutate(cleaned_abstract = clean_abstract(abstract)) }
二、统计年度词频
对每个年份的清洗后文本,生成词频统计表:
# 定义词频统计函数 get_word_freq <- function(cleaned_text) { # 创建文档-词项矩阵 dtm <- DocumentTermMatrix(VCorpus(VectorSource(cleaned_text))) # 转换为词频数据框 word_freq <- colSums(as.matrix(dtm)) word_freq <- sort(word_freq, decreasing = TRUE) return(data.frame(word = names(word_freq), freq = word_freq, stringsAsFactors = FALSE)) } # 生成各年份词频表,存储为列表 year_word_freqs <- lapply(year_dfs, function(df) { get_word_freq(df$cleaned_abstract) }) # 给词频列表命名对应年份(假设year_dfs的名称是年份) names(year_word_freqs) <- names(year_dfs)
三、绘制年度词频变化图
以Top10高频词为例,用ggplot2绘制跨年度词频变化:
library(ggplot2) # 整理成可视化所需的长格式数据 freq_long <- bind_rows(lapply(names(year_word_freqs), function(year) { year_word_freqs[[year]] %>% head(10) %>% mutate(year = as.integer(year)) })) # 绘制折线图 ggplot(freq_long, aes(x = year, y = freq, color = word, group = word)) + geom_line(size = 1) + geom_point(size = 2) + labs(title = "年度Top10高频词变化趋势", x = "年份", y = "词频") + theme_minimal() + theme(legend.position = "bottom") # 或者绘制年度柱状对比图(按年份分面) ggplot(freq_long, aes(x = reorder(word, -freq), y = freq, fill = word)) + geom_col() + facet_wrap(~year, scales = "free_x") + labs(title = "各年度Top10高频词", x = "词汇", y = "词频") + theme_minimal() + theme(axis.text.x = element_text(angle = 45, hjust = 1))
四、用wordcloud2生成年度词云
循环生成各年份的词云,可保存为图片:
library(wordcloud2) library(htmlwidgets) # 循环处理每个年份的词频表 for (year in names(year_word_freqs)) { freq_df <- year_word_freqs[[year]] # 生成词云(可调整size、color等参数) wc <- wordcloud2(freq_df, size = 1.5, color = "random-dark", backgroundColor = "white") # 保存词云为HTML文件(可后续转为图片) saveWidget(wc, paste0("platinum_resistant_cancer_wordcloud_", year, ".html")) }
内容的提问来源于stack exchange,提问作者Aidi
相关产品推荐
相关产品推荐

