如何在R语言中将多个文件上传至单个变量(Shiny场景)
我帮你把单文件分析的Shiny代码扩展成支持多文件的版本啦,这样就能对整个章节的所有文档一起分析了。下面是完整的改造思路和代码示例:
多文件文本分析Shiny应用改造方案
核心改造点:支持多文件上传与内容合并
首先把原来针对单个/两个文件的读取逻辑,改成循环读取所有上传的文件并合并内容,这样所有分析都会基于整个章节的文本数据:
library(shiny) library(tm) library(wordcloud) library(RColorBrewer) library(syuzhet) # 英文情感分析,中文可替换为SnowNLP/jiebaR ui <- fluidPage( titlePanel("章节级文本分析工具"), sidebarLayout( sidebarPanel( # 开启多文件上传功能 fileInput("files", "选择章节内的所有文本文件", multiple = TRUE, accept = c("text/plain")), textInput("v", "输入关键词做关联分析", placeholder = "比如:数据科学"), actionButton("run_analysis", "执行全量分析") ), mainPanel( tabsetPanel( tabPanel("高频词统计", tableOutput("freq_table")), tabPanel("词云可视化", plotOutput("wordcloud_plot")), tabPanel("词关联分析", tableOutput("assoc_table")), tabPanel("情感趋势分析", plotOutput("sentiment_plot")) ) ) ) ) server <- function(input, output) { # 响应式合并所有上传文件的文本内容 combined_text <- eventReactive(input$run_analysis, { req(input$files) # 确保有文件上传才执行 # 循环读取每个文件的内容并合并 all_texts <- lapply(input$files$datapath, function(file_path) { readLines(file_path, encoding = "UTF-8") %>% paste(collapse = "\n") }) paste(all_texts, collapse = "\n") }) # 1. 高频词统计 output$freq_table <- renderTable({ text_corpus <- VCorpus(VectorSource(combined_text())) # 文本预处理(根据语言调整,中文需替换停用词库) text_corpus <- text_corpus %>% tm_map(removePunctuation) %>% tm_map(removeNumbers) %>% tm_map(content_transformer(tolower)) %>% tm_map(removeWords, stopwords("english")) dtm <- DocumentTermMatrix(text_corpus) freq <- colSums(as.matrix(dtm)) # 按频率降序排列,取前20个高频词 data.frame(词 = names(freq), 出现次数 = freq) %>% arrange(desc(出现次数)) %>% head(20) }) # 2. 词云生成 output$wordcloud_plot <- renderPlot({ text_corpus <- VCorpus(VectorSource(combined_text())) text_corpus <- text_corpus %>% tm_map(removePunctuation) %>% tm_map(removeNumbers) %>% tm_map(content_transformer(tolower)) %>% tm_map(removeWords, stopwords("english")) dtm <- DocumentTermMatrix(text_corpus) freq <- colSums(as.matrix(dtm)) wordcloud(names(freq), freq, min.freq = 3, colors = brewer.pal(8, "Dark2"), random.order = FALSE) }) # 3. 词关联分析(基于input$v的关键词) output$assoc_table <- renderTable({ req(input$v) # 确保用户输入了关键词 text_corpus <- VCorpus(VectorSource(combined_text())) text_corpus <- text_corpus %>% tm_map(removePunctuation) %>% tm_map(removeNumbers) %>% tm_map(content_transformer(tolower)) %>% tm_map(removeWords, stopwords("english")) dtm <- DocumentTermMatrix(text_corpus) # 提取关联度≥0.3的词汇 assoc_result <- findAssocs(dtm, tolower(input$v), corlimit = 0.3) if(length(assoc_result[[1]]) > 0) { data.frame(关联词 = names(assoc_result[[1]]), 关联度 = assoc_result[[1]]) %>% arrange(desc(关联度)) } else { data.frame(提示信息 = "未找到与该关键词相关的内容") } }) # 4. 情感分析(趋势可视化) output$sentiment_plot <- renderPlot({ text <- combined_text() # 获取情感得分(中文可替换为SnowNLP的sentiment函数) sentiment_scores <- get_sentiment(text, method = "syuzhet") plot(sentiment_scores, type = "l", main = "章节文本情感趋势", xlab = "文本位置", ylab = "情感得分(正数=积极,负数=消极)", col = "steelblue", lwd = 2) abline(h = 0, col = "red", lty = 2) }) } shinyApp(ui = ui, server = server)
关键说明
- 多文件上传:通过
fileInput的multiple = TRUE开启批量上传,后台用lapply循环读取所有文件路径 - 内容合并:把所有文件的文本合并成一个大文本块,确保所有分析都是基于整个章节的全局数据
- 语言适配:如果是中文文本,需要替换停用词库(比如用
jiebaR的中文停用词),情感分析换成SnowNLP包的相关函数
内容的提问来源于stack exchange,提问作者BadLuckNick
相关产品推荐
相关产品推荐

