You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在R语言中将多个文件上传至单个变量(Shiny场景)

我帮你把单文件分析的Shiny代码扩展成支持多文件的版本啦,这样就能对整个章节的所有文档一起分析了。下面是完整的改造思路和代码示例:

多文件文本分析Shiny应用改造方案

核心改造点:支持多文件上传与内容合并

首先把原来针对单个/两个文件的读取逻辑,改成循环读取所有上传的文件并合并内容,这样所有分析都会基于整个章节的文本数据:

library(shiny)
library(tm)
library(wordcloud)
library(RColorBrewer)
library(syuzhet) # 英文情感分析,中文可替换为SnowNLP/jiebaR

ui <- fluidPage(
  titlePanel("章节级文本分析工具"),
  sidebarLayout(
    sidebarPanel(
      # 开启多文件上传功能
      fileInput("files", "选择章节内的所有文本文件", 
                multiple = TRUE, 
                accept = c("text/plain")),
      textInput("v", "输入关键词做关联分析", placeholder = "比如:数据科学"),
      actionButton("run_analysis", "执行全量分析")
    ),
    mainPanel(
      tabsetPanel(
        tabPanel("高频词统计", tableOutput("freq_table")),
        tabPanel("词云可视化", plotOutput("wordcloud_plot")),
        tabPanel("词关联分析", tableOutput("assoc_table")),
        tabPanel("情感趋势分析", plotOutput("sentiment_plot"))
      )
    )
  )
)

server <- function(input, output) {
  # 响应式合并所有上传文件的文本内容
  combined_text <- eventReactive(input$run_analysis, {
    req(input$files) # 确保有文件上传才执行
    # 循环读取每个文件的内容并合并
    all_texts <- lapply(input$files$datapath, function(file_path) {
      readLines(file_path, encoding = "UTF-8") %>% paste(collapse = "\n")
    })
    paste(all_texts, collapse = "\n")
  })

  # 1. 高频词统计
  output$freq_table <- renderTable({
    text_corpus <- VCorpus(VectorSource(combined_text()))
    # 文本预处理(根据语言调整,中文需替换停用词库)
    text_corpus <- text_corpus %>%
      tm_map(removePunctuation) %>%
      tm_map(removeNumbers) %>%
      tm_map(content_transformer(tolower)) %>%
      tm_map(removeWords, stopwords("english"))
    
    dtm <- DocumentTermMatrix(text_corpus)
    freq <- colSums(as.matrix(dtm))
    # 按频率降序排列,取前20个高频词
    data.frame(词 = names(freq), 出现次数 = freq) %>%
      arrange(desc(出现次数)) %>%
      head(20)
  })

  # 2. 词云生成
  output$wordcloud_plot <- renderPlot({
    text_corpus <- VCorpus(VectorSource(combined_text()))
    text_corpus <- text_corpus %>%
      tm_map(removePunctuation) %>%
      tm_map(removeNumbers) %>%
      tm_map(content_transformer(tolower)) %>%
      tm_map(removeWords, stopwords("english"))
    
    dtm <- DocumentTermMatrix(text_corpus)
    freq <- colSums(as.matrix(dtm))
    wordcloud(names(freq), freq, 
              min.freq = 3, 
              colors = brewer.pal(8, "Dark2"),
              random.order = FALSE)
  })

  # 3. 词关联分析(基于input$v的关键词)
  output$assoc_table <- renderTable({
    req(input$v) # 确保用户输入了关键词
    text_corpus <- VCorpus(VectorSource(combined_text()))
    text_corpus <- text_corpus %>%
      tm_map(removePunctuation) %>%
      tm_map(removeNumbers) %>%
      tm_map(content_transformer(tolower)) %>%
      tm_map(removeWords, stopwords("english"))
    
    dtm <- DocumentTermMatrix(text_corpus)
    # 提取关联度≥0.3的词汇
    assoc_result <- findAssocs(dtm, tolower(input$v), corlimit = 0.3)
    
    if(length(assoc_result[[1]]) > 0) {
      data.frame(关联词 = names(assoc_result[[1]]), 关联度 = assoc_result[[1]]) %>%
        arrange(desc(关联度))
    } else {
      data.frame(提示信息 = "未找到与该关键词相关的内容")
    }
  })

  # 4. 情感分析(趋势可视化)
  output$sentiment_plot <- renderPlot({
    text <- combined_text()
    # 获取情感得分(中文可替换为SnowNLP的sentiment函数)
    sentiment_scores <- get_sentiment(text, method = "syuzhet")
    plot(sentiment_scores, 
         type = "l", 
         main = "章节文本情感趋势",
         xlab = "文本位置", 
         ylab = "情感得分(正数=积极,负数=消极)",
         col = "steelblue",
         lwd = 2)
    abline(h = 0, col = "red", lty = 2)
  })
}

shinyApp(ui = ui, server = server)

关键说明

  • 多文件上传:通过fileInput的multiple = TRUE开启批量上传,后台用lapply循环读取所有文件路径
  • 内容合并:把所有文件的文本合并成一个大文本块,确保所有分析都是基于整个章节的全局数据
  • 语言适配:如果是中文文本,需要替换停用词库(比如用jiebaR的中文停用词),情感分析换成SnowNLP包的相关函数

内容的提问来源于stack exchange,提问作者BadLuckNick

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.26 08:45:12