R Shiny中实现textInput()与职位标题的模糊匹配需求
实现薪资对比工具的职位近似匹配功能
当前你的薪资对比工具仅支持职位名称的精确匹配,无法处理用户输入的类似"Senior Financial Accountant"这类和数据中"Senior Accountant"近似的职位名称。下面是实现近似匹配的具体方案:
方案思路
通过字符串相似度匹配实现近似匹配,这里使用R的stringdist包计算字符串间的编辑距离(衡量两个字符串的差异程度),搭配「先精确匹配、失败后触发近似匹配」的逻辑,既保证结果准确性,又提升工具的容错性。
修改后的完整代码
library(shiny) library(stringdist) # 示例数据(替换为你的salary_test_data) salary_test_data <- structure(list(grp = c("Senior Accountant", "Driver", "Admin Assistant", "Data scientist", "Receptionist", "Barista", "Accountant", "Data analyst", "Sales Executive", "Sales Representative", "Data analyst", "Account Manager", "Marketing Manager", "Data analyst", "Secretary", "Data analyst", "Barista", "Waiter", "Senior Accountant", "Senior Accountant", "Data analyst"), monthly_income = c(4000, 2500, 3500, 4900, 4500, 5000, 5000, 4000, 7000, 2500, 4000, 25688.33, 10000, 8000, 4000, 4500, 5500, 2500, 4500, 7000, 15000), location = c("Portland", "Portland", "Seattle", "Seattle", "Seattle", "Seattle", "Georgetown", "Georgetown", "Dammam", "Georgetown", "Georgetown", "Portland", "Portland", "Portland", "Georgetown", "Seattle", "Seattle", "Georgetown", "Portland", "Portland", "Portland"), qualifications = c("HS", "no_qual_preference", "BA", "BA", "no_qual_preference", "no_qual_preference", "BA", "no_qual_preference", "BA", "no_qual_preference", "no_qual_preference", "BA", "BA", "BA", "no_qual_preference", "no_qual_preference", "no_qual_preference", "no_qual_preference", "no_qual_preference", "no_qual_preference", "no_qual_preference")), row.names = c(NA, -21L), class = c("tbl_df", "tbl", "data.frame")) ui <- fluidPage( textInput("grp", "Occupation"), numericInput("monthly_income", "Monthly Pay", value = 0), textInput("location", "City"), textInput("qualifications", "Qualifications"), actionButton("compare_btn", "Compare Salary"), verbatimTextOutput("comparison_results") ) server <- function(input, output) { observeEvent(input$compare_btn, { # 统一字符串格式:转小写+去首尾空格,避免大小写/空格导致的匹配误差 user_Occupation <- trimws(tolower(input$grp)) user_monthly_pay <- input$monthly_income user_location <- trimws(tolower(input$location)) user_qualification <- trimws(tolower(input$qualifications)) # 第一步:尝试精确匹配 filtered_data_exact <- salary_test_data[trimws(tolower(salary_test_data$grp)) == user_Occupation & trimws(tolower(salary_test_data$location)) == user_location & trimws(tolower(salary_test_data$qualifications)) == user_qualification, ] if (nrow(filtered_data_exact) > 0) { median_salary <- median(filtered_data_exact$monthly_income) comparison_results <- ifelse(user_monthly_pay >= median_salary, paste0("你的月薪高于该城市同职位(", unique(filtered_data_exact$grp), ")的中位数薪资(", median_salary, ")"), paste0("你的月薪低于该城市同职位(", unique(filtered_data_exact$grp), ")的中位数薪资(", median_salary, ")")) } else { # 第二步:精确匹配失败,触发近似匹配 # 获取所有唯一职位并格式化 unique_jobs <- unique(trimws(tolower(salary_test_data$grp))) # 计算用户输入与每个职位的Jaro-Winkler距离(值越小越相似) dists <- stringdist(user_Occupation, unique_jobs, method = "jw") # 设置相似度阈值(可调整,0.2表示差异不超过20%) threshold <- 0.2 similar_jobs <- unique_jobs[dists <= threshold] if (length(similar_jobs) > 0) { # 转换回原数据中的职位名称(保留原大小写) matched_jobs <- unique(salary_test_data$grp[trimws(tolower(salary_test_data$grp)) %in% similar_jobs]) # 过滤近似匹配的职位+城市+资质数据 filtered_data_approx <- salary_test_data[trimws(tolower(salary_test_data$grp)) %in% similar_jobs & trimws(tolower(salary_test_data$location)) == user_location & trimws(tolower(salary_test_data$qualifications)) == user_qualification, ] if (nrow(filtered_data_approx) > 0) { median_salary <- median(filtered_data_approx$monthly_income) comparison_results <- paste0("未找到精确匹配的职位,已近似匹配到:", paste(matched_jobs, collapse = "、"), "\n", ifelse(user_monthly_pay >= median_salary, paste0("你的月薪高于该城市对应职位的中位数薪资(", median_salary, ")"), paste0("你的月薪低于该城市对应职位的中位数薪资(", median_salary, ")"))) } else { comparison_results <- paste0("未找到与「", input$grp, "」近似且匹配城市/资质的职位数据") } } else { comparison_results <- paste0("未找到与「", input$grp, "」近似的职位数据") } } output$comparison_results <- renderText({ comparison_results }) }) } shinyApp(ui, server)
关键修改说明
- 统一字符串格式:将用户输入和数据中的文本统一转小写并去除首尾空格,避免因大小写、空格差异导致的不必要匹配失败。
- 分层匹配逻辑:优先精确匹配保证结果准确,失败后再触发近似匹配,兼顾准确性和容错性。
- 相似度算法选择:使用Jaro-Winkler距离(
method = "jw"),该算法专门针对短字符串优化,适合职位名称这类文本的匹配。可通过调整threshold值控制匹配宽松度(值越小匹配越严格)。 - 友好提示:近似匹配成功时明确告知用户匹配到的职位,无匹配时给出具体原因,提升用户体验。
内容的提问来源于stack exchange,提问作者nesta1990
相关产品推荐
相关产品推荐

