You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何基于多列日期条件为R数据框分配原发癌类型

问题:为多发癌症患者根据最早诊断日期确定原发癌类型

我正在处理一个包含5种癌症类型的患者数据集,每个患者可能被诊断出1种或多种癌症,n.cancer列记录了患者的癌症确诊数量,每种癌症都对应一个诊断日期列。对于n.cancer > 1的患者,我希望根据最早诊断的癌症类型为primary_can列赋值,但多次尝试后,这类患者的primary_can始终被错误赋值为other或multiple,无法得到正确的原发癌类型。

数据示例

eligible_df <- structure(list(
      patid = 1:3,
      cancer.blood = c(1, 0, 1),
      cancer.lung = c(0, 1, 1),
      cancer.breast = c(0, 0, 0),
      date.cancer.blood = structure(c(14610, NA, 15579), class = "Date"),
      date.cancer.lung = structure(c(NA, 15059, 16618), class = "Date"),
      date.cancer.breast = structure(c(NA, NA, NA), class = "Date")
    ), row.names = c(NA, -3L), class = "data.frame")

eligible_df$n.cancer <- rowSums(eligible_df[, c("cancer.blood","cancer.lung","cancer.breast")])
# eligible_df$primary_can <- NA

期望输出

  • 患者1 → "blood"
  • 患者2 → "lung"
  • 患者3 → "blood"

我怀疑问题出在cancer_mapping对象上,但即使移除相关代码,依然无法得到正确结果。以下是我尝试过的三种方法:


尝试方法1

# -------------------------------
# Step 1: Setup
# -------------------------------
# Cancer types in the dataset
cancer_types <- c("blood","breast","colorectal","lung","prostate","melanoma","others")

# Initialize column
eligible_df$primary_can <- "other"

# -------------------------------
# Step 2: Patients with 1 cancer
# -------------------------------
for (i in 1:nrow(eligible_df)) {
  if (eligible_df$n.cancer[i] == 1) {
    for (cancer in cancer_types) {
      if (eligible_df[[paste0("cancer.", cancer)]][i] == 1) {
        eligible_df$primary_can[i] <- cancer
        break
      }
    }
  }
}

# -------------------------------
# Step 3: Patients with >1 cancer
# -------------------------------
for (i in 1:nrow(eligible_df)) {
  if (eligible_df$n.cancer[i] > 1) {
    # Collect all available diagnosis dates for this patient
    cancer_dates <- c(
      breast     = eligible_df$date.cancer.breast[i],
      blood      = eligible_df$date.cancer.blood[i],
      colorectal = eligible_df$date.cancer.colorectal[i],
      lung       = eligible_df$date.cancer.lung[i],
      prostate   = eligible_df$date.cancer.prostate[i],
      melanoma   = eligible_df$date.cancer.melanoma[i],
      others     = eligible_df$date.cancer.others[i]
    )
    
    # Drop NAs
    cancer_dates <- cancer_dates[!is.na(cancer_dates)]
    
    # If there’s at least one valid date, assign the earliest cancer
    if (length(cancer_dates) > 0) {
      earliest_cancer <- names(which.min(cancer_dates))
      eligible_df$primary_can[i] <- earliest_cancer
    }
  }
}

# -------------------------------
# Step 4: Add multiple cancer flag
# -------------------------------
eligible_df$multiple_cancer <- as.integer(eligible_df$n.cancer > 1)

# -------------------------------
# Step 5: Sanity check
# -------------------------------
cat("\nDistribution of primary cancers:\n")
print(table(eligible_df$primary_can))

cat("\nPatients with multiple cancers:", sum(eligible_df$multiple_cancer), "\n")

尝试方法2

# For multiple cancers, find earliest date
multi_cancer_indices <- which(cancer_mapping$n.cancer > 1)

# Step 1 & 2: For each patient with multiple cancers, find the earliest diagnosis date
for(i in multi_cancer_indices) {
  patid <- cancer_mapping$patid[i]
  
  # Extract candidate cancer dates
  cancer_dates <- c(
    breast    = eligible_df$date.cancer.breast[eligible_df$patid == patid],
    blood     = eligible_df$date.cancer.blood[eligible_df$patid == patid],
    colorectal= eligible_df$date.cancer.colorectal[eligible_df$patid == patid],
    lung      = eligible_df$date.cancer.lung[eligible_df$patid == patid],
    prostate  = eligible_df$date.cancer.prostate[eligible_df$patid == patid],
    melanoma  = eligible_df$date.cancer.melanoma[eligible_df$patid == patid],
    others    = eligible_df$date.cancer.others[eligible_df$patid == patid]
  )
  
  # Remove NAs
  cancer_dates <- cancer_dates[!is.na(cancer_dates)]
  
  if(length(cancer_dates) > 0) {
    # Step 3: Find earliest cancer
    earliest_cancer <- names(which.min(cancer_dates))
    
    # Assign the primary cancer type
    cancer_mapping$primary_can[i] <- earliest_cancer
  }
}

# Step 4: Verify the distributions
table(cancer_mapping$primary_can)

尝试方法3

# For multiple cancers, find earliest date
# Step 1: Check which records have n.cancer > 1
multi_cancer_indices <- which(cancer_mapping$n.cancer > 1)

if(length(multi_cancer_indices) > 0) {
  for(i in multi_cancer_indices) {
    patid <- cancer_mapping$patid[i]
    patient_data <- eligible_df[eligible_df$patid == patid, ][1, ]
    
    # Step 2: For each n.cancer > 1 check the date values of date.cancer.breast, date.cancer.blood, date.cancer.prostate,
    # date.cancer.colorectal, date.cancer.lung, date.cancer.others, date.cancer.melanoma and keep the
    # column with the earliest date
    date_columns <- c("date.cancer.breast", "date.cancer.blood", "date.cancer.prostate", 
                      "date.cancer.colorectal", "date.cancer.lung", "date.cancer.others", 
                      "date.cancer.melanoma")
    
    earliest_date <- as.Date("9999-12-31")
    earliest_column <- NA
    
    for(date_col in date_columns) {
      if(date_col %in% names(patient_data) && !is.na(patient_data[[date_col]])) {
        cancer_date <- as.Date(patient_data[[date_col]])
        if(cancer_date < earliest_date) {
          earliest_date <- cancer_date
          earliest_column <- date_col
        }
      }
    }
    
    # Step 3: use the column with the earliest date to assign a value to primary_can column
    # e.g. date.cancer.breast then primary_can = breast
    if(!is.na(earliest_column)) {
      primary_cancer <- strsplit(earliest_column, "\\.")[[1]][3]
      cancer_mapping$primary_can[i] <- primary_cancer
    } else {
      cancer_mapping$primary_can[i] <- "multiple"
    }
  }
}

# Create lookup tables
cancer_lookup <- setNames(cancer_mapping$primary_can, cancer_mapping$patid)

# Assign to eligible_df
eligible_df$primary_can <- cancer_lookup[as.character(eligible_df$patid)]
eligible_df$primary_can[is.na(eligible_df$primary_can)] <- "other"

# Create multiple cancer indicator
eligible_df$multiple_cancer <- as.integer(eligible_df$n.cancer > 1)

# Step 4: Verify the distributions
cancer_table <- table(eligible_df$primary_can)
cat("\nDistribution of cancer types:\n")
print(cancer_table)

内容的提问来源于stack exchange,提问作者Evridiki Georgaki

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.12 09:34:57