You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用PROC MDC预测测试样本选择概率?解决NCHOICE报错

解决PROC MDC中NCHOICE参数报错并实现选择概率预测

问题背景

尝试使用PROC MDC预测品牌选择概率,原代码运行后触发报错:

ERROR: The NCHOICE= option is not allowed when the number of choices for each individual (ID) is not the same.

原代码如下:

/*  Reformat data as necessary */
data brand_choice(keep=Brand Price Displ Feature Chosen PurchaseID);
array prices[4] priceprivate pricesunshine pricekeebler pricenabisco;
array displays[4] displprivate displsunshine displkeebler displnabisco;
array features[4] featprivate featsunshine featkeebler featnabisco;
array chosenbrand[4] private sunshine keebler nabisco;
array allbrands[4] $8 _temporary_ ('Private' 'Sunshine' 'Keebler' 'Nabisco');

 set reg.crackers_hw5;

 PurchaseID = _N_;

do i = 1 to 4;
     Brand = allbrands[i];
     Price = prices[i];
     Displ = displays[i];
     Feature = features[i];
     Chosen = chosenbrand[i];
                output;
end;

/* Step 1: Prepare the Dataset */
proc surveyselect data=brand_choice
    out=brand_choice_sampled
    outall
    samprate=0.75
    seed=1
    method=srs;
run;

data brand_choice_sampled;
    set brand_choice_sampled;
    if selected = 1 then choiceT = chosen;
    else choiceT = .;
run;

/* Step 2: Estimate the Model */
proc mdc data=brand_choice_sampled;
   class displ feature;
    model choiceT = price displ feature displ*feature / type = mprobit nchoice = 4;
    id PurchaseID;
    restrict Displ1 = 0, Feature1 = 0, Displ1Feature1 = 0;
    output out=pred_probs p=predicted_prob; 
run;

/* Step 3: View the Predicted Probabilities */
proc print data=pred_probs;
    var predicted_prob; 
    id PurchaseID;
run;

错误原因

原代码使用method=srs对单条观测进行简单随机抽样,导致部分PurchaseID对应的4条品牌选项记录被拆分(部分选中、部分未选中),使得每个ID的选项数不一致,违反了NCHOICE=4要求每个ID必须有固定数量选项的规则。

解决方案

按PurchaseID分组抽样,确保每个ID的所有4条记录要么全部进入训练集,要么全部进入测试集,保持每个ID的选项数统一为4。

修改后的完整代码

/*  Reformat data as necessary */
data brand_choice(keep=Brand Price Displ Feature Chosen PurchaseID);
array prices[4] priceprivate pricesunshine pricekeebler pricenabisco;
array displays[4] displprivate displsunshine displkeebler displnabisco;
array features[4] featprivate featsunshine featkeebler featnabisco;
array chosenbrand[4] private sunshine keebler nabisco;
array allbrands[4] $8 _temporary_ ('Private' 'Sunshine' 'Keebler' 'Nabisco');

 set reg.crackers_hw5;

 PurchaseID = _N_;

do i = 1 to 4;
     Brand = allbrands[i];
     Price = prices[i];
     Displ = displays[i];
     Feature = features[i];
     Chosen = chosenbrand[i];
                output;
end;

/* Step 1: 按PurchaseID分组抽样,确保每个ID整体被选中/未选中 */
proc sort data=brand_choice out=brand_choice_sorted;
    by PurchaseID;
run;

proc surveyselect data=brand_choice_sorted
    out=brand_choice_sampled
    outall
    samprate=0.75
    seed=1
    method=srs
    strata PurchaseID; /* 按PurchaseID分层,保证每个ID的所有记录同进同出 */
run;

/* 统一设置每个ID的choiceT:训练集(selected=1)保留Chosen值,测试集设为缺失 */
data brand_choice_sampled;
    set brand_choice_sampled;
    by PurchaseID;
    retain selected_flag;
    if first.PurchaseID then selected_flag = selected;
    if selected_flag = 1 then choiceT = chosen;
    else choiceT = .;
    drop selected selected_flag;
run;

/* Step 2: 估计模型并预测概率 */
proc mdc data=brand_choice_sampled;
   class displ feature;
    model choiceT = price displ feature displ*feature / type = mprobit nchoice = 4;
    id PurchaseID;
    restrict Displ1 = 0, Feature1 = 0, Displ1Feature1 = 0;
    output out=pred_probs p=predicted_prob; 
run;

/* Step 3: 查看预测概率 */
proc print data=pred_probs;
    var predicted_prob; 
    id PurchaseID;
run;

关键修改说明

  1. 按PurchaseID分层抽样:在PROC SURVEYSELECT中添加strata PurchaseID,确保每个ID的所有4条记录作为一个整体被抽样,避免拆分。
  2. 统一设置choiceT:使用by PurchaseID和retain语句,为同一个ID的所有记录统一设置choiceT值,保证训练集ID的所有记录都有选择标记,测试集ID的所有记录都为缺失。
  3. 排序预处理:抽样前对数据集按PurchaseID排序,确保分层抽样的正确性。

内容的提问来源于stack exchange,提问作者maha

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 06:36:04