You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C++读取Weka ARFF文件实现OneR算法时触发vector下标越界错误

解决VS 2018+中读取ARFF文件生成OneR规则时的vector下标越界断言错误

你遇到的Debug断言错误是vector下标越界,结合代码来看,主要是以下几个问题导致的,逐个修正即可:

1. 直接触发断言的:空vector的下标访问

代码中多处直接使用data.classLabels[0],如果parseARFF函数没有正确解析ARFF文件的类标签(比如解析逻辑缺失、ARFF文件格式不符合预期),data.classLabels会是空vector,此时访问下标0就会触发断言错误。

修正:

  • 所有访问data.classLabels的地方,先判断是否为空:
    if (data.classLabels.empty()) {
        cerr << "Error: No class labels parsed from ARFF file" << endl;
        return 1;
    }
    
  • 确保parseARFF函数正确提取ARFF文件中的类标签(通常是最后一个@attribute定义的属性)。

2. 未实现的parseARFF函数是核心问题

代码中parseARFF函数被省略,这是数据未正确加载的根源——如果实例、属性、类标签都没被正确解析到ARFFData结构体中,后续所有逻辑都会基于空数据报错。

补全parseARFF的基础实现:

#include <algorithm> // 补充头文件用于find函数

ARFFData parseARFF(const string& filename) {
    ARFFData data;
    ifstream file(filename);
    if (!file.is_open()) {
        cerr << "Failed to open file: " << filename << endl;
        return data;
    }

    string line;
    bool isDataSection = false;
    int classAttrIndex = -1;

    while (getline(file, line)) {
        // 跳过空行和注释
        if (line.empty() || line[0] == '%') continue;

        // 处理@attribute定义
        if (line.substr(0, 10) == "@attribute") {
            size_t pos = line.find(' ');
            pos = line.find(' ', pos + 1);
            string attrName = line.substr(11, pos - 11);
            string attrValues = line.substr(pos + 1);

            // 判断是否是类标签(假设最后一个@attribute是类)
            data.attributeNames.push_back(attrName);
            if (attrValues.find('{') != string::npos) {
                // 解析离散属性的取值
                vector<string> values;
                string val;
                for (char c : attrValues) {
                    if (c == '{' || c == '}' || c == ',') {
                        if (!val.empty()) {
                            values.push_back(val);
                            val.clear();
                        }
                    } else if (!isspace(c)) {
                        val += c;
                    }
                }
                data.attributeValues.push_back(values);
            } else {
                // 连续属性,暂时存空vector
                data.attributeValues.push_back({});
            }
            classAttrIndex = data.attributeNames.size() - 1;
        }
        // 处理@data部分
        else if (line.substr(0, 5) == "@data") {
            isDataSection = true;
        } else if (isDataSection) {
            // 解析实例数据
            Instance inst;
            string val;
            int attrIdx = 0;
            for (char c : line) {
                if (c == ',' || c == '\n') {
                    if (!val.empty()) {
                        if (attrIdx == classAttrIndex) {
                            inst.classLabel = val;
                            // 收集类标签(去重)
                            if (find(data.classLabels.begin(), data.classLabels.end(), val) == data.classLabels.end()) {
                                data.classLabels.push_back(val);
                            }
                        } else {
                            inst.attributes.push_back(val);
                        }
                        val.clear();
                        attrIdx++;
                    }
                } else if (!isspace(c)) {
                    val += c;
                }
            }
            // 处理最后一个值
            if (!val.empty()) {
                if (attrIdx == classAttrIndex) {
                    inst.classLabel = val;
                    if (find(data.classLabels.begin(), data.classLabels.end(), val) == data.classLabels.end()) {
                        data.classLabels.push_back(val);
                    }
                } else {
                    inst.attributes.push_back(val);
                }
            }
            data.instances.push_back(inst);
        }
    }

    file.close();
    return data;
}

3. findOneRRule函数的逻辑错误

当前代码中return make_pair(bestAttribute, minErrorRate);写在for循环内部,第一次迭代就会直接返回,根本没遍历所有属性寻找最优规则。另外,调用calculateErrorRate时直接传data.classLabels[0]完全不符合OneR算法逻辑。

修正后的findOneRRule:

pair<string, double> findOneRRule(const ARFFData& data) {
    if (data.instances.empty() || data.attributeNames.empty()) {
        return {"", 1.0};
    }

    string bestAttribute;
    double minErrorRate = 1.0;

    // 遍历每个属性,计算对应的最优错误率
    for (size_t attrIdx = 0; attrIdx < data.attributeNames.size(); attrIdx++) {
        const string& attrName = data.attributeNames[attrIdx];
        // 统计每个属性取值对应的类标签出现次数
        map<string, map<string, int>> valueClassCount;

        for (const Instance& inst : data.instances) {
            if (attrIdx >= inst.attributes.size()) continue; // 防止实例属性数量不匹配
            const string& val = inst.attributes[attrIdx];
            valueClassCount[val][inst.classLabel]++;
        }

        // 计算该属性对应的错误率
        int totalErrors = 0;
        for (const auto& entry : valueClassCount) {
            // 找到当前属性取值下出现次数最多的类标签
            int maxCount = 0;
            int total = 0;
            for (const auto& classCount : entry.second) {
                total += classCount.second;
                if (classCount.second > maxCount) {
                    maxCount = classCount.second;
                }
            }
            totalErrors += (total - maxCount);
        }

        double errorRate = static_cast<double>(totalErrors) / data.instances.size();
        // 更新最优属性
        if (errorRate < minErrorRate) {
            minErrorRate = errorRate;
            bestAttribute = attrName;
        }
    }

    return make_pair(bestAttribute, minErrorRate);
}

4. 移除无用的calculateErrorRate占位函数

原calculateErrorRate的逻辑完全不符合OneR需求,直接集成到findOneRRule中即可,不需要单独的占位函数。

修正后的main函数

int main() {
    ARFFData data = parseARFF("input.arff");

    if (data.instances.empty() || data.classLabels.empty()) {
        cerr << "Error: No valid data parsed from ARFF file" << endl;
        return 1;
    }

    pair<string, double> oneRRule = findOneRRule(data);

    if (oneRRule.first.empty()) {
        cerr << "Error: Failed to find valid OneR rule" << endl;
        return 1;
    }

    // 输出最优规则及完整的OneR规则集
    cout << "Best OneR Attribute: " << oneRRule.first << endl;
    cout << "Error Rate: " << oneRRule.second << endl;
    cout << "Full OneR Rules:" << endl;

    // 生成并输出每个属性取值对应的规则
    size_t bestAttrIdx = find(data.attributeNames.begin(), data.attributeNames.end(), oneRRule.first) - data.attributeNames.begin();
    map<string, map<string, int>> valueClassCount;
    for (const Instance& inst : data.instances) {
        if (bestAttrIdx >= inst.attributes.size()) continue;
        const string& val = inst.attributes[bestAttrIdx];
        valueClassCount[val][inst.classLabel]++;
    }
    for (const auto& entry : valueClassCount) {
        string bestClass;
        int maxCount = 0;
        for (const auto& classCount : entry.second) {
            if (classCount.second > maxCount) {
                maxCount = classCount.second;
                bestClass = classCount.first;
            }
        }
        cout << "  If " << oneRRule.first << " = " << entry.first << " then " << bestClass << endl;
    }

    return 0;
}

额外注意事项

  • 确保你的ARFF文件格式正确,包含@relation、@attribute(最后一个为类属性)、@data部分。
  • Debug模式下的断言是为了提前发现问题,Release模式下不会触发,但逻辑错误依然存在,所以必须修正根源问题。

内容的提问来源于stack exchange,提问作者kate bran

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.25 15:04:55