You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C++解析含嵌入逗号与引号的Kaggle CSV数据集问题求助

问题:C++ CSV解析无法处理嵌入逗号与引号导致数据错乱

我正在为学校项目处理一份电商笔记本销售数据集,需要用C++编写代码做排序和统计分析,但始终没法正确忽略数据里的嵌入逗号和随机引号。尝试把每个表头对应的数据存入vector后,打印时发现数据混杂,还有多余的逗号、括号和乱码长文本问题,以下是我写的代码:

#include <iostream>
#include <fstream>
#include <sstream>
#include <vector>
#include <string>
#include <set>

std::vector<std::string> parseCSVLine(const std::string& line) {
    std::vector<std::string> result;
    std::string cell;
    bool inQuotes = false;
    bool inBulletPoint = false; // New flag to track when we're within a bullet point
    for (auto it = line.begin(); it != line.end(); ++it) {
        const char nextChar = *it;

        // Check for bullet points
        if (!inQuotes && *it == '•') {
            inBulletPoint = true; // We're now inside a bullet point
            cell += nextChar; // Add the bullet point character to the cell
            continue;
        }

        // If we're in a bullet point, check for the end of the line or a comma (end of cell)
        if (inBulletPoint && (nextChar == ',' || it == line.end() - 1)) {
            inBulletPoint = false; // Exiting bullet point mode
            if (nextChar != ',') {
                cell += nextChar; // Ensure last character is included if not a comma
            }
            result.push_back(cell);
            cell.clear();
            continue;
        }
        else if (inBulletPoint) {
            // Simply add the character to the cell without interpreting it
            cell += nextChar;
            continue;
        }

        // Handle quotes (outside of bullet points)
        if (nextChar == '"') {
            if (inQuotes && (it + 1 != line.end()) && (*(it + 1) == '"')) {
                cell += nextChar; // Add a single quote to the cell value
                ++it; // Skip the next quote
            }
            else {
                inQuotes = !inQuotes;
            }
        }
        else if (nextChar == ',' && !inQuotes) {
            result.push_back(cell);
            cell.clear();
        }
        else {
            cell += nextChar;
        }
    }
    // Only check the last character if the line is not empty
    if (!cell.empty() || (!line.empty() && line.back() == ',')) {
        result.push_back(cell);
    }


    return result;
}

int main() {
    std::string filePath = "insert file path here";
    std::ifstream file(filePath);
    if (!file.is_open()) {
        std::cerr << "Failed to open file: " << filePath << std::endl;
        return 1;
    }

    std::string line;
    std::vector<std::string> headers;
    std::vector<std::vector<std::string>> columnData;

    if (getline(file, line)) {
        headers = parseCSVLine(line);
        columnData.resize(headers.size());
    }

    while (getline(file, line)) {
        auto data = parseCSVLine(line);
        for (size_t i = 0; i < data.size() && i < columnData.size(); ++i) {
            columnData[i].push_back(data[i]);
        }
    }

    file.close();

    //// Example output: Printing unique values for each heading for verification
    //for (size_t i = 0; i < headers.size(); ++i) {
    //    std::set<std::string> uniqueValues(columnData[i].begin(), columnData[i].end());
    //    std::cout << "Heading: " << headers[i] << " - Unique Values: " << uniqueValues.size() << std::endl;
    //    for (const auto& value : uniqueValues) {
    //        std::cout << value << std::endl;
    //    }
    //    std::cout << std::endl;
    //}


    // Make sure to define and fill 'columnData' and 'headers' as per your CSV parsing logic before this snippet

    // Here, the index is set to 2 since vector indices are 0-based and we want the third column (heading 3)
    size_t index = 2;

    // Check if the index is within the bounds of the 'columnData' vector
    if (index < columnData.size()) {
        std::cout << "Values under Heading 3 (" << headers[index] << "):" << std::endl;

        // Iterate over the vector at the given index and print each value
        for (const auto& value : columnData[index]) {
            std::cout << value << std::endl;
        }
    }
    else {
        std::cerr << "Index out of range. The columnData does not have a heading 3." << std::endl;
    }

    return 0;
}

问题根源

你添加的inBulletPoint状态逻辑破坏了标准CSV解析规则:

  • 遇到•就强制进入bullet模式,不管该符号是否在引号内,导致后续的逗号被错误判断为单元格结束符
  • bullet模式下直接跳过了引号处理逻辑,使得引号内的嵌入逗号无法被正确忽略

修正后的解析代码

核心是优先处理引号状态,再判断逗号分隔符,不需要单独处理bullet符号——因为如果bullet在引号内,会被当作单元格内容的一部分;如果在引号外,正常跟随文本即可。

#include <iostream>
#include <fstream>
#include <sstream>
#include <vector>
#include <string>
#include <set>

std::vector<std::string> parseCSVLine(const std::string& line) {
    std::vector<std::string> result;
    std::string cell;
    bool inQuotes = false;

    for (size_t i = 0; i < line.size(); ++i) {
        char c = line[i];

        // 处理双引号转义:"" 表示一个普通的"
        if (c == '"') {
            if (inQuotes && i + 1 < line.size() && line[i+1] == '"') {
                cell += '"';
                i++; // 跳过下一个引号
            } else {
                inQuotes = !inQuotes;
            }
        }
        // 只有不在引号内的逗号才是单元格分隔符
        else if (c == ',' && !inQuotes) {
            result.push_back(cell);
            cell.clear();
        }
        // 其他字符直接加入单元格
        else {
            cell += c;
        }
    }

    // 添加最后一个单元格
    result.push_back(cell);

    return result;
}

// main函数保持核心逻辑,增加格式校验
int main() {
    std::string filePath = "insert file path here";
    std::ifstream file(filePath);
    if (!file.is_open()) {
        std::cerr << "Failed to open file: " << filePath << std::endl;
        return 1;
    }

    std::string line;
    std::vector<std::string> headers;
    std::vector<std::vector<std::string>> columnData;

    if (getline(file, line)) {
        headers = parseCSVLine(line);
        columnData.resize(headers.size());
    }

    while (getline(file, line)) {
        auto data = parseCSVLine(line);
        // 确保数据列数和表头一致,避免越界导致的数据错位
        if (data.size() == columnData.size()) {
            for (size_t i = 0; i < data.size(); ++i) {
                columnData[i].push_back(data[i]);
            }
        } else {
            // 可选:打印错误行,方便排查格式问题
            std::cerr << "Warning: Line has mismatched column count: " << line << std::endl;
        }
    }

    file.close();

    size_t index = 2;
    if (index < columnData.size()) {
        std::cout << "Values under Heading 3 (" << headers[index] << "):" << std::endl;
        for (const auto& value : columnData[index]) {
            std::cout << value << std::endl;
        }
    } else {
        std::cerr << "Index out of range. The columnData does not have a heading 3." << std::endl;
    }

    return 0;
}

关键修改说明

  1. 删除了冗余的inBulletPoint状态逻辑,回归标准CSV解析规则
  2. 优化了循环索引方式,比迭代器更直观易维护
  3. 在main函数中增加了列数校验,避免因格式错误导致的数据错位
  4. 简化了末尾单元格的处理逻辑,确保所有情况下都能正确添加最后一个单元格

内容的提问来源于stack exchange,提问作者Shiva Ramlal

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.26 08:32:06