You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C++文本分析程序输出单词为何按字母序而非原文顺序

问题原因

你使用的std::set是有序关联容器,底层基于红黑树实现,插入元素时会自动按字典序排序,不会保留元素的插入顺序,因此输出的单词顺序和原文读取顺序完全无关,只会按字母序排列。
另外代码中还有一处逻辑bug:if(numbers = 2)是赋值操作而非等值判断,会导致该分支永远命中,只会输出第二个文件的单词。

解决方法

如果需要同时满足「单词去重」和「保留原文出现顺序」两个要求,可按以下逻辑修改:

  • 读取文件时,用vector<string>存储按原文顺序排列的唯一单词,同时搭配unordered_set<string>做存在性校验,避免重复单词插入vector
  • 做两个文件的交集、差集等集合运算时,可继续用std::set完成运算,得到结果集合后,遍历对应文件的原始顺序vector,只输出属于结果集合的单词,即可保留原文顺序

修改后的完整代码

#include <iostream>
#include <set>
#include <unordered_set>
#include <string>
#include <fstream>
#include <algorithm>
#include <vector>
#include <cstdlib>
#include <ctime>
using namespace std;

// 按原始顺序输出结果
void displayByOrder(const vector<string>& orderList, const set<string>& targetSet) {
    for (const auto& word : orderList) {
        if (targetSet.count(word)) {
            cout << word << endl;
        }
    }
}

void displayDifference(set<string> set1, set<string> set2, const vector<string>& order)
{
    vector<string> result(set1.size() + set2.size());
    
    auto iter = set_difference(set1.begin(), set1.end(),
                               set2.begin(), set2.end(),
                               result.begin());
                               
    result.resize(iter - result.begin());
    // 把差集转成set方便判断
    set<string> diffSet(result.begin(), result.end());
    displayByOrder(order, diffSet);
}

int main()
{
    ifstream inputFile;
    string name;
    const int MIN_VALUE = 1;
    const int MAX_VALUE = 2;
    int numbers;
    unsigned seed = time(0);
    
    srand(seed);
    
    set<string> firstSet;
    vector<string> firstOrder;
    unordered_set<string> firstDedup;

    set<string> secondSet;
    vector<string> secondOrder;
    unordered_set<string> secondDedup;
    
    inputFile.open("firstTextFile.txt");
    cout << "Reading data from the first file.\n" << endl;
    
    while (inputFile >> name)
    {
        firstSet.insert(name);
        // 去重同时保留顺序
        if (firstDedup.find(name) == firstDedup.end()) {
            firstDedup.insert(name);
            firstOrder.push_back(name);
        }
    }
    
    cout << "第一个文件按原文顺序的唯一单词列表:" << endl;
    for (const auto& word : firstOrder) {
        cout << word << endl;
    }
    
    inputFile.close();
    
    inputFile.open("secondTextFile.txt");
    cout << "\nReading data from the second file.\n" << endl;
    
    while (inputFile >> name)
    {
        secondSet.insert(name);
        if (secondDedup.find(name) == secondDedup.end()) {
            secondDedup.insert(name);
            secondOrder.push_back(name);
        }
    }
    
    cout << "第二个文件按原文顺序的唯一单词列表:" << endl;
    for (const auto& word : secondOrder) {
        cout << word << endl;
    }
    
    cout << "\n仅出现在第一个文件的单词(按原文顺序):\n" << endl;
    displayDifference(firstSet, secondSet, firstOrder);
    
    cout << "\n仅出现在第二个文件的单词(按原文顺序):\n" << endl;
    displayDifference(secondSet, firstSet, secondOrder);
    
    // 两个文件共有单词
    cout << "\n同时出现在两个文件的单词:\n" << endl;
    vector<string> commonRes(firstSet.size() + secondSet.size());
    auto commonIter = set_intersection(firstSet.begin(), firstSet.end(), secondSet.begin(), secondSet.end(), commonRes.begin());
    commonRes.resize(commonIter - commonRes.begin());
    set<string> commonSet(commonRes.begin(), commonRes.end());
    displayByOrder(firstOrder, commonSet);
    
    // 修复赋值bug
    numbers = rand() % 2 + 1;
    cout << "\n随机输出一个文件的单词列表:\n" << endl;
    if(numbers == 2)
    {
        for (const auto& word : secondOrder) {
            cout << word << endl;
        }
        cout << "\nThe Second File Words";
    }
    else
    {
        for (const auto& word : firstOrder) {
            cout << word << endl;
        }
        cout << "\nThe First File Words";
    }
    return 0;
}

内容的提问来源于stack exchange,提问作者Game Face

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.26 22:06:01