You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C#查找两个文本文件共有的10个最长单词并分别统计频次

核心修改思路

  1. 替换原方法的返回值结构,用C#值元组作为字典的Value,同时存储两个文件的独立频次
  2. 优化原有双重循环的低效率逻辑,改为分别统计两个文件的独立词频后取交集,大文件下性能提升显著

修改后的完整代码

词频统计方法

// 返回值的Value为元组,第一个元素是第一个文件的频次,第二个是第二个文件的频次
public static Dictionary<string, (int CountInFile1, int CountInFile2)> PopularWords(string data1, string data2, char[] punctuation)
{
    string[] book1 = data1.Split(punctuation, StringSplitOptions.RemoveEmptyEntries);
    string[] book2 = data2.Split(punctuation, StringSplitOptions.RemoveEmptyEntries);

    // 统计第一个文件的所有单词频次
    Dictionary<string, int> freq1 = new Dictionary<string, int>();
    foreach (string word in book1)
    {
        if (freq1.ContainsKey(word))
            freq1[word]++;
        else
            freq1[word] = 1;
    }

    // 统计第二个文件的所有单词频次
    Dictionary<string, int> freq2 = new Dictionary<string, int>();
    foreach (string word in book2)
    {
        if (freq2.ContainsKey(word))
            freq2[word]++;
        else
            freq2[word] = 1;
    }

    // 取两个文件的共有单词,合并两组频次
    Dictionary<string, (int, int)> matches = new Dictionary<string, (int, int)>();
    foreach (var kvp in freq1)
    {
        if (freq2.TryGetValue(kvp.Key, out int count2))
        {
            matches.Add(kvp.Key, (kvp.Value, count2));
        }
    }

    return matches;
}

文件读取与输出方法

public static void ProcessPopular(string data, string data1, string results)
{
    char[] punctuation = { ' ', '.', ',', '!', '?', ':', ';', '(', ')', '\n' };
    string lines = File.ReadAllText(data, Encoding.UTF8);
    string lines2 = File.ReadAllText(data1, Encoding.UTF8);

    var popular = PopularWords(lines, lines2, punctuation);

    // 按单词长度倒序排序,取前10个最长单词
    var sortedWords = popular
        .OrderByDescending(kvp => kvp.Key.Length)
        .Take(10)
        .ToArray();

    using (var writerF = File.CreateText(results))
    {
        writerF.WriteLine("{0, -25} | {1, -35} | {2, -35}", "Longest words", "Frequency in 1 .txt file", "Frequency in 2 .txt file");
        writerF.WriteLine(new string('-', 101));
        foreach (var item in sortedWords)
        {
            writerF.WriteLine("{0, -25} | {1, -35} | {2, -35}", 
                item.Key, 
                item.Value.CountInFile1, 
                item.Value.CountInFile2);
        }
    }
}

补充说明

如果需要忽略单词大小写统计(比如把Hello和hello视为同一个单词),可以在初始化字典的时候指定比较器:
new Dictionary<string, int>(StringComparer.OrdinalIgnoreCase)

内容的提问来源于stack exchange,提问作者rokenga

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.25 10:24:08