You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

LZ77编码处理ASCII特殊字符异常问题求助

LZ77压缩解压算法处理特殊字符异常问题修复

问题现象

实现的LZ77压缩解压算法在处理空格、换行符等ASCII特殊字符时出现以下异常:

  • 压缩输出存在格式错乱的空三元组
  • 解压后原特殊字符被错误替换(如空格变为')')

输入输出示例

原始输入

The quick brown fox jumps over the lazy dog.
ABCABCABCDABCDEFABCDEFGABCDEFGHABCDEFGHI

压缩输出

(0,0,T)(0,0,h)(0,0,e)(0,0, )(0,0,q)(0,0,u)(0,0,i)(0,0,c)(0,0,k)(6,1,b)(0,0,r)(0,0,o)(0,0,w)(0,0,n)(6,1,f)(5,1,x)(4,1,j)(16,1,m)(0,0,p)(0,0,s)(6,1,o)(0,0,v)(26,1,r)(5,1,t)(31,3,l)(0,0,a)(0,0,z)(0,0,y)(5,1,d)(15,1,g)(0,0,.)(0,0,
)(0,0,
)(0,0,A)(0,0,B)(0,0,C)(3,6,D)(4,4,E)(0,0,F)(6,6,G)(7,7,H)(8,8,I)

解压输出

The)quick)brown)fox)jumps)over)the)lazy)dog.))ABCABCABCDABCDEFABCDEFGABCDEFGHABCDEFGHI

参数配置

  • searchBuffer=256
  • lookAheadBuffer=4096

相关代码

Token结构

struct Token
{
    int offset;
    int length_of_match;
    char code_word;
};

压缩函数

std::vector<Token> compression_lz77(const std::string &input, int searchBuffer, int lookAheadBuffer)
{
    int inputLength = input.length();
    int position = 0;

    std::vector<Token> data;

    while (position < inputLength)
    {
        Token token{};
        token.offset = 0;
        token.length_of_match = 0;
        token.code_word = input[position];

        int max_offset = (position < searchBuffer) ? position : searchBuffer;
        int max_search_length = (position + lookAheadBuffer) > inputLength ? inputLength - position : lookAheadBuffer;

        for (int offset = 1; offset <= max_offset; offset++)
        {
            int len = 0;
            while (len < max_search_length && input[position - offset + len] == input[position + len])
            {
                len++;
            }

            if (len > token.length_of_match)
            {
                token.offset = offset;
                token.length_of_match = len;
                token.code_word = input[position + len];
            }
        }

        data.push_back(token);
        position += token.length_of_match + 1;
    }

    return data;
}

解压函数

std::string decompression_lz77(const std::vector<Token> &compressedData)
{
    std::string decompressed;

    for (const Token &token : compressedData)
    {
        if (token.offset == 0)
        {
                decompressed += token.code_word;
        }
        else
        {
            int startPos = decompressed.length() - token.offset;
            int endPos = startPos + token.length_of_match;

            for (int i = startPos; i < endPos; ++i)
            {
                decompressed += decompressed[i];
            }

            decompressed += token.code_word;
        }
    }

    return decompressed;
}

主处理函数(solve)

int solve(const CompressionParams &params)
{
    if (params.inputFileName.empty() || params.outputFileName.empty() || params.mode.empty() || params.inputBufferSize <= 0 || params.historyBufferSize <= 0)
    {
        std::cout << "Niepoprawne parametry linii polecen. Prosze podac wszystkie wymagane opcje." << std::endl;
        printInstructions();
        return 1;
    }

    std::ifstream inputFile(params.inputFileName, std::ios::binary);

    if (!inputFile.is_open())
    {
        std::cerr << "Blad podczas otwierania pliku wejsciowego: " << params.inputFileName << std::endl;
        return 1;
    }

    inputFile.seekg(0, std::ios::end);
    std::streampos fileSize = inputFile.tellg();
    inputFile.seekg(0, std::ios::beg);

    std::vector<char> fileContent(fileSize);
    inputFile.read(fileContent.data(), fileSize);
    inputFile.close();

    std::string data(fileContent.begin(), fileContent.end());
    std::vector<Token> arr;

    if (params.mode == "c")
    {
        if (!fileContent.empty())
        {
            arr = compression_lz77(data, params.inputBufferSize, params.historyBufferSize);

            std::ofstream outputFile(params.outputFileName);
            if (!outputFile.is_open())
            {
                std::cerr << "Wystapil blad podczas otwierania pliku wyjsciowego." << std::endl;
                return 1;
            }

            for (const auto &token : arr)
            {
                outputFile << "<" << token.offset << "," << token.length_of_match << "," << token.code_word << ">";
            }
            outputFile.close();

            auto startingSize = static_cast<double>(fileContent.size());
            double compressedSize = static_cast<double>(arr.size()) * (sizeof(Token) / sizeof(char));
            double wspolczynnik = (startingSize / compressedSize) * 100.0;

            std::cout << "Teoretyczny wspolczynnik kompresji: " << wspolczynnik << "%" << std::endl;
            std::cout << "Teoretyczny stopien kompresji " << (compressedSize / startingSize) << std::endl;
        }
    }
    else if (params.mode == "d")
    {
        std::ofstream outputFile(params.outputFileName, std::ios::binary);
        if (!outputFile.is_open())
        {
            std::cerr << "Wystapil blad podczas otwierania pliku wejsciowego." << std::endl;
            return 1;
        }

        std::string input(fileContent.begin(), fileContent.end());
        std::vector<Token> compressed_data;
        size_t pos = 0;

        while (pos < input.size())
        {
            size_t nextPos = input.find('>', pos);
            if (nextPos == std::string::npos)
            {
                break;
            }

            std::string tokenStr = input.substr(pos, nextPos - pos + 1);

            Token t{};
            std::istringstream tokenStream(tokenStr);
            char dummy;
            tokenStream >> dummy >> t.offset >> dummy >> t.length_of_match >> dummy >> t.code_word;
            compressed_data.push_back(t);
            pos = nextPos + 1;
        }

        for (const Token &token : compressed_data)
        {
            std::cout << '<' << token.offset << ',' << token.length_of_match << ',' << token.code_word << '>';
        }
        std::cout << std::endl;

        std::string decompressed_output = decompression_lz77(compressed_data);
        outputFile.write(decompressed_output.c_str(), decompressed_output.size());
        outputFile.close();
    }
    else
    {
        std::cerr << "Niepoprawny tryb. Proszę uzyc 'c' dla kompresji lub 'd' dla dekompresji." << std::endl;
        printInstructions();
        return 1;
    }
    return 0;
}

问题根源及修复方案

1. 特殊字符输出格式错乱问题

压缩时使用文本模式写入文件,导致换行符等特殊字符被自动转换(如Windows下\n转成\r\n),同时直接输出char类型特殊字符会破坏三元组结构(换行符直接换行拆分三元组)。

修复方法:

  • 压缩输出切换为二进制模式打开文件,避免字符转换:
    std::ofstream outputFile(params.outputFileName, std::ios::binary);
    
  • 对code_word进行十六进制转义存储,避免破坏输出结构:
    // 压缩时将字符转为十六进制字符串
    auto char_to_hex = [](char c) {
        std::stringstream ss;
        ss << std::hex << std::setw(2) << std::setfill('0') << static_cast<unsigned int>(static_cast<unsigned char>(c));
        return ss.str();
    };
    
    for (const auto &token : arr)
    {
        outputFile << "<" << token.offset << "," << token.length_of_match << "," << char_to_hex(token.code_word) << ">";
    }
    

2. 解压时特殊字符被错误替换问题

解压阶段用std::istringstream的>>运算符读取code_word时,>>会自动跳过空白字符(空格、换行等),导致特殊字符被错误替换。

修复方法:

  • 手动解析Token字段,不依赖>>运算符:
    // 解压时解析十六进制字符
    auto hex_to_char = [](const std::string &hex) {
        unsigned int val;
        std::stringstream ss(hex);
        ss >> std::hex >> val;
        return static_cast<char>(val);
    };
    
    while (pos < input.size())
    {
        size_t nextPos = input.find('>', pos);
        if (nextPos == std::string::npos) break;
    
        std::string tokenStr = input.substr(pos + 1, nextPos - pos - 1);
        size_t comma1 = tokenStr.find(',');
        size_t comma2 = tokenStr.find(',', comma1 + 1);
        if (comma1 == std::string::npos || comma2 == std::string::npos) {
            pos = nextPos + 1;
            continue;
        }
    
        Token t{};
        t.offset = std::stoi(tokenStr.substr(0, comma1));
        t.length_of_match = std::stoi(tokenStr.substr(comma1 + 1, comma2 - comma1 - 1));
        t.code_word = hex_to_char(tokenStr.substr(comma2 + 1));
    
        compressed_data.push_back(t);
        pos = nextPos + 1;
    }
    

3. 输入末尾越界访问问题

当position + len等于输入长度时,token.code_word = input[position + len]会访问越界,需添加边界判断:

修复方法:
修改压缩函数中的匹配逻辑:

if (len > token.length_of_match)
{
    token.offset = offset;
    token.length_of_match = len;
    if (position + len < inputLength)
    {
        token.code_word = input[position + len];
    }
    else
    {
        token.code_word = '\0'; // 用空字符标记结束
    }
}

解压时忽略结束标记:

if (token.offset == 0)
{
    if (token.code_word != '\0')
    {
        decompressed += token.code_word;
    }
}

总结

核心问题在于文本模式读写破坏特殊字符、>>运算符跳过空白字符、未处理输入末尾越界。通过切换二进制模式、手动解析Token、转义特殊字符、处理边界越界,即可解决特殊字符处理异常问题。


内容的提问来源于stack exchange,提问作者wolfie00

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.03 18:55:54