You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

UTF-16 LE BOM编码文件多线程字符替换乱码问题求助

UTF-16 LE BOM编码文件多线程字符替换乱码问题求助

我现在遇到一个编码相关的乱码问题,想请大家帮忙看看。我写了一个多线程程序,用来给UTF-16 LE BOM编码的文件每行末尾插入用户指定的字符,输出文件也保持UTF-16 LE BOM编码。现在的问题是:如果我在代码里直接硬编码替换字符(比如俄语字符),程序能正常工作;但如果从CMD命令行传入这个字符,输出的文件末尾就会变成乱码的象形文字,只有英文字符能正常传递。

我的源文件和输出文件都是UTF-16 LE BOM编码,VS的项目编码是多字节。看起来问题出在命令行参数的编码转换上,数据在传递过程中丢失了正确的编码信息。

下面是我的完整代码:

#define _CRT_SECURE_NO_WARNINGS
#include <windows.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <io.h>
#include <fcntl.h>

 #define threads 64
 #define BUF_SIZE 2
   #define BOM_SIZE 2 // Size of BOM for UTF-16 LE

   // Global mutex
   HANDLE hMutex;
    
      // Structure for data exchange
      struct MYDATA {
        int num;
wchar_t file[MAX_PATH];
wchar_t replacement; // Replacement character
int replacementsCount; // Replacement counter
    } data[threads];

  // Thread function
    DWORD WINAPI threadfunc(LPVOID param) {
// Set console encoding to UTF-16 LE
_setmode(_fileno(stdout), _O_U16TEXT);
_setmode(_fileno(stdin), _O_U16TEXT);

MYDATA* myData = (MYDATA*)param;
wchar_t* filename = myData->file;
wchar_t replacement = myData->replacement; // Replace space with the specified 
   character

HANDLE hIn = CreateFileW(filename, GENERIC_READ, 0, NULL, OPEN_EXISTING, 0, NULL);
if (hIn == INVALID_HANDLE_VALUE) {
    wprintf(L"(Thread) Can't open file %s!\n", filename);
    return 0; // Move to the next file
}

wchar_t outputFilename[MAX_PATH];
wcscpy(outputFilename, filename);
wchar_t* pos = wcsrchr(outputFilename, L'.');
if (pos == NULL) {
    wcscat(outputFilename, L".out");
}
else {
    wcscpy(pos, L".out");
}

HANDLE hOut = CreateFileW(outputFilename, GENERIC_WRITE, 0, NULL, CREATE_ALWAYS, 
   FILE_ATTRIBUTE_NORMAL, NULL);
if (hOut == INVALID_HANDLE_VALUE) {
    CloseHandle(hIn);
    wprintf(L"(Thread) Can't open output file %s!\n", outputFilename);
    return 0; // Move to the next file
}

// Write BOM to the output file
const CHAR bom[BOM_SIZE] = { 0xFF, 0xFE }; // BOM for UTF-16 LE
DWORD bytesWritten;
WriteFile(hOut, bom, BOM_SIZE, &bytesWritten, NULL);

wchar_t Buffer[BUF_SIZE];
WCHAR lineBuffer[BUF_SIZE * 1024]; // Assume that the line does not exceed 1024 
    characters
DWORD nIn, nOut;
INT lineLength = 0;

while (ReadFile(hIn, Buffer, BUF_SIZE, &nIn, NULL) && nIn > 0) {
    if (nIn == 1) {
        wprintf(L"Encoding error in the input file\nThe encoding must be UTF-16 LE");
        CloseHandle(hIn);
        CloseHandle(hOut);
        return -1;
    }

    // Check if the current character is a newline character
    if (Buffer[0] == '\r' && Buffer[1] == '\0') {
        // If this is a newline character, write it to the output file
        WriteFile(hOut, lineBuffer, lineLength * sizeof(WCHAR), &nOut, NULL);
        if (lineLength * sizeof(WCHAR) != nOut) {
            wprintf(L"Unrecoverable write error: %x\n", GetLastError());
            CloseHandle(hIn);
            CloseHandle(hOut);
            return -1;
        }

        // Add the replacement character at the end of the line
        WriteFile(hOut, &replacement, sizeof(WCHAR), &nOut, NULL);
        if (sizeof(WCHAR) != nOut) {
            wprintf(L"Unrecoverable write error: %x\n", GetLastError());
            CloseHandle(hIn);
            CloseHandle(hOut);
            return -1;
        }

        // Reset line length
        lineLength = 0;
        myData->replacementsCount++; // Increment the replacement counter
    }
    else {
        // Save the character to the line buffer
        lineBuffer[lineLength++] = (Buffer[1] << 8) | Buffer[0]; // Convert from UTF-16 
    LE
    }
}

// Write the remaining line if it is not empty
if (lineLength > 0) {
    WriteFile(hOut, lineBuffer, lineLength * sizeof(WCHAR), &nOut, NULL);
    if (lineLength * sizeof(WCHAR) != nOut) {
        wprintf(L"Unrecoverable write error: %x\n", GetLastError());
        CloseHandle(hIn);
        CloseHandle(hOut);
        return -1;
    }

           // Add the replacement character at the end of the last line
    WriteFile(hOut, &replacement, sizeof(WCHAR), &nOut, NULL);
    if (sizeof(WCHAR) != nOut) {
        wprintf(L"Unrecoverable write error: %x\n", GetLastError());
        CloseHandle(hIn);
        CloseHandle(hOut);
        return -1;
    }
}

CloseHandle(hIn);
CloseHandle(hOut);
wprintf(L"(Thread) Processed file %s with %d replacements.\n", filename, myData->replacementsCount);
return 0;
 }

 int main(int argc, char* argv[]) {
// Set console encoding to UTF-16 LE
_setmode(_fileno(stdout), _O_U16TEXT);
_setmode(_fileno(stdin), _O_U16TEXT);

// Thread handles
HANDLE hThreads[threads];
// Auxiliary variables
int i;

if (argc < 3) {
    wprintf(L"Usage: laba7.exe filename replacement_char ...\n");
    exit(-1);
}

// Create mutex
hMutex = CreateMutex(NULL, FALSE, NULL);

// Get the replacement character from command line arguments
wchar_t replacementChar = (wchar_t)argv[argc - 1][0]; // Take the last argument as the 
 replacement character

   for (i = 0; i < argc - 2; i++) {
    // Fill the data structure
    data[i].num = i;
    swprintf(data[i].file, MAX_PATH, L"%S", argv[i + 1]); // Convert from char to 
    wchar_t
    data[i].replacement = replacementChar; // Set the replacement character
    data[i].replacementsCount = 0; // Initialize the replacement counter

    // Create thread
    hThreads[i] = CreateThread(NULL, 0, threadfunc, (LPVOID)&data[i], 0, NULL);
    if (hThreads[i] == NULL) {
        wprintf(L"Can't create thread %d!\n", i);
        exit(-i);
    }
    else {
        wprintf(L"(Main) Created thread %d!\n", i);
    }
}

// Close thread handles
for (i = 0; i < argc - 2; i++) {
    // Ensure all threads have finished
    WaitForSingleObject(hThreads[i], INFINITE);
    CloseHandle(hThreads[i]);
}

// Release the mutex
CloseHandle(hMutex);
return 0;
 }

问题根源分析

问题的核心出在命令行参数的编码转换上:

  1. Windows的CMD默认使用系统区域设置的OEM编码(比如俄语系统是CP866),而你的项目用了多字节字符集,argv里的参数是单字节OEM编码的char类型。
  2. 你直接把argv[argc-1][0]强转成wchar_t,没有做正确的编码转换,非英文字符会因为编码不匹配被转成错误的Unicode值,最终写入文件就变成了乱码。
  3. 另外线程里重复设置控制台编码是冗余操作,控制台编码是全局的,只需要在主函数设置一次即可。

解决方案

1. 改用宽字符版主函数wmain

Windows原生支持wmain,它接收的argv是wchar_t*类型,直接对应UTF-16编码,能直接拿到命令行传入的正确Unicode字符,完全避免编码转换错误。

2. 修正参数处理逻辑

因为用了wmain,获取替换字符的代码可以直接写成:

wchar_t replacementChar = argv[argc - 1][0];

不需要再做错误的char转wchar_t操作。

3. 优化代码细节

  • 移除线程函数里的_setmode调用,只在wmain开头设置一次控制台编码。
  • 用安全的wcscpy_s替代wcscpy,避免缓冲区溢出风险。
  • 修正换行处理逻辑:UTF-16 LE的标准换行是\r\n(两个Unicode字符),之前的代码只处理了\r,可能导致换行异常。

修改后的完整可运行代码

#define _CRT_SECURE_NO_WARNINGS
#include <windows.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <io.h>
#include <fcntl.h>

#define threads 64
#define BUF_SIZE 2
#define BOM_SIZE 2 // Size of BOM for UTF-16 LE

// Global mutex
HANDLE hMutex;

// Structure for data exchange
struct MYDATA {
    int num;
    wchar_t file[MAX_PATH];
    wchar_t replacement; // Replacement character
    int replacementsCount; // Replacement counter
} data[threads];

// Thread function
DWORD WINAPI threadfunc(LPVOID param) {
    MYDATA* myData = (MYDATA*)param;
    wchar_t* filename = myData->file;
    wchar_t replacement = myData->replacement;

    HANDLE hIn = CreateFileW(filename, GENERIC_READ, 0, NULL, OPEN_EXISTING, 0, NULL);
    if (hIn == INVALID_HANDLE_VALUE) {
        wprintf(L"(Thread) Can't open file %s!\n", filename);
        return 0;
    }

    wchar_t outputFilename[MAX_PATH];
    wcscpy_s(outputFilename, MAX_PATH, filename);
    wchar_t* pos = wcsrchr(outputFilename, L'.');
    if (pos == NULL) {
        wcscat_s(outputFilename, MAX_PATH, L".out");
    }
    else {
        wcscpy_s(pos, outputFilename + MAX_PATH - pos, L".out");
    }

    HANDLE hOut = CreateFileW(outputFilename, GENERIC_WRITE, 0, NULL, CREATE_ALWAYS, FILE_ATTRIBUTE_NORMAL, NULL);
    if (hOut == INVALID_HANDLE_VALUE) {
        CloseHandle(hIn);
        wprintf(L"(Thread) Can't open output file %s!\n", outputFilename);
        return 0;
    }

    // Write BOM to the output file
    const BYTE bom[BOM_SIZE] = { 0xFF, 0xFE }; // BOM for UTF-16 LE
    DWORD bytesWritten;
    WriteFile(hOut, bom, BOM_SIZE, &bytesWritten, NULL);

    BYTE rawBuffer[BUF_SIZE]; // 存储原始字节,手动控制UTF-16 LE转wchar_t
    wchar_t lineBuffer[BUF_SIZE * 1024];
    DWORD nIn, nOut;
    INT lineLength = 0;

    while (ReadFile(hIn, rawBuffer, BUF_SIZE, &nIn, NULL) && nIn > 0) {
        if (nIn == 1) {
            wprintf(L"Encoding error in the input file\nThe encoding must be UTF-16 LE");
            CloseHandle(hIn);
            CloseHandle(hOut);
            return -1;
        }

        // 手动将UTF-16 LE字节转成wchar_t
        wchar_t currentChar = (wchar_t)((rawBuffer[1] << 8) | rawBuffer[0]);

        // 处理完整的\r\n换行对
        if (currentChar == L'\r') {
            // 写入当前行内容
            WriteFile(hOut, lineBuffer, lineLength * sizeof(wchar_t), &nOut, NULL);
            if (lineLength * sizeof(wchar_t) != nOut) {
                wprintf(L"Unrecoverable write error: %x\n", GetLastError());
                CloseHandle(hIn);
                CloseHandle(hOut);
                return -1;
            }

            // 写入替换字符
            WriteFile(hOut, &replacement, sizeof(wchar_t), &nOut, NULL);
            if (sizeof(wchar_t) != nOut) {
                wprintf(L"Unrecoverable write error: %x\n", GetLastError());
                CloseHandle(hIn);
                CloseHandle(hOut);
                return -1;
            }

            // 重置行缓冲区
            lineLength = 0;
            myData->replacementsCount++;

            // 读取并写入后续的\n字符
            ReadFile(hIn, rawBuffer, BUF_SIZE, &nIn, NULL);
            wchar_t newLineChar = (wchar_t)((rawBuffer[1] << 8) | rawBuffer[0]);
            WriteFile(hOut, &newLineChar, sizeof(wchar_t), &nOut, NULL);
        }
        else if (currentChar != L'\n') {
            // 普通字符存入行缓冲区
            lineBuffer[lineLength++] = currentChar;
        }
    }

    // 处理最后一行内容
    if (lineLength > 0) {
        WriteFile(hOut, lineBuffer, lineLength * sizeof(wchar_t), &nOut, NULL);
        if (lineLength * sizeof(wchar_t) != nOut) {
            wprintf(L"Unrecoverable write error: %x\n", GetLastError());
            CloseHandle(hIn);
            CloseHandle(hOut);
            return -1;
        }

        // 写入最后一行的替换字符
        WriteFile(hOut, &replacement, sizeof(wchar_t), &nOut, NULL);
        if (sizeof(wchar_t) != nOut) {
            wprintf(L"Unrecoverable write error: %x\n", GetLastError());
            CloseHandle(hIn);
            CloseHandle(hOut);
            return -1;
        }
        myData->replacementsCount++;
    }

    CloseHandle(hIn);
    CloseHandle(hOut);
    wprintf(L"(Thread) Processed file %s with %d replacements.\n", filename, myData->replacementsCount);
    return 0;
}

int wmain(int argc, wchar_t* argv[]) {
    // Set console encoding to UTF-16 LE
    _setmode(_fileno(stdout), _O_U16TEXT);
    _setmode(_fileno(stdin), _O_U16TEXT);

    HANDLE hThreads[threads];
    int i;

    if (argc < 3) {
        wprintf(L"Usage: laba7.exe filename replacement_char ...\n");
        exit(-1);
    }

    hMutex = CreateMutex(NULL, FALSE, NULL);

    // 直接从宽字符参数获取替换字符
    wchar_t replacementChar = argv[argc - 1][0];

    for (i = 0; i < argc - 2; i++) {
        data[i].num = i;
        wcscpy_s(data[i].file, MAX_PATH, argv[i + 1]);
        data[i].replacement = replacementChar;
        data[i].replacementsCount = 0;

        hThreads[i] = CreateThread(NULL, 0, threadfunc, (LPVOID)&data[i], 0, NULL);
        if (hThreads[i] == NULL) {
            wprintf(L"Can't create thread %d!\n", i);
            exit(-i);
        }
        else {
            wprintf(L"(Main) Created thread %d!\n", i);
        }
    }

    // 等待所有线程完成并关闭句柄
    for (i = 0; i < argc - 2; i++) {
        WaitForSingleObject(hThreads[i], INFINITE);
        CloseHandle(hThreads[i]);
    }

    CloseHandle(hMutex);
    return 0;
}

经过这些修改后,从CMD传入俄语等非英文字符时,就能正确传递到程序中,输出文件末尾的替换字符也不会再出现乱码了。

备注:内容来源于stack exchange,提问作者Marchenko

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.16 02:59:31