UTF-16 LE BOM编码文件多线程字符替换乱码问题求助
UTF-16 LE BOM编码文件多线程字符替换乱码问题求助
我现在遇到一个编码相关的乱码问题,想请大家帮忙看看。我写了一个多线程程序,用来给UTF-16 LE BOM编码的文件每行末尾插入用户指定的字符,输出文件也保持UTF-16 LE BOM编码。现在的问题是:如果我在代码里直接硬编码替换字符(比如俄语字符),程序能正常工作;但如果从CMD命令行传入这个字符,输出的文件末尾就会变成乱码的象形文字,只有英文字符能正常传递。
我的源文件和输出文件都是UTF-16 LE BOM编码,VS的项目编码是多字节。看起来问题出在命令行参数的编码转换上,数据在传递过程中丢失了正确的编码信息。
下面是我的完整代码:
#define _CRT_SECURE_NO_WARNINGS #include <windows.h> #include <stdio.h> #include <stdlib.h> #include <string.h> #include <io.h> #include <fcntl.h> #define threads 64 #define BUF_SIZE 2 #define BOM_SIZE 2 // Size of BOM for UTF-16 LE // Global mutex HANDLE hMutex; // Structure for data exchange struct MYDATA { int num; wchar_t file[MAX_PATH]; wchar_t replacement; // Replacement character int replacementsCount; // Replacement counter } data[threads]; // Thread function DWORD WINAPI threadfunc(LPVOID param) { // Set console encoding to UTF-16 LE _setmode(_fileno(stdout), _O_U16TEXT); _setmode(_fileno(stdin), _O_U16TEXT); MYDATA* myData = (MYDATA*)param; wchar_t* filename = myData->file; wchar_t replacement = myData->replacement; // Replace space with the specified character HANDLE hIn = CreateFileW(filename, GENERIC_READ, 0, NULL, OPEN_EXISTING, 0, NULL); if (hIn == INVALID_HANDLE_VALUE) { wprintf(L"(Thread) Can't open file %s!\n", filename); return 0; // Move to the next file } wchar_t outputFilename[MAX_PATH]; wcscpy(outputFilename, filename); wchar_t* pos = wcsrchr(outputFilename, L'.'); if (pos == NULL) { wcscat(outputFilename, L".out"); } else { wcscpy(pos, L".out"); } HANDLE hOut = CreateFileW(outputFilename, GENERIC_WRITE, 0, NULL, CREATE_ALWAYS, FILE_ATTRIBUTE_NORMAL, NULL); if (hOut == INVALID_HANDLE_VALUE) { CloseHandle(hIn); wprintf(L"(Thread) Can't open output file %s!\n", outputFilename); return 0; // Move to the next file } // Write BOM to the output file const CHAR bom[BOM_SIZE] = { 0xFF, 0xFE }; // BOM for UTF-16 LE DWORD bytesWritten; WriteFile(hOut, bom, BOM_SIZE, &bytesWritten, NULL); wchar_t Buffer[BUF_SIZE]; WCHAR lineBuffer[BUF_SIZE * 1024]; // Assume that the line does not exceed 1024 characters DWORD nIn, nOut; INT lineLength = 0; while (ReadFile(hIn, Buffer, BUF_SIZE, &nIn, NULL) && nIn > 0) { if (nIn == 1) { wprintf(L"Encoding error in the input file\nThe encoding must be UTF-16 LE"); CloseHandle(hIn); CloseHandle(hOut); return -1; } // Check if the current character is a newline character if (Buffer[0] == '\r' && Buffer[1] == '\0') { // If this is a newline character, write it to the output file WriteFile(hOut, lineBuffer, lineLength * sizeof(WCHAR), &nOut, NULL); if (lineLength * sizeof(WCHAR) != nOut) { wprintf(L"Unrecoverable write error: %x\n", GetLastError()); CloseHandle(hIn); CloseHandle(hOut); return -1; } // Add the replacement character at the end of the line WriteFile(hOut, &replacement, sizeof(WCHAR), &nOut, NULL); if (sizeof(WCHAR) != nOut) { wprintf(L"Unrecoverable write error: %x\n", GetLastError()); CloseHandle(hIn); CloseHandle(hOut); return -1; } // Reset line length lineLength = 0; myData->replacementsCount++; // Increment the replacement counter } else { // Save the character to the line buffer lineBuffer[lineLength++] = (Buffer[1] << 8) | Buffer[0]; // Convert from UTF-16 LE } } // Write the remaining line if it is not empty if (lineLength > 0) { WriteFile(hOut, lineBuffer, lineLength * sizeof(WCHAR), &nOut, NULL); if (lineLength * sizeof(WCHAR) != nOut) { wprintf(L"Unrecoverable write error: %x\n", GetLastError()); CloseHandle(hIn); CloseHandle(hOut); return -1; } // Add the replacement character at the end of the last line WriteFile(hOut, &replacement, sizeof(WCHAR), &nOut, NULL); if (sizeof(WCHAR) != nOut) { wprintf(L"Unrecoverable write error: %x\n", GetLastError()); CloseHandle(hIn); CloseHandle(hOut); return -1; } } CloseHandle(hIn); CloseHandle(hOut); wprintf(L"(Thread) Processed file %s with %d replacements.\n", filename, myData->replacementsCount); return 0; } int main(int argc, char* argv[]) { // Set console encoding to UTF-16 LE _setmode(_fileno(stdout), _O_U16TEXT); _setmode(_fileno(stdin), _O_U16TEXT); // Thread handles HANDLE hThreads[threads]; // Auxiliary variables int i; if (argc < 3) { wprintf(L"Usage: laba7.exe filename replacement_char ...\n"); exit(-1); } // Create mutex hMutex = CreateMutex(NULL, FALSE, NULL); // Get the replacement character from command line arguments wchar_t replacementChar = (wchar_t)argv[argc - 1][0]; // Take the last argument as the replacement character for (i = 0; i < argc - 2; i++) { // Fill the data structure data[i].num = i; swprintf(data[i].file, MAX_PATH, L"%S", argv[i + 1]); // Convert from char to wchar_t data[i].replacement = replacementChar; // Set the replacement character data[i].replacementsCount = 0; // Initialize the replacement counter // Create thread hThreads[i] = CreateThread(NULL, 0, threadfunc, (LPVOID)&data[i], 0, NULL); if (hThreads[i] == NULL) { wprintf(L"Can't create thread %d!\n", i); exit(-i); } else { wprintf(L"(Main) Created thread %d!\n", i); } } // Close thread handles for (i = 0; i < argc - 2; i++) { // Ensure all threads have finished WaitForSingleObject(hThreads[i], INFINITE); CloseHandle(hThreads[i]); } // Release the mutex CloseHandle(hMutex); return 0; }
问题根源分析
问题的核心出在命令行参数的编码转换上:
- Windows的CMD默认使用系统区域设置的OEM编码(比如俄语系统是CP866),而你的项目用了多字节字符集,
argv里的参数是单字节OEM编码的char类型。 - 你直接把
argv[argc-1][0]强转成wchar_t,没有做正确的编码转换,非英文字符会因为编码不匹配被转成错误的Unicode值,最终写入文件就变成了乱码。 - 另外线程里重复设置控制台编码是冗余操作,控制台编码是全局的,只需要在主函数设置一次即可。
解决方案
1. 改用宽字符版主函数wmain
Windows原生支持wmain,它接收的argv是wchar_t*类型,直接对应UTF-16编码,能直接拿到命令行传入的正确Unicode字符,完全避免编码转换错误。
2. 修正参数处理逻辑
因为用了wmain,获取替换字符的代码可以直接写成:
wchar_t replacementChar = argv[argc - 1][0];
不需要再做错误的char转wchar_t操作。
3. 优化代码细节
- 移除线程函数里的
_setmode调用,只在wmain开头设置一次控制台编码。 - 用安全的
wcscpy_s替代wcscpy,避免缓冲区溢出风险。 - 修正换行处理逻辑:UTF-16 LE的标准换行是
\r\n(两个Unicode字符),之前的代码只处理了\r,可能导致换行异常。
修改后的完整可运行代码
#define _CRT_SECURE_NO_WARNINGS #include <windows.h> #include <stdio.h> #include <stdlib.h> #include <string.h> #include <io.h> #include <fcntl.h> #define threads 64 #define BUF_SIZE 2 #define BOM_SIZE 2 // Size of BOM for UTF-16 LE // Global mutex HANDLE hMutex; // Structure for data exchange struct MYDATA { int num; wchar_t file[MAX_PATH]; wchar_t replacement; // Replacement character int replacementsCount; // Replacement counter } data[threads]; // Thread function DWORD WINAPI threadfunc(LPVOID param) { MYDATA* myData = (MYDATA*)param; wchar_t* filename = myData->file; wchar_t replacement = myData->replacement; HANDLE hIn = CreateFileW(filename, GENERIC_READ, 0, NULL, OPEN_EXISTING, 0, NULL); if (hIn == INVALID_HANDLE_VALUE) { wprintf(L"(Thread) Can't open file %s!\n", filename); return 0; } wchar_t outputFilename[MAX_PATH]; wcscpy_s(outputFilename, MAX_PATH, filename); wchar_t* pos = wcsrchr(outputFilename, L'.'); if (pos == NULL) { wcscat_s(outputFilename, MAX_PATH, L".out"); } else { wcscpy_s(pos, outputFilename + MAX_PATH - pos, L".out"); } HANDLE hOut = CreateFileW(outputFilename, GENERIC_WRITE, 0, NULL, CREATE_ALWAYS, FILE_ATTRIBUTE_NORMAL, NULL); if (hOut == INVALID_HANDLE_VALUE) { CloseHandle(hIn); wprintf(L"(Thread) Can't open output file %s!\n", outputFilename); return 0; } // Write BOM to the output file const BYTE bom[BOM_SIZE] = { 0xFF, 0xFE }; // BOM for UTF-16 LE DWORD bytesWritten; WriteFile(hOut, bom, BOM_SIZE, &bytesWritten, NULL); BYTE rawBuffer[BUF_SIZE]; // 存储原始字节,手动控制UTF-16 LE转wchar_t wchar_t lineBuffer[BUF_SIZE * 1024]; DWORD nIn, nOut; INT lineLength = 0; while (ReadFile(hIn, rawBuffer, BUF_SIZE, &nIn, NULL) && nIn > 0) { if (nIn == 1) { wprintf(L"Encoding error in the input file\nThe encoding must be UTF-16 LE"); CloseHandle(hIn); CloseHandle(hOut); return -1; } // 手动将UTF-16 LE字节转成wchar_t wchar_t currentChar = (wchar_t)((rawBuffer[1] << 8) | rawBuffer[0]); // 处理完整的\r\n换行对 if (currentChar == L'\r') { // 写入当前行内容 WriteFile(hOut, lineBuffer, lineLength * sizeof(wchar_t), &nOut, NULL); if (lineLength * sizeof(wchar_t) != nOut) { wprintf(L"Unrecoverable write error: %x\n", GetLastError()); CloseHandle(hIn); CloseHandle(hOut); return -1; } // 写入替换字符 WriteFile(hOut, &replacement, sizeof(wchar_t), &nOut, NULL); if (sizeof(wchar_t) != nOut) { wprintf(L"Unrecoverable write error: %x\n", GetLastError()); CloseHandle(hIn); CloseHandle(hOut); return -1; } // 重置行缓冲区 lineLength = 0; myData->replacementsCount++; // 读取并写入后续的\n字符 ReadFile(hIn, rawBuffer, BUF_SIZE, &nIn, NULL); wchar_t newLineChar = (wchar_t)((rawBuffer[1] << 8) | rawBuffer[0]); WriteFile(hOut, &newLineChar, sizeof(wchar_t), &nOut, NULL); } else if (currentChar != L'\n') { // 普通字符存入行缓冲区 lineBuffer[lineLength++] = currentChar; } } // 处理最后一行内容 if (lineLength > 0) { WriteFile(hOut, lineBuffer, lineLength * sizeof(wchar_t), &nOut, NULL); if (lineLength * sizeof(wchar_t) != nOut) { wprintf(L"Unrecoverable write error: %x\n", GetLastError()); CloseHandle(hIn); CloseHandle(hOut); return -1; } // 写入最后一行的替换字符 WriteFile(hOut, &replacement, sizeof(wchar_t), &nOut, NULL); if (sizeof(wchar_t) != nOut) { wprintf(L"Unrecoverable write error: %x\n", GetLastError()); CloseHandle(hIn); CloseHandle(hOut); return -1; } myData->replacementsCount++; } CloseHandle(hIn); CloseHandle(hOut); wprintf(L"(Thread) Processed file %s with %d replacements.\n", filename, myData->replacementsCount); return 0; } int wmain(int argc, wchar_t* argv[]) { // Set console encoding to UTF-16 LE _setmode(_fileno(stdout), _O_U16TEXT); _setmode(_fileno(stdin), _O_U16TEXT); HANDLE hThreads[threads]; int i; if (argc < 3) { wprintf(L"Usage: laba7.exe filename replacement_char ...\n"); exit(-1); } hMutex = CreateMutex(NULL, FALSE, NULL); // 直接从宽字符参数获取替换字符 wchar_t replacementChar = argv[argc - 1][0]; for (i = 0; i < argc - 2; i++) { data[i].num = i; wcscpy_s(data[i].file, MAX_PATH, argv[i + 1]); data[i].replacement = replacementChar; data[i].replacementsCount = 0; hThreads[i] = CreateThread(NULL, 0, threadfunc, (LPVOID)&data[i], 0, NULL); if (hThreads[i] == NULL) { wprintf(L"Can't create thread %d!\n", i); exit(-i); } else { wprintf(L"(Main) Created thread %d!\n", i); } } // 等待所有线程完成并关闭句柄 for (i = 0; i < argc - 2; i++) { WaitForSingleObject(hThreads[i], INFINITE); CloseHandle(hThreads[i]); } CloseHandle(hMutex); return 0; }
经过这些修改后,从CMD传入俄语等非英文字符时,就能正确传递到程序中,输出文件末尾的替换字符也不会再出现乱码了。
备注:内容来源于stack exchange,提问作者Marchenko
相关产品推荐
相关产品推荐

