You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Windows下C语言读写Unicode字符到文件乱码问题求助

Windows下MinGW32保存/读取中文Unicode字符异常的问题与修复

问题描述

编写C语言程序将Unicode字符保存到data.txt,重新加载后在终端打印。使用MinGW32编译器在Windows系统运行时,输入英文、北欧字符(åäö)正常,但输入中文“买买买”时读取结果乱码。

测试代码

#if defined(_MSC_VER) || defined(__MINGW32__) || defined(__MINGW64__)
#define ON_WINDOWS
#endif

#ifdef ON_WINDOWS
#define _CRT_SECURE_NO_WARNINGS
#include <io.h>     // _setmode
#include <fcntl.h>  // _O_U16TEXT

// just in case mingw doesn't define it after all:
#ifndef _O_U16TEXT
#define _O_U16TEXT (0x20000)
#endif
#endif

#include <stdio.h>
#include <locale.h>
#include <wchar.h>
#define SIZE 100

void set_locale_mode() {
#ifdef ON_WINDOWS
   // Unicode UTF-16, little endian byte order (BMP of ISO 10646)
   const char* CP_UTF_16LE = ".1200";

   setlocale(LC_ALL, CP_UTF_16LE);
   _setmode(_fileno(stdin), _O_U16TEXT);
   _setmode(_fileno(stdout), _O_U16TEXT);
#else
   setlocale(LC_ALL, "");
#endif
}

int main(void) {

   set_locale_mode();

   wchar_t myString[SIZE];
   wchar_t loadedString[SIZE];

   wprintf(L"Enter 3 characters: ");
   wscanf(L"%ls", myString);
   wprintf(L"Your input is %ls\n", myString);

   FILE *pFile;

   if(pFile=fopen("data.txt", "w")) {
      fwprintf(pFile, L"%ls", myString);
   } else {
      wprintf(L"Failed to write to file!\n");
   }

   fclose(pFile);

   if(pFile=fopen("data.txt", "r")) {
      fwscanf(pFile, L"%ls", loadedString);
   } else {
    wprintf(L"Failed to read from file!\n");
   } 

   fclose(pFile);

   wprintf(L"Loaded string is %ls", loadedString);

   return 0;
}

测试结果

  • 输入abc/åäö时输出正常:
Enter 3 characters: abc
Your input is abc
Loaded string is abc
  • 输入中文“买买买”时异常:
Enter 3 characters: 买买买
Your input is 买买买
Loaded string is 炨}ﺨa眬☺

问题原因

Windows系统下,C标准库处理文本模式的宽字符文件时,依赖UTF-16的BOM(字节顺序标记)识别文件编码。当前代码用fwprintf写入UTF-16LE格式的宽字符,但未写入BOM(0xFFFE),导致读取时fwscanf默认使用系统本地ANSI编码(如GBK)解析文件内容。中文的UTF-16LE编码与GBK编码不兼容,因此出现乱码;而åäö等字符的UTF-16LE编码恰好与GBK的部分编码重合,所以未触发异常。

修复方案

方案1:使用_wfopen指定文件编码(推荐)

通过_wfopen函数的ccs=UTF-16LE参数,强制文件以UTF-16LE编码读写,系统会自动处理BOM的写入和识别:

修改文件读写部分代码:

// 写入文件
if(pFile = _wfopen(L"data.txt", L"w, ccs=UTF-16LE")) {
    fwprintf(pFile, L"%ls", myString);
} else {
    wprintf(L"Failed to write to file!\n");
}

fclose(pFile);

// 读取文件
if(pFile = _wfopen(L"data.txt", L"r, ccs=UTF-16LE")) {
    fwscanf(pFile, L"%ls", loadedString);
} else {
    wprintf(L"Failed to read from file!\n");
}

方案2:手动写入/跳过BOM

如果不想使用_wfopen,可以手动写入UTF-16LE的BOM,并在读取时跳过:

// 写入文件(二进制模式确保字节不被转换)
if(pFile=fopen("data.txt", "wb")) {
    // 写入UTF-16LE BOM(小端字节序的BOM为0xFFFE)
    fputwc(0xFEFF, pFile);
    fwprintf(pFile, L"%ls", myString);
} else {
    wprintf(L"Failed to write to file!\n");
}

fclose(pFile);

// 读取文件
if(pFile=fopen("data.txt", "rb")) {
    wchar_t bom;
    // 读取BOM
    fread(&bom, sizeof(wchar_t), 1, pFile);
    // 如果BOM正确,继续读取;否则回到文件开头
    if(bom != 0xFEFF) {
        fseek(pFile, 0, SEEK_SET);
    }
    fwscanf(pFile, L"%ls", loadedString);
} else {
    wprintf(L"Failed to read from file!\n");
}

注意事项

  • MinGW32环境下_wfopen函数需要确保头文件包含完整,编译时无需额外参数。
  • 终端输出的正确性依赖set_locale_mode中设置的UTF-16LE locale和_setmode配置,当前代码这部分逻辑是正确的。

内容的提问来源于stack exchange,提问作者Alexander Jonsson

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.17 13:02:14