跨平台C++异常与基础日志设施的首选字符类型问询
嘿,作为有多年跨平台C++开发经验的人,我来聊聊针对你问题的实用方案——毕竟我也踩过不少跨平台编码的坑,尤其是在异常处理这块:
问题1:基础异常类的编码选择
我的建议是让异常类的内部存储与平台原生编码对齐,同时对外提供双接口,具体来说:
- Windows平台:内部用
std::wstring(UTF-16LE)存储异常信息,契合WinAPI、控制台/文件I/O的原生适配习惯,避免频繁转换带来的损耗和编码丢失。 - Unix/Linux/macOS平台:内部用
std::string(UTF-8)存储,贴合Unix生态对UTF-8的普遍偏好,无需额外转换就能对接系统工具和日志链路。
同时给异常类提供两类核心接口:
- 接受平台原生字符串类型的构造/访问接口(Windows用
std::wstring,Unix用std::string),让本地开发者写代码时更自然,不用额外适配。 - 接受UTF-8编码
std::string的通用接口,方便跨平台模块或通用代码调用,内部自动转换为平台原生编码存储。
这样既照顾了两边开发者的使用习惯,又能在跨场景下保持编码一致性,避免转译丢失。
问题2:无依赖的跨平台字符串转换(避开弃用的
std::wstring_convert) 既然C17弃用了std::wstring_convert,且你不想依赖平台API,最靠谱的方式是基于UTF编码规范手动实现转换逻辑——毕竟UTF-8、UTF-16、UTF-32的编码规则是公开标准,实现起来并不复杂,且完全依赖C标准库的基础类型。
首先明确各平台的字符串编码映射:
- Windows:
std::wstring= UTF-16LE(wchar_t为16位) - Unix/Linux/macOS:
std::wstring= UTF-32(wchar_t为32位) - 跨平台通用:
std::string= UTF-8(约定俗成的跨平台编码)
核心转换函数实现示例
可以封装一个跨平台的转换工具命名空间,用条件编译区分平台:
#include <string> #include <cstdint> #include <stdexcept> namespace utf_convert { // UTF-8转UTF-16LE(Windows专用) std::wstring utf8_to_utf16(const std::string& utf8) { std::wstring result; const uint8_t* data = reinterpret_cast<const uint8_t*>(utf8.data()); size_t len = utf8.size(); size_t i = 0; while (i < len) { uint32_t codepoint; if (data[i] <= 0x7F) { codepoint = data[i++]; } else if ((data[i] & 0xE0) == 0xC0) { if (i + 1 >= len) throw std::invalid_argument("Invalid UTF-8 sequence"); codepoint = ((data[i++] & 0x1F) << 6) | (data[i++] & 0x3F); } else if ((data[i] & 0xF0) == 0xE0) { if (i + 2 >= len) throw std::invalid_argument("Invalid UTF-8 sequence"); codepoint = ((data[i++] & 0x0F) << 12) | ((data[i++] & 0x3F) << 6) | (data[i++] & 0x3F); } else if ((data[i] & 0xF8) == 0xF0) { if (i + 3 >= len) throw std::invalid_argument("Invalid UTF-8 sequence"); codepoint = ((data[i++] & 0x07) << 18) | ((data[i++] & 0x3F) << 12) | ((data[i++] & 0x3F) << 6) | (data[i++] & 0x3F); } else { throw std::invalid_argument("Invalid UTF-8 sequence"); } // 处理UTF-16代理对 if (codepoint <= 0xFFFF) { result.push_back(static_cast<wchar_t>(codepoint)); } else { codepoint -= 0x10000; result.push_back(static_cast<wchar_t>(0xD800 + (codepoint >> 10))); result.push_back(static_cast<wchar_t>(0xDC00 + (codepoint & 0x3FF))); } } return result; } // UTF-16LE转UTF-8(Windows专用) std::string utf16_to_utf8(const std::wstring& utf16) { std::string result; const wchar_t* data = utf16.data(); size_t len = utf16.size(); size_t i = 0; while (i < len) { uint32_t codepoint; wchar_t c = data[i++]; if (c >= 0xD800 && c <= 0xDBFF) { if (i >= len) throw std::invalid_argument("Invalid UTF-16 sequence"); wchar_t d = data[i++]; if (d < 0xDC00 || d > 0xDFFF) throw std::invalid_argument("Invalid UTF-16 sequence"); codepoint = ((c - 0xD800) << 10) + (d - 0xDC00) + 0x10000; } else { codepoint = c; } // 转换为UTF-8字节序列 if (codepoint <= 0x7F) { result.push_back(static_cast<char>(codepoint)); } else if (codepoint <= 0x7FF) { result.push_back(static_cast<char>(0xC0 | (codepoint >> 6))); result.push_back(static_cast<char>(0x80 | (codepoint & 0x3F))); } else if (codepoint <= 0xFFFF) { result.push_back(static_cast<char>(0xE0 | (codepoint >> 12))); result.push_back(static_cast<char>(0x80 | ((codepoint >> 6) & 0x3F))); result.push_back(static_cast<char>(0x80 | (codepoint & 0x3F))); } else if (codepoint <= 0x10FFFF) { result.push_back(static_cast<char>(0xF0 | (codepoint >> 18))); result.push_back(static_cast<char>(0x80 | ((codepoint >> 12) & 0x3F))); result.push_back(static_cast<char>(0x80 | ((codepoint >> 6) & 0x3F))); result.push_back(static_cast<char>(0x80 | (codepoint & 0x3F))); } else { throw std::invalid_argument("Invalid Unicode codepoint"); } } return result; } // UTF-8转UTF-32(Unix专用) std::wstring utf8_to_utf32(const std::string& utf8) { std::wstring result; const uint8_t* data = reinterpret_cast<const uint8_t*>(utf8.data()); size_t len = utf8.size(); size_t i = 0; while (i < len) { uint32_t codepoint; if (data[i] <= 0x7F) { codepoint = data[i++]; } else if ((data[i] & 0xE0) == 0xC0) { if (i + 1 >= len) throw std::invalid_argument("Invalid UTF-8 sequence"); codepoint = ((data[i++] & 0x1F) << 6) | (data[i++] & 0x3F); } else if ((data[i] & 0xF0) == 0xE0) { if (i + 2 >= len) throw std::invalid_argument("Invalid UTF-8 sequence"); codepoint = ((data[i++] & 0x0F) << 12) | ((data[i++] & 0x3F) << 6) | (data[i++] & 0x3F); } else if ((data[i] & 0xF8) == 0xF0) { if (i + 3 >= len) throw std::invalid_argument("Invalid UTF-8 sequence"); codepoint = ((data[i++] & 0x07) << 18) | ((data[i++] & 0x3F) << 12) | ((data[i++] & 0x3F) << 6) | (data[i++] & 0x3F); } else { throw std::invalid_argument("Invalid UTF-8 sequence"); } result.push_back(static_cast<wchar_t>(codepoint)); } return result; } // UTF-32转UTF-8(Unix专用) std::string utf32_to_utf8(const std::wstring& utf32) { std::string result; for (wchar_t c : utf32) { uint32_t codepoint = static_cast<uint32_t>(c); if (codepoint <= 0x7F) { result.push_back(static_cast<char>(codepoint)); } else if (codepoint <= 0x7FF) { result.push_back(static_cast<char>(0xC0 | (codepoint >> 6))); result.push_back(static_cast<char>(0x80 | (codepoint & 0x3F))); } else if (codepoint <= 0xFFFF) { result.push_back(static_cast<char>(0xE0 | (codepoint >> 12))); result.push_back(static_cast<char>(0x80 | ((codepoint >> 6) & 0x3F))); result.push_back(static_cast<char>(0x80 | (codepoint & 0x3F))); } else if (codepoint <= 0x10FFFF) { result.push_back(static_cast<char>(0xF0 | (codepoint >> 18))); result.push_back(static_cast<char>(0x80 | ((codepoint >> 12) & 0x3F))); result.push_back(static_cast<char>(0x80 | ((codepoint >> 6) & 0x3F))); result.push_back(static_cast<char>(0x80 | (codepoint & 0x3F))); } else { throw std::invalid_argument("Invalid Unicode codepoint"); } } return result; } }
跨平台异常类实现
基于上面的转换工具,实现一个适配两边的异常类:
#include <exception> #include <string> #include "utf_convert.h" // 假设上面的转换函数放在这个头文件里 #ifdef _WIN32 using native_string = std::wstring; #else using native_string = std::string; #endif class BaseException : public std::exception { private: native_string native_msg_; std::string utf8_msg_; // 缓存UTF-8版本,避免重复转换 public: // 接受平台原生字符串的构造函数 explicit BaseException(native_string msg) : native_msg_(std::move(msg)) { #ifdef _WIN32 utf8_msg_ = utf_convert::utf16_to_utf8(native_msg_); #else utf8_msg_ = native_msg_; // Unix下原生就是UTF-8 #endif } // 接受UTF-8字符串的通用构造函数 explicit BaseException(const std::string& utf8_msg) : utf8_msg_(utf8_msg) { #ifdef _WIN32 native_msg_ = utf_convert::utf8_to_utf16(utf8_msg_); #else native_msg_ = utf8_msg_; #endif } // 重写std::exception的what(),统一返回UTF-8编码的字符串 const char* what() const noexcept override { return utf8_msg_.c_str(); } // 获取平台原生编码的异常信息 const native_string& native_msg() const noexcept { return native_msg_; } };
额外小贴士
- 日志系统:不管是Windows还是Unix,统一用UTF-8写入日志文件,避免编码混乱;Windows下如果要输出到控制台,可以将UTF-8转成UTF-16调用
WriteConsoleW,或者设置控制台代码页为65001(UTF-8)后用printf输出。 - GUI适配:Windows下GUI调用WinAPI时,直接用
native_msg()获取UTF-16LE字符串,无需额外转换;Unix下GUI(比如GTK、Qt)本身偏好UTF-8,直接用what()的结果即可。
内容的提问来源于stack exchange,提问作者Mordachai
相关产品推荐
相关产品推荐

