传递结构体元素时触发iconv: Invalid argument错误的原因排查
为什么传递track_data->title给to_utf8会触发iconv无效参数错误?
我实现了一个基于iconv库的to_utf8函数,用于将UTF-16字符串转换为UTF-8。手动构造字符串调用该函数时转换正常,但从Track结构体的title字段传入时,会触发iconv: Invalid argument错误。
相关代码
to_utf8函数实现
char* to_utf8(const unsigned int *encoding, char *source_str, unsigned int source_size){ iconv_t cd; // 设置源编码 if (*encoding == 0){ cd = iconv_open("UTF-8", "ISO-8859-1"); } else { cd = iconv_open("UTF-8", "UTF-16LE"); }; if (cd == (iconv_t)-1) { endwin(); perror("iconv_open:"); exit(1); }; // 跳过BOM int offset = 0; if (source_size >= 2 && ((unsigned char)source_str[0] == 0xFE && (unsigned char)source_str[1] == 0xFF) || ((unsigned char)source_str[0] == 0xFF && (unsigned char)source_str[1] == 0xFE)) { offset=2; source_size -= 2; }; size_t in_str_size = source_size, out_str_size = source_size; char *inbuf = (char *)source_str + offset; char *output = malloc(out_str_size); char *outbuf = output; size_t result = iconv(cd, &inbuf, &in_str_size, &outbuf, &out_str_size); if (result == (size_t)-1) { endwin(); perror("iconv"); free(output); iconv_close(cd); exit(1); return NULL; }; // 释放原内存 free(source_str); iconv_close(cd); return output; }
Track结构体调用场景
//... Track *track_data = calloc(sizeof(*track_data), 1); //... if (strcmp(tag_str, "TIT2") == 0){ track_data->title = calloc(sizeof(char), tag_size); memcpy(track_data->title, &id3_metadata_str[offset], tag_size-1); // 按需转换编码 if (encoding != 3){ track_data->title = to_utf8(&encoding, track_data->title, tag_size); };
Track结构体定义
struct Track{ char *path; char *artist; char *album; char *title; char *year; char *track; double duration; char *dur_str; int lyrics_size; char *lyrics; int shfl_num; long progress; } typedef Track;
正常工作的测试代码
#include <stdio.h> #include <stdlib.h> #include <string.h> #include <iconv.h> char* to_utf8(const unsigned int *encoding, char *source_str, unsigned int source_size){ iconv_t cd; if (*encoding == 0){ cd = iconv_open("UTF-8", "ISO-8859-1"); } else { cd = iconv_open("UTF-8", "UTF-16LE"); }; if (cd == (iconv_t)-1) { perror("iconv_open:"); exit(1); }; int offset = 0; if (source_size >= 2 && ((unsigned char)source_str[0] == 0xFE && (unsigned char)source_str[1] == 0xFF) || ((unsigned char)source_str[0] == 0xFF && (unsigned char)source_str[1] == 0xFE)) { offset=2; source_size -= 2; }; size_t in_str_size = source_size, out_str_size = source_size; char *inbuf = (char *)source_str + offset; char *output = malloc(out_str_size); char *outbuf = output; size_t result = iconv(cd, &inbuf, &in_str_size, &outbuf, &out_str_size); if (result == (size_t)-1) { perror("iconv"); free(output); iconv_close(cd); exit(1); return NULL; }; iconv_close(cd); return output; } int main() { unsigned int encoding = 1; // UTF-16带BOM char content_str[] = {0xFF, 0xFE, 0x42, 0x00, 0x61, 0x00, 0x64, 0x00, 0x20, 0x00, 0x44, 0x00, 0x61, 0x00, 0x79, 0x00, 0x20, 0x00, 0x66, 0x00, 0x6F, 0x00, 0x72, 0x00, 0x20, 0x00, 0x4D, 0x00, 0x79, 0x00, 0x20, 0x00, 0x45, 0x00, 0x6E, 0x00, 0x65, 0x00, 0x6D, 0x00, 0x69, 0x00, 0x65, 0x00, 0x73, 0x00 }; // UTF-16LE带BOM示例 unsigned int tag_size = sizeof(content_str); char* utf8_str = to_utf8(&encoding, content_str, tag_size); if (utf8_str != NULL) { printf("Converted UTF-8 string: %s\n", utf8_str); free(utf8_str); } else { printf("Conversion failed.\n"); } return 0; }
问题原因分析
两个场景的核心差异在于输入数据的完整性:
- 测试代码中,
content_str是完整的UTF-16LE字节序列(包含BOM和所有字符的双字节),tag_size是完整的字节数,满足UTF-16编码“字节数必须为偶数”的要求。 - 而Track调用场景中,
memcpy(track_data->title, &id3_metadata_str[offset], tag_size-1)只复制了tag_size-1个字节,导致传入的UTF-16数据不完整。iconv处理不完整的多字节序列时,会返回EINVAL(无效参数)错误。
另外,原to_utf8函数中存在一个隐藏问题:测试代码传入的是栈上的数组,函数内的free(source_str)会释放栈内存,这属于未定义行为,只是刚好没触发崩溃。
修复方案
1. 修复数据复制完整性
将memcpy的长度改为完整的tag_size,确保传入的UTF-16字节序列完整:
memcpy(track_data->title, &id3_metadata_str[offset], tag_size);
2. 修正to_utf8函数的内存管理
去掉函数内的free(source_str),由调用者负责释放原内存,避免释放栈内存的风险:
char* to_utf8(const unsigned int *encoding, char *source_str, unsigned int source_size){ iconv_t cd; if (*encoding == 0){ cd = iconv_open("UTF-8", "ISO-8859-1"); } else { cd = iconv_open("UTF-8", "UTF-16LE"); }; if (cd == (iconv_t)-1) { endwin(); perror("iconv_open:"); exit(1); }; int offset = 0; if (source_size >= 2 && ((unsigned char)source_str[0] == 0xFE && (unsigned char)source_str[1] == 0xFF) || ((unsigned char)source_str[0] == 0xFF && (unsigned char)source_str[1] == 0xFE)) { offset=2; source_size -= 2; }; // UTF-16转UTF-8最大字节比为1:3,扩容输出缓冲区避免溢出 size_t in_str_size = source_size, out_str_size = source_size * 3; char *inbuf = (char *)source_str + offset; char *output = malloc(out_str_size); if (!output) { iconv_close(cd); perror("malloc"); exit(1); } char *outbuf = output; size_t result = iconv(cd, &inbuf, &in_str_size, &outbuf, &out_str_size); if (result == (size_t)-1) { endwin(); perror("iconv"); free(output); iconv_close(cd); exit(1); return NULL; }; // 手动添加字符串结束符,iconv不会自动补充 *outbuf = '\0'; iconv_close(cd); return output; }
3. 修正调用逻辑
调用to_utf8后手动释放原title内存:
if (strcmp(tag_str, "TIT2") == 0){ track_data->title = calloc(sizeof(char), tag_size); memcpy(track_data->title, &id3_metadata_str[offset], tag_size); if (encoding != 3){ char *temp = to_utf8(&encoding, track_data->title, tag_size); free(track_data->title); track_data->title = temp; }; }
4. 修复测试代码的内存错误
测试代码中需将栈数组复制到堆内存后传入,避免释放栈内存:
int main() { unsigned int encoding = 1; char content_str[] = {0xFF, 0xFE, 0x42, 0x00, 0x61, 0x00, 0x64, 0x00, 0x20, 0x00, 0x44, 0x00, 0x61, 0x00, 0x79, 0x00, 0x20, 0x00, 0x66, 0x00, 0x6F, 0x00, 0x72, 0x00, 0x20, 0x00, 0x4D, 0x00, 0x79, 0x00, 0x20, 0x00, 0x45, 0x00, 0x6E, 0x00, 0x65, 0x00, 0x6D, 0x00, 0x69, 0x00, 0x65, 0x00, 0x73, 0x00 }; unsigned int tag_size = sizeof(content_str); // 复制到堆内存 char *heap_str = malloc(tag_size); memcpy(heap_str, content_str, tag_size); char* utf8_str = to_utf8(&encoding, heap_str, tag_size); if (utf8_str != NULL) { printf("Converted UTF-8 string: %s\n", utf8_str); free(utf8_str); } else { printf("Conversion failed.\n"); } return 0; }
内容的提问来源于stack exchange,提问作者Vulpes-Vulpeos
相关产品推荐
相关产品推荐

