You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

传递结构体元素时触发iconv: Invalid argument错误的原因排查

为什么传递track_data->title给to_utf8会触发iconv无效参数错误?

我实现了一个基于iconv库的to_utf8函数,用于将UTF-16字符串转换为UTF-8。手动构造字符串调用该函数时转换正常,但从Track结构体的title字段传入时,会触发iconv: Invalid argument错误。

相关代码

to_utf8函数实现

char* to_utf8(const unsigned int *encoding, char *source_str, unsigned int source_size){
    iconv_t cd;
    // 设置源编码
    if (*encoding == 0){
        cd = iconv_open("UTF-8", "ISO-8859-1");
    } else {
        cd = iconv_open("UTF-8", "UTF-16LE");
    };
    if (cd == (iconv_t)-1) {
        endwin();
        perror("iconv_open:");
        exit(1);
    };
    
    // 跳过BOM
    int offset = 0;
    if (source_size >= 2 &&
        ((unsigned char)source_str[0] == 0xFE && (unsigned char)source_str[1] == 0xFF) ||
        ((unsigned char)source_str[0] == 0xFF && (unsigned char)source_str[1] == 0xFE)) {
        offset=2;
        source_size -= 2;
    };
    
    size_t in_str_size = source_size,
           out_str_size = source_size;

    char *inbuf = (char *)source_str + offset;
    char *output = malloc(out_str_size);
    char *outbuf = output;

    size_t result = iconv(cd, &inbuf, &in_str_size, &outbuf, &out_str_size);
    if (result == (size_t)-1) {
        endwin();
        perror("iconv");
        free(output);
        iconv_close(cd);
        exit(1);
        return NULL;
    };
    
    // 释放原内存
    free(source_str);
    iconv_close(cd);

    return output;
}

Track结构体调用场景

//...
Track *track_data = calloc(sizeof(*track_data), 1);
//...
if (strcmp(tag_str, "TIT2") == 0){
    track_data->title = calloc(sizeof(char), tag_size);
    memcpy(track_data->title, &id3_metadata_str[offset], tag_size-1);
    // 按需转换编码
    if (encoding != 3){
        track_data->title = to_utf8(&encoding, track_data->title, tag_size); 
    };

Track结构体定义

struct Track{
    char   *path;
    char   *artist;
    char   *album;
    char   *title;
    char   *year;
    char   *track;
    double  duration;
    char   *dur_str;
    int     lyrics_size;
    char   *lyrics;
    int     shfl_num;
    long    progress;
} typedef Track;

正常工作的测试代码

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <iconv.h>

char* to_utf8(const unsigned int *encoding, char *source_str, unsigned int source_size){
    iconv_t cd;
    if (*encoding == 0){
        cd = iconv_open("UTF-8", "ISO-8859-1");
    } else {
        cd = iconv_open("UTF-8", "UTF-16LE");
    };
    if (cd == (iconv_t)-1) {
        perror("iconv_open:"); 
        exit(1); 
    };
    
    int offset = 0;
    if (source_size >= 2 &&
        ((unsigned char)source_str[0] == 0xFE && (unsigned char)source_str[1] == 0xFF) ||
        ((unsigned char)source_str[0] == 0xFF && (unsigned char)source_str[1] == 0xFE)) {
        offset=2;
        source_size -= 2;
    };
    
    size_t in_str_size = source_size,
           out_str_size = source_size;

    char *inbuf = (char *)source_str + offset;
    char *output = malloc(out_str_size);
    char *outbuf = output;

    size_t result = iconv(cd, &inbuf, &in_str_size, &outbuf, &out_str_size);
    if (result == (size_t)-1) {
        perror("iconv");
        free(output);
        iconv_close(cd);
        exit(1);
        return NULL;
    };
    
    iconv_close(cd);

    return output;
}

int main() {
    unsigned int encoding = 1; // UTF-16带BOM
    char content_str[] = {0xFF, 0xFE, 0x42, 0x00, 0x61, 0x00, 0x64, 0x00, 0x20, 0x00, 0x44, 0x00, 0x61, 0x00, 0x79,
                          0x00, 0x20, 0x00, 0x66, 0x00, 0x6F, 0x00, 0x72, 0x00, 0x20, 0x00, 0x4D, 0x00, 0x79, 0x00,
                          0x20, 0x00, 0x45, 0x00, 0x6E, 0x00, 0x65, 0x00, 0x6D, 0x00, 0x69, 0x00, 0x65, 0x00, 0x73, 0x00 }; // UTF-16LE带BOM示例
    unsigned int tag_size = sizeof(content_str);

    char* utf8_str = to_utf8(&encoding, content_str, tag_size);
    if (utf8_str != NULL) {
        printf("Converted UTF-8 string: %s\n", utf8_str);
        free(utf8_str);
    } else {
        printf("Conversion failed.\n");
    }

    return 0;
}

问题原因分析

两个场景的核心差异在于输入数据的完整性:

  • 测试代码中,content_str是完整的UTF-16LE字节序列(包含BOM和所有字符的双字节),tag_size是完整的字节数,满足UTF-16编码“字节数必须为偶数”的要求。
  • 而Track调用场景中,memcpy(track_data->title, &id3_metadata_str[offset], tag_size-1)只复制了tag_size-1个字节,导致传入的UTF-16数据不完整。iconv处理不完整的多字节序列时,会返回EINVAL(无效参数)错误。

另外,原to_utf8函数中存在一个隐藏问题:测试代码传入的是栈上的数组,函数内的free(source_str)会释放栈内存,这属于未定义行为,只是刚好没触发崩溃。

修复方案

1. 修复数据复制完整性

将memcpy的长度改为完整的tag_size,确保传入的UTF-16字节序列完整:

memcpy(track_data->title, &id3_metadata_str[offset], tag_size);

2. 修正to_utf8函数的内存管理

去掉函数内的free(source_str),由调用者负责释放原内存,避免释放栈内存的风险:

char* to_utf8(const unsigned int *encoding, char *source_str, unsigned int source_size){
    iconv_t cd;
    if (*encoding == 0){
        cd = iconv_open("UTF-8", "ISO-8859-1");
    } else {
        cd = iconv_open("UTF-8", "UTF-16LE");
    };
    if (cd == (iconv_t)-1) {
        endwin();
        perror("iconv_open:");
        exit(1);
    };
    
    int offset = 0;
    if (source_size >= 2 &&
        ((unsigned char)source_str[0] == 0xFE && (unsigned char)source_str[1] == 0xFF) ||
        ((unsigned char)source_str[0] == 0xFF && (unsigned char)source_str[1] == 0xFE)) {
        offset=2;
        source_size -= 2;
    };
    
    // UTF-16转UTF-8最大字节比为1:3,扩容输出缓冲区避免溢出
    size_t in_str_size = source_size,
           out_str_size = source_size * 3;

    char *inbuf = (char *)source_str + offset;
    char *output = malloc(out_str_size);
    if (!output) {
        iconv_close(cd);
        perror("malloc");
        exit(1);
    }
    char *outbuf = output;

    size_t result = iconv(cd, &inbuf, &in_str_size, &outbuf, &out_str_size);
    if (result == (size_t)-1) {
        endwin();
        perror("iconv");
        free(output);
        iconv_close(cd);
        exit(1);
        return NULL;
    };
    
    // 手动添加字符串结束符,iconv不会自动补充
    *outbuf = '\0';
    
    iconv_close(cd);

    return output;
}

3. 修正调用逻辑

调用to_utf8后手动释放原title内存:

if (strcmp(tag_str, "TIT2") == 0){
    track_data->title = calloc(sizeof(char), tag_size);
    memcpy(track_data->title, &id3_metadata_str[offset], tag_size);
    if (encoding != 3){
        char *temp = to_utf8(&encoding, track_data->title, tag_size);
        free(track_data->title);
        track_data->title = temp;
    };
}

4. 修复测试代码的内存错误

测试代码中需将栈数组复制到堆内存后传入,避免释放栈内存:

int main() {
    unsigned int encoding = 1;
    char content_str[] = {0xFF, 0xFE, 0x42, 0x00, 0x61, 0x00, 0x64, 0x00, 0x20, 0x00, 0x44, 0x00, 0x61, 0x00, 0x79,
                          0x00, 0x20, 0x00, 0x66, 0x00, 0x6F, 0x00, 0x72, 0x00, 0x20, 0x00, 0x4D, 0x00, 0x79, 0x00,
                          0x20, 0x00, 0x45, 0x00, 0x6E, 0x00, 0x65, 0x00, 0x6D, 0x00, 0x69, 0x00, 0x65, 0x00, 0x73, 0x00 };
    unsigned int tag_size = sizeof(content_str);

    // 复制到堆内存
    char *heap_str = malloc(tag_size);
    memcpy(heap_str, content_str, tag_size);
    
    char* utf8_str = to_utf8(&encoding, heap_str, tag_size);
    if (utf8_str != NULL) {
        printf("Converted UTF-8 string: %s\n", utf8_str);
        free(utf8_str);
    } else {
        printf("Conversion failed.\n");
    }

    return 0;
}

内容的提问来源于stack exchange,提问作者Vulpes-Vulpeos

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 12:25:55