You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C语言无编译错误但运行异常:解析器无法输出Token问题排查

C语言解析器无输出问题排查与修复

问题现象

作为C语言新手,我编写了一个解析文本文件的简单解析器,目标是解析内容为var a = 10的test.hex文件。程序可正常编译,但控制台无任何Token输出,编译器也未给出错误信息。经调试发现,执行buffer.init(&buffer);语句后,后续的printf命令均无法正常工作。完整代码如下:

#include <stdio.h>
#include <stdlib.h>
#include <wchar.h>
#include <wctype.h>
#include <stdbool.h>

typedef struct Str {
    size_t capacity; // allocated size of buffer
    size_t length;   // used size of buffer
    wchar_t* buffer; // character buffer

    void (*init)(struct Str*);                  // initialize with a string
    void (*push)(struct Str*, wchar_t);         // append a single character
    void (*trim)(struct Str*);                  // remove unused space
    void (*destroy)(struct Str*);                  // free memory
} Str;

// append a single character
void push_Str(Str* str, wchar_t ch) {
    if (str->length + 1 >= str->capacity) {
        str->capacity = str->capacity * 2;
        str->buffer = (wchar_t*)realloc(str->buffer, (str->capacity + 1) * sizeof(wchar_t));
    }
    str->buffer[str->length++] = ch;
    str->buffer[str->length] = L'\0';
}

// append a string to original string
Str* concat_Str(Str *str1, const Str *str2) {
    size_t new_len = str1->length + str2->length;
    if (new_len >= str1->capacity) {
        str1->capacity = new_len * 2;
        str1->buffer = (wchar_t*) realloc(str1->buffer, str1->capacity * sizeof(wchar_t));
    }
    wcscpy(&str1->buffer[str1->length], str2->buffer);
    str1->length = new_len;
    return str1;
}

// remove unused space
void trim_Str(Str* str) {
    str->buffer = (wchar_t*)realloc(str->buffer, (str->length + 1) * sizeof(wchar_t));
    str->capacity = str->length;
}

// free memory
void destroy_Str(Str* str) {
    free(str->buffer);
}

// initialize as empty
void init_Str(Str* str) {
    str->capacity = 8;
    str->length = 0;
    str->buffer = (wchar_t*)malloc((str->capacity + 1) * sizeof(wchar_t));
    if (str->buffer == NULL) {
        fprintf(stderr, "Failed to allocate memory\n");
        exit(EXIT_FAILURE);
    }
    str->buffer[0] = L'\0';

    str->init = &init_Str;
    str->push = &push_Str;
    str->trim = &trim_Str;
    str->destroy = &destroy_Str;
}

typedef struct Token
{
    enum
    {
        TOK_EOF = 1,
        TOK_ILLEGAL,
        TOK_SPACE,
        TOK_VAR,
        TOK_CONST,
        TOK_ASSIGN,
        TOK_INT_LIT,
        TOK_FLOAT_LIT,
        TOK_BOOL_LIT,
        TOK_IDENTIFIER,
    } type;
    Str value;
    // fpos_t atPos;
    // size_t atLine;
} Token;

void printToken(Token token) {
    printf("Printing Token: ");
    printf("{ type: %i, value: %ls }, ", token.type, token.value.buffer);
}

Token getNextToken(FILE *fp, size_t line) {
    printf("Getting next token\n");

    // get cursor position
    fpos_t cursor;
    fgetpos(fp, &cursor);

    // get the char at the position
    wchar_t ch;
    ch = fgetwc(fp);

    // create a buffer to hold the token value
    Str buffer;
    buffer.init(&buffer);
    // buffer.push(&buffer, ch);

    // if we are at end of fie, return a EOF token
    if (ch == WEOF) {
        Token token;
        token.type = TOK_EOF;
        token.value = buffer;
        // token.atPos = cursor;
        // token.atLine = line;
        return token;
    }

    // Create a token to return
    Token token;
    token.type = TOK_ILLEGAL;
    token.value = buffer;
    // token.atPos = cursor;
    // token.atLine = line;

  

    printToken(token);

    if (iswspace(ch)) {

        token.type = TOK_SPACE;

        // roll all spaces into a single space token
        while (iswspace(ch)) {
            ch = fgetwc(fp);
            // if (ch == L'\n') {
            //     token.atLine++;
            // }
        }
        token.value.push(&token.value, ch);
    
    } else if (iswdigit(ch)) {
        token.type = TOK_INT_LIT;
        token.value.push(&token.value, ch);

        // check for float lit first
        bool isFloat = false;

        // roll all digits into a single float 
        while (iswdigit(ch) || ch == L'.' || ch == L'_') {
            // push to buffer
            token.value.push(&token.value, ch);

            ch = fgetwc(fp);
            if (ch == L'.' && !isFloat) {
                isFloat = true;
                token.type = TOK_FLOAT_LIT;
            } else if (ch == L'.' && isFloat) {
                token.type = TOK_FLOAT_LIT;
                break;
            }
        }
    }
    else if (iswalpha(ch)) {
        token.type = TOK_IDENTIFIER;
        token.value.push(&token.value, ch);

        // roll all alphanum into the buffer
        while (iswalpha(ch) || ch == L'_') {
            // push to buffer
            // buffer.push(&buffer, ch);
            token.value.push(&token.value, ch);

            // check for keywords
            if (wcscmp(token.value.buffer, L"var") == 0) {
                token.type = TOK_VAR;
            } else if (wcscmp(token.value.buffer, L"const") == 0) {
                token.type = TOK_CONST;
            } else if (wcscmp(token.value.buffer, L"true") == 0 || wcscmp(token.value.buffer, L"false") == 0) {
                token.type = TOK_BOOL_LIT;
            } 

            // check for the next character
            ch = fgetwc(fp);
        }
    }
    // else {

    // }
    printf("Token: ");
    printToken(token);
    return token;
}


void parseFile (const char* filePath) {
    
    FILE* fp = fopen(filePath, "rb");
    if (fp == NULL) {
        fprintf(stderr, "Failed to open file: %s\n", filePath);
        exit(1);
    }
    // set file to wide character mode
    if (fwide(fp, 1) < 0) {
        fprintf(stderr, "Failed to set file to wide character mode\n");
        fclose(fp);
        exit(1);
    }

    size_t line = 0;

    Token token1 = getNextToken(fp, line);
    // line = token1.atLine;
    printToken(token1);
    
    Token token2 = getNextToken(fp, line);
    // line = token2.atLine;
    printToken(token2);
    
    Token token3 = getNextToken(fp, line);
    // line = token3.atLine;
    printToken(token3);
    
    Token token4 = getNextToken(fp, line);
    // line = token4.atLine;
    printToken(token4);
    
    Token token5 = getNextToken(fp, line);
    // line = token5.atLine;
    printToken(token5);
    
    Token token6 = getNextToken(fp, line);
    // line = token6.atLine;
    printToken(token6);
    
    Token token7 = getNextToken(fp, line);
    printToken(token7);
    
    // return 0;
}

int main (void) {
    parseFile("input/test.hex");
    return 0;
}

问题根源

  1. 未初始化函数指针调用:getNextToken中声明的Str buffer是局部变量,结构体内部的init等函数指针未被初始化,指向内存垃圾值。直接调用buffer.init(&buffer)会触发未定义行为,导致程序跳转至非法地址执行,破坏栈结构或程序流,这是后续printf失效的核心原因。
  2. 浅拷贝内存问题:Token的Str value通过token.value = buffer直接赋值属于浅拷贝,后续若重复释放内存会触发错误,不释放则造成内存泄漏。
  3. 字符未回退:读取Token时,遇到不属于当前Token的字符未放回文件流,导致后续Token读取错位,无法正确解析内容。
  4. 宽字符输出不兼容:使用printf配合%ls输出宽字符,在部分环境下无法正常显示。

修复方案

1. 修正Str初始化方式

去掉对未初始化函数指针的依赖,直接调用初始化函数:

// 替换原buffer.init(&buffer);
init_Str(&buffer);

2. 添加Str深拷贝函数

避免浅拷贝问题,新增拷贝函数:

Str copy_Str(const Str* src) {
    Str dest;
    init_Str(&dest);
    dest.capacity = src->length + 1;
    dest.length = src->length;
    dest.buffer = (wchar_t*)realloc(dest.buffer, dest.capacity * sizeof(wchar_t));
    if (dest.buffer) {
        wcscpy(dest.buffer, src->buffer);
    }
    return dest;
}

在getNextToken中赋值Token的value时使用深拷贝,并释放临时buffer:

token.value = copy_Str(&buffer);
buffer.destroy(&buffer);

3. 修复字符回退逻辑

在每个Token读取循环结束后,将非目标字符放回文件流:

// 以空格处理分支为例
if (iswspace(ch)) {
    token.type = TOK_SPACE;
    while (iswspace(ch)) {
        token.value.push(&token.value, ch);
        ch = fgetwc(fp);
    }
    if (ch != WEOF) {
        ungetwc(ch, fp);
    }
}

数字、标识符分支同理添加回退逻辑。

4. 修正宽字符输出

改用wprintf输出宽字符,确保兼容性:

void printToken(Token token) {
    wprintf(L"Printing Token: { type: %d, value: %ls }\n", token.type, token.value.buffer);
}

同时将getNextToken中的printf替换为wprintf。

5. 添加内存释放逻辑

在程序结束前释放每个Token的内存:

void parseFile(const char* filePath) {
    // ... 原代码 ...
    
    token1.value.destroy(&token1.value);
    token2.value.destroy(&token2.value);
    token3.value.destroy(&token3.value);
    token4.value.destroy(&token4.value);
    token5.value.destroy(&token5.value);
    token6.value.destroy(&token6.value);
    token7.value.destroy(&token7.value);
    
    fclose(fp);
}

修复后效果

完成上述修改后,程序能够正确解析var a = 10,控制台会输出各个Token的类型和值,宽字符输出也能正常工作。

内容的提问来源于stack exchange,提问作者Rishav Sharan

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.22 22:47:01