C语言无编译错误但运行异常:解析器无法输出Token问题排查
C语言解析器无输出问题排查与修复
问题现象
作为C语言新手,我编写了一个解析文本文件的简单解析器,目标是解析内容为var a = 10的test.hex文件。程序可正常编译,但控制台无任何Token输出,编译器也未给出错误信息。经调试发现,执行buffer.init(&buffer);语句后,后续的printf命令均无法正常工作。完整代码如下:
#include <stdio.h> #include <stdlib.h> #include <wchar.h> #include <wctype.h> #include <stdbool.h> typedef struct Str { size_t capacity; // allocated size of buffer size_t length; // used size of buffer wchar_t* buffer; // character buffer void (*init)(struct Str*); // initialize with a string void (*push)(struct Str*, wchar_t); // append a single character void (*trim)(struct Str*); // remove unused space void (*destroy)(struct Str*); // free memory } Str; // append a single character void push_Str(Str* str, wchar_t ch) { if (str->length + 1 >= str->capacity) { str->capacity = str->capacity * 2; str->buffer = (wchar_t*)realloc(str->buffer, (str->capacity + 1) * sizeof(wchar_t)); } str->buffer[str->length++] = ch; str->buffer[str->length] = L'\0'; } // append a string to original string Str* concat_Str(Str *str1, const Str *str2) { size_t new_len = str1->length + str2->length; if (new_len >= str1->capacity) { str1->capacity = new_len * 2; str1->buffer = (wchar_t*) realloc(str1->buffer, str1->capacity * sizeof(wchar_t)); } wcscpy(&str1->buffer[str1->length], str2->buffer); str1->length = new_len; return str1; } // remove unused space void trim_Str(Str* str) { str->buffer = (wchar_t*)realloc(str->buffer, (str->length + 1) * sizeof(wchar_t)); str->capacity = str->length; } // free memory void destroy_Str(Str* str) { free(str->buffer); } // initialize as empty void init_Str(Str* str) { str->capacity = 8; str->length = 0; str->buffer = (wchar_t*)malloc((str->capacity + 1) * sizeof(wchar_t)); if (str->buffer == NULL) { fprintf(stderr, "Failed to allocate memory\n"); exit(EXIT_FAILURE); } str->buffer[0] = L'\0'; str->init = &init_Str; str->push = &push_Str; str->trim = &trim_Str; str->destroy = &destroy_Str; } typedef struct Token { enum { TOK_EOF = 1, TOK_ILLEGAL, TOK_SPACE, TOK_VAR, TOK_CONST, TOK_ASSIGN, TOK_INT_LIT, TOK_FLOAT_LIT, TOK_BOOL_LIT, TOK_IDENTIFIER, } type; Str value; // fpos_t atPos; // size_t atLine; } Token; void printToken(Token token) { printf("Printing Token: "); printf("{ type: %i, value: %ls }, ", token.type, token.value.buffer); } Token getNextToken(FILE *fp, size_t line) { printf("Getting next token\n"); // get cursor position fpos_t cursor; fgetpos(fp, &cursor); // get the char at the position wchar_t ch; ch = fgetwc(fp); // create a buffer to hold the token value Str buffer; buffer.init(&buffer); // buffer.push(&buffer, ch); // if we are at end of fie, return a EOF token if (ch == WEOF) { Token token; token.type = TOK_EOF; token.value = buffer; // token.atPos = cursor; // token.atLine = line; return token; } // Create a token to return Token token; token.type = TOK_ILLEGAL; token.value = buffer; // token.atPos = cursor; // token.atLine = line; printToken(token); if (iswspace(ch)) { token.type = TOK_SPACE; // roll all spaces into a single space token while (iswspace(ch)) { ch = fgetwc(fp); // if (ch == L'\n') { // token.atLine++; // } } token.value.push(&token.value, ch); } else if (iswdigit(ch)) { token.type = TOK_INT_LIT; token.value.push(&token.value, ch); // check for float lit first bool isFloat = false; // roll all digits into a single float while (iswdigit(ch) || ch == L'.' || ch == L'_') { // push to buffer token.value.push(&token.value, ch); ch = fgetwc(fp); if (ch == L'.' && !isFloat) { isFloat = true; token.type = TOK_FLOAT_LIT; } else if (ch == L'.' && isFloat) { token.type = TOK_FLOAT_LIT; break; } } } else if (iswalpha(ch)) { token.type = TOK_IDENTIFIER; token.value.push(&token.value, ch); // roll all alphanum into the buffer while (iswalpha(ch) || ch == L'_') { // push to buffer // buffer.push(&buffer, ch); token.value.push(&token.value, ch); // check for keywords if (wcscmp(token.value.buffer, L"var") == 0) { token.type = TOK_VAR; } else if (wcscmp(token.value.buffer, L"const") == 0) { token.type = TOK_CONST; } else if (wcscmp(token.value.buffer, L"true") == 0 || wcscmp(token.value.buffer, L"false") == 0) { token.type = TOK_BOOL_LIT; } // check for the next character ch = fgetwc(fp); } } // else { // } printf("Token: "); printToken(token); return token; } void parseFile (const char* filePath) { FILE* fp = fopen(filePath, "rb"); if (fp == NULL) { fprintf(stderr, "Failed to open file: %s\n", filePath); exit(1); } // set file to wide character mode if (fwide(fp, 1) < 0) { fprintf(stderr, "Failed to set file to wide character mode\n"); fclose(fp); exit(1); } size_t line = 0; Token token1 = getNextToken(fp, line); // line = token1.atLine; printToken(token1); Token token2 = getNextToken(fp, line); // line = token2.atLine; printToken(token2); Token token3 = getNextToken(fp, line); // line = token3.atLine; printToken(token3); Token token4 = getNextToken(fp, line); // line = token4.atLine; printToken(token4); Token token5 = getNextToken(fp, line); // line = token5.atLine; printToken(token5); Token token6 = getNextToken(fp, line); // line = token6.atLine; printToken(token6); Token token7 = getNextToken(fp, line); printToken(token7); // return 0; } int main (void) { parseFile("input/test.hex"); return 0; }
问题根源
- 未初始化函数指针调用:
getNextToken中声明的Str buffer是局部变量,结构体内部的init等函数指针未被初始化,指向内存垃圾值。直接调用buffer.init(&buffer)会触发未定义行为,导致程序跳转至非法地址执行,破坏栈结构或程序流,这是后续printf失效的核心原因。 - 浅拷贝内存问题:
Token的Str value通过token.value = buffer直接赋值属于浅拷贝,后续若重复释放内存会触发错误,不释放则造成内存泄漏。 - 字符未回退:读取Token时,遇到不属于当前Token的字符未放回文件流,导致后续Token读取错位,无法正确解析内容。
- 宽字符输出不兼容:使用
printf配合%ls输出宽字符,在部分环境下无法正常显示。
修复方案
1. 修正Str初始化方式
去掉对未初始化函数指针的依赖,直接调用初始化函数:
// 替换原buffer.init(&buffer); init_Str(&buffer);
2. 添加Str深拷贝函数
避免浅拷贝问题,新增拷贝函数:
Str copy_Str(const Str* src) { Str dest; init_Str(&dest); dest.capacity = src->length + 1; dest.length = src->length; dest.buffer = (wchar_t*)realloc(dest.buffer, dest.capacity * sizeof(wchar_t)); if (dest.buffer) { wcscpy(dest.buffer, src->buffer); } return dest; }
在getNextToken中赋值Token的value时使用深拷贝,并释放临时buffer:
token.value = copy_Str(&buffer); buffer.destroy(&buffer);
3. 修复字符回退逻辑
在每个Token读取循环结束后,将非目标字符放回文件流:
// 以空格处理分支为例 if (iswspace(ch)) { token.type = TOK_SPACE; while (iswspace(ch)) { token.value.push(&token.value, ch); ch = fgetwc(fp); } if (ch != WEOF) { ungetwc(ch, fp); } }
数字、标识符分支同理添加回退逻辑。
4. 修正宽字符输出
改用wprintf输出宽字符,确保兼容性:
void printToken(Token token) { wprintf(L"Printing Token: { type: %d, value: %ls }\n", token.type, token.value.buffer); }
同时将getNextToken中的printf替换为wprintf。
5. 添加内存释放逻辑
在程序结束前释放每个Token的内存:
void parseFile(const char* filePath) { // ... 原代码 ... token1.value.destroy(&token1.value); token2.value.destroy(&token2.value); token3.value.destroy(&token3.value); token4.value.destroy(&token4.value); token5.value.destroy(&token5.value); token6.value.destroy(&token6.value); token7.value.destroy(&token7.value); fclose(fp); }
修复后效果
完成上述修改后,程序能够正确解析var a = 10,控制台会输出各个Token的类型和值,宽字符输出也能正常工作。
内容的提问来源于stack exchange,提问作者Rishav Sharan
相关产品推荐
相关产品推荐

