You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Lex扫描器编译报错:无法识别规则与致命解析错误排查求助

Lex扫描器编译错误排查方案

编译报错的直接原因是代码中存在多处语法错误,导致Lex无法正确解析规则,以下是具体问题和修复方法:

1. 宏定义语法错误

  • commentLeft宏:原定义commentLeft ({%)中,{是Lex的特殊字符(用于引用宏),匹配字面{必须转义,修正为:
    commentLeft     (\{%)
    
  • string宏:原正则"(""|[^"\n])*")存在引号嵌套错误,正确的字符串正则应匹配双引号包裹、内部双引号用两个双引号转义的格式,修正为:
    string          "(\"\"|[^"\n])*"
    

2. 规则段宏引用错误

规则中{delimiter;}、{arithmetic;}、{relational;}多了末尾的分号,Lex宏引用不需要加后缀分号,这会被当成正则的一部分导致无法识别规则,修正为:

{delimiter}       {tokenChar(yytext[0]);}
{arithmetic}      {tokenChar(yytext[0]);}
{relational}      {tokenChar(yytext[0]);}

3. 注释起始规则错误

原规则{% {语法错误,应引用定义好的commentLeft宏,修正为:

{commentLeft} { 
    LIST;
    BEGIN(COMMENT);
}

4. 潜在运行异常问题(非编译错误,但需修正)

  • token函数参数错误:token('<=>')中,单引号仅能包裹单个字符,<=>是三个字符,应改为字符串形式token("<=>")(若token接受字符串参数),或使用对应枚举常量(如ARRAY、BEGIN这类关键字的定义)。
  • main函数参数判断错误:原代码if (argc > 0)逻辑错误,argc至少为1(程序本身),应判断argc > 1;同时默认需将yyin指向标准输入,避免未初始化时调用fclose崩溃,修正后的main函数:
    int main(int argc, char **argv){
        FILE *yyin = stdin; // 默认使用标准输入
        if (argc > 1){
            yyin = fopen(argv[1], "r");
            if (!yyin){
                printf("Failed to open file %s\n", argv[1]);
                return 1;
            }
        }
        yylex();
        if (yyin != stdin) fclose(yyin); // 仅关闭手动打开的文件
        return 0;
    }
    

修正后的核心规则段

identifier              ([A-Za-z][0-9A-Za-z]*)
digit                   ([0-9])
integer                 ({digit}+)
float                   ({integer}"."[0-9]+)
delimiter               ([.,:;()\[\]{}])
arithmetic              ([+-*/])
relational              ([<>])
string                  "(\"\"|[^"\n])*"
commentLine             (\%[^\n]*)
commentLeft             (\{%)
commentRight            (%})

%x COMMENT

%option noyywrap

%%

{delimiter}       {tokenChar(yytext[0]);}
{arithmetic}      {tokenChar(yytext[0]);}
{relational}      {tokenChar(yytext[0]);}

"<="                    { token("<=>"); }
">="                    { token(">="); }
"not="                  { token("not="); }
":="                    { token(":="); }

"not"                   { token("not"); }
"and"                   { token("and"); }
"or"                    { token("or"); }

"array"                 { token(ARRAY); }
"begin"                 { token(BEGIN); }
"bool"                  { token(BOOL); }
"char"                  { token(CHAR); }
"const"                 { token(CONST); }
"decreasing"            { token(DECREASING); }
"default"               { token(DEFAULT); }
"do"                    { token(DO); }
"else"                  { token(ELSE); }
"end"                   { token(END); }
"exit"                  { token(EXIT); }
"false"                 { token(FALSE); }
"for"                   { token(FOR); }
"function"              { token(FUNCTION); }
"get"                   { token(GET); }
"if"                    { token(IF); }
"int"                   { token(INT); }
"loop"                  { token(LOOP); }
"of"                    { token(OF); }
"put"                   { token(PUT); }
"procedure"             { token(PROCEDURE); }
"real"                  { token(REAL); }
"result"                { token(RESULT); }
"return"                { token(RETURN); }
"skip"                  { token(SKIP); }
"string"                { token(STRING); }
"then"                  { token(THEN); }
"true"                  { token(TRUE); }
"var"                   { token(VAR); }
"when"                  { token(WHEN); }

{integer}  {
    tokenInteger("integer", atoi(yytext));
}

{identifier} {
    tokenString("identifier", yytext);
    table -> insert(yytext);
}

{float}    {
    tokenFloat("float", yytext);
 }

{string}    {
    char s[MAX_LINE_LENG] = {0};
    int idx = 0;
    for (int i = 1; i < yyleng - 1; ++i){
        if (yytext[i] == '"')
            ++i;
        s[idx++] = yytext[i];
    }
    tokenString("string", s);
}

{commentLine}  {
    LIST;
}

{commentLeft} { 
    LIST;
    BEGIN(COMMENT);
}

<COMMENT>[^\n]  {
    LIST;
}

<COMMENT>\n {   
    LIST;
    printf("%d: %s", linenum, buf);
    linenum++;
    buf[0] = '\0';
}

<COMMENT>{commentRight}    {
        LIST;
        BEGIN(INITIAL);
    }

\n  {
    LIST;
    printf("%d: %s", linenum++, buf);
    buf[0] = '\0';
}

[ \t]*  {
    LIST;
}

.   {
    LIST;
    printf("%d:%s\n", linenum+1, buf);
    printf("bad character:'%s'\n",yytext);
    exit(-1);
}
%%

修复以上错误后,重新执行lex scanner.l即可正常编译。

内容的提问来源于stack exchange,提问作者letmesleepplz

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.23 00:39:59