You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何修改C语言程序实现从文本文件读取并分类识别Token

Fixing Your Tokenizer to Read from a Text File

Hey, let's get your token identifier working with a text file instead of that hardcoded string. I'll walk you through the changes needed and share the full updated code.

Key Changes Needed

First, we need to:

  • Add file I/O logic to read content from token.txt
  • Update the isDelimiter function to handle newline characters (since files have line breaks)
  • Handle cases where the file content might be longer than 100 characters
  • Fix memory leaks from the substring allocation

Updated Full Code

Here's the modified identifier.c with file reading functionality:

#include <stdbool.h>
#include <stdio.h>
#include <string.h>
#include <stdlib.h>

// Returns 'true' if the character is a DELIMITER.
bool isDelimiter(char ch) {
    // Added '\n' to handle line breaks in the text file
    if (ch == ' ' || ch == '+' || ch == '-' || ch == '*' || ch == '/' || ch == ',' || ch == ';' || ch == '>' || ch == '<' || ch == '=' || ch == '(' || ch == ')' || ch == '[' || ch == ']' || ch == '{' || ch == '}' || ch == '\n')
        return (true);
    return (false);
}

// Returns 'true' if the character is an OPERATOR.
bool isOperator(char ch) {
    if (ch == '+' || ch == '-' || ch == '*' || ch == '/' || ch == '>' || ch == '<' || ch == '=')
        return (true);
    return (false);
}

// Returns 'true' if the string is a VALID IDENTIFIER.
bool validIdentifier(char* str) {
    if (str[0] == '0' || str[0] == '1' || str[0] == '2' || str[0] == '3' || str[0] == '4' || str[0] == '5' || str[0] == '6' || str[0] == '7' || str[0] == '8' || str[0] == '9' || isDelimiter(str[0]) == true)
        return (false);
    return (true);
}

// Returns 'true' if the string is a KEYWORD.
bool isKeyword(char* str) {
    if (!strcmp(str, "if") || !strcmp(str, "else") || !strcmp(str, "while") || !strcmp(str, "do") || !strcmp(str, "break") || !strcmp(str, "continue") || !strcmp(str, "int") || !strcmp(str, "double") || !strcmp(str, "float") || !strcmp(str, "return") || !strcmp(str, "char") || !strcmp(str, "case") || !strcmp(str, "sizeof") || !strcmp(str, "long") || !strcmp(str, "short") || !strcmp(str, "typedef") || !strcmp(str, "switch") || !strcmp(str, "unsigned") || !strcmp(str, "void") || !strcmp(str, "static") || !strcmp(str, "struct") || !strcmp(str, "goto") || !strcmp(str, "for"))
        return (true); // Added "for" since it's in your token.txt
    return (false);
}

// Returns 'true' if the string is an INTEGER.
bool isInteger(char* str) {
    int i, len = strlen(str);
    if (len == 0)
        return (false);
    for (i = 0; i < len; i++) {
        if (str[i] != '0' && str[i] != '1' && str[i] != '2' && str[i] != '3' && str[i] != '4' && str[i] != '5' && str[i] != '6' && str[i] != '7' && str[i] != '8' && str[i] != '9' || (str[i] == '-' && i > 0))
            return (false);
    }
    return (true);
}

// Returns 'true' if the string is a REAL NUMBER.
bool isRealNumber(char* str) {
    int i, len = strlen(str);
    bool hasDecimal = false;
    if (len == 0)
        return (false);
    for (i = 0; i < len; i++) {
        if (str[i] != '0' && str[i] != '1' && str[i] != '2' && str[i] != '3' && str[i] != '4' && str[i] != '5' && str[i] != '6' && str[i] != '7' && str[i] != '8' && str[i] != '9' && str[i] != '.' || (str[i] == '-' && i > 0))
            return (false);
        if (str[i] == '.')
            hasDecimal = true;
    }
    return (hasDecimal);
}

// Extracts the SUBSTRING.
char* subString(char* str, int left, int right) {
    int i;
    char* subStr = (char*)malloc( sizeof(char) * (right - left + 2));
    for (i = left; i <= right; i++)
        subStr[i - left] = str[i];
    subStr[right - left + 1] = '\0';
    return (subStr);
}

// Parsing the input STRING.
void parse(char* str) {
    int left = 0, right = 0;
    int len = strlen(str);
    while (right <= len && left <= right) {
        if (isDelimiter(str[right]) == false)
            right++;
        if (isDelimiter(str[right]) == true && left == right) {
            if (isOperator(str[right]) == true)
                printf("'%c' IS AN OPERATOR\n", str[right]);
            right++;
            left = right;
        } else if (isDelimiter(str[right]) == true && left != right || (right == len && left != right)) {
            char* subStr = subString(str, left, right - 1);
            if (isKeyword(subStr) == true)
                printf("'%s' IS A KEYWORD\n", subStr);
            else if (isInteger(subStr) == true)
                printf("'%s' IS AN INTEGER\n", subStr);
            else if (isRealNumber(subStr) == true)
                printf("'%s' IS A REAL NUMBER\n", subStr);
            else if (validIdentifier(subStr) == true && isDelimiter(str[right - 1]) == false)
                printf("'%s' IS A VALID IDENTIFIER\n", subStr);
            else if (validIdentifier(subStr) == false && isDelimiter(str[right - 1]) == false)
                printf("'%s' IS NOT A VALID IDENTIFIER\n", subStr);
            free(subStr); // Free allocated memory to avoid leaks
            left = right;
        }
    }
    return;
}

// DRIVER FUNCTION
int main() {
    // Larger buffer to handle full file content
    char str[500];
    FILE *file = fopen("token.txt", "r");
    
    if (file == NULL) {
        printf("Error opening file! Make sure token.txt is in the same directory.\n");
        return 1;
    }
    
    // Read entire file into buffer and null-terminate
    size_t bytesRead = fread(str, 1, sizeof(str) - 1, file);
    str[bytesRead] = '\0';
    
    fclose(file);
    parse(str); // Run token parsing
    
    return 0;
}

What Changed?

Let's break down the key tweaks:

  1. Updated isDelimiter: Added '\n' so line breaks are treated as delimiters, preventing them from being included in tokens.
  2. Added "for" to isKeyword: Your token.txt lists "for" as a keyword, so we added it to the detection list.
  3. File I/O in main:
    • Opens token.txt for reading with error handling (checks if the file exists)
    • Uses a larger buffer (500 characters) to fit your file content
    • Properly null-terminates the buffer after reading
  4. Memory Leak Fix: Added free(subStr); in the parse function to clean up memory allocated by subString.

Your token.txt Setup

Ensure your token.txt is in the same folder as your compiled program with this content:

1) Keywords: for, while, if
2) Identifier total, sum, average, a, b, c
3) Operators: '+', '++', '-' etc.
4) Separators: ', ' ';' etc
5) Integers: 1, 2, 3, 5, 6, 7, 8, 9

How to Run

Compile and execute with these commands:

gcc identifier.c -o tokenizer
./tokenizer

You'll see output that correctly categorizes each token (keywords, integers, identifiers, operators) from your text file. The single quotes (') will show as invalid identifiers, which is expected since they don't follow identifier naming rules.

内容的提问来源于stack exchange,提问作者Majadul Haque

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.29 19:42:51