You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C语言如何将字符串分割为单词数组 兼容多空格、制表符分隔

C语言按空白拆分字符串为单词数组的实现方法

strtok完全可以处理制表符等多类型空白分隔符的场景,只要正确传入分隔符集合即可,下面给出两种可直接运行的实现方案。

方案1:基于strtok的最简实现

strtok的核心特性是会自动跳过连续的分隔符,只要你把所有需要识别的空白字符都放到分隔符参数字符串中,就能实现任意数量、任意类型空白分隔的单词拆分。
注意两个使用前提:

  • strtok会修改传入的输入字符串,因此不能直接传入只读的字符串字面量,需要先将原字符串拷贝到可写的内存缓冲区
  • 分隔符集合需要覆盖你需要识别的所有空白类型,也可以直接用ctype.h的isspace做判断(见方案2)

完整实现代码:

#include <stdio.h>
#include <string.h>
#include <stdlib.h>

int main() {
    const char *raw_str = "This  is\t\ta cool\nsentence"; // 混合多空格、制表符、换行
    // 拷贝原字符串到可写缓冲区
    char *buf = malloc(strlen(raw_str) + 1);
    strcpy(buf, raw_str);

    const char *delimiters = " \t\n\r"; // 要识别的分隔符:空格、制表符、换行、回车
    // 第一次遍历统计单词总数
    int word_cnt = 0;
    char *tok = strtok(buf, delimiters);
    while (tok) {
        word_cnt++;
        tok = strtok(NULL, delimiters);
    }
    // 重置buf后第二次遍历填充单词数组
    strcpy(buf, raw_str);
    char **words = malloc(sizeof(char*) * (word_cnt + 1)); // 多分配1位存NULL结束标记
    int idx = 0;
    tok = strtok(buf, delimiters);
    while (tok) {
        words[idx++] = tok; // 如果需要独立的单词副本,替换为words[idx++] = strdup(tok);
        tok = strtok(NULL, delimiters);
    }
    words[idx] = NULL; // 数组末尾加NULL标记,和argv格式一致,方便遍历判断结束

    // 测试输出
    for (int i = 0; words[i]; i++) {
        printf("words[%d] = %s\n", i, words[i]);
    }

    // 释放内存,如果用了strdup需要先逐个释放words[i]
    free(words);
    free(buf);
    return 0;
}

运行输出和你预期完全一致:

words[0] = This
words[1] = is
words[2] = a
words[3] = cool
words[4] = sentence

方案2:手动遍历实现(无侵入、支持全类型空白)

如果你不想修改原字符串,或者需要覆盖所有标准空白字符(包括垂直制表符、换页符等),可以手动逐字符遍历实现,不需要依赖strtok:

#include <stdio.h>
#include <string.h>
#include <stdlib.h>
#include <ctype.h>

int main() {
    const char *S = "This  is     a cool\t\rsentence";
    const int len = strlen(S);
    char **words = NULL;
    int word_cnt = 0;
    int i = 0;

    while (i < len) {
        // 跳过所有连续空白
        while (i < len && isspace((unsigned char)S[i])) i++;
        if (i >= len) break;
        // 定位当前单词的起止位置
        int start = i;
        while (i < len && !isspace((unsigned char)S[i])) i++;
        int wlen = i - start;
        // 拷贝单词到数组
        char *word = malloc(wlen + 1);
        memcpy(word, S + start, wlen);
        word[wlen] = '\0';
        word_cnt++;
        words = realloc(words, sizeof(char*) * word_cnt);
        words[word_cnt - 1] = word;
    }
    // 追加结束标记
    words = realloc(words, sizeof(char*) * (word_cnt + 1));
    words[word_cnt] = NULL;

    // 测试输出
    for (int j = 0; words[j]; j++) {
        printf("words[%d] = %s\n", j, words[j]);
    }

    // 释放内存
    for (int j = 0; words[j]; j++) free(words[j]);
    free(words);
    return 0;
}

这个方案的优势:

  • 不会修改原输入字符串,直接传入字符串字面量也不会触发内存错误
  • 用标准库isspace判断空白,覆盖所有C标准定义的空白字符,不需要手动维护分隔符集合
  • 逻辑完全可控,方便后续扩展自定义分隔规则

内容的提问来源于stack exchange,提问作者tvojamamkadevlol

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.30 06:12:13