You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用OpenMP并行化C语言逐行读取词表文件的函数?

并行读取词表文件的OpenMP实现方案

核心思路

文本文件并行读取的核心是先将文件分割为多个独立块,每个线程负责处理一块,同时确保块边界落在单词的换行分隔处,避免拆分单词。之后每个线程用私有存储收集块内的单词,最后由主线程合并所有线程的结果到主数组。

具体实现步骤

1. 预计算与块边界调整

先获取文件总大小,按线程数拆分出大致均等的块,再调整块的起始和结束位置到最近的换行符,保证每个块内的单词都是完整的。

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <omp.h>

#define MAX_WORD_LEN 56

// 调整块起始位置到下一个换行符(跳过不完整单词)
void adjust_start(FILE *fp, off_t *start) {
    fseek(fp, *start, SEEK_SET);
    char c;
    while ((c = fgetc(fp)) != EOF && c != '\n') {
        (*start)++;
    }
    if (c == '\n') (*start)++; // 跳过换行符,从下一行开始
}

// 调整块结束位置到上一个换行符(截断不完整单词)
void adjust_end(FILE *fp, off_t *end) {
    if (*end == 0) return;
    fseek(fp, *end - 1, SEEK_SET);
    char c;
    while ((c = fgetc(fp)) != EOF && c != '\n') {
        (*end)--;
        if (*end == 0) break;
        fseek(fp, *end - 1, SEEK_SET);
    }
}

2. 线程私有单词存储结构

为每个线程定义私有动态数组,用来存储读取到的单词,支持自动扩容避免内存浪费:

typedef struct {
    char **words;
    int count;       // 当前已存储单词数
    int capacity;    // 数组容量
} ThreadWordList;

// 初始化线程私有单词列表
ThreadWordList init_thread_list() {
    ThreadWordList list;
    list.count = 0;
    list.capacity = 1024; // 初始容量可根据实际情况调整
    list.words = malloc(list.capacity * sizeof(char*));
    return list;
}

// 向私有列表添加单词
void add_word(ThreadWordList *list, const char *word) {
    if (list->count >= list->capacity) {
        list->capacity *= 2;
        list->words = realloc(list->words, list->capacity * sizeof(char*));
    }
    list->words[list->count] = malloc(MAX_WORD_LEN + 1);
    strncpy(list->words[list->count], word, MAX_WORD_LEN);
    list->words[list->count][MAX_WORD_LEN] = '\0'; // 确保字符串终止符
    list->count++;
}

3. 并行读取与结果合并

使用OpenMP并行区域,每个线程打开独立的文件指针(避免共享指针的竞争),处理自己的块并收集单词,最后主线程合并所有线程的私有列表:

char** parallel_read_wordlist(const char *filename, int *total_count) {
    FILE *fp = fopen(filename, "r");
    if (!fp) {
        perror("Failed to open file");
        return NULL;
    }

    // 获取文件总大小
    fseek(fp, 0, SEEK_END);
    off_t file_size = ftell(fp);
    fseek(fp, 0, SEEK_SET);

    int num_threads = omp_get_max_threads();
    off_t block_size = file_size / num_threads;

    // 存储所有线程的私有单词列表
    ThreadWordList *thread_lists = malloc(num_threads * sizeof(ThreadWordList));

    #pragma omp parallel num_threads(num_threads)
    {
        int tid = omp_get_thread_num();
        ThreadWordList list = init_thread_list();
        FILE *thread_fp = fopen(filename, "r"); // 每个线程独立打开文件
        if (!thread_fp) {
            perror("Thread failed to open file");
            #pragma omp cancel parallel
        }

        // 计算当前线程的块范围
        off_t start = tid * block_size;
        off_t end = (tid == num_threads - 1) ? file_size : (tid + 1) * block_size;

        // 调整块边界到完整单词
        if (tid != 0) adjust_start(thread_fp, &start);
        adjust_end(thread_fp, &end);

        // 定位到块起始位置
        fseek(thread_fp, start, SEEK_SET);

        char buffer[MAX_WORD_LEN + 2]; // 多留一位存储换行符
        while (ftell(thread_fp) < end && fgets(buffer, sizeof(buffer), thread_fp)) {
            // 去掉换行符
            buffer[strcspn(buffer, "\n")] = '\0';
            if (strlen(buffer) > 0) { // 跳过空行
                add_word(&list, buffer);
            }
        }

        fclose(thread_fp);
        thread_lists[tid] = list;
    }

    // 合并所有线程的单词到主数组
    *total_count = 0;
    for (int i = 0; i < num_threads; i++) {
        *total_count += thread_lists[i].count;
    }

    char **result = malloc(*total_count * sizeof(char*));
    int idx = 0;
    for (int i = 0; i < num_threads; i++) {
        for (int j = 0; j < thread_lists[i].count; j++) {
            result[idx++] = thread_lists[i].words[j];
        }
        free(thread_lists[i].words); // 释放线程私有数组的外层指针
    }
    free(thread_lists);
    fclose(fp);

    return result;
}

关键优化与注意事项

  • 独立文件指针:每个线程单独打开文件,避免共享指针的锁竞争,大幅提升并行效率。
  • 动态扩容:线程私有列表初始容量设为合理值,满了自动翻倍,平衡内存占用和扩容开销。
  • 空行处理:跳过文件中的空行,确保结果数组只包含有效单词。
  • 编译选项:编译时需添加OpenMP支持,例如GCC使用-fopenmp参数。
  • 超长单词处理:如果存在超过MAX_WORD_LEN的单词,可调整缓冲区大小,或在读取时做截断处理(根据业务需求)。

使用示例

int main() {
    int total_words;
    char **wordlist = parallel_read_wordlist("words.txt", &total_words);
    if (!wordlist) {
        return 1;
    }

    // 业务逻辑处理...

    // 释放内存
    for (int i = 0; i < total_words; i++) {
        free(wordlist[i]);
    }
    free(wordlist);
    return 0;
}

内容的提问来源于stack exchange,提问作者ric

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.15 10:08:14