You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C语言并行编辑超2GB的mmap文件时性能骤降问题排查

并行加密算法大文件处理性能骤降问题

核心实现逻辑

每次通过void *fileBytes = mmap(NULL, loadSize, PROT_READ | PROT_WRITE, MAP_SHARED, fd, offset);将待加密文件按1GB大小做mmap映射,创建8个线程把1GB块拆分为4096字节的独立块分配给线程处理,完成后munmap再处理下一块。

线程创建核心代码:

for(t = 0; t < threadCount; t++) {
    pthread_create(&threads[t], NULL, function, &assignments[t]);
} 
for(t = 0; t < threadCount; t++) {
    pthread_join(threads[t], NULL);
}

性能现象

在4核Intel Linux机器上:

  • 处理2GB以下文件时速度极快(1GB约1秒),CPU使用率达250-300%;
  • 处理超2GB文件时,速度大幅下降,CPU使用率降至30-40%,例如8GB文件需约1分24秒。

排查测试

为定位问题,做了简化测试:仅将每个4096字节块的首字节设为0,结果:

  • 1GB文件耗时约0.2秒;
  • 8GB文件耗时约30秒。

测试代码如下:

void *simpleMod(void *data) {
    //get assignment info
    //assignments are a struct that hold an array of jobs (which are pointers) and the job count
    assignment *assignmentPtr = data;
    assignment currentAssignment = *assignmentPtr;
    job *jobList = currentAssignment.jobs;
    unsigned long long jobCount = currentAssignment.jobCount;

    //setup tracking vars
    job currentJob;
    union block *chunk;

    //BYTE_COUNT is 4096
    /*
    union block {
        uint_fast64_t longs[LONG_COUNT];
        char bytes[BYTE_COUNT];
    } block;
    */

    //go through jobs one by one and set first byte to 0
    for(unsigned long long j = 0; j < jobCount; j++) {
        currentJob = jobList[j];
        chunk = currentJob.start;
        chunk->bytes[0] = 0;
    }//end for each job

    return NULL;
}

int runFunction(unsigned long long loadSize, unsigned long long offset, int fd, 
        assignment assignments[], pthread_t threads[], void *(*function)(void *)) {
    //get pointer to location in file
    void *fileBytes =  mmap(NULL, loadSize, PROT_READ | PROT_WRITE, MAP_SHARED, fd, offset);
    if(fileBytes == MAP_FAILED) { //check for error
        printf("Could not open file to be encrypted\n");
        return 1;
    }

    // malloc the jobs
    unsigned long long chunkCount = loadSize / BYTE_COUNT;
    unsigned long long chunksPerThread = chunkCount / threadCount;
    unsigned long long remainingChunks = chunkCount - chunksPerThread * threadCount;
    unsigned long long addPerAssignemnt = (remainingChunks + (threadCount - 1)) / threadCount;

    for(int i = 0; i < threadCount; i++) {
        assignments[i].jobCount = chunksPerThread;
        if(remainingChunks == 0) {
             assignments[i].jobCount += remainingChunks;
            remainingChunks = 0;
        } else if(remainingChunks >= addPerAssignemnt) {
            assignments[i].jobCount += addPerAssignemnt;
            remainingChunks -= addPerAssignemnt;
        }
        assignments[i].jobs = malloc(sizeof(job) * assignments[i].jobCount);
    }

    //then assignments need to be filled with jobs
    unsigned int t = 0;
    unsigned long long index = 0; //the job index within assignment;
    
    //assign chunks as jobs to assignments
    for(unsigned long long c = 0; c < chunkCount; c++) { //cycle through dimensions
        //setup new job
        job *newJob = &assignments[t].jobs[index];
        newJob->start = PTR_PLUS_BYTES(fileBytes, c * BYTE_COUNT);
        //update assignment
        if(t == threadCount - 1) {
            t = 0;
            index++;
        } else {
            t++;
        }
    }

    //setup threads for fucntion
    for(t = 0; t < threadCount; t++) {
        pthread_create(&threads[t], NULL, function, &assignments[t]);
    } 
    for(t = 0; t < threadCount; t++) {
        pthread_join(threads[t], NULL);
    }

    //free the jobs
    for(t = 0; t < threadCount; t++) {
        free(assignments[t].jobs);
    }

    //unmap mem
    munmap(fileBytes, loadSize);

}//end runFunction

int disperseFunction(void *(*function)(void *), pthread_t threads[], 
        assignment assignments[], int fd, unsigned long long fileSize) {

    struct timeval start, end;

    //only load up to certain amount of file at time
    for(unsigned long long lc = 0; lc < fileSize / maxLoad; lc++) {
        runFunction(maxLoad, maxLoad * lc, fd, assignments, threads, function);
    }

    //run remainder
    unsigned long long lastSize = fileSize % maxLoad;
    runFunction(lastSize, fileSize - lastSize, fd, assignments, threads, function);
    
    return 0;
}

int fileSimpleEdit(char *targetFile) {

    //get file size
    int fd = open(targetFile, O_RDWR);                          
    struct stat statBuf;                      
    fstat(fd, &statBuf);    
    unsigned long long originalSize = statBuf.st_size;

    threadCount = 8;
    pthread_t threads[threadCount];
    
    //pad the file
    unsigned short paddingSize = BYTE_COUNT - (originalSize % BYTE_COUNT);
    FILE *unpaddedFile = fopen(targetFile, "a");
    if(unpaddedFile == NULL) {
        printf("Could not open file to be encrypted\n");
        return 1;
    }
    //create padding
    uint_fast8_t padding[paddingSize];
    memset(&padding, 1, paddingSize);
    //write padding
    fwrite(&padding, 1, paddingSize, unpaddedFile);
    fclose(unpaddedFile);
    close(fd);

    //open file after its padded
    fd = open(targetFile, O_RDWR);                                               
    fstat(fd, &statBuf);    
    unsigned long long fileSize = statBuf.st_size;

    //next setup assignments array
    assignment assignments[threadCount];

    disperseFunction(simpleMod, threads, assignments, fd, fileSize);
    
    return 0;
} //end fileEncrypt

原因分析

  1. 页缓存失效与磁盘I/O瓶颈:小文件能完全被系统页缓存容纳,mmap操作直接命中内存;大文件超过系统可用内存后,后续1GB块的mmap会触发页缓存置换,频繁的磁盘读写(尤其是脏页写回)导致CPU等待I/O,使用率骤降。
  2. 任务分配的缓存局部性差:当前轮询式分配分散的4096字节块,线程处理的内存地址不连续,破坏CPU缓存行局部性。小文件时缓存尚可命中,大文件下缓存命中率急剧下降,叠加I/O瓶颈导致性能暴跌。
  3. mmap/munmap累积开销:每处理1GB就执行一次映射解除与重新映射,大文件下重复操作次数多,加上页表维护开销增大,拖慢整体速度。
  4. 内存不足触发Swap交换:若系统物理内存不足,大文件处理时会触发磁盘Swap交换,交换速度远低于内存,导致CPU空闲等待。

优化建议

  • 优化任务分配局部性:给每个线程分配连续的内存块区间,而非分散小批量块,提升CPU缓存命中率。
  • 减少映射操作次数:尝试一次映射更大范围(系统允许的情况下),或复用映射区域,避免频繁mmap/munmap。
  • 调整脏页回写策略:使用msync(MS_ASYNC)批量回写脏页,或修改系统vm.dirty_ratio等参数,优化脏页刷新时机。
  • 监控系统状态:用iostat查看磁盘I/O使用率,free查看内存与Swap占用,perf分析CPU缓存命中率,精准定位瓶颈。

内容的提问来源于stack exchange,提问作者DBlake44

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.19 17:01:08