You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

为何通过cat管道读取stdin比直接读取文件更快?

管道读文件比双线程直接读取更快的奇怪现象

现象概述

我发现以下命令:

cat 1GB.txt | ./program_that_reads_stdin

比直接读文件的命令更快:

./program_that_reads_file_directly

注:每次计时前都会刷新页缓存:

echo 1 > /proc/sys/vm/drop_caches

更新:测试单线程直接读文件实现后,发现它比cat管道版本和我的多线程实现都慢:

cat | ./read from stdin: 0.57s to 0.61s
./read_from_file_multithreaded: 0.60s to 0.64s
./read_from_file_singlethreaded: 0.64s to 0.71s

调整缓冲区大小也没能提升单线程直接读文件的速度。

实际上我试过多款单线程直接读文件实现,它们都比我的多线程实现慢——所以我决定优化多线程方案,而非改用单线程。

程序说明

  • program_that_reads_stdin即下文的example程序:单线程,仅循环调用read读取stdin并计算哈希。
  • program_that_reads_file_directly即下文的b3sum程序:双线程生产者-消费者模式,直接打开1GB.txt读取并处理。

b3sum程序(双线程直接读文件)

源码

#include "blake3.h"

#include <stdio.h>
#include <stdlib.h>
#include <unistd.h>
#include <fcntl.h>
#include <malloc.h>
#include <errno.h>
#include <string.h>
#include <iostream>
#include <chrono>
#include <queue>
#include <thread>
#include <mutex>
#include <condition_variable>

static constexpr std::size_t MAXFILESIZE = 1024 * 1024 * 1024;
static constexpr std::size_t BUFSIZE = 65536; // 128 * 1024 * 1024;
static constexpr std::size_t NUMBUFS = MAXFILESIZE / BUFSIZE + 1;


// globals
std::mutex mut;
unsigned char* buffers[NUMBUFS];
std::queue<char*> available_bufs; // queue of buffers which are available for consumers to consume
int bytes_read[NUMBUFS] = {0};
std::condition_variable cv;
// write_head is the shared variable associated with cv
int write_head = 0; // index of buffer currently being written to - hasn't been fully written yet.

void producer_thread()
{
    int fd;
    int i, n=0;
    const char* fname = "1GB.txt";

    if ((fd = open(fname, O_RDONLY)) < 0) {
        printf("%s: cannot open %s\n", fname);
        exit(2);
    }

    for (int i = 0; i < NUMBUFS; ++i){
        unsigned char* buf = buffers[i];
        int n = read(fd,buf,BUFSIZE);
        bytes_read[i] = n;
        // signal to consumer thread
        {
            std::lock_guard<std::mutex> lk(mut);
            write_head = i + 1;
            //std::cout << "new write_head:" << write_head << std::endl;
        }
        cv.notify_all();

        if ( n == 0 ){ // if we have reached end of file
            std::cout << "Read to end of file" << std::endl;
            std::cout << "Buffers used: " << i;
            return;
        }
    }
}


void consumer_thread(int thread_id){
  // Initialize the hasher.
  blake3_hasher hasher;
  blake3_hasher_init(&hasher);

    // read write_head
    for (int i = 0; i < NUMBUFS; ++i){
        // wait for buffer to become available for reading]
        //puts("waiting...");
        {
            std::unique_lock<std::mutex> lk(mut);
            cv.wait(lk, [&]() { return i < write_head; });
        }
        //puts("finished waiting");
        int n = bytes_read[i];
        //std::cout << "bytes read: " << n << std::endl;
        if ( n == 0 ) {
            // print output
            // Finalize the hash. BLAKE3_OUT_LEN is the default output length, 32 bytes.
            uint8_t output[BLAKE3_OUT_LEN];
            blake3_hasher_finalize(&hasher, output, BLAKE3_OUT_LEN);

            // Print the hash as hexadecimal.
            printf("blake3 hash: ");
            for (size_t i = 0; i < BLAKE3_OUT_LEN; i++) {
                printf("%02x", output[i]);
            }
            printf("\n");
            return ;
        }
        // now process the data
        unsigned char* buf = buffers[i];
        blake3_hasher_update(&hasher, buf, n);
    }

}


int main (int argc, char* argv[]) {
    using std::chrono::high_resolution_clock;
    using std::chrono::duration_cast;
    using std::chrono::duration;
    using std::chrono::milliseconds;

    // create shared buffers
    puts("Allocating buffers");
    auto start = high_resolution_clock::now();
    int alignment = 4096;
    
    // allocate the buffers and put them into the global buffers array
    for (int i = 0; i < NUMBUFS; ++i){
        unsigned char* buf = (unsigned char*) memalign(alignment, BUFSIZE);
        buffers[i] = buf;
    }
    auto end = high_resolution_clock::now();
    /* Getting number of milliseconds as a double. */
    duration<double, std::milli> ms_double = end - start;
    std::cout << "time taken" << ms_double.count() << "ms\n";
    puts("finished allocating buffers");

    // start producer and consumer threads
    std::thread t1(producer_thread), t2(consumer_thread, 1);
    t1.join();
    t2.join();

    return 0;
}

编译与运行

需先构建BLAKE3的静态库,编译命令:

g++ -c b3sum.cc -O3; g++ b3sum.o -o b3sum libblake3.a

运行命令:

echo 1 > /proc/sys/vm/drop_caches
time ./b3sum

测试结果:

  • BUFSIZE设为65536时,平均耗时0.85秒;
  • BUFSIZE改为128 * 1024 * 1024时,平均耗时0.63秒。

example程序(单线程读stdin)

源码

#include "blake3.h"
#include <errno.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h>

int main(void) {
  // Initialize the hasher.
  blake3_hasher hasher;
  blake3_hasher_init(&hasher);

  // Read input bytes from stdin.
  unsigned char buf[65536];
  while (1) {
    ssize_t n = read(STDIN_FILENO, buf, sizeof(buf));
    if (n > 0) {
      blake3_hasher_update(&hasher, buf, n);
    } else if (n == 0) {
      break; // end of file
    } else {
      fprintf(stderr, "read failed: %s\n", strerror(errno));
      exit(1);
    }
  }

  // Finalize the hash. BLAKE3_OUT_LEN is the default output length, 32 bytes.
  uint8_t output[BLAKE3_OUT_LEN];
  blake3_hasher_finalize(&hasher, output, BLAKE3_OUT_LEN);

  // Print the hash as hexadecimal.
  for (size_t i = 0; i < BLAKE3_OUT_LEN; i++) {
    printf("%02x", output[i]);
  }
  printf("\n");
  return 0;
}

运行命令

echo 1 > /proc/sys/vm/drop_caches
time cat 1GB.txt | ./example

测试结果:约0.59秒,约50%概率低于0.6秒(最低0.57秒),比./b3sum快约0.3秒,且./b3sum耗时从未低于0.6秒。

排查与疑问

  • 原本以为是系统调用数量导致性能差异,但strace显示cat管道版本的系统调用更多(片段如下),说明系统调用数量不是原因:
read(3, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
write(1, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
read(3, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
write(1, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
read(3, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
write(1, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
read(3, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
write(1, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
read(3, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
write(1, "\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0\0"..., 131072) = 131072
  • 不同设备表现不同:树莓派上cat file | openssl sha1比sha1sum file慢,但当前机器上前者更快(0.9秒 vs 1.2秒),这可能是因为openssl sha1实现更优。但我的场景中使用的是相同的BLAKE3哈希代码,管道版本不应比双线程实现更快。

  • 调整BUFSIZE仅在设为128MB时明显提速,其他参数无显著效果。

内容的提问来源于stack exchange,提问作者1f604

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.17 02:41:59