You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

为何io_uring轮询模式下文件读取缓慢且需多次运行?

问题概述

我正在开发一个需使用io_uring轮询模式的项目,但遇到两个问题:

  1. 需多次运行程序才能输出文件内容,否则会陷入长时间等待;
  2. 文件读取速度极慢,例如读取仅含“Hello world!”的小文件,耗时仍超10秒。

补充信息

  • Linux系统内核版本:6.9.5-1-default;
  • 必须使用IORING_SETUP_SQPOLL标志,无需调用io_uring_enter系统调用,需启用提交队列轮询(submission queue polling)功能;
  • 不使用任何liburing API;
  • 即使移除代码中的sleep(10),读取速度仍很慢且无输出。

疑问

  1. 为何需多次运行程序才能输出文件内容?
  2. 为何io_uring在轮询模式下读取文件如此缓慢?

代码

#include <stdio.h>
#include <stdlib.h>
#include <stdatomic.h>
#include <sys/stat.h>
#include <sys/ioctl.h>
#include <sys/syscall.h>
#include <sys/mman.h>
#include <sys/uio.h>
#include <linux/fs.h>
#include <fcntl.h>
#include <time.h>
#include <unistd.h>
#include <string.h>
#include <inttypes.h>  // For PRIdMAX
#include <sys/types.h> // For off_t

/* If your compilation fails because the header file below is missing,
 * your kernel is probably too old to support io_uring.
 * */
#include <linux/io_uring.h>

#define QUEUE_DEPTH 128
#define BLOCK_SZ    4096
int first;
/* This is x86 specific */
#define read_barrier()  __asm__ __volatile__("":::"memory")
#define write_barrier() __asm__ __volatile__("":::"memory")

/* Macros for barriers needed by io_uring */
#define io_uring_smp_store_release(p, v)            
    atomic_store_explicit((_Atomic typeof(*(p)) *)(p), (v), 
                  memory_order_release)
#define io_uring_smp_load_acquire(p)                
    atomic_load_explicit((_Atomic typeof(*(p)) *)(p),   
                 memory_order_acquire)
struct app_io_sq_ring {
    unsigned *head;
    unsigned *tail;
    unsigned *ring_mask;
    unsigned *ring_entries;
    unsigned *flags;
    unsigned *array;
};

struct app_io_cq_ring {
    unsigned *head;
    unsigned *tail;
    unsigned *ring_mask;
    unsigned *ring_entries;
    struct io_uring_cqe *cqes;
};

struct submitter {
    int ring_fd;
    struct app_io_sq_ring sq_ring;
    struct io_uring_sqe *sqes;
    struct app_io_cq_ring cq_ring;
};

struct file_info {
    off_t file_sz;
    struct iovec iovecs[];      /* Referred by readv/writev */
};

typedef struct {
  FILE *fp;
  struct submitter *s;
  struct file_info *fi;
} my_file;


/*
 * This code is written in the days when io_uring-related system calls are not
 * part of standard C libraries. So, we roll our own system call wrapper
 * functions.
 * */

int io_uring_setup(unsigned entries, struct io_uring_params *p)
{
    return (int) syscall(__NR_io_uring_setup, entries, p);
}

int io_uring_enter(int ring_fd, unsigned int to_submit,
                          unsigned int min_complete, unsigned int flags)
{
    return (int) syscall(__NR_io_uring_enter, ring_fd, to_submit, min_complete,
                   flags, NULL, 0);
}

int io_uring_register(unsigned int fd, unsigned int opcode,
                      const void *arg, unsigned int nr_args)
{
    return (int) syscall(__NR_io_uring_register, fd, opcode, arg, nr_args);
}


off_t get_file_size(int fd) {
    struct stat st;

    if(fstat(fd, &st) < 0) {
        perror("fstat");
        return -1;
    }
    if (S_ISBLK(st.st_mode)) {
        unsigned long long bytes;
        if (ioctl(fd, BLKGETSIZE64, &bytes) != 0) {
            perror("ioctl");
            return -1;
        }
        return bytes;
    } else if (S_ISREG(st.st_mode))
        return st.st_size;

    return -1;
}

off_t get_file_size2(FILE *file) {
    struct stat st;
    int fd = fileno(file);  // 获取与 FILE* 关联的文件描述符

    if (fd == -1) {
        perror("fileno");
        return -1;
    }

    if (fstat(fd, &st) < 0) {
        perror("fstat");
        return -1;
    }

    if (S_ISBLK(st.st_mode)) {
        unsigned long long bytes;
        if (ioctl(fd, BLKGETSIZE64, &bytes) != 0) {
            perror("ioctl");
            return -1;
        }
        return bytes;
    } else if (S_ISREG(st.st_mode)) {
        return st.st_size;
    }

    return -1;
}


int app_setup_uring(struct submitter *s) {
    memset(s, 0, sizeof(*s));
    struct app_io_sq_ring *sring = &s->sq_ring;
    struct app_io_cq_ring *cring = &s->cq_ring;
    struct io_uring_params p;
    void *sq_ptr, *cq_ptr;

    memset(&p, 0, sizeof(p));
    p.flags |= IORING_SETUP_SQPOLL;
    p.flags |= IORING_SETUP_SQ_AFF;
    p.sq_thread_idle = 20000;
    p.sq_thread_cpu = 4;
    s->ring_fd = io_uring_setup(QUEUE_DEPTH, &p);
    if (s->ring_fd < 0) {
      perror("io_uring_setup");
      return 1;
    }

    int sring_sz = p.sq_off.array + p.sq_entries * sizeof(unsigned);
    int cring_sz = p.cq_off.cqes + p.cq_entries * sizeof(struct io_uring_cqe);

    if (p.features & IORING_FEAT_SINGLE_MMAP) {
        if (cring_sz > sring_sz) {
            sring_sz = cring_sz;
        }
        cring_sz = sring_sz;
    }

    sq_ptr = mmap(0, sring_sz, PROT_READ | PROT_WRITE, 
            MAP_SHARED | MAP_POPULATE,
            s->ring_fd, IORING_OFF_SQ_RING);
    if (sq_ptr == MAP_FAILED) {
        perror("mmap");
        return 1;
    }

    if (p.features & IORING_FEAT_SINGLE_MMAP) {
        cq_ptr = sq_ptr;
    } else {
        /* Map in the completion queue ring buffer in older kernels separately */
        cq_ptr = mmap(0, cring_sz, PROT_READ | PROT_WRITE, 
                MAP_SHARED | MAP_POPULATE,
                s->ring_fd, IORING_OFF_CQ_RING);
        if (cq_ptr == MAP_FAILED) {
            perror("mmap");
            return 1;
        }
    }

    sring->head = sq_ptr + p.sq_off.head;
    sring->tail = sq_ptr + p.sq_off.tail;
    sring->ring_mask = sq_ptr + p.sq_off.ring_mask;
    sring->ring_entries = sq_ptr + p.sq_off.ring_entries;
    sring->flags = sq_ptr + p.sq_off.flags;
    sring->array = sq_ptr + p.sq_off.array;

    s->sqes = mmap(0, p.sq_entries * sizeof(struct io_uring_sqe),
            PROT_READ | PROT_WRITE, MAP_SHARED | MAP_POPULATE,
            s->ring_fd, IORING_OFF_SQES);
    if (s->sqes == MAP_FAILED) {
        perror("mmap");
        return 1;
    }

    cring->head = cq_ptr + p.cq_off.head;
    cring->tail = cq_ptr + p.cq_off.tail;
    cring->ring_mask = cq_ptr + p.cq_off.ring_mask;
    cring->ring_entries = cq_ptr + p.cq_off.ring_entries;
    cring->cqes = cq_ptr + p.cq_off.cqes;

    return 0;
}

void output_to_console(char *buf, int len) {
    while (len--) {
        fputc(*buf++, stdout);
    }
}


void my_fread(size_t size, size_t count, my_file *restrict mf) {
    struct submitter*s = mf->s;
    struct file_info *fi;
    struct app_io_cq_ring *cring = &s->cq_ring;
    struct io_uring_cqe *cqe;
    size_t total_bytes = size * count;
    unsigned head, reaped = 0;

    head = *cring->head;

    do {
        read_barrier();

        if (head == *cring->tail)
            break;

        /* Get the entry */
        cqe = &cring->cqes[head & *s->cq_ring.ring_mask];
        fi = (struct file_info*) cqe->user_data;
        if (cqe->res < 0)
            fprintf(stderr, "Error: %s\n", strerror(abs(cqe->res)));

        int blocks = (int) total_bytes / BLOCK_SZ;
        if (total_bytes % BLOCK_SZ) blocks++;

        for (int i = 0; i < blocks; i++)
            output_to_console(fi->iovecs[i].iov_base, fi->iovecs[i].iov_len);

        head++;
    } while (1);

    *cring->head = head;
    write_barrier();
}

my_file *my_fopen(const char *filename, const char *mode) {
    struct submitter *s = malloc(sizeof(struct submitter));
    struct file_info *fi;
    if(app_setup_uring(s)) {
        fprintf(stderr, "Unable to setup uring!\n");
        free(s);
        return NULL;
    }

    // TODO: Write file
    FILE *fp = fopen(filename, mode);
    if (!fp) {
        printf("Fopen Failed!");
        return NULL;
    }
    int fd = fileno(fp);

    if (fd < 0) {
       printf("Fopen fileno");
        return NULL;
    }

    //printf("fd is %d, fd1 is %d\n", fd, fd1);
    struct app_io_sq_ring *sring = &s->sq_ring;
    unsigned index = 0, current_block = 0, tail = 0, next_tail = 0;

    off_t file_sz = get_file_size2(fp);
    printf("The size of the file is: %" PRIdMAX " bytes\n", (intmax_t)file_sz);
    if (file_sz < 0)
      return NULL;
    off_t bytes_remaining = file_sz;
    int blocks = (int)file_sz / BLOCK_SZ;
    if (file_sz % BLOCK_SZ)
      blocks++;
    
    fi = malloc(sizeof(*fi) + sizeof(struct iovec) * blocks);
    if (!fi) {
         fprintf(stderr, "Unable to allocate memory\n");
         return NULL;
    }
    fi->file_sz = file_sz;
    my_file *mf = malloc(sizeof(my_file));;
    mf->s = s;
    mf->fi = fi;
    mf->fp = fp;

    while (bytes_remaining) {
      off_t bytes_to_read = bytes_remaining;
      if (bytes_to_read > BLOCK_SZ)
        bytes_to_read = BLOCK_SZ;

      fi->iovecs[current_block].iov_len = bytes_to_read;

      void *buf;
      if (posix_memalign(&buf, BLOCK_SZ, BLOCK_SZ)) {
        perror("posix_memalign");
        return NULL;
      }
      fi->iovecs[current_block].iov_base = buf;

      current_block++;
      bytes_remaining -= bytes_to_read;
    }

    /* Add our submission queue entry to the tail of the SQE ring buffer */
    next_tail = tail = *sring->tail;
    next_tail++;
    read_barrier();
    index = tail & *s->sq_ring.ring_mask;
    struct io_uring_sqe *sqe = &s->sqes[index];
    sqe->fd = fd;
    sqe->flags = 0;
    sqe->opcode = IORING_OP_READV;
    sqe->addr = (unsigned long)fi->iovecs;
    sqe->len = blocks;
    sqe->off = 0;
    sqe->user_data = (unsigned long long)fi;
    sring->array[index] = index;
    tail = next_tail;

    if (*sring->tail != tail) {
      *sring->tail = tail;
      write_barrier();
    }

    if ((*sring->flags) & IORING_SQ_NEED_WAKEUP) {
      first++;
      int ret = io_uring_enter(s->ring_fd, 1, 1, IORING_ENTER_GETEVENTS);
      if (ret < 0) {
        perror("io_uring_enter");
        return NULL;
      }
    }
    return mf;
}

int main(int argc, char *argv[]) {
    first = 0;
    if (argc < 2) {
        fprintf(stderr, "Usage: %s <filename>\n", argv[0]);
        return 1;
    }

    for (int i = 1; i < argc; i++)
    {
        my_file *mf = my_fopen(argv[i], "r");
        if (mf != NULL) {
            sleep(10);
            my_fread(mf->fi->file_sz, 1,mf);
        }
        else {
            printf("Fopen Fail!");
            break;
        }
    }

    printf("io_uring_enter times = %d\n", first);
    return 0;
}

运行命令

$ gcc test.c -o example
$ ./example file_path1 file_path2

期望

该示例能在轮询模式下快速读取文件。


问题分析与解决方案

1. 多次运行才输出内容的原因

  • SQ线程唤醒逻辑缺失:代码仅在IORING_SQ_NEED_WAKEUP标志位触发时才唤醒SQPOLL线程,但SQ线程可能已进入idle休眠状态,此时标志位未必会被设置,导致提交的SQE无法被及时处理。多次运行时可能恰好赶上SQ线程处于活跃状态,任务才会被调度执行。
  • CQE读取逻辑不完善:my_fread仅尝试读取一次CQE就退出,如果任务还未完成,就会直接跳过输出;只有当多次运行时延迟完成的任务刚好被读取到,才会有输出。

2. 读取速度极慢的原因

  • 文件描述符冲突:使用fopen打开的FILE*带有用户态缓冲,而IORING_OP_READV直接操作底层文件描述符,两者缓存机制冲突,导致额外IO开销和数据同步问题。
  • SQ线程idle时间过长:p.sq_thread_idle = 20000(20秒)意味着SQ线程空闲20秒才会休眠,但任务提交时线程若已休眠,唤醒延迟极高;同时绑定的CPU核心可能繁忙,导致SQ线程无法及时调度。
  • CQE轮询缺失:未添加循环等待CQE完成的逻辑,sleep(10)属于盲等,无法保证任务已处理完成。

修复方案

  1. 替换文件打开方式:用open直接获取无缓冲的文件描述符,避免与FILE*缓存冲突:
    // 替换原fopen代码
    int fd = open(filename, O_RDONLY);
    if (fd < 0) {
        perror("open");
        return NULL;
    }
    
  2. 调整SQ线程参数:缩短sq_thread_idle时间,避免线程长时间休眠:
    p.sq_thread_idle = 1000; // 1秒后进入idle
    
  3. 强制唤醒SQ线程:提交SQE后直接调用io_uring_enter唤醒SQ线程,无需依赖标志位判断:
    // 替换原唤醒逻辑
    io_uring_s
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 00:09:59