You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

为何tasklet比workqueue更快?实验异常求分析与优化建议

Tasklet与Workqueue延迟对比实验问题分析与改进建议

实验背景

开展了tasklet与workqueue的延迟对比实验,编写内核模块模拟测试,workqueue测试代码如下。为规避缓存影响,加入了缓存刷新与随机化处理。实验结果显示:多次测试中tasklet耗时始终稳定,但workqueue的耗时在不同测试中差异可达一倍。最初推测是workqueue的上下文切换导致,但添加上下文切换统计后发现,任务为串行执行,仅加入msleep时才会触发上下文切换。目前无法定位问题根源,怀疑存在其他外部影响因素,需分析问题并给出实验改进建议。

Workqueue测试代码

#include <linux/delay.h>
#include <linux/fs.h>
#include <linux/init.h>
#include <linux/kernel.h>
#include <linux/ktime.h>
#include <linux/module.h>
#include <linux/proc_fs.h>
#include <linux/random.h>
#include <linux/sched.h>
#include <linux/workqueue.h>

#include <asm/cacheflush.h>

#define NUM_WORKITEMS 1  // Number of work items to schedule

static ktime_t work_start_time[NUM_WORKITEMS], work_end_time[NUM_WORKITEMS];
static struct workqueue_struct *my_workqueue; // Single workqueue
static struct work_struct works[NUM_WORKITEMS];

static void clear_cpu_cache(void)
{
#if defined(__x86_64__) || defined(__i386__)
    smp_call_function_single(0, (smp_call_func_t)wbinvd, NULL, 1); // Trigger on all CPUs
#elif defined(CONFIG_ARM) || defined(CONFIG_ARM64)
    smp_call_function(clear_cache_all, NULL, 1);  // Call flush_cache_all() on all CPUs
#else
    pr_warn("Cache clearing not implemented for this architecture\n");
#endif
}

static void flush_disk_cache(void)
{
    struct file *f;

    f = filp_open("/proc/sys/vm/drop_caches", O_WRONLY, 0);
    if (IS_ERR(f)) {
        pr_warn("Failed to open /proc/sys/vm/drop_caches\n");
        return;
    }

    kernel_write(f, "3\n", 2, &f->f_pos);
    filp_close(f, NULL);

    pr_info("Disk cache flushed via /proc/sys/vm/drop_caches\n");
}

static void clear_all_caches(void)
{
    clear_cpu_cache();     // Clear CPU caches
    flush_disk_cache();    // Clear disk I/O caches
}

// Function to get the voluntary and involuntary context switches of the current task
static void get_context_switches(struct task_struct *task, long *voluntary, long *involuntary)
{
    *voluntary = task->nvcsw;      // Voluntary context switches
    *involuntary = task->nivcsw;   // Involuntary context switches
}

static void work_handler(struct work_struct *work)
{
    int work_id = (int)((uintptr_t)work - (uintptr_t)works) / sizeof(struct work_struct);
    pid_t pid = current->pid;  // Get the current process ID
    long pre_voluntary, pre_involuntary;
    long post_voluntary, post_involuntary;

    // Get context switch counts before execution
    get_context_switches(current, &pre_voluntary, &pre_involuntary);
    
    // Simulate heavy computation
    long long i, counter = 0;
    unsigned long cpu_start_time = ktime_to_ns(ktime_get()); // Start CPU time
    
    int index;
    for (i = 0; i < 1000000; ++i) {
        get_random_bytes(&index, sizeof(int));
        index %= 1000000;
        counter += index * index;
    }
    
    // msleep(100);

    work_end_time[work_id] = ktime_get(); // End wall time
    unsigned long cpu_end_time = ktime_to_ns(work_end_time[work_id]); // End CPU time
    
    pr_info("Work item %d executed on CPU %d\n", work_id, smp_processor_id());
    pr_info("Work item %d latency: %lld ns, "
            "computation cost (CPU time): %llu ns, counter = %lld\n",
            work_id,
            ktime_to_ns(ktime_sub(work_end_time[work_id], work_start_time[work_id])),
            cpu_end_time - cpu_start_time, counter);

    // Get context switch counts after execution
    get_context_switches(current, &post_voluntary, &post_involuntary);

    pr_info("Context switches before work: Voluntary = %ld, Involuntary = %ld\n", pre_voluntary, pre_involuntary);
    pr_info("Context switches after work: Voluntary = %ld, Involuntary = %ld\n", post_voluntary, post_involuntary);
}

static int __init single_workqueue_init(void)
{
    int i;

    // Create a single workqueue
    my_workqueue = alloc_workqueue("my_single_workqueue", WQ_UNBOUND, 0);
    if (!my_workqueue) {
        pr_err("Failed to create workqueue\n");
        return -ENOMEM;
    }

    clear_all_caches();

    // Initialize and queue multiple work items
    for (i = 0; i < NUM_WORKITEMS; i++) {
        INIT_WORK(&works[i], work_handler);
        work_start_time[i] = ktime_get(); // Record start time
        queue_work(my_workqueue, &works[i]); // Schedule work item on the single workqueue
    }

    pr_info("Single workqueue test module loaded\n");
    return 0;
}

static void __exit single_workqueue_exit(void)
{
    if (my_workqueue) {
        flush_workqueue(my_workqueue); // Ensure all work items are completed
        destroy_workqueue(my_workqueue); // Destroy the workqueue
    }

    pr_info("Single workqueue test module unloaded\n");
}

module_init(single_workqueue_init);
module_exit(single_workqueue_exit);

MODULE_LICENSE("GPL");
MODULE_AUTHOR("Your Name");
MODULE_DESCRIPTION("Single workqueue with multiple work items");

问题根源分析

  • Workqueue调度特性差异:当前使用的是WQ_UNBOUND类型的workqueue,这类workqueue的工作线程属于内核线程池,可被调度到任意CPU执行;而tasklet是绑定在触发时的CPU上以软中断上下文执行,调度延迟更稳定。unbound workqueue的线程可能面临CPU负载不均衡、调度器调度延迟的影响,导致执行时间波动。
  • 缓存刷新的局限性:当前的clear_cpu_cache函数仅在CPU0执行缓存刷新(x86下用smp_call_function_single(0)),但workqueue线程可能在其他CPU上运行,这些CPU的缓存未被清理,导致不同测试中缓存命中率不同,进而影响计算耗时。
  • 随机数生成的不确定性:get_random_bytes在内核中依赖熵池,如果熵池不足,可能会等待熵生成或使用伪随机数生成器,这会导致每次调用的耗时不一致,进而影响整体计算时间。
  • 系统负载干扰:即使没有上下文切换,系统中其他进程/内核任务的CPU占用、中断处理等,也会抢占workqueue线程的CPU时间;而tasklet作为软中断,优先级比普通内核线程高,受到的干扰更小。

实验改进建议

  • 改用绑定型Workqueue:创建workqueue时使用WQ_BOUND标志,指定绑定到特定CPU,确保workqueue线程固定在触发时的CPU执行,和tasklet的执行CPU一致,减少调度带来的波动。示例代码:
    my_workqueue = alloc_workqueue("my_bound_workqueue", WQ_BOUND, 0, smp_processor_id());
    
  • 完善缓存刷新逻辑:确保所有在线CPU的缓存都被清理,x86下遍历所有CPU执行wbinvd:
    int cpu;
    for_each_online_cpu(cpu) {
        smp_call_function_single(cpu, (smp_call_func_t)wbinvd, NULL, 1);
    }
    
  • 替换随机数生成:用确定性的计算替代get_random_bytes,比如固定的伪随机序列(线性同余生成器),避免熵池带来的不确定性:
    static unsigned int seed = 12345;
    // 线性同余生成器(符合POSIX标准)
    unsigned int get_deterministic_random(void) {
        seed = seed * 1103515245 + 12345;
        return (unsigned int)(seed / 65536) % 32768;
    }
    
    之后将循环内的get_random_bytes替换为该函数,确保计算过程耗时稳定。
  • 控制系统负载:测试时尽量降低系统其他负载,关闭不必要的服务、进程;或使用taskset将测试相关的内核线程绑定到特定CPU,减少外部干扰。
  • 增加样本量与统计分析:重复测试100次以上,计算耗时的平均值、方差,通过统计方法区分是随机波动还是系统性问题。
  • 更精确的时间统计:使用local_clock()(x86下等价于rdtsc)获取更精确的CPU时间,区分墙钟时间和CPU执行时间:如果CPU执行时间波动小而墙钟时间波动大,说明是调度延迟导致的波动。
  • 监测CPU调度事件:使用ftrace跟踪workqueue线程的调度事件(如sched_switch、sched_wakeup),查看线程是否被抢占及抢占来源,定位外部干扰因素。

内容的提问来源于stack exchange,提问作者Hoang Huy

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.15 13:24:56