RCU链表list_del_rcu与list_for_each_entry使用问题:NOPTI错误排查
RCU链表操作触发NOPTI错误的排查与修复
问题描述
我正在学习RCU链表内核API,程序设置了2个写线程和1个读线程:每隔1秒,两个写线程执行添加新节点操作;当节点总数能被5整除时,删除整个链表。尽管用自旋锁做了保护,仍出现NOPTI错误,怀疑是重复访问已删除节点,但困惑的是调用list_del_rcu()前并未释放锁。
相关代码
#include <linux/init.h> #include <linux/spinlock.h> #include <linux/rcupdate.h> #include <linux/kthread.h> #include <linux/string.h> #include <linux/slab.h> #include <linux/delay.h> #define RD_THREAD 1 #define WR_THREAD 2 #define THREAD_NAME 16 static DEFINE_SPINLOCK(spl); static struct task_struct *task_read[RD_THREAD], *task_write[WR_THREAD]; typedef struct { struct list_head node; int value; } node_t; static LIST_HEAD(head); static atomic_t total; static int read_func(void *arg) { node_t *curNode; while (!kthread_should_stop()) { ssleep(10); rcu_read_lock(); list_for_each_entry_rcu(curNode, &head, node) { printk(KERN_CONT "-> %d ", curNode->value); } printk(KERN_INFO ""); rcu_read_unlock(); } return 0; } static int write_func(void *arg) { int counter = 0; node_t *newNode, *oldNode, *temp; int choice = 0; while (!kthread_should_stop()) { ssleep(1); switch (choice) { case 0: newNode = kmalloc(sizeof (*newNode), GFP_KERNEL); if (!newNode) { printk(KERN_ERR "No memory left.\n"); break; } newNode->value = counter++; spin_lock(&spl); list_add_rcu(&newNode->node, &head); atomic_inc(&total); spin_unlock(&spl); break; case 1: if (counter % 5) { break; } spin_lock(&spl); list_for_each_entry_safe(oldNode, temp, &head, node) { /* * Print node content. */ printk(KERN_INFO "Add %p %d %s %d\n", (void *)oldNode, oldNode->value, current->comm, atomic_read(&total)); list_del_rcu(&oldNode->node); atomic_dec(&total); spin_unlock(&spl); synchronize_rcu(); kfree(oldNode); spin_lock(&spl); } spin_unlock(&spl); break; default: choice = -1; break; } choice++; } spin_lock(&spl); list_for_each_entry_safe(oldNode, temp, &head, node) { list_del_rcu(&oldNode->node); spin_unlock(&spl); synchronize_rcu(); kfree(oldNode); spin_lock(&spl); } spin_unlock(&spl); return 0; } static void end(void) { int rc; unsigned int counter; for (counter = 0; counter < RD_THREAD; ++counter) { if (task_read[counter] && !IS_ERR(task_read[counter])) { rc = kthread_stop(task_read[counter]); printk(KERN_INFO "read_func_%u stopped with rc (%d)\n", counter, rc); } } for (counter = 0; counter < WR_THREAD; ++counter) { if (task_write[counter] && !IS_ERR(task_write[counter])) { rc = kthread_stop(task_write[counter]); printk(KERN_INFO "write_func_%u stopped with rc (%d)\n", counter, rc); } } printk(KERN_INFO "Module unloaded.\n"); return; } static int __init start(void) { unsigned int counter; char thread_name[THREAD_NAME] = { 0 }; for (counter = 0; counter < WR_THREAD; ++counter) { snprintf(thread_name, THREAD_NAME, "write_func_%d", counter); task_write[counter] = kthread_create(write_func, NULL, thread_name); if (IS_ERR(task_write[counter])) { end(); printk(KERN_ERR "Failed to create %s (%ld)\n", thread_name, PTR_ERR(task_write[counter])); return PTR_ERR(task_write[counter]); } else { wake_up_process(task_write[counter]); } } for (counter = 0; counter < RD_THREAD; ++counter) { snprintf(thread_name, THREAD_NAME, "read_func_%d", counter); task_read[counter] = kthread_create(read_func, NULL, thread_name); if (IS_ERR(task_read[counter])) { end(); printk(KERN_ERR "Failed to create %s (%ld)\n", thread_name, PTR_ERR(task_read[counter])); return PTR_ERR(task_read[counter]); } else { wake_up_process(task_read[counter]); } } printk(KERN_INFO "Module started.\n"); return 0; } module_init(start); module_exit(end); MODULE_LICENSE("GPL"); MODULE_AUTHOR("MP");
dmesg错误日志
[ 6124.396574] Module started. [ 6134.590443] -> 2 -> 2 -> 1 -> 1 -> 0 -> 0 [ 6134.590451] [ 6138.750431] Add 00000000f8506a7d 4 write_func_1 10 [ 6138.750435] Add 0000000089a4bcde 4 write_func_0 9 [ 6138.756419] Add 0000000089a4bcde 4 write_func_1 8 [ 6138.756423] list_del corruption, ffff905962895a40->prev is LIST_POISON2 (dead000000000200) [ 6138.756464] ------------[ cut here ]------------ [ 6138.756464] kernel BUG at lib/list_debug.c:50! [ 6138.756484] invalid opcode: 0000 [#1] SMP NOPTI [ 6138.756501] CPU: 3 PID: 19796 Comm: write_func_1 Kdump: loaded Tainted: P OE --------- - - 4.18.0-372.9.1.el8.x86_64 #1
问题根源
- 自旋锁的错误释放:在删除节点的循环中,每删除一个节点就临时释放自旋锁,调用
synchronize_rcu()后再加锁。这会导致其他写线程在解锁期间修改链表结构,当前线程的遍历指针temp可能失效,甚至同一节点被多个线程重复删除,触发链表 corruption。 list_for_each_entry_safe的误用:该宏依赖自旋锁保护下的链表结构稳定,但解锁后链表被其他线程修改,temp指向的下一个节点可能已被删除或篡改,导致遍历逻辑混乱。- 删除条件的错误判断:使用每个写线程本地的
counter判断是否触发删除,多线程下counter值不一致,导致多个线程同时进入删除逻辑,重复操作同一链表节点。
修复方案
核心修改点
- 保持自旋锁直到所有节点从链表中移除,避免遍历期间链表结构被修改。
- 先完成所有
list_del_rcu()操作,统一调用synchronize_rcu()等待读线程退出临界区后,再释放节点内存。 - 使用全局原子变量
total判断删除条件,确保多线程下条件判断一致。
修改后的write_func核心代码
static int write_func(void *arg) { int counter = 0; node_t *newNode, *oldNode, *temp; int choice = 0; static LIST_HEAD(temp_del_list); // 临时存储待释放的节点 while (!kthread_should_stop()) { ssleep(1); switch (choice) { case 0: newNode = kmalloc(sizeof (*newNode), GFP_KERNEL); if (!newNode) { printk(KERN_ERR "No memory left.\n"); break; } newNode->value = counter++; spin_lock(&spl); list_add_rcu(&newNode->node, &head); atomic_inc(&total); spin_unlock(&spl); break; case 1: // 用全局原子变量判断删除条件,而非本地counter if (atomic_read(&total) % 5 != 0) { break; } spin_lock(&spl); // 在锁保护下遍历并移除所有节点到临时链表 list_for_each_entry_safe(oldNode, temp, &head, node) { printk(KERN_INFO "Del %p %d %s %d\n", (void *)oldNode, oldNode->value, current->comm, atomic_read(&total)); list_del_rcu(&oldNode->node); list_add(&oldNode->node, &temp_del_list); } atomic_set(&total, 0); // 直接置0,因为所有节点都已移除 spin_unlock(&spl); // 等待所有读线程退出RCU临界区 synchronize_rcu(); // 释放所有待删除节点内存 list_for_each_entry_safe(oldNode, temp, &temp_del_list, node) { list_del(&oldNode->node); kfree(oldNode); } break; default: choice = -1; break; } choice++; } // 模块退出时的清理逻辑同样修改 spin_lock(&spl); list_for_each_entry_safe(oldNode, temp, &head, node) { list_del_rcu(&oldNode->node); list_add(&oldNode->node, &temp_del_list); } atomic_set(&total, 0); spin_unlock(&spl); synchronize_rcu(); list_for_each_entry_safe(oldNode, temp, &temp_del_list, node) { list_del(&oldNode->node); kfree(oldNode); } return 0; }
修复说明
- 自旋锁仅在修改链表结构(添加/移除节点)时持有,确保遍历期间链表结构稳定,避免多线程干扰。
- 所有待删除节点先移到临时链表,统一等待RCU读临界区结束后再释放内存,符合RCU的使用规范。
- 使用全局原子变量
total判断删除条件,确保多线程下触发删除的时机一致,避免重复删除。
内容的提问来源于stack exchange,提问作者MankPan
相关产品推荐
相关产品推荐

