使用pthreads时程序退出随机挂起问题排查及优化咨询
问题描述
熟练使用pthread_create、pthread_join和pthread_detach接口,但对mutex(互斥锁)和条件变量实践经验较少。为避免多次创建/销毁线程的开销,尝试仅用mutex和条件变量实现类似join/barrier的线程同步结构,但编写的代码会随机出现退出挂起的情况,有时又能正常退出。自查认为所有mutex已解锁且线程已完成join,同时性能表现较差。虽不打算用于生产,但希望找出挂起原因,同时咨询是否有更高效的实现方式。
补充说明:主线程创建dt(drinktea)、db(drinkbeer)、ec(eatcrisps)三个线程,db在同步点阻塞ec,dt阻塞db,主线程阻塞dt。已知pthreads有barrier扩展,但希望基于标准组件实现。
问题代码
#include <stdio.h> #include <stdint.h> #include <stdlib.h> #include <string.h> #include <stdbool.h> #include <unistd.h> #include <math.h> #include <time.h> #include <pthread.h> // gcc mutex.c -o mutex.bin -lm -O3 -Wall -ffast-math -march=native -std=c11 -fstrict-aliasing -fasm // using gcc (Ubuntu 9.4.0-1ubuntu1~20.04.1) 9.4.0 uint64_t get_cycles () { uint32_t lo,hi; asm volatile("rdtsc":"=a"(lo),"=d"(hi)); return (( uint64_t)hi<<32 | lo); } pthread_mutex_t mutex_dt = PTHREAD_MUTEX_INITIALIZER; // Est. 1,000,000 cycle min for body pthread_mutex_t mutex_db = PTHREAD_MUTEX_INITIALIZER; pthread_mutex_t mutex_ec = PTHREAD_MUTEX_INITIALIZER; pthread_cond_t cond_dt = PTHREAD_COND_INITIALIZER; pthread_cond_t cond_db = PTHREAD_COND_INITIALIZER; pthread_cond_t cond_ec = PTHREAD_COND_INITIALIZER; uint32_t iterations = 20; uint32_t num_threads; void *drinktea() { uint32_t i, j, k; pthread_mutex_lock(&mutex_dt); for (i=0; i<iterations; i++) { for (j=0; j<iterations; j++) { for(k=0; k<100; k++) { if (!get_cycles()) printf("Drinking Tea...\n"); } } pthread_mutex_lock(&mutex_db); pthread_cond_signal(&cond_dt); if (i<iterations-1) { pthread_cond_wait(&cond_dt, &mutex_dt); } pthread_cond_signal(&cond_db); pthread_mutex_unlock(&mutex_db); } pthread_mutex_unlock(&mutex_dt); return NULL; } void *drinkbeer() { uint32_t i, j, k; pthread_mutex_lock(&mutex_db); pthread_cond_signal(&cond_db); pthread_cond_wait(&cond_db, &mutex_db); for (i=0; i<iterations; i++) { for (j=0; j<iterations; j++) { for(k=0; k<100; k++) { if (!get_cycles()) printf("Drinking Beer...\n"); } } pthread_mutex_lock(&mutex_ec); if (i<iterations-1) { pthread_cond_wait(&cond_db, &mutex_db); } pthread_cond_signal(&cond_ec); pthread_mutex_unlock(&mutex_ec); } pthread_mutex_unlock(&mutex_db); return NULL; } void *eatcrisps() { uint32_t i, j, k; pthread_mutex_lock(&mutex_ec); pthread_cond_signal(&cond_ec); pthread_cond_wait(&cond_ec, &mutex_ec); for (i=0; i<iterations; i++) { for (j=0; j<iterations; j++) { for(k=0; k<100; k++) { if (!get_cycles()) printf("Eating Crisps...\n"); } } if (i<iterations-1) pthread_cond_wait(&cond_ec, &mutex_ec); } pthread_mutex_unlock(&mutex_ec); return NULL; } int warmup() { uint32_t i; for(i=0; i<100000000; i++) { if (!get_cycles()) { printf("Warm up message\n"); } }; return 0; } int main(int argc, char **argv) { uint64_t i, cyclesstart, cyclesend, sumcycles; sumcycles = 0; pthread_t thread_id[3]; if (warmup() != 0) exit(1); // Simulate join/barrier pthread_mutex_lock(&mutex_dt); pthread_mutex_lock(&mutex_db); pthread_mutex_lock(&mutex_ec); cyclesstart = get_cycles(); pthread_create(&thread_id[0], NULL, drinktea, NULL); // approx. 500k cycles for both pthread_create(&thread_id[2], NULL, eatcrisps, NULL); cyclesend = get_cycles(); printf("Pthread Create x2 Cycles = %li \n", cyclesend - cyclesstart); pthread_cond_wait(&cond_ec, &mutex_ec); pthread_mutex_unlock(&mutex_ec); pthread_cond_signal(&cond_ec); pthread_create(&thread_id[1], NULL, drinkbeer, NULL); pthread_cond_wait(&cond_db, &mutex_db); pthread_mutex_unlock(&mutex_db); pthread_cond_signal(&cond_db); for (i=0; i<iterations; i++) { pthread_cond_wait(&cond_dt, &mutex_dt); pthread_cond_signal(&cond_dt); if (i>0) { cyclesend = get_cycles(); sumcycles += cyclesend - cyclesstart; printf("%li \n", cyclesend - cyclesstart); } cyclesstart = get_cycles(); } pthread_mutex_unlock(&mutex_dt); printf("Pthread Mean Cycles = %f \n", (float)sumcycles/(iterations-1)); pthread_join(thread_id[2], NULL); pthread_join(thread_id[1], NULL); pthread_join(thread_id[0], NULL); pthread_exit(0); //exit(0); }
挂起原因分析
1. 信号丢失与时序不确定性
条件变量的信号没有绑定明确的状态标记,存在信号提前发送的问题:比如主线程在子线程还未进入pthread_cond_wait时就发送了信号,导致子线程后续进入等待后永远收不到唤醒信号,从而挂起。这种时序的不确定性直接导致了随机挂起的现象。
2. 未处理虚假唤醒
pthread_cond_wait存在虚假唤醒(无信号触发也返回)的可能,你的代码没有检查唤醒的触发条件是否满足,仅依赖信号触发,一旦出现虚假唤醒,会打破原有的同步链逻辑,导致后续流程错乱。
3. 锁的嵌套与无序操作
drinktea线程中,持有mutex_dt的同时嵌套获取mutex_db,而主线程和drinkbeer线程也会操作mutex_db,这种锁的嵌套和无序操作增加了死锁风险,当线程调度顺序变化时,容易出现互相等待锁释放的情况。
更高效的标准组件实现方式
基于标准mutex和条件变量可以实现可靠的barrier结构,核心是用计数器记录到达同步点的线程数,配合条件变量实现所有线程到达后再继续执行。以下是通用实现:
#include <pthread.h> typedef struct { pthread_mutex_t mutex; pthread_cond_t cond; int count; // 当前到达同步点的线程数 int total; // 参与同步的总线程数 int generation; // 区分同步轮次,避免虚假唤醒 } barrier_t; // 初始化barrier void barrier_init(barrier_t *barrier, int total_threads) { pthread_mutex_init(&barrier->mutex, NULL); pthread_cond_init(&barrier->cond, NULL); barrier->count = 0; barrier->total = total_threads; barrier->generation = 0; } // 等待所有线程到达barrier void barrier_wait(barrier_t *barrier) { pthread_mutex_lock(&barrier->mutex); int gen = barrier->generation; barrier->count++; if (barrier->count == barrier->total) { // 最后一个线程到达,重置计数器并唤醒所有线程 barrier->count = 0; barrier->generation++; pthread_cond_broadcast(&barrier->cond); } else { // 等待直到本轮所有线程到达,处理虚假唤醒 while (gen == barrier->generation) { pthread_cond_wait(&barrier->cond, &barrier->mutex); } } pthread_mutex_unlock(&barrier->mutex); } // 销毁barrier void barrier_destroy(barrier_t *barrier) { pthread_mutex_destroy(&barrier->mutex); pthread_cond_destroy(&barrier->cond); }
实现优势
- 可靠性:通过
generation字段区分同步轮次,彻底避免虚假唤醒的干扰;所有线程到达后才会继续执行,不会出现信号丢失问题。 - 高效性:使用
pthread_cond_broadcast一次性唤醒所有等待线程,减少多次信号发送的开销;锁仅在同步点操作,降低锁竞争。 - 通用性:支持任意数量的线程同步,可直接适配你的多线程迭代场景。
场景适配
在你的代码中,主线程和三个工作线程共用一个barrier,每次迭代完成后调用barrier_wait,即可实现所有线程同步后再进入下一轮,彻底解决原有的挂起和性能问题。
内容的提问来源于stack exchange,提问作者Simon Goater

