Linux sys/shm.h共享内存延迟20us?求代码排查与优化建议
共享内存IPC延迟过高问题排查与优化
我需要实现进程间数据交换,选择了shared memory(共享内存),因为它是最快的IPC方式,但实测结果却不符预期。目前测得发送与接收的延迟约20us,远高于预期的纳秒级,请求排查代码问题或提供优化方案。
实现代码
string_worker.hpp(基础类,确保读写进程均可创建shm)
#pragma once #include <sys/ipc.h> #include <sys/shm.h> #include <sys/types.h> #include <sys/time.h> #include <thread> #include <chrono> #include <iostream> #include <cstring> #include <error.h> #include <errno.h> #include <fcntl.h> #include <unistd.h> struct Header { // Header is visable for reader and writer, to know where is the last data position const int32_t size_; volatile int32_t tail_; // writable for writer, read by reader Header(int32_t size, int32_t tail) : size_(size), tail_(tail) {} void Show(const std::string & tag = "") const { printf("%s size = %d, tail = %d\n", tag.c_str(), size_, tail_); } }; class StringWorker { public: virtual ~StringWorker() { shmdt(m_data); shmctl(shmid, IPC_RMID, 0); } protected: key_t get_keyid(const std::string & name) { // name the shared memory const std::string & s = "./" + name; if (access(s.c_str(), F_OK) == -1) { const std::string & s1 = "touch " + s; system(s1.c_str()); } key_t semkey = ftok(name.c_str(), 1); if (semkey == -1) { printf("shm_file:%s not existed\n", s.c_str()); exit(1); } return semkey; } void init(const std::string& name, int size) { // create or connect to the shm key_t m_key = get_keyid(name); char * p = nullptr; shmid = shmget(m_key, 0, 0); // TODO: check m_key existed if (shmid == -1) { if (errno != ENOENT && errno != EINVAL) { printf("errno is %s\n", strerror(errno)); exit(1); } shmid = shmget(m_key, sizeof(Header) + 1024 * 64 * size, 0666 | IPC_CREAT | O_EXCL); if (shmid == -1) { printf("both connet and create are failed for shm\n"); exit(1); } p = (char*)shmat(shmid, NULL, 0); printf("creating new shm %s %p\n", name.c_str(), p); Header h(size, 0); memcpy(p, &h, sizeof(Header)); } else { p = (char*)shmat(shmid, NULL, 0); printf("existed m_data = %p\n", p); } if (p == nullptr) { printf("shmat failed"); exit(1); } header_ = (Header*)p; m_data = p + sizeof(Header); } int shmid; char* m_data; Header* header_; };
string_sender.hpp(发送端)
#pragma once #include <mutex> #include <fcntl.h> #include <fstream> #include <sstream> #include "./string_worker.hpp" class StringSender: public StringWorker { public: StringSender(const std::string& key, int size = 4096) { init(key, size); tail_ = &header_->tail_; } void Send(const char * p, size_t size) { char * cur_data = m_data + *tail_ % header_->size_; memcpy(cur_data, p, size + 1); // '\0' *tail_ += 1; } private: volatile int32_t * tail_; }; int main() { timeval t; StringSender sender("a"); std::string a; while (std::cin >> a && a != "quit") { sender.Send(a.data(), a.size()); gettimeofday(&t, NULL); std::cout << a << " sent " << t.tv_usec << std::endl; } }
string_recver.hpp(接收端)
#pragma once #include "./string_worker.hpp" class StringRecver : public StringWorker { public: StringRecver(const std::string & key, int size = 4096) { init(key, size); read_index_ = header_->tail_; } std::string_view Recv() { while (read_index_ == header_->tail_) {} read_index_ += 1; return m_data + read_index_ % header_->size_; } private: int32_t read_index_; }; int main() { timeval t; StringRecver recver("a"); while (true) { std::string_view a = recver.Recv(); gettimeofday(&t, NULL); std::cout << "recv " << a << " " << t.tv_usec << std::endl; } }
编译命令
#!/bin/bash g++ -std=c++2a -O3 sender.cpp -o s g++ -std=c++2a -O3 recver.cpp -o r
程序输出
发送端
existed m_data = 0x7f0612d08000 nasd nasd sent 453488 asd asd sent 989748 nasd nasd sent 355481
接收端
existed m_data = 0x7fa263818000 recv asd 453509 recv sd 989769 recv asd 355502
硬件信息
Intel(R) Core(TM) i9-10900K CPU @ 3.70GHz
问题分析与优化方案
核心问题点
忙等待导致调度开销
接收端的空循环while (read_index_ == header_->tail_) {}会持续占用CPU,触发操作系统进程调度切换,带来数十微秒的延迟。内存可见性保障不足
仅用volatile无法保证多进程场景下的内存屏障正确性,编译器和CPU的指令重排可能导致接收端无法及时感知tail_的更新。计时方式不准确
gettimeofday的tv_usec仅记录微秒部分,跨秒时数值跳变,无法准确计算单次传输延迟。共享内存销毁逻辑错误
读写进程的析构函数都会调用shmctl(IPC_RMID),导致其中一个进程销毁共享内存后,另一个进程访问出错。
具体优化措施
1. 替换忙等待为轻量级同步机制
用POSIX信号量替代空循环,减少CPU空转和调度延迟:
- 在
Header中添加信号量:struct Header { const int32_t size_; std::atomic<int32_t> tail_; sem_t sem; Header(int32_t size, int32_t tail) : size_(size), tail_(tail) {} }; - 创建共享内存时初始化信号量:
sem_init(&h.sem, 1, 0); // 跨进程共享,初始值0 - 发送端写完数据后发信号:
sem_post(&header_->sem); - 接收端等待信号:
sem_wait(&header_->sem);
2. 用原子类型保障内存可见性
替换volatile int32_t为std::atomic<int32_t>,自动插入内存屏障:
// 发送端更新 tail_->fetch_add(1, std::memory_order_release); // 接收端读取 while (read_index_ == tail_->load(std::memory_order_acquire)) {}
3. 优化计时方式
使用std::chrono::high_resolution_clock计算准确延迟,需将发送时间戳写入共享内存:
// 发送端 auto send_ts = std::chrono::high_resolution_clock::now().time_since_epoch().count(); // 将send_ts写入共享内存对应位置 sender.Send(a.data(), a.size()); // 接收端 std::string_view a = recver.Recv(); auto recv_ts = std::chrono::high_resolution_clock::now().time_since_epoch().count(); auto delay = recv_ts - send_ts; // 纳秒级差值 std::cout << "recv " << a << " delay: " << delay << "ns" << std::endl;
4. 修复共享内存销毁逻辑
仅让创建共享内存的进程负责销毁:
class StringWorker { protected: bool is_creator_ = false; void init(...) { if (shmid == -1) { // 创建共享内存逻辑 is_creator_ = true; } else { is_creator_ = false; } } public: virtual ~StringWorker() { shmdt(m_data); if (is_creator_) shmctl(shmid, IPC_RMID, 0); } };
5. 其他优化点
- 用
mlock锁定共享内存到物理内存,避免页交换延迟:mlock(p, sizeof(Header) + 1024*64*size); - 对齐数据结构到64字节缓存行,避免伪共享:在
Header中添加填充字段补全64字节。
内容的提问来源于stack exchange,提问作者kevin h
相关产品推荐
相关产品推荐

