如何在不违反严格别名规则的前提下优化util_memswap函数?
我在对以下util_memswap函数实现进行代码评审时,收到了优化建议:
void util_memswap(size_t psize, void *restrict p1, void *restrict p2) { unsigned char *a = p1; unsigned char *b = p2; unsigned char tmp; while (psize--) { tmp = *a; *a++ = *b; *b++ = tmp; } }
建议指出,该逐字节交换的实现可针对大对象优化,类似标准库memcpy/memmove会根据目标平台架构采用更大的内存单元操作,仅在首尾或非对齐数据时逐字节处理;且编译器难以自动识别此模式进行合并优化。
我发现部分memcpy实现会将void*转为long*以按字拷贝,但这似乎违反严格别名规则(因为void*指向的原对象可能并非long*类型)。而用三次memcpy调用替代的方案,据Peter Cordes所述可能会影响编译器优化效果。
该util_memswap函数用于如下SWAP宏中:
#include <assert.h> #include <string.h> /** * Like C11's _Static_assert() except that it can be used in an expression. * * EXPR - The expression to check. * MSG - The string literal of the error message to print only if EXPR evalutes * to false. * * Always return true. */ #define STATIC_ASSERT_EXPR(EXPR, MSG) \ (!!sizeof( struct { static_assert ( (EXPR), MSG ); char c; } )) /* OP is aware that this would not work for VLAs, or variables that are * declared with the `register` storage class, or identical variables. */ #define SWAP(A, B) \ util_memswap((sizeof *(1 ? &(A) : &(B)) \ * STATIC_ASSERT_EXPR(sizeof (A) == sizeof (B), \ "Arguments of SWAP() must have same size and compatible types.")), \ &(A), \ &(B))
请问在不违反严格别名规则的前提下,我该如何优化上述util_memswap函数?
在不违反严格别名规则的前提下,可通过以下几种实用方式优化util_memswap:
1. 分场景处理:小对象逐字节,大对象用memcpy分块交换
对于小尺寸交换(比如小于等于平台L1缓存行大小),保留原逐字节逻辑避免额外开销;大对象则用栈上缓冲区分块调用memcpy完成交换,完全符合标准规则,同时利用memcpy的平台优化特性:
#include <string.h> #include <stddef.h> // 缓冲区大小可根据平台调整,比如设为64字节(常见L1缓存行大小) #define SWAP_BUF_SIZE 64 void util_memswap(size_t psize, void *restrict p1, void *restrict p2) { if (psize == 0) return; // 小对象直接逐字节交换 if (psize <= SWAP_BUF_SIZE) { unsigned char *a = p1; unsigned char *b = p2; unsigned char tmp; while (psize--) { tmp = *a; *a++ = *b; *b++ = tmp; } return; } // 大对象分块处理,避免栈溢出 unsigned char buf[SWAP_BUF_SIZE]; size_t remaining = psize; unsigned char *a = p1; unsigned char *b = p2; while (remaining > SWAP_BUF_SIZE) { memcpy(buf, a, SWAP_BUF_SIZE); memcpy(a, b, SWAP_BUF_SIZE); memcpy(b, buf, SWAP_BUF_SIZE); a += SWAP_BUF_SIZE; b += SWAP_BUF_SIZE; remaining -= SWAP_BUF_SIZE; } // 处理剩余字节 memcpy(buf, a, remaining); memcpy(a, b, remaining); memcpy(b, buf, remaining); }
2. 利用_Generic让宏匹配类型选择最优交换
既然SWAP宏能获取操作对象的类型,可借助C11的_Generic特性,针对内置类型直接调用字级交换函数,自定义大类型回退到逐字节或缓冲区方案,完全不触碰严格别名规则:
#include <stddef.h> #include <string.h> // 基础逐字节交换 static inline void util_memswap_byte(size_t psize, void *restrict p1, void *restrict p2) { unsigned char *a = p1; unsigned char *b = p2; unsigned char tmp; while (psize--) { tmp = *a; *a++ = *b; *b++ = tmp; } } // 针对long类型的字交换 static inline void util_swap_long(long *a, long *b) { long tmp = *a; *a = *b; *b = tmp; } // 针对long long类型的双字交换 static inline void util_swap_llong(long long *a, long long *b) { long long tmp = *a; *a = *b; *b = tmp; } // 扩展SWAP宏,自动匹配最优交换方式 #define SWAP(A, B) do { \ STATIC_ASSERT_EXPR(sizeof(A) == sizeof(B), "SWAP arguments must have same size"); \ _Generic((A), \ long: util_swap_long, \ long long: util_swap_llong, \ int: util_swap_long, // 可根据平台调整类型映射 \ default: util_memswap_byte \ )(&A, &B); \ } while(0)
3. 编译器扩展:用may_alias属性实现对齐优化(谨慎使用)
如果只针对GCC/Clang等编译器,可使用__attribute__((may_alias))标记指针类型,允许其访问任意类型对象,从而实现类似标准库的对齐优化:
#include <stddef.h> void util_memswap(size_t psize, void *restrict p1, void *restrict p2) { if (psize == 0) return; unsigned char *a_byte = p1; unsigned char *b_byte = p2; // 处理开头未对齐的字节 size_t align = _Alignof(long); size_t unaligned = (size_t)a_byte % align; if (unaligned != 0) { unaligned = align - unaligned; unaligned = unaligned > psize ? psize : unaligned; for (size_t i = 0; i < unaligned; i++) { unsigned char tmp = *a_byte; *a_byte++ = *b_byte; *b_byte++ = tmp; } psize -= unaligned; } // 用带may_alias属性的long指针做字交换 typedef long __attribute__((may_alias)) alias_long; alias_long *a_word = (alias_long*)a_byte; alias_long *b_word = (alias_long*)b_byte; size_t num_words = psize / sizeof(long); for (size_t i = 0; i < num_words; i++) { alias_long tmp = *a_word; *a_word++ = *b_word; *b_word++ = tmp; } // 处理剩余字节 psize %= sizeof(long); for (size_t i = 0; i < psize; i++) { unsigned char tmp = *a_byte; *a_byte++ = *b_byte; *b_byte++ = tmp; } }
注意:此方式依赖编译器扩展,可移植性稍差,需根据目标平台评估使用。
内容的提问来源于stack exchange,提问作者Madagascar

