You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在十万步以上模拟中高效估算终止游戏的平均得分?

如何在大规模GPU模拟中高效估算终止游戏的平均得分?

问题背景

我有一个存储整数类型得分的device array,以及一个存储int8_t类型标记的数组。第i个得分对应向量化模拟器中的第i个游戏,若第i个游戏需重启,则第i个标记非零。数组会在模拟的每一步更新,终止的游戏会被重启。

核心需求:在包含10万+步骤的模拟过程中,估算终止游戏的平均得分,且运行时性能优先级高于统计正确性。

当前仅实现得分重置逻辑的内核代码如下:

__global__ void reset_score(const int game_count, const int8_t *terminated_flags, int *scores)
{
    int idx = blockIdx.x * blockDim.x + threadIdx.x;
    if (idx >= game_count)
        return;

    if (terminated_flags[idx] == 0)
        return;

    scores[idx] = 0;
    return;
}

在之前的主机端实现中,我通过单个工作线程使用指数移动平均值(EMA)来更新终止游戏的得分。


高效解决方案(性能优先)

1. 核内嵌入统计逻辑+块级聚合原子操作

直接在重置得分的内核中嵌入统计逻辑,避免额外启动单独统计内核。通过块级共享内存聚合减少全局原子操作的次数,降低性能开销:

__global__ void reset_score_with_stats(const int game_count, const int8_t *terminated_flags, 
                                      int *scores, unsigned long long *global_total_score, 
                                      unsigned int *global_terminated_count)
{
    __shared__ unsigned long long block_total;
    __shared__ unsigned int block_count;

    // 初始化块共享内存
    if (threadIdx.x == 0) {
        block_total = 0;
        block_count = 0;
    }
    __syncthreads();

    int idx = blockIdx.x * blockDim.x + threadIdx.x;
    if (idx < game_count && terminated_flags[idx] != 0) {
        // 记录终止得分并重置
        int final_score = scores[idx];
        scores[idx] = 0;

        // 先累加至块级共享内存
        atomicAdd(&block_total, (unsigned long long)final_score);
        atomicAdd(&block_count, 1u);
    }
    __syncthreads();

    // 块内单个线程统一更新全局统计
    if (threadIdx.x == 0) {
        if (block_count > 0) {
            atomicAdd(global_total_score, block_total);
            atomicAdd(global_terminated_count, block_count);
        }
    }
}

2. 延迟更新EMA(减少主机-设备交互)

由于性能优先,无需每步计算EMA,可每隔固定步数(如1000步)批量更新,大幅降低主机与设备的数据交互频率:

// 主机端示例逻辑
float ema_avg = 0.0f;
const float alpha = 0.05f; // 平滑系数,值越小平滑效果越强
unsigned long long host_total = 0;
unsigned int host_count = 0;
const unsigned long long zero_total = 0;
const unsigned int zero_count = 0;

// 预分配设备端统计变量
unsigned long long *d_total_score;
unsigned int *d_terminated_count;
cudaMalloc(&d_total_score, sizeof(unsigned long long));
cudaMalloc(&d_terminated_count, sizeof(unsigned int));
cudaMemcpy(d_total_score, &zero_total, sizeof(unsigned long long), cudaMemcpyHostToDevice);
cudaMemcpy(d_terminated_count, &zero_count, sizeof(unsigned int), cudaMemcpyHostToDevice);

for (int step = 0; step < 100000; step++) {
    // 运行模拟步骤...
    dim3 grid((game_count + 255) / 256);
    dim3 block(256);
    reset_score_with_stats<<<grid, block>>>(game_count, d_terminated_flags, d_scores, d_total_score, d_terminated_count);

    // 每隔1000步更新一次EMA
    if (step % 1000 == 0) {
        // 异步拷贝统计数据到主机
        cudaMemcpyAsync(&host_total, d_total_score, sizeof(unsigned long long), cudaMemcpyDeviceToHost);
        cudaMemcpyAsync(&host_count, d_terminated_count, sizeof(unsigned int), cudaMemcpyDeviceToHost);
        cudaDeviceSynchronize();

        if (host_count > 0) {
            float current_avg = (float)host_total / host_count;
            ema_avg = alpha * current_avg + (1.0f - alpha) * ema_avg;

            // 重置设备端统计变量,避免数值溢出
            cudaMemcpyAsync(d_total_score, &zero_total, sizeof(unsigned long long), cudaMemcpyHostToDevice);
            cudaMemcpyAsync(d_terminated_count, &zero_count, sizeof(unsigned int), cudaMemcpyHostToDevice);
        }
    }
}

// 释放资源
cudaFree(d_total_score);
cudaFree(d_terminated_count);

3. 额外性能优化点

  • 跳过小额统计:若某一步终止游戏数量极少(如<5),可跳过该步的块级统计,进一步降低原子操作开销
  • 流并行隐藏延迟:将模拟内核与统计拷贝操作放在不同CUDA流中,隐藏数据拷贝的等待时间
  • 动态调整平滑系数:根据模拟总步数调整alpha值,步数越多可适当调小alpha,提升平滑效果

内容的提问来源于stack exchange,提问作者Dudly01

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.05 16:12:32