CUDA粒子仿真遇cudaErrorIllegalAddress(700)错误求助
解决CUDA粒子仿真中的cudaErrorIllegalAddress(700)错误
问题描述
我正在完成大学作业,用CUDA/C++实现基础粒子仿真(此前已用Rust线程完成同系统)。但当前GPU版本持续出现cudaErrorIllegalAddress(700)错误,怀疑源于update_particles_GPU函数实现。
最初使用设备托管内存,在60fps循环、多喷雾罐、数千粒子场景下触发该错误,简化为单喷雾罐、少量粒子、手动内存拷贝后问题仍未解决。尝试在核函数中添加打印排查线程无效,且不同线程配置表现异常:1块3333线程单次更新正常,循环场景中会触发无效内存访问;1块2000线程表现类似;2块1000线程立即报错;使用num_blocks+block_size配置也立即报错。
代码执行流程:
- 创建
AerosolCan实例red_can - 初始化空指针
dev_red_particles - 调用喷雾方法生成粒子并添加到
red_can的粒子数组(疑似错误点) - 在GPU上分配内存
- 将主机端
red_can.particles数组拷贝到设备端 - 调用核函数
update_particles_GPU(疑似错误点) - 执行
cudaDeviceSynchronize() - 将设备端更新后的数组拷贝回主机端
- 打印粒子更新后的位置
错误原因分析
- 核函数传入主机内存指针:调用
update_particles_GPU时,直接传递了主机端的&time_step和&red_can.particles_created指针。GPU核函数运行在设备上,无法直接访问主机内存空间,这会直接触发非法内存访问错误。 - 边界判断逻辑错误:核函数中
if (i > *particles_created)的判断会允许线程处理索引等于particles_created的元素,而实际有效的粒子索引范围是0到particles_created-1,越界访问会导致非法内存访问。 - 无效的空指针检查:
&particles[i] == NULL的判断完全无效,因为particles是通过cudaMalloc分配的有效设备数组指针,只要i在合法范围内,&particles[i]不可能为空,这段代码属于冗余且错误的判断。
修复方案
核心修复点
- 为
time_step和particles_created分配设备内存,将主机端值拷贝到设备后再传递给核函数 - 修正核函数中的边界判断条件为
i >= *particles_created - 移除核函数中无效的空指针检查代码
修复后完整代码
#define _USE_MATH_DEFINES #include <iostream> #include <math.h> #include "cuda_runtime.h" #include "device_launch_parameters.h" #include <chrono> #include <thread> #include <stdio.h> #include <vector> #include <windows.h> const float DRAG = 0.05; const float GRAVITY = 9.8; const float BLEND = 0.1; struct Colour { float red = 1; float green = 1; float blue = 1; }; struct Particle { Colour colour = Colour(); float x = 0; float y = 0; float z = 0; float velocity_x = 0; float velocity_y = 0; float velocity_z = 0; bool collided = false; bool landed_on_paper = false; }; const int max_particles_per_can = 3333; __global__ void update_particles_GPU(Particle* particles, const float* time_step, const uint32_t* particles_created) { int i = blockIdx.x * blockDim.x + threadIdx.x; if (i >= *particles_created) { return; } Particle* particle = &particles[i]; if (!particle->collided) { float GRAVITY = 9.8; float DRAG = 0.05; float acceleration_x = -DRAG * (particle->velocity_x * particle->velocity_x); float distance_x = particle->velocity_x * *time_step + 0.5 * acceleration_x * (*time_step * *time_step); float acceleration_y = GRAVITY - DRAG * (particle->velocity_y * particle->velocity_y); float distance_y = particle->velocity_y * *time_step + 0.5 * acceleration_y * (*time_step * *time_step); float acceleration_z = -DRAG * (particle->velocity_z * particle->velocity_z); float distance_z = particle->velocity_z * *time_step + 0.5 * acceleration_z * (*time_step * *time_step); particle->x += distance_x; particle->y += distance_y; particle->z += distance_z; if (particle->y < 0) { particle->y = 0; } if (particle->velocity_x < 0) { particle->velocity_x += -acceleration_x * *time_step; } else { particle->velocity_x += acceleration_x * *time_step; } particle->velocity_y += -acceleration_y * *time_step; if (particle->velocity_z < 0) { particle->velocity_z += -acceleration_z * *time_step; } else { particle->velocity_z += acceleration_z * *time_step; } // Collision if (particle->y == 0) { particle->collided = true; } } } struct AerosolCan { Particle* particles = new Particle[max_particles_per_can]; float x = 0; float y = 0; float z = 0; float base_velocity_x = 0; float base_velocity_y = 0; float base_velocity_z = 0; Colour colour = Colour(); float spray_radius = 0; uint32_t particles_created = 0; void print_particles() { for (int i = 0; i < particles_created; i++) { Particle* particle = &particles[i]; std::cout << "Particle " << i + 1 << " | X: " << particle->x << " | Y: " << particle->y << " | Z: " << particle->z << " | Hit = " << particle->landed_on_paper << std::endl; } std::cout << "" << std::endl; } void spray(Particle* particles, uint32_t number_of_particles) { float radius = spray_radius; while (radius > 0.0 && number_of_particles > 0) { // Create new particles for (int i = 0; i < number_of_particles; i++) { float horizontal_angle = i / number_of_particles * 2.0 * M_PI; float vertical_angle = i / number_of_particles * M_PI; float new_x = x + radius * cos(horizontal_angle) * sin(vertical_angle); float new_y = y + radius * sin(horizontal_angle) * sin(vertical_angle); float new_z = z + radius * cos(vertical_angle); Particle new_particle = Particle(); new_particle.colour = colour; new_particle.x = new_x; new_particle.y = new_y; new_particle.z = new_z; new_particle.velocity_x = base_velocity_x; new_particle.velocity_y = base_velocity_y; new_particle.velocity_z = base_velocity_z; particles[particles_created] = new_particle; particles_created++; } radius = radius / 2.0; number_of_particles = number_of_particles / 2.0; } } }; const int block_size = 256; int main() { int num_blocks = (max_particles_per_can + block_size - 1) / block_size; std::cout << "Blocks: " << num_blocks << " | Block size: " << block_size << std::endl; Colour red_colour; red_colour.red = 1; red_colour.green = 0; red_colour.blue = 0; AerosolCan red_can = AerosolCan(); red_can.colour = red_colour; red_can.x = -25; red_can.y = 30; red_can.z = 60; red_can.base_velocity_x = 125; red_can.base_velocity_y = 10; red_can.base_velocity_z = 0; red_can.spray_radius = 15; Particle* dev_red_particles = nullptr; float* dev_time_step = nullptr; uint32_t* dev_particles_created = nullptr; // Spray particles red_can.spray(red_can.particles, 5); // Copy data to GPU cudaError_t cuda_status; cuda_status = cudaMalloc((void**)&dev_red_particles, max_particles_per_can * sizeof(Particle)); if (cuda_status != cudaSuccess) { fprintf(stderr, "cudaMalloc failed"); goto Error; } cuda_status = cudaMemcpy(dev_red_particles, red_can.particles, max_particles_per_can * sizeof(Particle), cudaMemcpyHostToDevice); if (cuda_status != cudaSuccess) { fprintf(stderr, "cudaMemcpy failed"); goto Error; } float time_step = 0.001; cuda_status = cudaMalloc((void**)&dev_time_step, sizeof(float)); if (cuda_status != cudaSuccess) { fprintf(stderr, "cudaMalloc for time_step failed"); goto Error; } cuda_status = cudaMemcpy(dev_time_step, &time_step, sizeof(float), cudaMemcpyHostToDevice); if (cuda_status != cudaSuccess) { fprintf(stderr, "cudaMemcpy for time_step failed"); goto Error; } cuda_status = cudaMalloc((void**)&dev_particles_created, sizeof(uint32_t)); if (cuda_status != cudaSuccess) { fprintf(stderr, "cudaMalloc for particles_created failed"); goto Error; } cuda_status = cudaMemcpy(dev_particles_created, &red_can.particles_created, sizeof(uint32_t), cudaMemcpyHostToDevice); if (cuda_status != cudaSuccess) { fprintf(stderr, "cudaMemcpy for particles_created failed"); goto Error; } // Update particles update_particles_GPU <<<num_blocks, block_size>>> (dev_red_particles, dev_time_step, dev_particles_created); cuda_status = cudaDeviceSynchronize(); if (cuda_status != cudaSuccess) { fprintf(stderr, "cudaSync failed"); goto Error; } // Copy data back to host cuda_status = cudaMemcpy(red_can.particles, dev_red_particles, max_particles_per_can * sizeof(Particle), cudaMemcpyDeviceToHost); if (cuda_status != cudaSuccess) { fprintf(stderr, "cudaMemcpy failed"); goto Error; } // Print particles red_can.print_particles(); Error: cudaFree(dev_red_particles); cudaFree(dev_time_step); cudaFree(dev_particles_created); return 0; }
内容的提问来源于stack exchange,提问作者user18365833
相关产品推荐
相关产品推荐

