ILGPU内核静默编译失败求助:批量清除等效角色位置内核无法编译
ILGPU内核静默编译失败求助:批量清除等效角色位置内核无法编译
各位好,我现在在调试一个基于ILGPU编写的内核,但它一直静默编译失败,完全没有给出错误提示,这可把我愁坏了。我的应用里有两个大型内核,第一个能正常加载并运行,但下面这个负责批量处理角色位置的PurgeEquivalentManPositionsBatchKernel内核就是无法通过编译,代码如下:
/// <summary> /// Unified GPU kernel for processing multiple groups in a single batch call. Each thread /// handles one (group, position) combination using BFS to determine reachability /// </summary> private static void PurgeEquivalentManPositionsBatchKernel( Index1D index, ArrayView1D<ushort, Stride1D.Dense> manPositions, // All man positions across all groups (padded) ArrayView1D<int, Stride1D.Dense> stateIndices, // Corresponding state indices for man positions ArrayView1D<ushort, Stride1D.Dense> diamondPositions, // Diamond positions grouped by group index ArrayView1D<int, Stride1D.Dense> groupSizes, // Number of valid positions per group ArrayView2D<ushort, Stride2D.DenseY> layout, // Map layout (TileContent enum values) int mapHeight, int mapWidth, int currentOrder, // Number of diamonds per group int maxPositionsPerGroup, // Maximum positions in any group (for padding) int totalGroups, // Total number of groups ArrayView3D<byte, Stride3D.DenseXY> bfsWorkspace, // BFS workspace [thread][height][width] ArrayView1D<int, Stride1D.Dense> resultArray) // Output: 0=keep, 1=skip { // BOUNDS CHECK: Ensure we don't exceed the allocated array size if (index >= manPositions.Length) return; // THREAD MAPPING: Convert linear thread index to (group, position) coordinates Each // group has maxPositionsPerGroup slots (some may be padding) int groupIndex = index / maxPositionsPerGroup; // Which group this thread handles int positionIndex = index % maxPositionsPerGroup; // Which position within that group // VALIDATION: Skip if this thread maps to invalid data This handles: // - groupIndex beyond actual number of groups // - positionIndex beyond actual positions in this group // - stateIndices[index] == -1 (padding marker) if (groupIndex >= totalGroups || positionIndex >= groupSizes[groupIndex] || stateIndices[index] < 0) return; // WORKSPACE INITIALIZATION: Set up the BFS grid for this thread Each thread gets its // own slice of the 3D workspace: bfsWorkspace[index, *, *] // Values: 0=free, 1=queued (current wavefront), 2=visited, 4=obstacle(wall/diamond) for (int i = 0; i < mapHeight; i++) { for (int j = 0; j < mapWidth; j++) { // Check for Wall (2) or Outside (32) using TileContent enum values Wall | // Outside = 2 | 32 = 34 (WallOrOutside) if ((layout[i, j] & 34) != 0) // TileContent.WallOrOutside bfsWorkspace[index, i, j] = 4; // Wall obstacle else bfsWorkspace[index, i, j] = 0; // Free space } } // DIAMOND OBSTACLE PLACEMENT: Mark diamonds as impassable Diamond positions for this // group start at (groupIndex * currentOrder) int diamondOffset = groupIndex * currentOrder; for (int d = 0; d < currentOrder; d++) { ushort diamondPos = diamondPositions[diamondOffset + d]; // DECODE POSITION: Extract row/col from packed ushort // Format: high byte = row, low byte = col byte row = (byte)(diamondPos >> 8); byte col = (byte)(diamondPos & 0xFF); // BOUNDS CHECK: Ensure diamond position is valid if (row < mapHeight && col < mapWidth) { bfsWorkspace[index, row, col] = 4; // Diamond obstacle } } // STARTING POSITION: Get the man position for this thread ushort manPos = manPositions[index]; byte manRow = (byte)(manPos >> 8); byte manCol = (byte)(manPos & 0xFF); // BOUNDS CHECK: Ensure man position is valid if (manRow >= mapHeight || manCol >= mapWidth) return; // BOUNDARY TRACKING int minRow = manRow; int maxRow = manRow; int minCol = manCol; int maxCol = manCol; // BFS INITIALIZATION: Mark starting position as current wavefront bfsWorkspace[index, manRow, manCol] = 1; // Mark as queued (current wavefront) // DIRECTION VECTORS: Up, Right, Down, Left int[] dr = [-1, 0, 1, 0]; int[] dc = [0, 1, 0, -1]; // WAVEFRONT BFS: Process wavefront until no more expansions possible bool hasWavefront = true; while (hasWavefront) { hasWavefront = false; // Will be set to true if we find any wavefront cells // Use bounded scan instead of full matrix scan int nextMinRow = mapHeight; int nextMaxRow = -1; int nextMinCol = mapWidth; int nextMaxCol = -1; // BOUNDED MATRIX SCAN: Only scan the active region for (int i = minRow; i <= maxRow; i++) { for (int j = minCol; j <= maxCol; j++) { if (bfsWorkspace[index, i, j] == 1) // Current wavefront cell { // MARK AS VISITED: This cell is now processed bfsWorkspace[index, i, j] = 2; // Mark as visited // EXPAND TO NEIGHBORS: Check all 4 directions for (int dir = 0; dir < 4; dir++) { int nr = i + dr[dir]; int nc = j + dc[dir]; // BOUNDS CHECK: Stay within map boundaries if (nr >= 0 && nr < mapHeight && nc >= 0 && nc < mapWidth) { // QUEUE FREE NEIGHBORS: Add unvisited free cells to wavefront if (bfsWorkspace[index, nr, nc] == 0) // If free and unvisited { bfsWorkspace[index, nr, nc] = 1; // Add to next wavefront // Update boundary tracking nextMinRow = Math.Min(nextMinRow, nr); nextMaxRow = Math.Max(nextMaxRow, nr); nextMinCol = Math.Min(nextMinCol, nc); nextMaxCol = Math.Max(nextMaxCol, nc); hasWavefront = true; } } } } } } // Update boundaries for next iteration minRow = nextMinRow; maxRow = nextMaxRow; minCol = nextMinCol; maxCol = nextMaxCol; } // TODO: 后续逻辑(原代码此处截断) }
我已经检查了数组边界、视图的 stride 配置,还有线程映射的逻辑,但完全找不到编译失败的原因。有没有大佬遇到过类似的ILGPU内核静默编译失败的情况?或者能帮我看看代码里有没有明显的ILGPU不兼容的写法?
内容来源于stack exchange
相关产品推荐
相关产品推荐

