You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于Rust的内核触发Invalid Opcode中断问题求助

Rust内核触发「Invalid Opcode」中断问题排查

问题概述

基于Rust开发的内核触发了「Invalid Opcode」无效操作码中断,该内核来自MimosaOS仓库的src/kernel/core/main.rs,Bootloader来自同一仓库的src/bootloader/bootloader.c。

Bootloader执行步骤

  • 执行UEFI功能检查
  • 通过图形输出协议(Graphics Output Protocol)获取帧缓冲区
  • 从存储介质加载内核
  • 获取内存映射
  • 调用ExitBootServices退出引导服务
  • 执行多项检查确保内核合法
  • 跳转到kernelAddress + prog_entry计算的地址,传入帧缓冲区与内存映射信息

内核预期执行步骤

  • 清空屏幕
  • 绘制从(20,1)到(20,8)的直线
  • 无限循环

Bootloader加载内核代码

// ...

// Get a volume handle to the volume the bootloader has been loaded from

EFI_LOADED_IMAGE *loaded_image = NULL;                  // Image interface
EFI_GUID lipGuid = EFI_LOADED_IMAGE_PROTOCOL_GUID;      // Image interface GUID
EFI_FILE_IO_INTERFACE *IOVolume;                        // File system interface
EFI_GUID fsGuid = EFI_SIMPLE_FILE_SYSTEM_PROTOCOL_GUID; // File system interface GUID
EFI_FILE_HANDLE volumeHandle;                           // The volume's interface

// Get the loaded image protocol interface for the "image"
uefi_call_wrapper(BS->HandleProtocol, 3, ImageHandle, &lipGuid, (void**) &loaded_image);
// Get the volume handle
uefi_call_wrapper(BS->HandleProtocol, 3, loaded_image->DeviceHandle, &fsGuid, (void*)&IOVolume);
// VolumeHandle = LibOpenRoot(loaded_image->DeviceHandle); // Not sure if this works, so this is commented out
uefi_call_wrapper(IOVolume->OpenVolume, 2, IOVolume, &volumeHandle);


EFI_FILE_HANDLE kernelHandle; // Kernel file handle

// Get a handle to the kernel file
uefi_call_wrapper(volumeHandle->Open, 5, volumeHandle, &kernelHandle, L"kernel.elf", EFI_FILE_MODE_READ, EFI_FILE_READ_ONLY | EFI_FILE_HIDDEN | EFI_FILE_SYSTEM);

// Read from the kernel file and load it into memory

//initPageDir();
//initPage();
//loadPageDir(pageDirectory);

uint64_t kernelSize = FileSize(kernelHandle);

#define PAGE_SIZE 4096
#define PRE_ALLOC 100

// Allocate (kernelSize / PAGE_SIZE) + PRE_ALLOC pages for the kernel. UEFI initilizes the VAS as identity mapped,
// so I can't use AllocatePool, as the kernel address will be inconsistent otherwise.

UINTN numPages = (kernelSize / (PAGE_SIZE)) + PRE_ALLOC;

Print(L"Pages to allocate: %d", numPages);

void* kernelAddress = (void*)0x10000;

Status = uefi_call_wrapper(BS->AllocatePages, 4, AllocateAnyPages, EfiLoaderCode, numPages, (EFI_PHYSICAL_ADDRESS*)&kernelAddress);
if (EFI_ERROR(Status) || kernelAddress == NULL)
{
    Print(L"Could not allocate pages for the kernel! Stopping boot!\n");
    Print(L"Failure Reason: ");

    if (Status == EFI_OUT_OF_RESOURCES)
    {
        Print(L"EFI_OUT_OF_RESOURCES\n");
    }
    else if (Status == EFI_INVALID_PARAMETER)
    {
        Print(L"EFI_INVALID_PARAMETER\n");
    }
    else if (Status == EFI_NOT_FOUND)
    {
        Print(L"EFI_NOT_FOUND\n");
    }
    else if (Status != EFI_SUCCESS && kernelAddress == NULL)
    {
        Print(L"UNKNOWN_K_PTR_NULL\n");
    }

    Print(L"Please power off your machine. (I am too lazy to implement ACPI drivers)\n");

    while (true) { }
}

uint8_t *kernelBuf = kernelAddress;

uefi_call_wrapper(kernelHandle->Read, 3, kernelHandle, &kernelSize, kernelBuf);


Print(L"Loaded kernel at address %x\n", kernelBuf);



// ...



terminal_writestring("Loading kernel...\n");

// Parse the kernel ELF file

uint32_t header_magic = 0;

header_magic |= (uint32_t)kernelBuf[0] << 24;
header_magic |= (uint32_t)kernelBuf[1] << 16;
header_magic |= (uint32_t)kernelBuf[2] << 8;
header_magic |= (uint32_t)kernelBuf[3];

// If header_magic == .ELF
if (header_magic != 0x7F454C46) 
{
    terminal_writestring("Fatal Error: Kernel is not an ELF file! Stopping boot!\n");
    terminal_writestring("Please power off your machine. (I am too lazy to implement ACPI drivers)\n");

    // Stop here, so the user can power off without causing problems.
    while (true) { }
}

// 1 = 32 bit, 2 = 64 bit
uint8_t header_32_or_64 = kernelBuf[4];

// The kernel must be 64 bit
if (header_32_or_64 != 2)
{
    terminal_writestring("Fatal Error: Kernel is not a 64 bit executable! Stopping boot!\n");
    terminal_writestring("Please power off your machine. (I am too lazy to implement ACPI drivers)\n");

    // Stop here, so the user can power off without causing problems.
    while (true) { }
}

// 1 = relocatable, 2 = executable, 3 = shared, 4 = core
uint16_t header_elf_type = 0;

header_elf_type |= (uint16_t)kernelBuf[16];
header_elf_type |= (uint16_t)kernelBuf[17] << 8;

// The kernel must be executable
if (header_elf_type != 2)
{
    terminal_writestring("Fatal Error: Kernel is not executable! Stopping boot!\n");
    terminal_writestring("Please power off your machine. (I am too lazy to implement ACPI drivers)\n");

    // Stop here, so the user can power off without causing problems.
    while (true) { }
}

// 0x3E means x86-64
uint16_t header_instruction_set = 0;

header_instruction_set |= (uint16_t)kernelBuf[18];
header_instruction_set |= (uint16_t)kernelBuf[19] << 8;

// The kernel must be x86-64
if (header_instruction_set != (uint16_t)0x3E)
{
    terminal_writestring("Fatal Error: Kernel is not x86-64! Stopping boot!\n");
    terminal_writestring("Please power off your machine. (I am too lazy to implement ACPI drivers)\n");

    // Stop here, so the user can power off without causing problems.
    while (true) { }
}

// Quick hack so the kernel can actually execute. I don't know what could go wrong from this,
// but I know it will mess up something, I just don't know what yet.

uint64_t prog_entry = 0;

prog_entry |= (uint64_t)kernelBuf[24];
prog_entry |= (uint64_t)kernelBuf[25] << 8;
prog_entry |= (uint64_t)kernelBuf[26] << 16;
prog_entry |= (uint64_t)kernelBuf[27] << 24;
prog_entry |= (uint64_t)kernelBuf[28] << 32;
prog_entry |= (uint64_t)kernelBuf[29] << 40;
prog_entry |= (uint64_t)kernelBuf[30] << 48;
prog_entry |= (uint64_t)kernelBuf[31] << 56;

// Initilize a struct with the required framebuffer info to pass to the kernel, so drawing is still possible

framebuffer_info_s framebuf;
framebuf.base_address = gop->Mode->FrameBufferBase;
framebuf.width = gop->Mode->Info->HorizontalResolution;
framebuf.height = gop->Mode->Info->VerticalResolution;
framebuf.pitch = gop->Mode->Info->PixelsPerScanLine;

terminal_writestring("Jumping to kernel address\n");

// Jump to the kernel address
typedef int k_main(framebuffer_info_s framebuffer, EFI_MEMORY_DESCRIPTOR* memory_map);
k_main* k = (k_main*)kernelAddress + prog_entry;
k(framebuf, memoryMap);

Rust内核源码

#![no_std]
#![no_main]

use core::panic::PanicInfo;

#[repr(C)]
pub struct FramebufferInfo {
    base_address: u64,
    width: u32,
    height: u32,
    pitch: u32
}

#[repr(C)]
pub struct MemoryDescriptor {
    r#type: u32,           // Field size is 32 bits followed by 32 bit pad
    pad: u32,
    physical_start: u64,  // Field size is 64 bits
    virtual_start: u64,   // Field size is 64 bits
    number_of_pages: u64,  // Field size is 64 bits
    attribute: u64       // Field size is 64 bits
}

#[no_mangle]
pub extern "C" fn _start(framebuffer_info: FramebufferInfo, memory_map: *const MemoryDescriptor) -> ! {
    let framebuffer = framebuffer_info.base_address as *mut u32;
    let pitch = framebuffer_info.pitch;

    // Clear screen
    for y in 0..framebuffer_info.height {
        for x in 0..framebuffer_info.width {
            unsafe {
                *framebuffer.offset((pitch * y + x) as isize) = 0x00000000;
            }
        }
    }

    unsafe {
        *framebuffer.offset((pitch * 20 + 1) as isize) = 0xFFFFFFFF;
        *framebuffer.offset((pitch * 20 + 2) as isize) = 0xFFFFFFFF;
        *framebuffer.offset((pitch * 20 + 3) as isize) = 0xFFFFFFFF;
        *framebuffer.offset((pitch * 20 + 4) as isize) = 0xFFFFFFFF;
        *framebuffer.offset((pitch * 20 + 5) as isize) = 0xFFFFFFFF;
        *framebuffer.offset((pitch * 20 + 6) as isize) = 0xFFFFFFFF;
        *framebuffer.offset((pitch * 20 + 7) as isize) = 0xFFFFFFFF;
        *framebuffer.offset((pitch * 20 + 8) as isize) = 0xFFFFFFFF;
    }

    loop {}
}

/// This function is called on panic.
#[panic_handler]
fn panic(_info: &PanicInfo) -> ! {
    loop {}
}

补充信息

此前C内核也出现类似问题,但仅在调用函数时触发。尝试QEMU+GDB调试但无法命中断点,怀疑是Bootloader地址计算错误,或内核代码异常。

错误截图:Invalid Opcode错误截图

还测试了极简ASM内核,同样触发该错误,RIP地址在文件范围内,可能与汇编方式有关:

ASM内核源码

.global _start

.text
_start:
    # Loop indefinitely
    loop:
    jmp loop

编译命令

gcc -c src/kernel/core/test.s
ld test.o

核心问题排查与解决

1. 致命错误:入口地址计算逻辑错误

Bootloader中跳转地址的计算使用了指针算术,而非直接的地址数值相加:

k_main* k = (k_main*)kernelAddress + prog_entry;

kernelAddress是void*类型,指针加法会以sizeof(k_main)为单位计算偏移,而非字节偏移。正确做法是先转换为数值类型再相加:

k_main* k = (k_main*)((uintptr_t)kernelAddress + prog_entry);

这是导致跳转地址错误、执行无效指令的最可能原因。

2. ELF文件加载不规范

当前代码仅读取整个ELF文件到内存,未按照ELF规范解析程序头(Program Header),将各个段加载到指定的物理地址。必须解析程序头表,根据p_paddr/p_vaddr/p_filesz/p_memsz字段,将段内容复制到对应内存地址,并清零多余内存区域。

3. 汇编内核编译流程问题

用gcc+ld默认编译汇编代码会生成包含冗余初始化代码的ELF文件,需使用链接脚本指定内核加载地址和入口点:

ENTRY(_start)
SECTIONS {
    . = 0x100000;
    .text : { *(.text) }
    .data : { *(.data) }
    .bss : { *(.bss) }
}

编译命令改为:

nasm -f elf64 src/kernel/core/test.s -o test.o
ld -T kernel.ld test.o -o kernel.elf

4. UEFI环境切换后的CPU状态

调用ExitBootServices后,需确保CPU处于长模式,段寄存器配置正确(如CS指向64位代码段),分页设置符合内核需求(UEFI默认恒等映射,若内核需自定义分页需重新配置)。

内容的提问来源于stack exchange,提问作者Nikki

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.10 21:50:22