基于Rust的内核触发Invalid Opcode中断问题求助
问题概述
基于Rust开发的内核触发了「Invalid Opcode」无效操作码中断,该内核来自MimosaOS仓库的src/kernel/core/main.rs,Bootloader来自同一仓库的src/bootloader/bootloader.c。
Bootloader执行步骤
- 执行UEFI功能检查
- 通过图形输出协议(Graphics Output Protocol)获取帧缓冲区
- 从存储介质加载内核
- 获取内存映射
- 调用ExitBootServices退出引导服务
- 执行多项检查确保内核合法
- 跳转到
kernelAddress + prog_entry计算的地址,传入帧缓冲区与内存映射信息
内核预期执行步骤
- 清空屏幕
- 绘制从(20,1)到(20,8)的直线
- 无限循环
Bootloader加载内核代码
// ... // Get a volume handle to the volume the bootloader has been loaded from EFI_LOADED_IMAGE *loaded_image = NULL; // Image interface EFI_GUID lipGuid = EFI_LOADED_IMAGE_PROTOCOL_GUID; // Image interface GUID EFI_FILE_IO_INTERFACE *IOVolume; // File system interface EFI_GUID fsGuid = EFI_SIMPLE_FILE_SYSTEM_PROTOCOL_GUID; // File system interface GUID EFI_FILE_HANDLE volumeHandle; // The volume's interface // Get the loaded image protocol interface for the "image" uefi_call_wrapper(BS->HandleProtocol, 3, ImageHandle, &lipGuid, (void**) &loaded_image); // Get the volume handle uefi_call_wrapper(BS->HandleProtocol, 3, loaded_image->DeviceHandle, &fsGuid, (void*)&IOVolume); // VolumeHandle = LibOpenRoot(loaded_image->DeviceHandle); // Not sure if this works, so this is commented out uefi_call_wrapper(IOVolume->OpenVolume, 2, IOVolume, &volumeHandle); EFI_FILE_HANDLE kernelHandle; // Kernel file handle // Get a handle to the kernel file uefi_call_wrapper(volumeHandle->Open, 5, volumeHandle, &kernelHandle, L"kernel.elf", EFI_FILE_MODE_READ, EFI_FILE_READ_ONLY | EFI_FILE_HIDDEN | EFI_FILE_SYSTEM); // Read from the kernel file and load it into memory //initPageDir(); //initPage(); //loadPageDir(pageDirectory); uint64_t kernelSize = FileSize(kernelHandle); #define PAGE_SIZE 4096 #define PRE_ALLOC 100 // Allocate (kernelSize / PAGE_SIZE) + PRE_ALLOC pages for the kernel. UEFI initilizes the VAS as identity mapped, // so I can't use AllocatePool, as the kernel address will be inconsistent otherwise. UINTN numPages = (kernelSize / (PAGE_SIZE)) + PRE_ALLOC; Print(L"Pages to allocate: %d", numPages); void* kernelAddress = (void*)0x10000; Status = uefi_call_wrapper(BS->AllocatePages, 4, AllocateAnyPages, EfiLoaderCode, numPages, (EFI_PHYSICAL_ADDRESS*)&kernelAddress); if (EFI_ERROR(Status) || kernelAddress == NULL) { Print(L"Could not allocate pages for the kernel! Stopping boot!\n"); Print(L"Failure Reason: "); if (Status == EFI_OUT_OF_RESOURCES) { Print(L"EFI_OUT_OF_RESOURCES\n"); } else if (Status == EFI_INVALID_PARAMETER) { Print(L"EFI_INVALID_PARAMETER\n"); } else if (Status == EFI_NOT_FOUND) { Print(L"EFI_NOT_FOUND\n"); } else if (Status != EFI_SUCCESS && kernelAddress == NULL) { Print(L"UNKNOWN_K_PTR_NULL\n"); } Print(L"Please power off your machine. (I am too lazy to implement ACPI drivers)\n"); while (true) { } } uint8_t *kernelBuf = kernelAddress; uefi_call_wrapper(kernelHandle->Read, 3, kernelHandle, &kernelSize, kernelBuf); Print(L"Loaded kernel at address %x\n", kernelBuf); // ... terminal_writestring("Loading kernel...\n"); // Parse the kernel ELF file uint32_t header_magic = 0; header_magic |= (uint32_t)kernelBuf[0] << 24; header_magic |= (uint32_t)kernelBuf[1] << 16; header_magic |= (uint32_t)kernelBuf[2] << 8; header_magic |= (uint32_t)kernelBuf[3]; // If header_magic == .ELF if (header_magic != 0x7F454C46) { terminal_writestring("Fatal Error: Kernel is not an ELF file! Stopping boot!\n"); terminal_writestring("Please power off your machine. (I am too lazy to implement ACPI drivers)\n"); // Stop here, so the user can power off without causing problems. while (true) { } } // 1 = 32 bit, 2 = 64 bit uint8_t header_32_or_64 = kernelBuf[4]; // The kernel must be 64 bit if (header_32_or_64 != 2) { terminal_writestring("Fatal Error: Kernel is not a 64 bit executable! Stopping boot!\n"); terminal_writestring("Please power off your machine. (I am too lazy to implement ACPI drivers)\n"); // Stop here, so the user can power off without causing problems. while (true) { } } // 1 = relocatable, 2 = executable, 3 = shared, 4 = core uint16_t header_elf_type = 0; header_elf_type |= (uint16_t)kernelBuf[16]; header_elf_type |= (uint16_t)kernelBuf[17] << 8; // The kernel must be executable if (header_elf_type != 2) { terminal_writestring("Fatal Error: Kernel is not executable! Stopping boot!\n"); terminal_writestring("Please power off your machine. (I am too lazy to implement ACPI drivers)\n"); // Stop here, so the user can power off without causing problems. while (true) { } } // 0x3E means x86-64 uint16_t header_instruction_set = 0; header_instruction_set |= (uint16_t)kernelBuf[18]; header_instruction_set |= (uint16_t)kernelBuf[19] << 8; // The kernel must be x86-64 if (header_instruction_set != (uint16_t)0x3E) { terminal_writestring("Fatal Error: Kernel is not x86-64! Stopping boot!\n"); terminal_writestring("Please power off your machine. (I am too lazy to implement ACPI drivers)\n"); // Stop here, so the user can power off without causing problems. while (true) { } } // Quick hack so the kernel can actually execute. I don't know what could go wrong from this, // but I know it will mess up something, I just don't know what yet. uint64_t prog_entry = 0; prog_entry |= (uint64_t)kernelBuf[24]; prog_entry |= (uint64_t)kernelBuf[25] << 8; prog_entry |= (uint64_t)kernelBuf[26] << 16; prog_entry |= (uint64_t)kernelBuf[27] << 24; prog_entry |= (uint64_t)kernelBuf[28] << 32; prog_entry |= (uint64_t)kernelBuf[29] << 40; prog_entry |= (uint64_t)kernelBuf[30] << 48; prog_entry |= (uint64_t)kernelBuf[31] << 56; // Initilize a struct with the required framebuffer info to pass to the kernel, so drawing is still possible framebuffer_info_s framebuf; framebuf.base_address = gop->Mode->FrameBufferBase; framebuf.width = gop->Mode->Info->HorizontalResolution; framebuf.height = gop->Mode->Info->VerticalResolution; framebuf.pitch = gop->Mode->Info->PixelsPerScanLine; terminal_writestring("Jumping to kernel address\n"); // Jump to the kernel address typedef int k_main(framebuffer_info_s framebuffer, EFI_MEMORY_DESCRIPTOR* memory_map); k_main* k = (k_main*)kernelAddress + prog_entry; k(framebuf, memoryMap);
Rust内核源码
#![no_std] #![no_main] use core::panic::PanicInfo; #[repr(C)] pub struct FramebufferInfo { base_address: u64, width: u32, height: u32, pitch: u32 } #[repr(C)] pub struct MemoryDescriptor { r#type: u32, // Field size is 32 bits followed by 32 bit pad pad: u32, physical_start: u64, // Field size is 64 bits virtual_start: u64, // Field size is 64 bits number_of_pages: u64, // Field size is 64 bits attribute: u64 // Field size is 64 bits } #[no_mangle] pub extern "C" fn _start(framebuffer_info: FramebufferInfo, memory_map: *const MemoryDescriptor) -> ! { let framebuffer = framebuffer_info.base_address as *mut u32; let pitch = framebuffer_info.pitch; // Clear screen for y in 0..framebuffer_info.height { for x in 0..framebuffer_info.width { unsafe { *framebuffer.offset((pitch * y + x) as isize) = 0x00000000; } } } unsafe { *framebuffer.offset((pitch * 20 + 1) as isize) = 0xFFFFFFFF; *framebuffer.offset((pitch * 20 + 2) as isize) = 0xFFFFFFFF; *framebuffer.offset((pitch * 20 + 3) as isize) = 0xFFFFFFFF; *framebuffer.offset((pitch * 20 + 4) as isize) = 0xFFFFFFFF; *framebuffer.offset((pitch * 20 + 5) as isize) = 0xFFFFFFFF; *framebuffer.offset((pitch * 20 + 6) as isize) = 0xFFFFFFFF; *framebuffer.offset((pitch * 20 + 7) as isize) = 0xFFFFFFFF; *framebuffer.offset((pitch * 20 + 8) as isize) = 0xFFFFFFFF; } loop {} } /// This function is called on panic. #[panic_handler] fn panic(_info: &PanicInfo) -> ! { loop {} }
补充信息
此前C内核也出现类似问题,但仅在调用函数时触发。尝试QEMU+GDB调试但无法命中断点,怀疑是Bootloader地址计算错误,或内核代码异常。
错误截图:
还测试了极简ASM内核,同样触发该错误,RIP地址在文件范围内,可能与汇编方式有关:
ASM内核源码
.global _start .text _start: # Loop indefinitely loop: jmp loop
编译命令
gcc -c src/kernel/core/test.s ld test.o
核心问题排查与解决
1. 致命错误:入口地址计算逻辑错误
Bootloader中跳转地址的计算使用了指针算术,而非直接的地址数值相加:
k_main* k = (k_main*)kernelAddress + prog_entry;
kernelAddress是void*类型,指针加法会以sizeof(k_main)为单位计算偏移,而非字节偏移。正确做法是先转换为数值类型再相加:
k_main* k = (k_main*)((uintptr_t)kernelAddress + prog_entry);
这是导致跳转地址错误、执行无效指令的最可能原因。
2. ELF文件加载不规范
当前代码仅读取整个ELF文件到内存,未按照ELF规范解析程序头(Program Header),将各个段加载到指定的物理地址。必须解析程序头表,根据p_paddr/p_vaddr/p_filesz/p_memsz字段,将段内容复制到对应内存地址,并清零多余内存区域。
3. 汇编内核编译流程问题
用gcc+ld默认编译汇编代码会生成包含冗余初始化代码的ELF文件,需使用链接脚本指定内核加载地址和入口点:
ENTRY(_start) SECTIONS { . = 0x100000; .text : { *(.text) } .data : { *(.data) } .bss : { *(.bss) } }
编译命令改为:
nasm -f elf64 src/kernel/core/test.s -o test.o ld -T kernel.ld test.o -o kernel.elf
4. UEFI环境切换后的CPU状态
调用ExitBootServices后,需确保CPU处于长模式,段寄存器配置正确(如CS指向64位代码段),分页设置符合内核需求(UEFI默认恒等映射,若内核需自定义分页需重新配置)。
内容的提问来源于stack exchange,提问作者Nikki

