diff --git a/Cargo.lock b/Cargo.lock index 1c6b07c793b..4263f62325b 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1322,6 +1322,10 @@ dependencies = [ "zerocopy", ] +[[package]] +name = "toyos-libc-copies" +version = "0.1.0" + [[package]] name = "toyos-logstream" version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index 16388297212..153e4cb3453 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -37,6 +37,7 @@ members = [ "toyos-inspect", "toyos-keymap", "toyos-ld", + "toyos-libc-copies", "toyos-logstream", "toyos-manifest", "toyos-mdns", diff --git a/NOTICE b/NOTICE index 4204a285fd6..cec1ae65e4c 100644 --- a/NOTICE +++ b/NOTICE @@ -267,6 +267,32 @@ instead — which would put it inside the "comes with QEMU" allowance and delete 6,291,456 bytes from the repository. + +aavmf/*.fd — EDK II firmware for QEMU `virt`, BSD-2-Clause-Patent AND Apache-2.0 +------------------------------------------------------------ + + AAVMF_CODE.fd 67,108,864 bytes + sha256 47765fe344818cbc464b1c14ae658fb4b854f5c2ceffa982411731eb4865594d + AAVMF_VARS.fd 67,108,864 bytes + sha256 b3b855c5a80310168051164986855692d1bdb06e67619856177965cd87c6774f + + Copyright (c) 2019, TianoCore and contributors + Licence text: licenses/BSD-2-Clause-Patent-EDK2.txt + Copyright 1995-2023 The OpenSSL Project Authors + Licence text: licenses/Apache-2.0-OpenSSL.txt + SPDX-License-Identifier: BSD-2-Clause-Patent AND Apache-2.0 + +QEMU's own prebuilt ArmVirtQemu firmware, unmodified: `edk2-aarch64-code.fd` +and `edk2-arm-vars.fd` as QEMU 11.1.0 installs them under `share/qemu/`, +whose version string reads `edk2-stable202408-prebuilt.qemu.org` (a +`DEBUG_GCC5` build of 2024-09-12). The variable store is the template: every +boot gives it a writable snapshot QEMU discards, because this `DEBUG` build +asserts on a read-only store. QEMU builds it with `NETWORK_TLS_ENABLE` +(`roms/edk2-build.config`, `[opts.common]`), and edk2-stable202408's +ArmVirtQemu links edk2's bundled OpenSSL either way (`OpensslLib` with TLS, +`OpensslLibCrypto` without): OpenSSL 3.0.9, whose `LICENSE.txt` is the +Apache-2.0 text above and which carries no `NOTICE`. + tests/fixtures/gbae-v0.2.0-* — gbae, MIT ----------------------------------------- diff --git a/aavmf/AAVMF_CODE.fd b/aavmf/AAVMF_CODE.fd new file mode 100644 index 00000000000..89924f4b509 Binary files /dev/null and b/aavmf/AAVMF_CODE.fd differ diff --git a/aavmf/AAVMF_VARS.fd b/aavmf/AAVMF_VARS.fd new file mode 100644 index 00000000000..a71658f9882 Binary files /dev/null and b/aavmf/AAVMF_VARS.fd differ diff --git a/bootloader/.cargo/config.toml b/bootloader/.cargo/config.toml index 6abe5e339da..e00810caadd 100644 --- a/bootloader/.cargo/config.toml +++ b/bootloader/.cargo/config.toml @@ -10,4 +10,4 @@ rustflags = [ ] [target.x86_64-unknown-uefi] -linker = "toyos-ld" \ No newline at end of file +linker = "toyos-ld" diff --git a/bootloader/rust-toolchain.toml b/bootloader/rust-toolchain.toml index 2549413f0dd..b58d6aa2bd1 100644 --- a/bootloader/rust-toolchain.toml +++ b/bootloader/rust-toolchain.toml @@ -1,2 +1,2 @@ [toolchain] -targets = ["x86_64-unknown-uefi"] \ No newline at end of file +targets = ["x86_64-unknown-uefi", "aarch64-unknown-uefi"] diff --git a/bootloader/src/arch/aarch64.rs b/bootloader/src/arch/aarch64.rs new file mode 100644 index 00000000000..dbb0b327022 --- /dev/null +++ b/bootloader/src/arch/aarch64.rs @@ -0,0 +1,138 @@ +//! AArch64: the loader's every instruction Rust has no portable spelling for. + +use toyos_abi::boot::KernelArgs; +use toyos_bootmap::Typing; + +/// The machine the kernel image must be built for: the loader's own. +pub const ELF_MACHINE: toyos_elf::Machine = toyos_elf::Machine::Aarch64; + +/// How the boot map's descriptors are encoded. +pub use toyos_bootmap::aarch64 as encoding; + +/// How the boot map types memory: by firmware's map, since AArch64 has no +/// range registers to do it and every descriptor names its own type. +pub fn typing(write_back: &[(u64, u64)]) -> Typing<'_> { + Typing::ByMap(write_back) +} + +/// The generic timer's virtual count, `CNTVCT_EL0`. +pub fn counter() -> u64 { + let count: u64; + // SAFETY: reads a counter EL1 and EL2 may always read; the `ISB` keeps the + // read in program order. + unsafe { core::arch::asm!("isb", "mrs {}, cntvct_el0", out(reg) count, options(nomem, nostack, preserves_flags)) }; + count +} + +/// What the loader's report says beside the counter: its rate. Where it counts +/// from is firmware's to say and no register here does. +pub fn counter_origin() -> alloc::string::String { + let hz: u64; + // SAFETY: reads a register EL1 and EL2 may always read. + unsafe { core::arch::asm!("mrs {}, cntfrq_el0", out(reg) hz, options(nomem, nostack, preserves_flags)) }; + alloc::format!("CNTFRQ_EL0 {hz} Hz; the counter's origin is firmware's") +} + +/// What the loader says about the CPU as firmware handed it over, or why the +/// kernel cannot run on it: entered at EL2 on a CPU without FEAT_E2H0, +/// `HCR_EL2.E2H` is RES1, so the kernel's entry cannot clear it and every +/// `_el1` register its drop programs would be EL2's own. Refused here, where +/// the console still prints; the entry's read-back of `HCR_EL2` stays as the +/// last line of defence. +pub fn cpu_as_entered() -> Result, alloc::string::String> { + let current: u64; + // SAFETY: reads `CurrentEL`, which EL1 and above may read. + unsafe { core::arch::asm!("mrs {}, currentel", out(reg) current, options(nomem, nostack, preserves_flags)) }; + let el = (current >> 2) & 0b11; + if el != 2 { + return Ok(Some(alloc::format!("CPU: entered at EL{el}"))); + } + let (hcr, mmfr4): (u64, u64); + // SAFETY: at EL2 both are readable. `ID_AA64MMFR4_EL1` by its encoding, + // `S3_0_C0_C7_4`, which sits in the ID space an older CPU reads as zero. + unsafe { + core::arch::asm!("mrs {}, hcr_el2", out(reg) hcr, options(nomem, nostack, preserves_flags)); + core::arch::asm!("mrs {}, S3_0_C0_C7_4", out(reg) mmfr4, options(nomem, nostack, preserves_flags)); + } + let e2h = (hcr >> 34) & 1; + let e2h0 = (mmfr4 >> 24) & 0xF; + let state = alloc::format!("CPU: entered at EL2, HCR_EL2.E2H {e2h}, ID_AA64MMFR4_EL1.E2H0 {e2h0:#x}"); + if e2h0 != 0 { + return Err(alloc::format!( + "{state}: REFUSED, HCR_EL2.E2H is RES1 on a CPU without FEAT_E2H0, and the kernel's \ + drop to EL1 needs it clear" + )); + } + Ok(Some(alloc::format!("{state}: the kernel's entry writes E2H clear"))) +} + +/// The smallest data cache line on this machine, from `CTR_EL0.DminLine`. +fn line() -> u64 { + let ctr: u64; + // SAFETY: reads `CTR_EL0`, which every level may read. + unsafe { core::arch::asm!("mrs {}, ctr_el0", out(reg) ctr, options(nomem, nostack, preserves_flags)) }; + 4 << ((ctr >> 16) & 0xF) +} + +/// Every line of `[at, at + len)` cleaned to the point of coherency and +/// invalidated, then `DSB SY` for their completion. +pub fn write_back(at: u64, len: usize) { + let step = line(); + let mut addr = at & !(step - 1); + while addr < at + len as u64 { + // SAFETY: `DC CIVAC` cleans and invalidates the line holding an + // address the caller allocated; it changes no memory's contents. + unsafe { core::arch::asm!("dc civac, {}", in(reg) addr, options(nostack, preserves_flags)) }; + addr += step; + } + // SAFETY: a barrier; waits for the maintenance above to complete. + unsafe { core::arch::asm!("dsb sy", options(nostack, preserves_flags)) }; +} + +/// AArch64 has no I/O port space: every caller checks [`pio::EXISTS`] first. +pub mod pio { + pub const EXISTS: bool = false; + + /// # Safety + /// Never called: [`EXISTS`] is false. + pub unsafe fn outw(_port: u16, _value: u16) { + unreachable!("AArch64 has no I/O port space") + } + + pub fn inw(_port: u16) -> u16 { + unreachable!("AArch64 has no I/O port space") + } +} + +/// Hand the CPU to the kernel as firmware left it — its exception level, its +/// identity tables — at the image's physical entry, with `x0 = args`. The +/// kernel's entry switches to the boot map (`args.boot_pml4_addr`) itself (`kernel/src/arch/aarch64/boot.rs`), +/// because at EL2 only the kernel's own drop to EL1 can install it. +/// +/// The image is cleaned to the point of coherency first: that entry fetches +/// instructions with the MMU off for a few of them, straight from memory. +/// +/// # Safety +/// `args.boot_pml4_addr` is the boot map, `image` is the relocated kernel image, `entry_offset` +/// is its entry point's offset in it, and `args` stays where it is until the +/// kernel copies it. +pub unsafe fn enter_kernel(image: (u64, u64), entry_offset: u64, args: &KernelArgs) -> ! { + write_back(image.0, image.1 as usize); + let entry = image.0 + entry_offset; + // SAFETY: interrupts masked for good — the kernel's vectors are not + // installed yet — then every instruction cache line invalidated against + // the image just cleaned, and a branch to its entry with `x0 = args`, the + // boot protocol `toyos-abi::boot` and the kernel's `_start` define. + unsafe { + core::arch::asm!( + "msr daifset, #0xf", + "ic iallu", + "dsb ish", + "isb", + "br {entry}", + entry = in(reg) entry, + in("x0") args as *const KernelArgs, + options(noreturn), + ); + } +} diff --git a/bootloader/src/arch/mod.rs b/bootloader/src/arch/mod.rs new file mode 100644 index 00000000000..29a2241ec9a --- /dev/null +++ b/bootloader/src/arch/mod.rs @@ -0,0 +1,7 @@ +//! The machine the loader runs on, and the only part of it that knows which one. + +#[cfg_attr(target_arch = "x86_64", path = "x86_64.rs")] +#[cfg_attr(target_arch = "aarch64", path = "aarch64.rs")] +mod imp; + +pub use imp::*; diff --git a/bootloader/src/arch/x86_64.rs b/bootloader/src/arch/x86_64.rs new file mode 100644 index 00000000000..6a34eea1420 --- /dev/null +++ b/bootloader/src/arch/x86_64.rs @@ -0,0 +1,119 @@ +//! x86-64: the loader's every instruction Rust has no portable spelling for. + +use toyos_abi::boot::KernelArgs; + +/// The machine the kernel image must be built for: the loader's own. +pub const ELF_MACHINE: toyos_elf::Machine = toyos_elf::Machine::X86_64; + +/// How the boot map's entries are encoded. +pub use toyos_bootmap::x86_64 as encoding; + +/// How the boot map types memory: by the MTRRs firmware programmed, beneath +/// entries that select plain memory. +pub fn typing(_write_back: &[(u64, u64)]) -> toyos_bootmap::Typing<'_> { + toyos_bootmap::Typing::Firmware +} + +/// The time-stamp counter, which counts from reset. +pub fn counter() -> u64 { + // SAFETY: RDTSC reads a counter and nothing else; every x86-64 has it. + unsafe { core::arch::x86_64::_rdtsc() } +} + +/// What the loader's report says beside the counter: `IA32_TSC_ADJUST`, +/// where CPUID says the CPU has it — every write to the TSC since reset is +/// added to it (Intel SDM Vol. 3B, "Time-Stamp Counter Adjustment"), so zero +/// is a counter firmware never wrote and the TSC is time since power-on. +pub fn counter_origin() -> alloc::string::String { + let max = core::arch::x86_64::__cpuid(0).eax; + // Leaf 7 exists when the maximum leaf reaches it. + if max < 7 || core::arch::x86_64::__cpuid_count(7, 0).ebx & (1 << 1) == 0 { + return alloc::string::String::from("IA32_TSC_ADJUST not on this CPU"); + } + let (lo, hi): (u32, u32); + // SAFETY: the loader runs at CPL 0, and CPUID.07H:EBX[1] says the MSR exists. + unsafe { + core::arch::asm!("rdmsr", in("ecx") 0x3bu32, out("eax") lo, out("edx") hi, options(nomem, nostack)) + }; + alloc::format!("IA32_TSC_ADJUST {}", ((u64::from(hi) << 32) | u64::from(lo)) as i64) +} + +/// `CLFLUSH`'s line on every x86-64 part. +const LINE: u64 = 64; + +/// Every line of `[at, at + len)` written back out of this CPU's caches and +/// every other CPU's, before this returns. +pub fn write_back(at: u64, len: usize) { + let mut line = at & !(LINE - 1); + while line < at + len as u64 { + // SAFETY: `CLFLUSH` writes back and invalidates the line containing the + // address and touches nothing else; the caller names memory it + // allocated, and the instruction faults on nothing a canonical address + // can be. + unsafe { + core::arch::asm!("clflush [{addr}]", addr = in(reg) line as *const u8, options(nostack, preserves_flags)); + } + line += LINE; + } + // SAFETY: `SFENCE` orders those writebacks ahead of whatever ends this + // machine; it touches no memory or register. + unsafe { core::arch::asm!("sfence", options(nostack, preserves_flags)) }; +} + +/// What the loader says about the CPU as firmware handed it over, or why the +/// kernel cannot run on it. A UEFI x86-64 loader runs in long mode at CPL 0, +/// the state the kernel's entry takes, so there is nothing to say or refuse. +pub fn cpu_as_entered() -> Result, alloc::string::String> { + Ok(None) +} + +/// The I/O port space the chipset's TCO block answers in. +pub mod pio { + /// Whether this architecture has an I/O port space at all. + pub const EXISTS: bool = true; + + /// # Safety + /// No fault in Ring 0; the caller owns which device answers at `port` and + /// what the word commands it to do. `kernel/src/arch/x86_64/cpu.rs` states + /// the same contract for the same instruction. + pub unsafe fn outw(port: u16, value: u16) { + // SAFETY: the caller's contract. + unsafe { + core::arch::asm!("out dx, ax", in("dx") port, in("ax") value, options(nomem, nostack, preserves_flags)) + }; + } + + /// One word from an I/O port; safe because a read has no value a caller can + /// get wrong, as `kernel/src/arch/x86_64/cpu.rs`'s `inw` is. + pub fn inw(port: u16) -> u16 { + let value: u16; + // SAFETY: one instruction into the declared output, no memory operand. + unsafe { + core::arch::asm!("in ax, dx", out("ax") value, in("dx") port, options(nomem, nostack, preserves_flags)); + } + value + } +} + +/// Switch to the boot map at `args.boot_pml4_addr` and jump to the kernel +/// image's entry through the high half, handing it `args`. +/// +/// # Safety +/// The boot map identity-maps the memory this code and its stack run from and +/// maps the kernel image at `PHYS_OFFSET`; `image` is that relocated image, +/// `entry_offset` its entry point's offset in it, and `args` stays where it is +/// until the kernel copies it. +pub unsafe fn enter_kernel(image: (u64, u64), entry_offset: u64, args: &KernelArgs) -> ! { + let root = args.boot_pml4_addr; + let entry = crate::PHYS_OFFSET + image.0 + entry_offset; + // SAFETY: the caller's contract: the switch keeps this code and stack + // mapped, and the jump lands in the image it mapped. + unsafe { core::arch::asm!("mov cr3, {}", in(reg) root, options(nostack)) }; + // SAFETY: `kernel.elf`'s entry point takes `&KernelArgs` in `rdi` by the boot + // protocol `toyos-abi::boot` and the kernel side of it define between them — + // `sysv64`, because this target's own `"C"` is the Microsoft convention — + // and this bootloader has no way to check the callee's signature, only to + // keep its own side of that contract. + let entry: extern "sysv64" fn(&KernelArgs) -> ! = unsafe { core::mem::transmute(entry) }; + entry(args) +} diff --git a/bootloader/src/blackbox.rs b/bootloader/src/blackbox.rs index 6dfc7851c92..85a25013d35 100644 --- a/bootloader/src/blackbox.rs +++ b/bootloader/src/blackbox.rs @@ -252,30 +252,11 @@ fn when(stamp: u64) -> String { alloc::format!("{HEAD} the record below is from the boot armed at {}", Civil::from_unix_secs(stamp).stem()) } -/// Write the page back out of this CPU's caches, and every other CPU's. -/// -/// The kernel's `blackbox::flush` is the same loop for the same reason; the -/// instruction is each binary's because `toyos-blackbox` forbids unsafe code, -/// and `toyos_blackbox::CACHE_LINE` is the one decision they share. +/// Write the page back out of every CPU's caches, as the kernel's +/// `blackbox::flush` does and for the same reason: a reset does not write dirty +/// lines back. fn flush(page: Page) { - let mut line = 0usize; - while line < BYTES { - // SAFETY: `CLFLUSH` writes back and invalidates the line containing the - // address and touches nothing else; the address is inside the page this - // image allocated, and the instruction faults on nothing a canonical - // address can be. - unsafe { - core::arch::asm!( - "clflush [{addr}]", - addr = in(reg) (page.0 + line as u64) as *const u8, - options(nostack, preserves_flags), - ); - } - line += toyos_blackbox::CACHE_LINE; - } - // SAFETY: `SFENCE` orders those writebacks ahead of whatever ends this - // machine; it touches no memory or register. - unsafe { core::arch::asm!("sfence", options(nostack, preserves_flags)) }; + crate::arch::write_back(page.0, BYTES); } /// A sealed [`toyos_blackbox::Fault`] as lines for the log. diff --git a/bootloader/src/main.rs b/bootloader/src/main.rs index 9031a8cc2e3..40d1bdf5c37 100644 --- a/bootloader/src/main.rs +++ b/bootloader/src/main.rs @@ -16,11 +16,11 @@ use uefi::{ proto::device_path::{media::{PartitionFormat, PartitionSignature}, DevicePath, DevicePathNode, DeviceType, DeviceSubType}, proto::loaded_image::LoadedImage, proto::media::file::{File, FileAttribute, FileInfo, FileMode}, - table::{boot::{MemoryType, OpenProtocolAttributes, OpenProtocolParams, PAGE_SIZE}, cfg::ACPI2_GUID, runtime::ResetType}, + table::{boot::{MemoryAttribute, MemoryType, OpenProtocolAttributes, OpenProtocolParams, PAGE_SIZE}, cfg::ACPI2_GUID, runtime::ResetType}, Event, }; use toyos_abi::boot::{KernelArgs, MemoryMapEntry, RootBridgeWindow, MAX_ROOT_BRIDGE_WINDOWS}; -use toyos_bootmap::{Plan, BOOT_MAP_BYTES, MAX_PAGES, PML4_HIGH_HALF, PML4_IDENTITY}; +use toyos_bootmap::{Plan, BOOT_MAP_BYTES, MAX_PAGES, ROOT_HIGH_HALF, ROOT_IDENTITY}; use toyos_update::policy; use toyos_update::record::{self, Booted, Ended, Record}; @@ -38,6 +38,7 @@ macro_rules! println { }; } +mod arch; mod attempt; mod blackbox; mod bootnext; @@ -75,10 +76,8 @@ const MAP_MARGIN: usize = 64; fn alloc_kernel_memory(size: usize) -> vec::Vec { const KERNEL_ALIGN: usize = 2 * 1024 * 1024; // 2MB let layout = Layout::from_size_align(size, KERNEL_ALIGN).expect("invalid layout"); - // SAFETY: `layout` has non-zero size — `size` is `vaddr_max + stack_size` - // at the one call site, and `stack_size` alone is a fixed 8 MiB — so - // `alloc_zeroed`'s "layout must have non-zero size" precondition always - // holds. + // SAFETY: `layout` has non-zero size so `alloc_zeroed`'s "layout must have + // non-zero size" precondition always holds. let ptr = unsafe { alloc::alloc::alloc_zeroed(layout) }; assert!(!ptr.is_null(), "kernel allocation failed"); // SAFETY: `ptr` was just returned by the global allocator for exactly @@ -336,7 +335,7 @@ fn load_kernel_elf(kernel_elf_bytes: &[u8]) -> LoadedKernel { // every program image with reads the kernel's own image here. Refused by // name before anything is allocated — ELF32, big-endian, a version that is // not `EV_CURRENT`, an `e_type` that is not `ET_DYN`, a machine that is not - // x86-64, no program headers or a table outside the file, more than + // this loader's own, no program headers or a table outside the file, more than // `toyos_elf::MAX_LOAD_SEGMENTS` `PT_LOAD`s or none at all, a `PT_LOAD` // with `p_filesz > p_memsz` or a `p_vaddr + p_memsz` or `p_offset + // p_filesz` that overflows, and an `e_entry` no segment covers. @@ -345,7 +344,7 @@ fn load_kernel_elf(kernel_elf_bytes: &[u8]) -> LoadedKernel { // loader: the pair is a (copy length, destination size) pair here too, as // the image is sized from every `p_memsz` and each segment is then copied // in at `p_filesz`. - let layout = toyos_elf::Layout::parse(kernel_elf_bytes) + let layout = toyos_elf::Layout::parse(kernel_elf_bytes, arch::ELF_MACHINE) .unwrap_or_else(|e| panic!("kernel.elf: {e}")); // Section headers are optional to `toyos-elf`, which loads programs whose @@ -363,10 +362,16 @@ fn load_kernel_elf(kernel_elf_bytes: &[u8]) -> LoadedKernel { println!("Kernel stack size: {}", stack_size); // `vaddr_max` is the largest `p_vaddr + p_memsz` over the `PT_LOAD` // segments, and the image is laid out at its own vaddrs — so it is what the - // kernel's memory has to cover before the stack is added to it. + // kernel's memory has to cover before the stack is added to it. The stack + // starts on the next page: where the image ends is the linker's choice, and + // both ABIs want the stack pointer 16-byte aligned, which AArch64's + // `SCTLR_EL1.SA` enforces on every access through it. + const PAGE: u64 = 4096; let mem_size = layout .vaddr_max - .checked_add(stack_size as u64) + .checked_add(PAGE - 1) + .map(|end| end & !(PAGE - 1)) + .and_then(|stack_base| stack_base.checked_add(stack_size as u64)) .and_then(|n| usize::try_from(n).ok()) .expect("kernel.elf: image plus stack does not fit an allocation"); @@ -394,7 +399,7 @@ fn load_kernel_elf(kernel_elf_bytes: &[u8]) -> LoadedKernel { for section in sections.iter().filter(|section| section.kind == SHT_RELA) { let table = file_range(kernel_elf_bytes, section.offset, section.size) .expect("kernel.elf: SHT_RELA section is past the end of the file"); - for rela in RelaTable::new(table).iter() { + for rela in RelaTable::new(table, arch::ELF_MACHINE).iter() { match rela.kind { RelocKind::Relative => { // Both fields index the image and both come out of the @@ -515,14 +520,13 @@ fn query_gop(system_table: &SystemTable) -> Option { }) } -/// Write `plan` into `pt_mem` and return the PML4's physical address. +/// Write `plan` into `pt_mem` and return its root table's physical address. /// /// # Safety /// `pt_mem` is [`toyos_bootmap::MAX_PAGES`] pages of zeroed memory, 4096-aligned. unsafe fn build_boot_page_tables(pt_mem: *mut u8, plan: &Plan) -> u64 { - const PAGE_PRESENT: u64 = 1 << 0; - const PAGE_WRITE: u64 = 1 << 1; - const PAGE_SIZE_BIT: u64 = 1 << 7; + use arch::encoding::{block, page, table}; + use toyos_bootmap::Slot; let mut next_page = 0usize; let mut alloc_page = || -> *mut u64 { @@ -531,26 +535,53 @@ unsafe fn build_boot_page_tables(pt_mem: *mut u8, plan: &Plan) -> u64 { page }; - let pml4 = alloc_page(); + let root = alloc_page(); let identity_pdpt = alloc_page(); let high_pdpt = alloc_page(); let mut directories = [core::ptr::null_mut::(); toyos_bootmap::MAX_DIRECTORIES]; for (slot, gib) in plan.directories().iter().enumerate() { let pd = alloc_page(); directories[slot] = pd; - *identity_pdpt.add(*gib as usize) = pd as u64 | PAGE_PRESENT | PAGE_WRITE; - *high_pdpt.add(*gib as usize) = pd as u64 | PAGE_PRESENT | PAGE_WRITE; + *identity_pdpt.add(*gib as usize) = table(pd as u64); + *high_pdpt.add(*gib as usize) = table(pd as u64); + } + + let mut fine = [core::ptr::null_mut::(); toyos_bootmap::MAX_PAGES]; + for (table_at, (directory, index)) in plan.fine_slots().enumerate() { + let leaves = alloc_page(); + fine[table_at] = leaves; + *directories[directory].add(index) = table(leaves as u64); } for entry in plan.entries() { - *directories[entry.directory].add(entry.index) = - entry.phys | PAGE_PRESENT | PAGE_WRITE | PAGE_SIZE_BIT | entry.cache.bits(); + match entry.slot { + Slot::Directory { directory, index } => { + *directories[directory].add(index) = block(entry.phys, entry.cache) + } + Slot::Fine { table, index } => *fine[table].add(index) = page(entry.phys, entry.cache), + } } - *pml4.add(PML4_IDENTITY) = identity_pdpt as u64 | PAGE_PRESENT | PAGE_WRITE; - *pml4.add(PML4_HIGH_HALF) = high_pdpt as u64 | PAGE_PRESENT | PAGE_WRITE; + *root.add(ROOT_IDENTITY) = table(identity_pdpt as u64); + *root.add(ROOT_HIGH_HALF) = table(high_pdpt as u64); + + root as u64 +} - pml4 as u64 +/// Every range firmware's map says is write-back memory, `(base, length)`: +/// a descriptor carrying `EFI_MEMORY_WB` and not one of the two I/O types, +/// which a firmware may give the attribute without meaning memory. +fn write_back_memory(system_table: &SystemTable) -> vec::Vec<(u64, u64)> { + let boot_services = system_table.boot_services(); + let sizes = boot_services.memory_map_size(); + // Room for the descriptors allocating this buffer itself may add. + let mut buffer = vec![0u8; sizes.map_size + 8 * sizes.entry_size]; + let map = boot_services.memory_map(&mut buffer).expect("the memory map, before the exit"); + map.entries() + .filter(|d| d.att.contains(MemoryAttribute::WRITE_BACK)) + .filter(|d| d.ty != MemoryType::MMIO && d.ty != MemoryType::MMIO_PORT_SPACE) + .map(|d| (d.phys_start, d.page_count * PAGE_SIZE as u64)) + .collect() } /// Say whether `at .. at + len` is inside the boot map, and refuse the boot @@ -572,30 +603,23 @@ fn report_reach(what: &str, at: u64, len: u64) { ); } -/// The time-stamp counter, which counts from reset. +/// The CPU's free-running counter, which counts from reset. fn tsc() -> u64 { - // SAFETY: RDTSC reads a counter and nothing else; every x86-64 has it. - unsafe { core::arch::x86_64::_rdtsc() } -} - -/// `IA32_TSC_ADJUST`, where CPUID says the CPU has it: every write to the TSC -/// since reset is added to it (Intel SDM Vol. 3B, "Time-Stamp Counter -/// Adjustment"), so zero is a counter firmware never wrote and the TSC is time -/// since power-on. -fn tsc_adjust() -> Option { - let max = core::arch::x86_64::__cpuid(0).eax; - // Leaf 7 exists when the maximum leaf reaches it. - if max < 7 || core::arch::x86_64::__cpuid_count(7, 0).ebx & (1 << 1) == 0 { - return None; - } - let (lo, hi): (u32, u32); - // SAFETY: the loader runs at CPL 0, and CPUID.07H:EBX[1] says the MSR exists. - unsafe { core::arch::asm!("rdmsr", in("ecx") 0x3bu32, out("eax") lo, out("edx") hi, options(nomem, nostack)) }; - Some(((u64::from(hi) << 32) | u64::from(lo)) as i64) + arch::counter() } #[allow(clippy::too_many_arguments)] fn start_kernel(kernel: LoadedKernel, kernel_elf_bytes: vec::Vec, cmdline: vec::Vec, rsdp_addr: u64, gop: Option, boot_part: Option, log_partition_guid: [u8; 16], rtc_utc_offset: Option, root_image: Option, entry_tsc: u64, system_table: SystemTable) -> ! { + // Said before it is refused, for `report_reach`'s reason. + match arch::cpu_as_entered() { + Ok(None) => {} + Ok(Some(line)) => println!("{line}"), + Err(why) => { + println!("{why}"); + panic!("{why}"); + } + } + // The last of the firmware questions, and asked here for the same reason // the GOP's was asked before this: the protocol dies with boot services. // @@ -619,11 +643,16 @@ fn start_kernel(kernel: LoadedKernel, kernel_elf_bytes: vec::Vec, cmdline: v // // Said before it is applied: a machine this refuses leaves the refusal in // `loader.log`, which is the artifact a machine with no console has. - let planned = Plan::new(gop.as_ref().map(|g| (g.framebuffer, g.framebuffer_size))); + // Firmware's write-back memory, for an architecture whose boot map types + // pages by it. Before the exit, while the map can still be asked for; the + // attributes a descriptor carries do not change across it. + let write_back = write_back_memory(&system_table); + let planned = + Plan::new(gop.as_ref().map(|g| (g.framebuffer, g.framebuffer_size)), arch::typing(&write_back)); match &planned { Ok(plan) => match plan.scanout() { Some((at, len)) => println!( - "Scanout: {at:#x}+{len:#x} mapped uncacheable in 2 MiB pages at identity and at \ + "Scanout: {at:#x}+{len:#x} mapped as the scanout in 2 MiB pages at identity and at \ PHYS_OFFSET, in {} page directories", plan.directories().len() ), @@ -636,7 +665,7 @@ fn start_kernel(kernel: LoadedKernel, kernel_elf_bytes: vec::Vec, cmdline: v // SAFETY: `pt_mem` is the `MAX_PAGES * 4096`-byte, 4096-aligned, zeroed // allocation above, and a `Plan` never names more pages than that. let pml4_phys = unsafe { build_boot_page_tables(pt_mem, &plan) }; - println!("Boot map: PML4 {pml4_phys:#x}, {BOOT_MAP_BYTES:#x} bytes at identity and at PHYS_OFFSET"); + println!("Boot map: root {pml4_phys:#x}, {BOOT_MAP_BYTES:#x} bytes at identity and at PHYS_OFFSET"); let kernel_phys = kernel.memory.as_ptr() as u64; report_reach("Kernel image", kernel_phys, kernel.memory.len() as u64); @@ -709,12 +738,9 @@ fn start_kernel(kernel: LoadedKernel, kernel_elf_bytes: vec::Vec, cmdline: v kernel_args.loader_handoff_tsc = tsc(); println!( - "Loader TSC: {entry_tsc} at entry, {} at the handoff; IA32_TSC_ADJUST {}", + "Loader TSC: {entry_tsc} at entry, {} at the handoff; {}", kernel_args.loader_handoff_tsc, - match tsc_adjust() { - Some(adjust) => alloc::format!("{adjust}"), - None => alloc::string::String::from("not on this CPU"), - } + arch::counter_origin(), ); // Last, and after every line above: a console write, a FAT write and a @@ -756,29 +782,19 @@ fn start_kernel(kernel: LoadedKernel, kernel_elf_bytes: vec::Vec, cmdline: v kernel_args.memory_map_size = memory_map.len() as u64 * mem::size_of::() as u64; - // Switch to new page tables. SAFETY: `pml4_phys` is the table built above, - // identity-mapping low memory (so the code and stack this instruction - // itself runs from stay mapped across the switch) and high-half-mapping - // the same range at `PHYS_OFFSET` for the jump below. The assert before the - // exit proved the whole kernel image is inside that range. - unsafe { core::arch::asm!("mov cr3, {}", in(reg) pml4_phys, options(nostack)) }; - - let entry_virt = PHYS_OFFSET + kernel_phys + kernel.entry_offset as u64; - mem::forget(memory_map); mem::forget(kernel.memory); mem::forget(kernel_elf_bytes); mem::forget(cmdline); - // SAFETY: `entry_virt` is `kernel_phys + entry_offset` read through the - // high-half mapping just switched to, which the assert above proved - // covers the whole kernel image. `kernel.elf`'s entry point is `extern - // "sysv64" fn(&KernelArgs) -> !` by the boot protocol `toyos-abi::boot` - // and the kernel side of it define between them — this bootloader has no - // way to check the callee's signature, only to keep its own side of that - // contract. - let entry: extern "sysv64" fn(&KernelArgs) -> ! = unsafe { mem::transmute(entry_virt) }; - entry(&kernel_args); + let image = (kernel_phys, kernel_args.kernel_memory_size); + // SAFETY: `kernel_args.boot_pml4_addr` is the table built above, + // identity-mapping low memory (so the code and stack a switch to it runs + // from stay mapped across it) and mapping the same range at `PHYS_OFFSET`; + // the assert before the exit proved the whole kernel image is inside that + // range, and `image` is that image, relocated, with its entry at + // `entry_offset`. + unsafe { arch::enter_kernel(image, kernel.entry_offset as u64, &kernel_args) } } /// When this pass armed the page, in Unix seconds, or 0 where firmware would diff --git a/bootloader/src/watchdog.rs b/bootloader/src/watchdog.rs index 2cdbf0b1c51..798a0f15b18 100644 --- a/bootloader/src/watchdog.rs +++ b/bootloader/src/watchdog.rs @@ -54,6 +54,9 @@ pub fn arm(system_table: &SystemTable, rsdp_addr: u64, cmdline: &str) { if !toyos_abi::boot::actuators(cmdline).any(|token| token == toyos_tco::PARAM) { return; } + if !crate::arch::pio::EXISTS { + return refused(format_args!("this architecture has no I/O port space, where a TCO block answers")); + } let ecam = match toyos_acpi::ecam_base(Identity, rsdp_addr) { Ok((_, base)) => base, Err(e) => return refused(format_args!("this machine's tables name no ECAM ({e:?})")), @@ -81,14 +84,14 @@ pub fn arm(system_table: &SystemTable, rsdp_addr: u64, cmdline: &str) { // SAFETY: `port` is `toyos_tco`'s answer for the row this machine's own PCI // ids matched, and every offset written is inside that row's block. unsafe { - outw(port + TCO_TMR, toyos_tco::TIMER); - outw(port + TCO1_CNT, TCO1_CNT_RUN); + crate::arch::pio::outw(port + TCO_TMR, toyos_tco::TIMER); + crate::arch::pio::outw(port + TCO1_CNT, TCO1_CNT_RUN); // Reloading is also what returns the expiry count to zero. - outw(port + TCO_RLD, 1); + crate::arch::pio::outw(port + TCO_RLD, 1); } // `TCO1_CNT` is judged whole apart from `TCO1_CNT_LOCK`, which the datasheet // says no write clears: one bit of it is not the register. - let cnt = inw(port + TCO1_CNT); + let cnt = crate::arch::pio::inw(port + TCO1_CNT); if !toyos_tco::cnt_took_the_write(cnt) { return refused(format_args!( "{port:#x} did not take TCO1_CNT={TCO1_CNT_RUN:#06x}, it reads {cnt:#06x}" @@ -107,10 +110,10 @@ pub fn arm(system_table: &SystemTable, rsdp_addr: u64, cmdline: &str) { /// What the block holds once it is armed, as whole words. fn report(port: u16, cnt: u16) { - let rld = inw(port + TCO_RLD); - let tmr = inw(port + TCO_TMR); - let sts1 = inw(port + TCO1_STS); - let sts2 = inw(port + TCO2_STS); + let rld = crate::arch::pio::inw(port + TCO_RLD); + let tmr = crate::arch::pio::inw(port + TCO_TMR); + let sts1 = crate::arch::pio::inw(port + TCO1_STS); + let sts2 = crate::arch::pio::inw(port + TCO2_STS); println!( "watchdog: read back TCO_RLD={rld:#06x} TCO_TMR={tmr:#06x} TCO1_CNT={cnt:#06x} \ TCO1_STS={sts1:#06x} TCO2_STS={sts2:#06x}" @@ -187,25 +190,6 @@ fn config_u32(ecam: u64, device: u8, function: u8, offset: u16) -> u32 { unsafe { read_volatile(at as *const u32) } } -/// # Safety -/// No fault in Ring 0; the caller owns which device answers at `port` and what -/// the word commands it to do. `kernel/src/arch/cpu.rs` states the same -/// contract for the same instruction. -unsafe fn outw(port: u16, value: u16) { - core::arch::asm!("out dx, ax", in("dx") port, in("ax") value, options(nomem, nostack, preserves_flags)); -} - -/// One word from an I/O port; safe because a read has no value a caller can get -/// wrong, as `kernel/src/arch/cpu.rs`'s `inw` is. -fn inw(port: u16) -> u16 { - let value: u16; - // SAFETY: one instruction into the declared output, no memory operand. - unsafe { - core::arch::asm!("in ax, dx", out("ax") value, in("dx") port, options(nomem, nostack, preserves_flags)); - } - value -} - /// Why this machine is not watched, and that the boot goes on anyway. fn refused(why: core::fmt::Arguments) { println!("watchdog: {why}. This boot is unwatched until the kernel arms its own"); diff --git a/issues/boot-media/the-boot-map-reaches-4-gib-and-firmware-decides-what-lands-in-it.md b/issues/boot-media/the-boot-map-reaches-4-gib-and-firmware-decides-what-lands-in-it.md new file mode 100644 index 00000000000..d9bc0ef517c --- /dev/null +++ b/issues/boot-media/the-boot-map-reaches-4-gib-and-firmware-decides-what-lands-in-it.md @@ -0,0 +1,41 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# The boot map reaches 4 GiB, and firmware decides what lands in it + +`toyos-bootmap`'s plan maps physical `0..BOOT_MAP_BYTES` (4 GiB) plus the +scanout, and everything the kernel touches before `mm::init` has to be inside +it. The loader places none of it there: the kernel image +(`alloc_kernel_memory`), the memory map it hands over, the boot parameter and +ROOT's image all come from the pool allocator or `AnyPages`, wherever firmware +puts them. `report_reach` then refuses the boot by name for the three it +checks — and does not check the memory map at all, which the kernel reads +through the direct map before `mm::init` too. + +On q35 OVMF allocates below 4 GiB, so the x86 suite has never met it. On QEMU +`virt` RAM starts at 1 GiB and AAVMF allocates from its top: with `-m 4G` +the kernel image lands at 0x13b400000 and the loader refuses the boot +("Kernel image at 0x13b400000+0xb07000 is outside the boot map"). The +harness's `Profile::Virt` boots with 2 GiB for that reason +(`tests/common/qemu.rs`, `qemu_command`). A real machine of either +architecture with RAM above 4 GiB and a firmware that allocates top-down is +the same defect. + +**Exit condition**: the loader allocates what the kernel reads before +`mm::init` with `AllocateType::MaxAddress` inside the map (or the plan grows to +reach all of firmware's write-back memory), the memory map is among what +`report_reach` checks, and `Profile::Virt` boots with the 4 GiB every other +profile has. + +**The AArch64 typing refuses a part-typed page.** `Typing::of` +(`toyos-bootmap/src/lib.rs`) refuses the boot (`Refusal::Mixed`) for any 2 MiB +page of the low 4 GiB that firmware's memory map types in part: written back +for some of it and not the rest. QEMU `virt`'s map has no such page. A +SystemReady machine whose map has 4 KiB holes there, which the PR #524 review +expects and nothing here has measured, is refused at stage 9 of +`issues/kernel/toyos-runs-on-arm64.md`; that stage owes the split of such a +page into 4 KiB leaves typed by the map, as the scanout's partial pages +already are. diff --git a/issues/build/a-defect-only-contention-exposes-is-classified-as-a-wrong-sched.md b/issues/build/a-defect-only-contention-exposes-is-classified-as-a-wrong-sched.md new file mode 100644 index 00000000000..3a63445eca4 --- /dev/null +++ b/issues/build/a-defect-only-contention-exposes-is-classified-as-a-wrong-sched.md @@ -0,0 +1,36 @@ +--- +status: open +kind: tooling +opened: 2026-09-26 +--- + +# A defect only contention exposes is classified as a wrong `Sched::Parallel` + +When a test fails in the wide run and passes in the lone re-run, and the wide +run shared the host, `alone_line` (`tests/toyos.rs`) prints one verdict: +`GREEN — it fails only beside other guests, so its Sched::Parallel is wrong`. +That names a cause. What the two runs establish is only that the failure needs +the timing a loaded host produces, and a kernel race that only a slow or +preempted vCPU opens produces exactly that pair. + +Recorded case, PR #524's fast tier at `f67863c2`: 270 passed, 130 failed, and +254 of the failures were one kernel panic, `vconsole: no tx slot`. Each of the +130 was green alone, so each got the scheduling verdict, 130 times. The cause +was one kernel defect: `serial::BackendGuard::try_lock` built, and so dropped +and unlocked, a guard for the CPU that lost the exchange (fixed in +`1e5ef5c1`). Changing any test's `Sched` would have hidden it. + +What reads it wrong: + +- The per-test verdict names a cause the evidence does not decide. The + `tests/CLAUDE.md` caveat already says it is a hypothesis; the line itself + says it is a finding. +- Nothing groups the failures. Tests that fail wide on one shared headline (the + same panic message, `toyos_build::alone::same_failure`) are one finding, and + the report prints them as N separate classifications. + +**Exit condition**: the shared-host green arm states the observation (fails +only beside other guests) without naming `Sched` as the cause, and the run's +summary groups wide failures that share a headline, printing the group's count +before any per-test classification. The staged pair for the grouping is the +`f67863c2` shape: many tests, one panic headline. diff --git a/issues/build/a-guest-lane-leaves-the-toolchain-tarball-in-its-tmpdir.md b/issues/build/a-guest-lane-leaves-the-toolchain-tarball-in-its-tmpdir.md new file mode 100644 index 00000000000..ec08efc3e01 --- /dev/null +++ b/issues/build/a-guest-lane-leaves-the-toolchain-tarball-in-its-tmpdir.md @@ -0,0 +1,19 @@ +--- +status: open +kind: tooling +opened: 2026-09-26 +--- + +# A guest lane leaves the toolchain tarball in its `$TMPDIR` + +`src/ci.rs`'s guest lane points `$TMPDIR` at a `TempDir` and ends on +`nothing left in $TMPDIR`. Its `the toolchain` step is `release::install`, which +downloads `toyos-toolchain.tar.zst` to `std::env::temp_dir()`, unpacks it, +and never removes it. So every guest, tcg and audio job reds on +`left … : toyos-toolchain.tar.zst` whatever its suite did. Seen on all fifteen +of those jobs in the nightly on PR #524's branch (run 36266825579). Neither +side is that branch's: both are main's, the step from `ba68664b` (#529), the +install from before it. Main has had no nightly since #529 landed. + +**Exit**: the tarball goes into a `TempDir`, or is removed once it is +unpacked, and a guest lane's last step is green on a nightly. diff --git a/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md b/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md new file mode 100644 index 00000000000..2aa83a71564 --- /dev/null +++ b/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md @@ -0,0 +1,36 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# Assembly outside an arch module: two framebuffers and the guest probes + +The owner's ruling of 2026-09-26 puts every `asm!`, `global_asm!`, +`naked_asm!`, naked function and `core::arch::*` intrinsic inside an +architecture's own module: `kernel/src/arch//`, the bootloader's +`src/arch/`, and `toyos-abi`'s per-arch syscall entry. `src/sourcegate.rs`'s +`ARCH_RULES` enforces it. The kernel and the loader now hold none outside +those; what is left is declared in that table as an exception, each row +pointing here: + +- `userland/toyos-window/src/framebuffer.rs` and `userland/metalprobe/src/fb.rs`: + `_mm_sfence` after writing a write-combining framebuffer. Userland has no + portable way to say "drain my stores to the scanout"; the SDK (`toyos/src`) + owes one, and it is also only changed under an ABI brief. +- Seventeen guest probes in `tests/toyos-rust-tests/src/bin/`, whose subject + is an x86 instruction (`rdgsbase`, `fxsave64`, `int1`, x87 control words) + or the raw `syscall` gate with arguments no SDK call will pass. They run in + the x86-64 suite, which is the only suite until the harness gains its arch + axis (the port's stage 8). +- The userland drivers `netd` (`virtio_net.rs`, `i219.rs`) and `soundd` + (`virtio.rs`) order their DMA rings with `fence(Release)`/`fence(Acquire)`. + That is not assembly and no rule reds on it, but it is the same missing + interface: on AArch64 a `fence` is `dmb ish`, which orders nothing a device + outside the inner-shareable domain observes (the kernel's `arch::barrier` + says why, and `Mmio` carries `writel`/`readl` ordering there). + +**Exit condition**: the SDK gains a per-architecture module (as `toyos-abi`'s +syscall entry and libc's `arch/` are) holding a scanout flush and DMA barriers; the guest probes move under a per-arch +directory the harness selects by `Arch`; and every row in `ARCH_RULES` that +cites this file is deleted. diff --git a/issues/build/blocking-read-window-completed-26-of-500-beside-other-guests.md b/issues/build/blocking-read-window-completed-26-of-500-beside-other-guests.md new file mode 100644 index 00000000000..45649d3551b --- /dev/null +++ b/issues/build/blocking-read-window-completed-26-of-500-beside-other-guests.md @@ -0,0 +1,32 @@ +--- +status: open +kind: finding +opened: 2026-09-26 +--- + +# `blocking_read_window` completed 26 of 500 round trips beside other guests + +Fast tier at `1e5ef5c1` (PR #524's branch; the host carried the branch's own +twelve guest slots and another worktree's suite at the same time): +`blocking_read_stress: only 26 of 500 round trips completed inside 3s — a wake +was not delivered`. The harness's re-run alone was green in 2 s +(`at least 193 held windows a post landed in (0 -> 256)`). `cargo run -- +--known-red blocking_read_window` answers NO. + +What the red run's own log says against its sentence: the two processes spent +`cpu=1639ms` and `cpu=1998ms` of the 3.5 s they ran (`syscall_wall=3535ms` and +`3599ms`), and cpu0 took 522 interrupts, 456 of them xHCI — a guest that was +running and slow, not one parked on a wake that never came. So the verdict's +cause is unread: the test waits host seconds and names a lost wake when they +run out, which a starved guest satisfies as well as a lost wake does. + +Main reproduces it. A same-session A/B, runs of main (`d65446cc`) and of the +branch started together so both arms carried one load (six suites at once): +main 16 of 17 green, the branch 17 of 18, and each arm's one red is this +sentence (`only 26 of 500` on main, `only 27 of 500` on the branch; the +branch's red spent `cpu=1568ms` and `cpu=1532ms` of its window). The rate is +the same on both arms, so the branch did not move it. + +**Exit**: the verdict tells a lost wake from a slow guest (the round trips' +progress over the window, not only the count at its end), and a cause for +this run. diff --git a/issues/build/issue-files-cite-paths-that-moved.md b/issues/build/issue-files-cite-paths-that-moved.md new file mode 100644 index 00000000000..7f8760805cf --- /dev/null +++ b/issues/build/issue-files-cite-paths-that-moved.md @@ -0,0 +1,40 @@ +--- +status: open +kind: tooling +opened: 2026-09-26 +--- + +# Issue files cite source paths that no longer exist + +An issue names the site it is about by path, and a path is the claim a reader +checks first. Moving a source file leaves every issue that cites it pointing at +nothing, and nothing notices. + +Measured on PR #524's branch at `73a89365` (the port's stage 0 moved the +kernel's x86 code under `kernel/src/arch/x86_64/`, the syscall handlers out of +`arch/`, and `hw.rs`): every `kernel/`, `src/`, `userland/`, `tests/`, +`bootloader/` and `toyos-*/` path written in `issues/`, checked against that +tree and against `origin/main`: + +- 46 citations in 35 files name a path that exists on `main` and not on the + branch: the branch moved them. Among them `kernel/src/arch/syscall/*.rs`, + `kernel/src/hw.rs`, `kernel/src/mm/paging.rs`, `kernel/src/arch/apic.rs`, + `kernel/src/arch/idt/*.rs` and `kernel/src/drivers/watchdog.rs`. +- Already on `main`, before this branch: at least 14 citations in 11 files of + a whole `kernel/src/` path that does not exist (`kernel/src/arch/syscall.rs`, + `kernel/src/inbox.rs`, `kernel/src/log_file.rs`, + `kernel/src/completion/mod.rs` among them). The wider count, 153 in 95 files, + also holds crate-relative spellings (`sched/dump.rs`) that the check cannot + tell from rot. + +**Whether a gate belongs with it.** `src/CLAUDE.md` says documentation carries +no gates, and an issue is prose; a red gate over `issues/` contradicts that +rule, so it is the owner's to decide. What the tree already requires of a +deleted document (its citations go in the same merge) is the rule a move needs +too, and an on-demand check that lists every cited path that does not resolve +(offline, beside `--check-forks`) would let the mover do it. A gate is what +makes the mover's duty hold. The on-demand check only makes it cheap. + +**Exit condition**: every whole-repository path written in `issues/` resolves +in the tree that holds it, and the owner has ruled whether a move that strands +one is refused by a gate or caught on demand. diff --git a/issues/build/ovmf-s-licence-record-names-no-openssl.md b/issues/build/ovmf-s-licence-record-names-no-openssl.md new file mode 100644 index 00000000000..dd99fbafc47 --- /dev/null +++ b/issues/build/ovmf-s-licence-record-names-no-openssl.md @@ -0,0 +1,22 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# OVMF's licence record names no OpenSSL + +`NOTICE`'s `ovmf/*.fd` section and its three `src/licence.rs` rows say +`BSD-2-Clause-Patent`. EDK II's platforms link its bundled OpenSSL into the +firmware through `CryptoPkg` (`BaseCryptLib`, and `TlsLib` when +`NETWORK_TLS_ENABLE` is set), and OpenSSL 3 is Apache-2.0. For the AArch64 +firmware this is established and recorded (PR #524: edk2-stable202408's +ArmVirtQemu links OpenSSL 3.0.9 whether TLS is on or off, and QEMU builds it +with TLS on). For OVMF nothing is: the section itself says the files came +with no version record or build recipe, so which CryptoPkg libraries they +link is unknown, and the record claims a single licence nobody has checked. + +**Exit condition**: the OVMF images' crypto content is read out of the +binaries or their upstream build, and the section and rows state every +licence it carries, or the images are replaced by ones whose build is known +(the section's own open question to the owner). diff --git a/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md b/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md index 7b442238f29..2c18c525d26 100644 --- a/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md +++ b/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md @@ -22,3 +22,10 @@ its cause. **Exit**: a cause for the empty uart on a boot that rebooted as designed, or the marker waited for where the boot's reboot cannot race it. + +Again in the fast tier at `8846c021` (PR #524's branch, alone on the host): +the same `QEMU died before ===READY=== (status: exit 0)` with `uart: nothing +at all`, after `stop: 5 of 5 userland thread(s) stopped`, `usb-quiesce: disk 0 +SYNCHRONIZE CACHE ok` and `Rebooting.`; this time no usb-storage transport +break before it. The re-run alone was green. `cargo run -- --known-red` +answers NO. diff --git a/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md b/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md index 342c8c26a86..8e268f16a72 100644 --- a/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md +++ b/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md @@ -37,3 +37,9 @@ with the same rule; if the readings split by runner model, the gate names the model it judges and refuses to judge the rest. Until then, a CI red under this name is this file and the redlist row, and the landing it dequeues is re-queued once, not re-run until green. + +A second signature, which this file's quarantine row does not cover. The fast +tier on PR #524's branch at `235c5a5b` reds `sched_check_build` on `cpu0: 85 +passes … a 90th percentile needs at least 100 samples behind it and this has +85`. The host was loaded then by another worktree's spinner at 397% CPU. The +harness's re-run alone was green. diff --git a/issues/build/the-primary-rebuilds-its-compiler-on-compiler-alone.md b/issues/build/the-primary-rebuilds-its-compiler-on-compiler-alone.md new file mode 100644 index 00000000000..39616126a87 --- /dev/null +++ b/issues/build/the-primary-rebuilds-its-compiler-on-compiler-alone.md @@ -0,0 +1,24 @@ +--- +status: assigned +kind: tooling +opened: 2026-09-26 +--- + +# The primary rebuilds its compiler on `compiler/` alone + +`src/toolchain.rs` rebuilds the primary's toolchain when the stamp over +`rust/compiler/` changes, and `compiler::record` writes that tree as the +compiler the primary's `stage2` is. A fork commit that moves only +`src/bootstrap`, `src/tools`, `src/stage0`, `Cargo.lock` or the LLVM submodule +leaves the primary on the compiler it had, and a worktree whose `compiler/` +matches the record is handed that compiler too. A worktree's own compiler is +keyed on all of them (`compiler::key`, PR #524), so the two answers to "which +compiler do these sources name" differ. + +**Owner**: the ARM64 track (`issues/kernel/toyos-runs-on-arm64.md`), whose +PR #524 made the worktree key; it is closed before that track's stage 4 +lands. + +**Exit condition**: the primary's rebuild and its record read the same sources +`compiler::key` does, and a worktree compares against that, shown by a test +in which a fork moving only `src/tools` gets a compiler of its own. diff --git a/issues/build/the-toolkit-forks-resolve-an-x86-only-toyos-window.md b/issues/build/the-toolkit-forks-resolve-an-x86-only-toyos-window.md new file mode 100644 index 00000000000..072726d8c1d --- /dev/null +++ b/issues/build/the-toolkit-forks-resolve-an-x86-only-toyos-window.md @@ -0,0 +1,27 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# The toolkit forks resolve an x86-only toyos-window, so calc, snake and doom do not build for AArch64 + +`softbuffer`'s and `winit`'s toyos forks (the `…-sdk-0.2` branches +`userland/Cargo.toml` patches in) name `toyos-window = "0.2"`, and cargo +resolves that to the published `toyos-window 0.2.0`, not to the path crate: +the `[patch]` redirect only applies to a compatible version. That release's +`framebuffer.rs` calls `core::arch::x86_64::_mm_sfence` unconditionally, so it +does not compile for `aarch64-unknown-toyos`, and neither does anything that +pulls either fork in: `calc`, `snake` and `doom`. `cpal`'s fork, `mio`, `tokio`, +`getrandom` and `libloading` name `toyos-abi`/`toyos` 0.1/0.2, which already +select their syscall entry per architecture and build. + +`src/build.rs`'s `NOT_YET_BUILT` leaves the three off an AArch64 ROOT and says +so at build time. Everything else in the userland workspace builds and links +for AArch64 (PR #524). + +**Exit condition**: the forks name a `toyos-window` version the path crate +satisfies (PR #528 moves them to `…-sdk-0.15` branches for its own reasons, and +the path crate is at 0.16.0 on #524), the three build for +`aarch64-unknown-toyos`, and their rows in `NOT_YET_BUILT` are deleted; doom's +row goes when its C is also shown to compile for AArch64 through `toyos-cc`. diff --git a/issues/build/two-suites-in-one-worktree-race-on-the-c-corpus-libc.md b/issues/build/two-suites-in-one-worktree-race-on-the-c-corpus-libc.md new file mode 100644 index 00000000000..4eeef14b07e --- /dev/null +++ b/issues/build/two-suites-in-one-worktree-race-on-the-c-corpus-libc.md @@ -0,0 +1,27 @@ +--- +status: open +kind: tooling +opened: 2026-09-26 +--- + +# Two suites in one worktree race on the C corpus's libc archive + +`tests/common/compile.rs`'s `libc_archive_toyos` runs `cargo rustc +--crate-type staticlib` into `userland/libc/target/sysroot-/` once per +process, then every C test links `-ltoyos_libc` from there. Nothing holds the +archive across the build and its use, so a second `cargo test` in the same +worktree, running its own `cargo rustc` into the same directory, removes and +re-creates the archive while the first suite's C links read it. `src/build.rs`'s +`build_programs` holds `buildlock::artifact` across build and read for exactly +this reason; this path does not. + +Seen on PR #524's A/B, three `cargo test --test toyos-build -- +blocking_read_window` at once per worktree, on both main (`d65446cc`) and the +branch: four of thirty runs panicked before any guest booted, each naming a C +case that "stopped building" (`resolve_libs failed: cannot parse -ltoyos_libc: +library not found`, or `expected staticlib at …/libtoyos_libc.a` from the +`NOT_RUN` check). The harness reports it as a C test that no longer builds. + +**Exit condition**: the archive is built and read under one hold, as +`build_programs` does, and two suites started at once in one worktree both get +past the C corpus's build. diff --git a/issues/kernel/fpu-isolation-s-thread-join-answers-not-found-on-main.md b/issues/kernel/fpu-isolation-s-thread-join-answers-not-found-on-main.md new file mode 100644 index 00000000000..22f81548621 --- /dev/null +++ b/issues/kernel/fpu-isolation-s-thread-join-answers-not-found-on-main.md @@ -0,0 +1,31 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# `fpu_isolation`'s probe thread is joined as not found, on main + +`fpu_isolation` (nightly tier) is red on `main` at `17eb66a4`, and on every +branch built on it. Each `check` child spawns `thread_probe`, which records +its entry state and calls `SYS_THREAD_EXIT`. `thread_join` on that tid then +answers `18446744073709551614` (−2), where 0 is expected +(`tests/toyos-rust-tests/src/bin/fpu_isolation.rs:346`). The kernel logs the +thread's `exit … code=0`, so the thread ran, and `sys_thread_join` answers +`NotFound` from `wait_thread_zombie`'s `Err(())`: no zombie for a thread that +existed. The leak arm therefore reports `a process started with the previous +one's FP registers` in all three rounds; that is the arm's wording, and not +what failed. + +Measured by `cargo test --test toyos-build -- fpu_isolation --nightly`: +- at `17eb66a4`: EXIT=1, the harness's alone re-run red the same way; +- on PR #524's branch at `235c5a5b`, and on its nightly (run 36266825579, + `guest (12)`): the same. + +It passed on main's nightly at `e8d7c9c0` (run 36228604597, `guest (11)`). +Thirteen landings fall between the two, and none is bisected. One of them, +#513 ("One way to wait", `46056bfa`), rewrote `sys_thread_join`'s wait. + +**Exit**: `fpu_isolation` green, and a test that joins a thread which has +already exited by the time of the join, red on the tree this issue was +filed against. diff --git a/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md b/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md new file mode 100644 index 00000000000..6c9e88011c1 --- /dev/null +++ b/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md @@ -0,0 +1,22 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# MSI and pin routing take an x86 vector and APIC id in generic code + +`kernel/src/drivers/pci.rs`'s `enable_msix` and `enable_msi` take +`vector: u8`, and `kernel/src/iommu/mod.rs`'s `remap_pin` takes +`apic_id: u8` beside it. Both are x86's interrupt addressing: an 8-bit IDT +vector delivered to an xAPIC id. A GICv3 message names an LPI through the ITS +(a 32-bit event id, translated to an INTID above 8191) and a redistributor, +and neither fits a `u8`. + +Owned by stage 4 of `issues/kernel/toyos-runs-on-arm64.md`, which brings up +the GIC and its ITS. + +**Exit condition**: a PCI function's interrupt is programmed from an +arch-provided message (address and data, as `arch::msi` already provides the +doorbell), the routing names its target by the arch's own type, and no +generic signature carries `vector: u8` or `apic_id`. diff --git a/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md b/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md index 8e4cd9c9953..4faff8fa22d 100644 --- a/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md +++ b/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md @@ -18,3 +18,16 @@ The boot's console shows the stick's transport breaking twice on `SCSI 0x2a` console, not the USB stack, quiesce or its config. Exit: a cause, or a recurrence that shows it is not load-bound. + +**Recurrence, PR #524's fast tier at `396f5b4d`, dev host.** Red wide after +325 s with the same `QEMU never reported stopping: the guest asked for a +reboot and stayed up`, then `ALONE ... GREEN` in 4 s. The capture has the same +`quiesce_writers: 4 of 6 writers reached their loop in 5s` and +`test_rs_quiesce_writers exit=1`, and this time no transport break on the +stick. While it ran, the host's 12 guest slots were full, shared with the +suites of five other worktrees (`toyos-lld`, `toyos-tcp`, `toyos-update`, +`toyos-logtrack`, `toyos-guiplat`). +The branch reorders xHCI ring writes (`dma_wmb`) but touches no quiesce code. +A same-session A/B then gave 5 of 5 green on each arm, main at `e48604c0` and +the branch: the red did not come back alone or beside the other arm, so it +is still not shown to be anything but load-bound. diff --git a/issues/kernel/quiesce-stops-the-machine-stayed-up-beside-other-guests.md b/issues/kernel/quiesce-stops-the-machine-stayed-up-beside-other-guests.md index f4b57f44be9..436dffd5fe4 100644 --- a/issues/kernel/quiesce-stops-the-machine-stayed-up-beside-other-guests.md +++ b/issues/kernel/quiesce-stops-the-machine-stayed-up-beside-other-guests.md @@ -20,3 +20,9 @@ Not shown: where the loaded run's stop went, since its capture has no `stop:` record. **Exit**: the stop's record, or its absence, explained on a loaded run. + +Again in the fast tier on PR #524's branch at `235c5a5b`: the same `QEMU +never reported stopping: the guest asked for a reboot and stayed up`, after +266 s, and the harness's re-run alone was green. The host was loaded +throughout by another worktree's spinner at 397% CPU, with the load average +between 20 and 28. diff --git a/issues/kernel/the-aarch64-kernel-builds-with-dead-code-allowed.md b/issues/kernel/the-aarch64-kernel-builds-with-dead-code-allowed.md new file mode 100644 index 00000000000..67ae56c3762 --- /dev/null +++ b/issues/kernel/the-aarch64-kernel-builds-with-dead-code-allowed.md @@ -0,0 +1,21 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# The AArch64 kernel builds with dead code allowed, blanket + +`kernel/.cargo/config.toml` gives `aarch64-unknown-none-softfloat` +`-Adead_code -Aunfulfilled_lint_expectations`: until the syscall gate, the +interrupt controller and the scheduler have a way in, every item only they +reach is unreachable on AArch64. The allowance covers the whole kernel, so a +genuinely dead item written anywhere, for either architecture's shared code, +is invisible on this target; only the x86 build, which keeps +`-Dwarnings` whole, still sees it. + +Owned by stage 7 of `issues/kernel/toyos-runs-on-arm64.md`, when userland +boots and the gate reaches everything. + +**Exit condition**: both `-A` flags are gone from the AArch64 target's +`rustflags` and the AArch64 kernel builds with `-Dwarnings` alone. diff --git a/issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md b/issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md new file mode 100644 index 00000000000..8da9acbaaed --- /dev/null +++ b/issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md @@ -0,0 +1,21 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# The boot-timing handoff is named for the TSC + +`toyos-abi/src/boot.rs`'s `KernelArgs` carries `loader_entry_tsc` and +`loader_handoff_tsc`, and `kernel/src/main.rs`'s `report_power_on` reports +"the TSC went backwards". On AArch64 the loader fills both from `CNTVCT_EL0`, +the generic timer's count, so the ABI and the message name a counter the +machine does not have. + +Owned by stage 4 of `issues/kernel/toyos-runs-on-arm64.md`, which makes the +generic timer the clock. `KernelArgs` is the ABI, so the rename lands with +that stage's other ABI work. + +**Exit condition**: the two fields and the report name the CPU's counter +(`cpu::counter`) rather than the TSC, and both architectures' loaders fill +them. diff --git a/issues/kernel/the-crash-evidence-records-x86-fault-registers.md b/issues/kernel/the-crash-evidence-records-x86-fault-registers.md new file mode 100644 index 00000000000..5492759a8b6 --- /dev/null +++ b/issues/kernel/the-crash-evidence-records-x86-fault-registers.md @@ -0,0 +1,30 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# The crash evidence records x86 fault registers, in slots an AArch64 CPU aliases + +`kernel/src/panic.rs`'s `Evidence` stores `apic`, `rip` and `cr2`, and +`record_fault` takes them by those names: the x86 APIC id, the faulting +instruction and x86's fault address. On AArch64 the same three are the MPIDR +affinity, `ELR_EL1` and `FAR_EL1`, and the report would print them under x86 +labels. + +The slot is `FIRST[cpu::hardware_id() & (SLOTS - 1)]`, `SLOTS = 64`, as is +`PANIC_DEPTH`'s. An APIC id below 64 is one slot per CPU. AArch64's +`hardware_id` is MPIDR's `Aff3:Aff2:Aff1:Aff0`, so a second cluster's CPU 0 +(`Aff1 = 1`, `0x100`) lands on slot 0 beside the first CPU of all. QEMU `virt` +with 17 CPUs (TCG, `-cpu max`, GICv3) names CPU 16 `mpidr=0x100` in its MADT, +so there two CPUs share one record and one panic depth, which the `apic` +field was left unmasked to make visible and not to prevent. + +Owned by stage 5 of `issues/kernel/toyos-runs-on-arm64.md`, the first stage +that starts a second CPU. + +**Exit condition**: the evidence names the fault by role (the CPU, the +faulting PC, the faulting address) and each architecture's report labels +them; the slot is the CPU's dense index, not its hardware id masked, shown +by a test that panics a CPU whose `Aff1` is not zero beside CPU 0 and reads +both records. diff --git a/issues/kernel/the-saved-kernel-context-names-x86-registers.md b/issues/kernel/the-saved-kernel-context-names-x86-registers.md new file mode 100644 index 00000000000..7984c7af405 --- /dev/null +++ b/issues/kernel/the-saved-kernel-context-names-x86-registers.md @@ -0,0 +1,20 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# The saved kernel context names x86 registers in generic code + +`kernel/src/sched/payload.rs`'s `KernelCtx`, which every CPU's switch loads, +carries `rsp` and `fs_base`: the stack pointer and the user thread pointer by +their x86 names, in code outside `arch/`. AArch64 saves `sp` and +`TPIDR_EL0` in the same two roles, so the port either writes AArch64 state +into fields named for x86 or grows a second copy of the struct. + +Owned by stage 4 of `issues/kernel/toyos-runs-on-arm64.md`, which writes the +AArch64 context switch. + +**Exit condition**: the fields are named by role (the saved kernel stack +pointer, the user thread pointer), the x86 and AArch64 switches both load +them, and nothing outside `kernel/src/arch/` names `rsp` or `fs_base`. diff --git a/issues/kernel/toyos-runs-on-arm64.md b/issues/kernel/toyos-runs-on-arm64.md index 33ed6054363..abf5a9b7218 100644 --- a/issues/kernel/toyos-runs-on-arm64.md +++ b/issues/kernel/toyos-runs-on-arm64.md @@ -239,29 +239,34 @@ before any aarch64 file exists, with x86 as its only user: in its own crate, as `toyos-bootmap` does: it grows a TTBR plan, and `toyos-pcid` becomes an ASID/PCID allocator. +## x86 left in generic code + +Each is its own issue, owned by the stage that removes it: + +- `issues/kernel/the-saved-kernel-context-names-x86-registers.md` (stage 4) +- `issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md` (stage 4) +- `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md` (stage 4) +- `issues/kernel/the-crash-evidence-records-x86-fault-registers.md` (stage 5) +- `issues/kernel/the-aarch64-kernel-builds-with-dead-code-allowed.md` (stage 7) + ## Stages Each stage names its exit; "measured" means a number from a run. 0. **Rulings and shared seams, before any aarch64 file exists.** `src/` gains an `Arch`, replacing the 57 triple mentions. `arch/syscall/` minus - `gate.rs` moves out of `arch/`. `toyos-elf` and `toyos-ld` take a machine: - `EM_AARCH64`, `R_AARCH64_RELATIVE`, PE `0xAA64`. The MSI doorbell becomes + `gate.rs` moves out of `arch/`. The MSI doorbell becomes one arch-provided constant. `IrqGuard`/`LogCommitGuard` become one arch primitive. The loom model owed by `issues/kernel/the-stops-no-lost-wake-claim-rests-on-x86-locked-rmws.md` lands, and the `Mmio` barrier contract above is written and asserted. **Exit**: x86 builds and passes unchanged. `rg 'x86_64-unknown' src/` - names one `Arch` table. `toyos-elf` and `toyos-ld` host tests cover an - aarch64 PIE with relocations. + names one `Arch` table. -1. **`aarch64-unknown-toyos` in the rust fork and `toyos-ld`.** A target - spec: `aarch64-unknown-none-elf`, PIC, frame pointers, `toyos-ld` as the - linker. Std pal `_start` and TLS for variant I with TLSDESC. `toyos-ld` - handles the TLSDESC/TLSLE relocation families and `aarch64` stubs in ELF - output. **Exit**: `cargo +toyos build --target aarch64-unknown-toyos` - builds `std` plus a hello-world and the whole Rust userland. `toyos-ld`'s - output passes `toyos-elf` and `llvm-readobj` with no unknown relocations. +1. **`aarch64-unknown-toyos` in the rust fork.** A target spec: + `aarch64-unknown-none-elf`, PIC, frame pointers. Std pal `_start` and TLS + for variant I. **Exit**: `cargo +toyos build --target aarch64-unknown-toyos` + builds `std` plus a hello-world and the whole Rust userland. 2. **The UEFI loader on AArch64** (`aarch64-unknown-uefi`, tier 2 upstream, no fork work needed). Entry is `extern "C"` rather than `sysv64`. @@ -288,12 +293,21 @@ Each stage names its exit; "measured" means a number from a run. **Exit**: a user process takes a page fault and a syscall on one CPU; the timer drives preemption; an interrupt storm test ends with no lost timer tick; the longest interrupts-off window is measured against x86's. + The entry's EL2 writes of `CNTHCTL_EL2`, `CNTVOFF_EL2` and `CPTR_EL2` + (`kernel/src/arch/aarch64/boot.rs`) are untested until here: deleting any + one stays green in `virt_el2_drop`, because stage 3 reads no counter and + runs no FP. This stage's timer and FP tests run under that EL2 profile too, + and each of the three deletions is shown red. 5. **SMP through PSCI.** `CPU_ON` from MADT GICC entries, SGIs as the IPI, broadcast TLBI behind the machine-wide invalidation contract. **Exit**: `-smp 8` boots all CPUs; every CPU asserts the control-register declaration; the shootdown stress test and the loom-checked stop both pass; `CPU_OFF`/`SYSTEM_RESET`/`SYSTEM_OFF` replace ACPI reset and PM1a. + The TLS-descriptor resolver lands here, in std with its loader half and a + `dlopen` test; until it does the kernel refuses `R_AARCH64_TLSDESC` by name + (`toyos_elf::rela::ExeRefusal::TlsDescriptor` for an executable, + `toyos_elf::RelocError::TlsDescriptor` for a library). 6. **Virtio on `virt`.** virtio-pci (ECAM from MCFG) for blk, net, gpu, sound, input and rng. virtio-input replaces the i8042 as the @@ -301,6 +315,11 @@ Each stage names its exit; "measured" means a number from a run. alongside RNDR. SMMUv3 is on `virt` (`-M virt,iommu=smmuv3`), decoded from IORT. **Exit**: netd claims its NIC through an SMMUv3 domain; a foreign-DMA test faults into a `DMA FAULT` record, not a crash. + Stage 0's three DMA-ordering fixes (NVMe's phase before its body, xHCI + `TrbRing::put`'s body before its cycle bit, and the event ring's cycle bit + before its body) get a test here that reds with `put` written back as one + `self.buf.write(off, trb)`; x86's TSO hides all three from every guest + test until then. 7. **Userland boots.** `init`, `logd`, the compositor, netd, soundd and sshd, built for `aarch64-unknown-toyos`. C programs stay x86-only per the diff --git a/kernel-loom/Cargo.toml b/kernel-loom/Cargo.toml index bd27abf6aed..4983c147759 100644 --- a/kernel-loom/Cargo.toml +++ b/kernel-loom/Cargo.toml @@ -178,6 +178,30 @@ device-irq-lossy = [] # the same name only so `cfg` checking knows it. dump-report-relaxed = [] +# The negative control for the panic console's published descriptor. It drops +# the publisher's `Release` fence between its odd mark and its words, so a +# reader can keep words of a publication whose mark it never saw, and +# `panic_console_publish.rs` must red: +# +# cargo test --manifest-path kernel-loom/Cargo.toml --features seqlock-writer-fence-off \ +# --test panic_console_publish +# +# Never on by default and never reachable from a kernel build, which declares +# the same name only so `cfg` checking knows it. +seqlock-writer-fence-off = [] + +# The negative control for the console backend's lock. It builds +# `serial_lock.rs`'s `try_lock` with `then_some`, whose argument is built on a +# lost exchange too and whose drop then releases the holder's lock, and +# `serial_lock.rs` must red: +# +# cargo test --manifest-path kernel-loom/Cargo.toml --features serial-try-lock-then-some \ +# --test serial_lock +# +# Never on by default and never reachable from a kernel build, which declares +# the same name only so `cfg` checking knows it. +serial-try-lock-then-some = [] + [dependencies] loom = "0.7" # The record layout the shard is a ring of. `no_std`, no atomics of its own, diff --git a/kernel-loom/src/lib.rs b/kernel-loom/src/lib.rs index d89c36569bc..36bc127f92f 100644 --- a/kernel-loom/src/lib.rs +++ b/kernel-loom/src/lib.rs @@ -1,7 +1,7 @@ //! Loom harness for the kernel's memory-ordering primitives. //! //! `kernel/src/sync.rs`, `kernel/src/shootdown.rs`, -//! `kernel/src/sched/reap_gate.rs` and `kernel/src/drivers/i8042/tally.rs` are +//! `kernel/src/sched/reap_gate.rs` and `kernel/src/arch/x86_64/i8042/tally.rs` are //! compiled into this crate with `feature = "loom"` on, so their atomics and //! cells resolve to loom's instrumented ones and the models drive the real //! primitives rather than transliterations of them — a transliteration is @@ -89,9 +89,9 @@ pub mod arch { /// The kernel implementation masks IF and TF across reservation and /// publication. Loom has no per-CPU flags; the model's sole-writer /// precondition is the corresponding witness here. - pub struct LogCommitGuard; + pub struct IrqGuard; - impl LogCommitGuard { + impl IrqGuard { pub fn close() -> Self { Self } @@ -119,7 +119,7 @@ pub mod arch { #[cfg(feature = "loom")] pub unsafe fn percpu_fetch_add( counter: &loom::sync::atomic::AtomicU64, - _guard: &LogCommitGuard, + _guard: &IrqGuard, ) -> u64 { counter.fetch_add(1, loom::sync::atomic::Ordering::Relaxed) } @@ -128,7 +128,7 @@ pub mod arch { #[cfg(not(feature = "loom"))] pub unsafe fn percpu_fetch_add( counter: &core::sync::atomic::AtomicU64, - _guard: &LogCommitGuard, + _guard: &IrqGuard, ) -> u64 { counter.fetch_add(1, core::sync::atomic::Ordering::Relaxed) } @@ -350,7 +350,7 @@ pub mod sleeplock; /// subject is a *driver*, and it is here for the reason the others are: the /// property is "no reader ever sees this pair disagree", which is a claim about /// instants that no guest test can express and that x86's TSO hides. -#[path = "../../kernel/src/drivers/i8042/tally.rs"] +#[path = "../../kernel/src/arch/x86_64/i8042/tally.rs"] pub mod i8042_tally; /// The panic snapshot's owner and access state, driven together by @@ -360,3 +360,12 @@ pub mod capture_latch; #[path = "../../kernel/src/drivers/panic_console/access.rs"] pub mod capture_access; + +/// The panic console's published framebuffer descriptor, driven by +/// `tests/panic_console_publish.rs`. +#[path = "../../kernel/src/drivers/panic_console/published.rs"] +pub mod panic_console_published; + +/// The console backend's lock, driven by `tests/serial_lock.rs`. +#[path = "../../kernel/src/drivers/serial_lock.rs"] +pub mod serial_lock; diff --git a/kernel-loom/tests/log_body_words.rs b/kernel-loom/tests/log_body_words.rs index 3ed72828722..d49d78f97eb 100644 --- a/kernel-loom/tests/log_body_words.rs +++ b/kernel-loom/tests/log_body_words.rs @@ -41,7 +41,7 @@ fn record(seq: u64, len: usize) -> LogRecord { fn round_trip(len: usize) { let shard = Shard::new(); - let guard = kernel_loom::arch::LogCommitGuard::close(); + let guard = kernel_loom::arch::IrqGuard::close(); // SAFETY: this host thread is the shard's only producer, and each sequence // number is committed once. let seq = unsafe { shard.reserve(&guard) }; @@ -86,7 +86,7 @@ fn every_field_survives_the_word_packing() { #[test] fn a_shorter_record_does_not_inherit_the_longer_one_it_replaced() { let shard = Shard::new(); - let guard = kernel_loom::arch::LogCommitGuard::close(); + let guard = kernel_loom::arch::IrqGuard::close(); // SAFETY: sole producer, each number committed once. let first = unsafe { shard.reserve(&guard) }; unsafe { shard.commit(first, &record(first, 64), &guard) }; diff --git a/kernel-loom/tests/log_record.rs b/kernel-loom/tests/log_record.rs index e24127e3415..80b8712e671 100644 --- a/kernel-loom/tests/log_record.rs +++ b/kernel-loom/tests/log_record.rs @@ -47,7 +47,7 @@ use toyos_abi::log::{LogRecord, MAX_RECORD_MESSAGE}; /// The model's witness for the production IF/TF-off bracket. fn commit_one(shard: &Shard) -> u64 { - let guard = kernel_loom::arch::LogCommitGuard::close(); + let guard = kernel_loom::arch::IrqGuard::close(); // SAFETY: every caller is this model's sole producer for the shard. The // real witness additionally prevents that producer being preempted between // these two operations. @@ -106,7 +106,7 @@ fn a_committed_record_is_whole_or_absent() { let w = loom::thread::spawn(move || { // SAFETY: this thread is the shard's sole writer, which is the // precondition the shim's own doc states. - let guard = kernel_loom::arch::LogCommitGuard::close(); + let guard = kernel_loom::arch::IrqGuard::close(); let seq = unsafe { writer.reserve(&guard) }; let r = record(seq); unsafe { writer.commit(seq, &r, &guard) }; @@ -188,7 +188,7 @@ fn a_recycled_slot_does_not_answer_for_the_record_it_replaced() { // from the moment `head` moves, before a byte of its body is written — // which is the guarantee the reader's lower bound rests on. // SAFETY: sole writer. - let guard = kernel_loom::arch::LogCommitGuard::close(); + let guard = kernel_loom::arch::IrqGuard::close(); let recycling = unsafe { shard.reserve(&guard) }; assert_eq!(recycling % SHARD_RECORDS as u64, FIRST_SEQ % SHARD_RECORDS as u64); assert!( @@ -250,7 +250,7 @@ fn a_reader_racing_a_recycle_gets_nothing_rather_than_a_mixture() { let writer = shard.clone(); let w = loom::thread::spawn(move || { // SAFETY: this thread is the shard's sole writer. - let guard = kernel_loom::arch::LogCommitGuard::close(); + let guard = kernel_loom::arch::IrqGuard::close(); let seq = unsafe { writer.reserve(&guard) }; let r = record(seq); unsafe { writer.commit(seq, &r, &guard) }; @@ -344,7 +344,7 @@ fn a_key_and_the_record_it_names_come_from_one_generation() { let writer = shard.clone(); let w = loom::thread::spawn(move || { // SAFETY: this thread is the shard's sole writer. - let guard = kernel_loom::arch::LogCommitGuard::close(); + let guard = kernel_loom::arch::IrqGuard::close(); let seq = unsafe { writer.reserve(&guard) }; let r = record(seq); unsafe { writer.commit(seq, &r, &guard) }; diff --git a/kernel-loom/tests/log_wake.rs b/kernel-loom/tests/log_wake.rs index 58927f6bb3c..4cc42094f1b 100644 --- a/kernel-loom/tests/log_wake.rs +++ b/kernel-loom/tests/log_wake.rs @@ -35,7 +35,7 @@ #![cfg(feature = "loom")] -use kernel_loom::arch::LogCommitGuard; +use kernel_loom::arch::IrqGuard; use kernel_loom::log_shard::{arm_waiter, signal_after_commit, waiter, Shard, FIRST_SEQ}; use loom::sync::atomic::{AtomicBool, Ordering}; use loom::sync::Arc; @@ -72,13 +72,13 @@ fn a_commit_and_an_arm_cannot_both_miss() { let producer = m.clone(); let p = loom::thread::spawn(move || { - let guard = LogCommitGuard::close(); + let guard = IrqGuard::close(); // SAFETY: one producer, and the guard is the model's stand-in for // the kernel's sole-writer bracket (`percpu_fetch_add`'s shim // argument in `lib.rs` is the other half). let seq = unsafe { producer.shard.reserve(&guard) }; unsafe { producer.shard.commit(seq, &record(seq), &guard) }; - // The kernel's `LogCommitGuard` has a `Drop` that reopens interrupts + // The kernel's `IrqGuard` has a `Drop` that reopens interrupts // here; the model's stand-in has nothing to restore, so this reads // as a no-op drop. It stays because the bracket closing before the // wake signal is the edge this model is about. diff --git a/kernel-loom/tests/log_zeroed_init.rs b/kernel-loom/tests/log_zeroed_init.rs index edcaeb0e284..be7d25c56f5 100644 --- a/kernel-loom/tests/log_zeroed_init.rs +++ b/kernel-loom/tests/log_zeroed_init.rs @@ -43,7 +43,7 @@ fn a_zero_allocated_ap_shard_issues_first_seq_first() { "zero must remain the empty-slot state" ); - let guard = kernel_loom::arch::LogCommitGuard::close(); + let guard = kernel_loom::arch::IrqGuard::close(); // SAFETY: this host thread is the allocation's only producer. let first = unsafe { shard.reserve(&guard) }; assert_eq!( diff --git a/kernel-loom/tests/panic_console_publish.rs b/kernel-loom/tests/panic_console_publish.rs new file mode 100644 index 00000000000..2544e850c39 --- /dev/null +++ b/kernel-loom/tests/panic_console_publish.rs @@ -0,0 +1,35 @@ +//! Loom: a panic console reader keeps one publication whole or none. +//! +//! The descriptor is four words a painter draws through; half of one +//! publication and half of the next is a pointer with another mode's stride. +//! One publisher replaces it (the boot, a mode set's window) while a painter +//! on another CPU snapshots it with no lock. The negative control is the +//! `seqlock-writer-fence-off` feature, which must red this file. +#![cfg(feature = "loom")] + +use kernel_loom::panic_console_published::{Published, WORDS}; +use loom::sync::Arc; +use loom::thread; + +#[test] +fn a_snapshot_is_one_publication_whole() { + loom::model(|| { + let fb = Arc::new(Published::new()); + fb.publish([1; WORDS]); + + let publisher = { + let fb = fb.clone(); + thread::spawn(move || fb.publish([2; WORDS])) + }; + if let Some(words) = fb.snapshot() { + assert!( + words == [1; WORDS] || words == [2; WORDS], + "a snapshot kept two publications' words: {words:?}" + ); + } + publisher.join().unwrap(); + + // Quiescent: the second publication, whole. + assert_eq!(fb.snapshot(), Some([2; WORDS])); + }); +} diff --git a/kernel-loom/tests/serial_lock.rs b/kernel-loom/tests/serial_lock.rs new file mode 100644 index 00000000000..58613469a37 --- /dev/null +++ b/kernel-loom/tests/serial_lock.rs @@ -0,0 +1,51 @@ +//! Loom: a lost `try_lock` on the console backend leaves its holder holding. +//! +//! The console drain takes the backend with `try_lock` from every CPU that +//! logs, so a losing attempt is the common case beside another CPU's write. +//! A loss that released the lock let a third writer in under the holder. The +//! negative control is the `serial-try-lock-then-some` feature, which must red +//! this file. +#![cfg(feature = "loom")] + +use kernel_loom::serial_lock::BackendLock; +use loom::cell::UnsafeCell; +use loom::sync::Arc; +use loom::thread; + +/// One CPU: a loss while it holds the lock is a loss again on the next try. +#[test] +fn a_lost_try_lock_leaves_the_lock_held() { + loom::model(|| { + let lock = BackendLock::new(); + let held = lock.try_lock().expect("an unheld lock refused its first taker"); + assert!(lock.try_lock().is_none(), "a held lock was taken twice"); + assert!(lock.try_lock().is_none(), "a lost try_lock released the lock its holder still has"); + drop(held); + assert!(lock.try_lock().is_some(), "a released lock stayed held"); + }); +} + +/// Two CPUs, each trying twice: whoever gets in writes alone. +#[test] +fn two_writers_never_overlap() { + loom::model(|| { + let lock = Arc::new(BackendLock::new()); + let line = Arc::new(UnsafeCell::new(0u32)); + let writer = |lock: Arc, line: Arc>| { + move || { + for _ in 0..2 { + if let Some(held) = lock.try_lock() { + // SAFETY: the backend is held, so no other writer is in here. + line.with_mut(|n| unsafe { *n += 1 }); + drop(held); + return; + } + thread::yield_now(); + } + } + }; + let other = thread::spawn(writer(lock.clone(), line.clone())); + writer(lock, line)(); + other.join().unwrap(); + }); +} diff --git a/kernel/.cargo/config.toml b/kernel/.cargo/config.toml index 579ded638a7..8817cd2f548 100644 --- a/kernel/.cargo/config.toml +++ b/kernel/.cargo/config.toml @@ -3,4 +3,19 @@ target = "x86_64-unknown-none" rustflags = ["-Dwarnings", "-Cforce-frame-pointers=yes"] [target.x86_64-unknown-none] -linker = "toyos-ld" \ No newline at end of file +linker = "toyos-ld" + +# A target's own `rustflags` replace `build.rustflags`, so both flags repeat. +# `-Adead_code` is the AArch64 port's declared debt: until its stage 7 gives +# the syscall gate, the interrupt controller and the scheduler a way in, every +# item only those reach is unreachable there and nowhere else — and an +# `#[expect(dead_code)]` cannot be met under it, hence the second `-A`. Exit: +# both go when `issues/kernel/toyos-runs-on-arm64.md`'s stage 7 lands. The +# loader takes a static PIE at base 0 (`ET_DYN`, `R_AARCH64_RELATIVE` only), +# which rust-lld makes of the target's static-model code with `-pie`: `-z +# notext` lets absolute words stay in read-only sections, which the loader +# relocates before anything runs. Position-independent codegen is not an +# option: it reaches symbols through GOT slots holding virtual addresses, +# which the entry cannot use while the MMU is off. +[target.aarch64-unknown-none-softfloat] +rustflags = ["-Dwarnings", "-Cforce-frame-pointers=yes", "-Adead_code", "-Aunfulfilled_lint_expectations", "-Clink-arg=-pie", "-Clink-arg=-znotext"] diff --git a/kernel/CLAUDE.md b/kernel/CLAUDE.md index 0a247839023..6fffea753de 100644 --- a/kernel/CLAUDE.md +++ b/kernel/CLAUDE.md @@ -1,6 +1,6 @@ # Kernel -The module header at the site owns its subsystem — read it before changing a module. The scheduler core is `toyos-sched/`, driven from `kernel/src/sched/`; every Ring 3 transition's machine state is `kernel/src/arch/fpu.rs`; every syscall is `kernel/src/arch/syscall/`, and `dispatch.rs` decodes every user pointer the ABI takes. +The module header at the site owns its subsystem — read it before changing a module. The scheduler core is `toyos-sched/`, driven from `kernel/src/sched/`; every Ring 3 transition's machine state is `kernel/src/arch/x86_64/fpu.rs`; every syscall is `kernel/src/syscall/`, and `dispatch.rs` decodes every user pointer the ABI takes. ## Caveats that bite every agent @@ -12,7 +12,7 @@ The module header at the site owns its subsystem — read it before changing a m - **No disk wait in this kernel can park** — at the moment a transfer is waited for, the CPU is four ticket spinlocks deep, each disabling preemption. - **`crate::log!` may not be called inside `with_cpu`'s exclusive region unless a `panic!` follows it** — the log's readiness path re-enters `driver::pass` and wedges the machine. - **`drain_irqs` is the drivers' engine and nothing on it may wait** — a blocking call there empties the audio pipeline on every plug. -- **A syscall that can block resolves its handle and clones the object out before it blocks** — a `with_object`/`with_process_data` guard held across a park is a runtime panic no compile check catches; the `SYS_FSYNC` arm in `arch/syscall/dispatch.rs` is the pattern. +- **A syscall that can block resolves its handle and clones the object out before it blocks** — a `with_object`/`with_process_data` guard held across a park is a runtime panic no compile check catches; the `SYS_FSYNC` arm in `syscall/dispatch.rs` is the pattern. - **A block-layer `BudgetExpired` is not-durable-yet and never a loss** — it is retried on a fresh budget above every lock; a flush that discards its pages on one splits a FAT mirror. - **A decision the process table makes lives in `toyos-proclife`, never in `process.rs`** — its defects are interleavings and that crate is the only machine that can enumerate one. - **A task holds at most one watch registration** — a standing registration across a loop must not call anything that registers again. A double registration panics only at attempt ≥ 2, so the contention depth is the coverage. diff --git a/kernel/Cargo.lock b/kernel/Cargo.lock index edf3a3f0abd..68940e86c2f 100644 --- a/kernel/Cargo.lock +++ b/kernel/Cargo.lock @@ -41,6 +41,7 @@ dependencies = [ "toyos-acpi", "toyos-blackbox", "toyos-blockhold", + "toyos-bootmap", "toyos-dma", "toyos-elf", "toyos-fat32", @@ -92,6 +93,10 @@ version = "0.1.0" name = "toyos-blockhold" version = "0.1.0" +[[package]] +name = "toyos-bootmap" +version = "0.1.0" + [[package]] name = "toyos-dma" version = "0.1.0" diff --git a/kernel/Cargo.toml b/kernel/Cargo.toml index 08036c8c6d6..58c72cb9182 100644 --- a/kernel/Cargo.toml +++ b/kernel/Cargo.toml @@ -19,6 +19,15 @@ loom = [] # with either fence removed behaves identically, which is the whole reason the # obligation is a model. wake-fence-off = [] +# The same for the panic console's published descriptor: `kernel-loom` turns it on +# to drop the publisher's `Release` fence from `drivers/panic_console/published.rs`, +# and `kernel-loom/tests/panic_console_publish.rs` must red — a reader keeping a +# descriptor that is half one publication and half the next. +seqlock-writer-fence-off = [] +# The same for the console backend's lock: `kernel-loom` turns it on to build +# `drivers/serial_lock.rs`'s `try_lock` with `then_some`, whose lost exchange +# releases the holder's lock, and `kernel-loom/tests/serial_lock.rs` must red. +serial-try-lock-then-some = [] # The same arrangement for a poll ring's one-shot answer: `kernel-loom` turns it # on to split `inbox/once.rs`'s exchange into a load and a store, and # `kernel-loom/tests/poll_once.rs` must red when it does — a poll two posts @@ -95,7 +104,7 @@ boot-actuators = [] # One name rather than seven, because a feature set is a whole kernel build and # builds that differ only in which unreachable arm they carry are several builds # of one kernel. The per-action justification — what nothing else can reach — now -# lives beside each arm in `arch/syscall/dispatch.rs`, which is the file that +# lives beside each arm in `syscall/dispatch.rs`, which is the file that # has to stay true when an arm changes. # # The one place it is not confined to the match: `arch::tlb::serve_ipi` loads one @@ -268,7 +277,7 @@ stack-witness = [] # lands eight bytes below the frame, so the frame has to be inside a real stack # before the call happens. `stack-witness`'s third test is what establishes that. switch-witness = ["stack-witness"] -# The negative controls — `kernel/src/hw.rs`'s `switch_witness_mutate` carries +# The negative controls — `kernel/src/arch/x86_64/hw.rs`'s `switch_witness_mutate` carries # what each stages. One write per boot, at the ninth switch (`MUTATE_AT`, and # small because it was measured): `frame` writes the `rbx` slot of a frame the # check has just validated, and is the control for the instrument — a witness that @@ -386,6 +395,7 @@ toyos-acpi = { path = "../toyos-acpi" } toyos-dma = { path = "../toyos-dma" } toyos-blackbox = { path = "../toyos-blackbox" } toyos-blockhold = { path = "../toyos-blockhold" } +toyos-bootmap = { path = "../toyos-bootmap" } toyos-fat32 = { path = "../toyos-fat32" } toyos-elf = { path = "../toyos-elf" } toyos-gpt = { path = "../toyos-gpt" } diff --git a/kernel/rust-toolchain.toml b/kernel/rust-toolchain.toml index 76653840601..e3ab4867b0a 100644 --- a/kernel/rust-toolchain.toml +++ b/kernel/rust-toolchain.toml @@ -1,2 +1,2 @@ [toolchain] -targets = ["x86_64-unknown-none"] \ No newline at end of file +targets = ["x86_64-unknown-none", "aarch64-unknown-none-softfloat"] diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 36ca7e2d453..768fac482b1 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -46,6 +46,12 @@ actuators! { /// Panic between arming the on-screen console and `mm::init`. test_early_panic = "test-early-panic"; + /// Take an undefined-instruction exception right after the architecture's + /// console step, where it has one to take there: the earliest fault the + /// exception vectors must report. AArch64 installs its vectors in the entry; + /// x86-64 loads its IDT later, and panics by name instead. + test_early_fault = "test-early-fault"; + /// Panic inside `percpu::init_bsp`, one statement after it loads the IDT: /// the earliest point a panic is reportable at all, and the window the T14 /// stops in. What it judges is that the reset register is decoded by then. diff --git a/kernel/src/arch/aarch64/barrier.rs b/kernel/src/arch/aarch64/barrier.rs new file mode 100644 index 00000000000..91612c0d8f4 --- /dev/null +++ b/kernel/src/arch/aarch64/barrier.rs @@ -0,0 +1,49 @@ +//! The orderings a device's view of memory needs, as AArch64 gives them. +//! +//! The same contract x86-64's `barrier` implements, and here every function is +//! an instruction: the Arm memory model is weakly ordered, and a device sits +//! outside the inner-shareable domain `fence` orders (`DMB ISH`). The outer +//! shareable `DMB OSH*` forms are what a DMA master observes (Arm ARM K.a, +//! B2.3.10 and D8.2.2); `DSB` is needed only where completion, not order, is +//! the point. + +/// Every store to memory this CPU made before this call is visible to a +/// device's DMA reads before any store it makes after it — a descriptor before +/// the index that publishes it. `DMB OSHST`. +#[inline(always)] +pub fn dma_wmb() { + // SAFETY: a barrier; reads and writes nothing. + unsafe { core::arch::asm!("dmb oshst", options(nostack, preserves_flags)) }; +} + +/// Every load of device-written memory this CPU makes after this call sees at +/// least what the loads before it saw — a completion's index before the entry +/// it counts. `DMB OSHLD`. +#[inline(always)] +pub fn dma_rmb() { + // SAFETY: a barrier; reads and writes nothing. + unsafe { core::arch::asm!("dmb oshld", options(nostack, preserves_flags)) }; +} + +/// What an MMIO store is preceded by: [`dma_wmb`], so a register write that +/// starts a device's work comes after the memory that work reads. +#[inline(always)] +pub fn before_mmio_write() { + dma_wmb(); +} + +/// What an MMIO load is followed by: [`dma_rmb`], so memory read after a +/// status register is at least as new as the status. +#[inline(always)] +pub fn after_mmio_read() { + dma_rmb(); +} + +/// Every store this CPU made to the scanout has completed before this returns: +/// the scanout is Normal non-cacheable, whose stores may be gathered and held, +/// and `DSB ST` waits for every earlier store to complete. +#[inline(always)] +pub fn scanout_flush() { + // SAFETY: a barrier; reads and writes nothing. + unsafe { core::arch::asm!("dsb st", options(nostack, preserves_flags)) }; +} diff --git a/kernel/src/arch/aarch64/boot.rs b/kernel/src/arch/aarch64/boot.rs new file mode 100644 index 00000000000..37fb11c9eab --- /dev/null +++ b/kernel/src/arch/aarch64/boot.rs @@ -0,0 +1,274 @@ +//! The AArch64 steps of the boot: the entry the loader jumps to, and what +//! `kernel_main` asks of this architecture at the points where one differs +//! from another. +//! +//! **The entry.** The loader leaves the CPU as firmware ran it — at EL2 or +//! EL1, on firmware's identity tables — cleans the kernel image and its own +//! tables to the point of coherency, and jumps to [`_start`]'s physical +//! address with `x0 = &KernelArgs`. `_start` writes the +//! [`control_regs`](super::control_regs) declaration whole: at EL2 it writes +//! `HCR_EL2` first, halts in a named refusal unless it reads back as declared, +//! then programs EL1's registers with the MMU already on and drops with `ERET`; at EL1 it +//! turns the MMU off first, so no translation register changes under a live +//! walk. Either way it arrives at the kernel's link address in the view at +//! `PHYS_OFFSET`, on the kernel's own stack, with the vectors installed, and +//! calls `kernel_main`. + +use core::mem::offset_of; + +use toyos_abi::boot::{KernelArgs, MemoryMapEntry}; +use toyos_acpi::MadtEntry; + +use super::control_regs as regs; +use crate::drivers::acpi::DirectPhys; +use crate::log; +use crate::mm::Region; + +/// Entry point: the loader jumps here at its physical address, MMU on under +/// firmware's identity map, with `x0 = &KernelArgs`. +/// # Safety +/// Only the loader may call this, fresh from firmware, with `x0` holding a live [`KernelArgs`]. +#[unsafe(naked)] +#[no_mangle] +pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { + core::arch::naked_asm!( + "mov x19, x0", + // x20 = the stack's top, physical: kernel image + stack offset + stack size. + "ldr x20, [x19, #{kernel_memory}]", + "ldr x1, [x19, #{stack_offset}]", + "add x20, x20, x1", + "ldr x1, [x19, #{stack_size}]", + "add x20, x20, x1", + // x2 = TCR_EL1 whole, x3 = the loader's L0 table, x4 = MAIR_EL1. + "ldr x2, ={tcr}", + "mrs x1, id_aa64mmfr0_el1", + "and x1, x1, #0xf", + "orr x2, x2, x1, lsl #{ips}", + "ldr x3, [x19, #{root}]", + "ldr x4, ={mair}", + "ldr x5, ={sctlr}", + "mrs x21, CurrentEL", + "lsr x21, x21, #2", + "cmp x21, #2", + "b.eq 2f", + "cmp x21, #1", + "b.ne 9f", + // EL1: translation off, then the declaration, then translation on. + "ldr x1, ={sctlr_off}", + "msr sctlr_el1, x1", + "isb", + "msr mair_el1, x4", + "msr tcr_el1, x2", + "msr ttbr0_el1, x3", + "msr ttbr1_el1, x3", + "msr cpacr_el1, xzr", + "tlbi vmalle1", + "dsb nsh", + "isb", + "msr sctlr_el1, x5", + "isb", + "ldr x1, =3f", + "br x1", + // EL2: `HCR_EL2` first, since with `E2H` set every `_el1` name + // below is an EL2 register; refused unless it reads back as declared. + // Then EL1's registers with the MMU on, EL2's others, and the drop. + "2:", + "ldr x1, ={hcr}", + "msr hcr_el2, x1", + "isb", + "mrs x6, hcr_el2", + "cmp x6, x1", + "b.ne {refuse_hcr}", + "msr mair_el1, x4", + "msr tcr_el1, x2", + "msr ttbr0_el1, x3", + "msr ttbr1_el1, x3", + "msr cpacr_el1, xzr", + "msr sctlr_el1, x5", + "mov x1, #{cnthctl}", + "msr cnthctl_el2, x1", + "msr cntvoff_el2, xzr", + "ldr x1, ={cptr}", + "msr cptr_el2, x1", + "tlbi vmalle1", + "dsb nsh", + "mov x1, #{spsr}", + "msr spsr_el2, x1", + "ldr x1, =3f", + "msr elr_el2, x1", + "isb", + "eret", + // At the link address, EL1, MMU on. + "3:", + "ldr x1, ={phys_offset}", + "add x20, x20, x1", + "mov sp, x20", + "add x19, x19, x1", + "adrp x1, {entry_el}", + "str x21, [x1, :lo12:{entry_el}]", + "bl {install}", + "mov x0, x19", + "mov x29, xzr", + "mov x30, xzr", + "bl {kernel_main}", + // EL3, or anything else: nothing here may run there. + "9:", + "wfe", + "b 9b", + kernel_memory = const offset_of!(KernelArgs, kernel_memory_addr), + stack_offset = const offset_of!(KernelArgs, kernel_stack_addr), + stack_size = const offset_of!(KernelArgs, kernel_stack_size), + root = const offset_of!(KernelArgs, boot_pml4_addr), + tcr = const regs::TCR, + ips = const regs::TCR_IPS_SHIFT, + mair = const regs::MAIR, + sctlr = const regs::SCTLR, + sctlr_off = const regs::SCTLR_MMU_OFF, + hcr = const regs::HCR_EL2, + refuse_hcr = sym refused_hcr_el2_readback, + cnthctl = const regs::CNTHCTL_EL2, + cptr = const regs::CPTR_EL2, + spsr = const regs::SPSR_EL2_TO_EL1, + phys_offset = const crate::PHYS_OFFSET, + entry_el = sym regs::ENTRY_EL, + install = sym super::trap::install, + kernel_main = sym crate::kernel_main, + ); +} + +/// Where a CPU entered at EL2 halts when `HCR_EL2` reads back anything but the +/// declaration the entry wrote: `E2H` held set, which the loader refuses by +/// name first (`bootloader/src/arch/aarch64.rs`), or any other bit. There the +/// registers the drop programs are not the ones the declaration names, so the +/// kernel does not run. Nothing can report yet: the console is found +/// after the drop, so the refusal is this symbol, which the halted PC names. +#[unsafe(naked)] +#[no_mangle] +unsafe extern "C" fn refused_hcr_el2_readback() -> ! { + core::arch::naked_asm!("1:", "wfe", "b 1b") +} + +/// The ACPI tables this architecture decodes: the MADT for its GIC and CPUs, +/// the FADT for PSCI and reset, the GTDT for the timer, the SPCR for the +/// console, and the MCFG for ECAM. +pub const ACPI_TABLES: &[&[u8; 4]] = &[b"APIC", b"FACP", b"GTDT", b"SPCR", b"MCFG"]; + +/// Before the panel is armed: nothing. The loader mapped the scanout with its +/// final memory type, Normal non-cacheable, and `MAIR_EL1` names that type +/// since the entry. +pub fn before_panel() {} + +/// Once the console and the boot parameter exist: the declaration read back, +/// and what this boot found — the memory map and the tables the rest of the +/// port reads. +pub fn after_console(args: &KernelArgs, maps: &[MemoryMapEntry]) { + regs::check(); + for entry in maps { + log!("memory: {:#014x}..{:#014x} uefi type {}", entry.start, entry.end, entry.uefi_type); + } + log!("memory: {} ranges, as the loader handed them over", maps.len()); + survey(args.rsdp_addr); + + // After the vectors and before anything else can fault: the earliest + // exception they must report. + if crate::actuator::test_early_fault() { + super::cpu::undefined_instruction(); + } +} + +/// The MADT's GIC structures and the GTDT's timers, decoded and said: what +/// the interrupt controller and the timer of stage 4 are built from. +fn survey(rsdp_addr: u64) { + match toyos_acpi::find_table(DirectPhys, rsdp_addr, b"APIC", toyos_acpi::MADT_ENTRIES) { + Ok(madt) => { + let (mut cpus, mut enabled) = (0u32, 0u32); + for item in toyos_acpi::madt_entries(&madt) { + match item { + Ok(MadtEntry::Gicc(gicc)) => { + cpus += 1; + enabled += u32::from(gicc.enabled); + log!( + "ACPI: MADT GICC uid={} mpidr={:#x} enabled={} gicr={:#x}", + gicc.uid, + gicc.mpidr, + gicc.enabled, + gicc.gicr_base + ); + } + Ok(MadtEntry::Gicd { base, version }) => { + log!("ACPI: MADT GICD at {base:#x}, GIC version {version}") + } + Ok(MadtEntry::Gicr { base, length }) => { + log!("ACPI: MADT GICR range {base:#x}+{length:#x}") + } + Ok(MadtEntry::Its { id, base }) => log!("ACPI: MADT ITS {id} at {base:#x}"), + Ok(MadtEntry::LocalApic { .. } + | MadtEntry::IoApic(_) + | MadtEntry::SourceOverride(_) + | MadtEntry::Other(_)) => {} + Err(halt) => { + log!( + "ACPI: MADT entry at +{} declares {} bytes of a {}-byte list — stopping", + halt.at, + halt.declared, + halt.list_len + ); + break; + } + } + } + log!("ACPI: MADT names {cpus} GIC CPU interfaces, {enabled} enabled"); + } + Err(e) => log!("ACPI: MADT unusable: {e:?}"), + } + match toyos_acpi::gtdt(DirectPhys, rsdp_addr) { + Ok(gtdt) => log!( + "ACPI: GTDT timers: EL1 physical GSIV {}, EL1 virtual GSIV {}, EL2 GSIV {} ({})", + gtdt.non_secure_el1.gsiv, + gtdt.virtual_el1.gsiv, + gtdt.el2.gsiv, + if gtdt.virtual_el1.edge() { "edge" } else { "level" }, + ), + Err(e) => log!("ACPI: GTDT unusable: {e:?}"), + } +} + +/// Physical memory only this architecture's boot uses: none. +pub fn reserved() -> Region { + Region { start: 0, end: 0 } +} + +/// What the boot learns bringing interrupts up and hands later steps. +pub struct Platform { + never: core::convert::Infallible, +} + +/// Interrupt delivery, this CPU's per-CPU block and the syscall gate. +pub fn interrupts(_rsdp_addr: u64) -> Platform { + owed!("interrupt delivery", "stage 4") +} + +/// The clock: the generic timer's counter at `CNTFRQ_EL0`. +pub fn clock(_args: &KernelArgs) { + owed!("the clock", "stage 4") +} + +/// The per-CPU timer. +pub fn timer() { + owed!("the timer", "stage 4") +} + +/// The platform's own devices that are not PCI functions: none this kernel +/// drives on an ACPI Arm machine. +pub fn platform_devices(_rsdp_addr: u64) {} + +/// Every other CPU, running. +pub fn start_other_cpus(platform: &Platform, _args: &KernelArgs) { + match platform.never {} +} + +/// The interrupt-controller selftests an actuator asks for. +#[cfg(feature = "boot-actuators")] +pub fn interrupt_selftests() { + owed!("the interrupt controller", "stage 4") +} diff --git a/kernel/src/arch/aarch64/cache.rs b/kernel/src/arch/aarch64/cache.rs new file mode 100644 index 00000000000..0b0e9df4d6e --- /dev/null +++ b/kernel/src/arch/aarch64/cache.rs @@ -0,0 +1,30 @@ +//! Writing memory back out of the caches, for the one reader that is not a +//! CPU: DRAM across a reset. + +/// The smallest data cache line on this machine, from `CTR_EL0.DminLine` +/// (log2 of the word count), so a walk by it misses no line. +fn line() -> u64 { + let ctr: u64; + // SAFETY: reads `CTR_EL0`, which EL1 may always read. + unsafe { core::arch::asm!("mrs {}, ctr_el0", out(reg) ctr, options(nomem, nostack, preserves_flags)) }; + 4 << ((ctr >> 16) & 0xF) +} + +/// Every line of `[at, at + len)` cleaned to the point of coherency and +/// invalidated before this returns: `DC CIVAC` by line, then `DSB SY` for +/// their completion. +/// +/// A reset invalidates the caches without writing them back, so a byte that +/// must outlive one has to reach DRAM first. +pub fn write_back(at: u64, len: usize) { + let step = line(); + let mut addr = at & !(step - 1); + while addr < at + len as u64 { + // SAFETY: `DC CIVAC` cleans and invalidates the line holding a mapped + // address the caller owns; it changes no memory's contents. + unsafe { core::arch::asm!("dc civac, {}", in(reg) addr, options(nostack, preserves_flags)) }; + addr += step; + } + // SAFETY: a barrier; waits for the maintenance above to complete. + unsafe { core::arch::asm!("dsb sy", options(nostack, preserves_flags)) }; +} diff --git a/kernel/src/arch/aarch64/console_uart.rs b/kernel/src/arch/aarch64/console_uart.rs new file mode 100644 index 00000000000..1241613b2cc --- /dev/null +++ b/kernel/src/arch/aarch64/console_uart.rs @@ -0,0 +1,77 @@ +//! The console UART: whichever PL011-shaped UART the SPCR places firmware's +//! console on. Firmware configured it (rate, framing, enable), and SPCR +//! describes that configuration, so this file programs nothing and only moves +//! bytes: a PL011's `UARTDR` and `UARTFR`, which the SBSA generic UART keeps +//! at the same offsets (Arm PL011 TRM r1p5, §3.3; Arm BSA 1.0, Appendix B). + +use core::sync::atomic::{AtomicU64, Ordering}; + +use toyos_acpi::{SerialInterface, GAS_SYSTEM_MEMORY}; + +use crate::drivers::acpi::DirectPhys; +use crate::log; +use crate::mm::{DirectMap, Mmio}; + +/// `UARTDR`, the data register, and `UARTFR`, the flag register. +const DR: u64 = 0x000; +const FR: u64 = 0x018; +/// `UARTFR.RXFE`: the receive FIFO is empty. `UARTFR.TXFF`: the transmit FIFO is full. +const FR_RXFE: u32 = 1 << 4; +const FR_TXFF: u32 = 1 << 5; +/// One 4 KiB register frame, which is what both UARTs decode. +const FRAME: u64 = 0x1000; + +/// The register frame's physical address; zero until [`init`] found one. +static BASE: AtomicU64 = AtomicU64::new(0); + +fn regs() -> Mmio { + let base = BASE.load(Ordering::Relaxed); + assert!(base != 0, "console UART: a byte moved before `init` found the UART"); + Mmio::new(DirectMap::from_phys(base), FRAME) +} + +/// Find the UART SPCR names and answer whether it is one this file drives. +pub fn init(rsdp_addr: u64) -> bool { + let spcr = match toyos_acpi::spcr(DirectPhys, rsdp_addr) { + Ok(spcr) => spcr, + Err(e) => { + log!("serial: no console UART, because the SPCR is unusable: {e:?}"); + return false; + } + }; + let kind = match spcr.interface { + SerialInterface::Pl011 => "PL011", + SerialInterface::SbsaGeneric | SerialInterface::SbsaGeneric32 => "SBSA generic UART", + SerialInterface::Ns16550 | SerialInterface::Other(_) => { + log!("serial: SPCR names a {:?} UART, which this kernel does not drive on AArch64", spcr.interface); + return false; + } + }; + if spcr.base.space != GAS_SYSTEM_MEMORY || spcr.base.address == 0 { + log!("serial: SPCR places the {kind} at {:?}, which is not a memory-mapped frame", spcr.base); + return false; + } + BASE.store(spcr.base.address, Ordering::Relaxed); + log!("serial: {kind} at {:#x} (SPCR, GSIV {})", spcr.base.address, spcr.gsiv); + true +} + +/// Whether a received byte waits. +pub fn rx_ready() -> bool { + regs().read_u32(FR) & FR_RXFE == 0 +} + +/// The received byte; only after [`rx_ready`] said one waits. +pub fn read_byte() -> u8 { + regs().read_u32(DR) as u8 +} + +/// Whether the transmitter will take a byte. +pub fn tx_ready() -> bool { + regs().read_u32(FR) & FR_TXFF == 0 +} + +/// Put one byte in the transmitter; only after [`tx_ready`], or the byte may be lost. +pub fn write_byte(byte: u8) { + regs().write_u32(DR, u32::from(byte)); +} diff --git a/kernel/src/arch/aarch64/control_regs.rs b/kernel/src/arch/aarch64/control_regs.rs new file mode 100644 index 00000000000..9c82addbb2a --- /dev/null +++ b/kernel/src/arch/aarch64/control_regs.rs @@ -0,0 +1,105 @@ +//! What the EL1 system registers hold on every CPU in this machine, and what +//! EL2 is left holding when the kernel is entered there. One declaration: +//! [`super::boot`]'s entry writes every register here whole, from these +//! constants, before the MMU is on; [`check`] reads the EL1 ones back and +//! refuses a CPU whose registers say anything else. Nothing else writes any +//! of them. +//! +//! Field positions are Arm ARM K.a, chapter D24 (the register descriptions). + +use core::sync::atomic::{AtomicU64, Ordering}; + +use crate::log; + +/// `MAIR_EL1`: the three memory types the loader's tables and this kernel +/// map with, by `AttrIndx` — declared once, beside the descriptors that name them. +pub use toyos_bootmap::aarch64::MAIR; + +/// `SCTLR_EL1`: MMU on (`M`), data and instruction caches on (`C`, `I`), +/// stack alignment checked at EL1 and EL0 (`SA`, `SA0`), AArch32 EL0's `IT` +/// and `SETEND` disabled (`ITD`, `SED` — RES1 on a CPU with no AArch32 EL0, +/// and nothing this kernel runs is AArch32), over the bits Armv8.0 makes +/// RES1 (29, 28, 23, 22, 20, 11). `WXN` stays clear: the +/// loader's blocks are writable and executable both until the kernel owns +/// its tables. Little-endian at both levels; EL0's cache maintenance and +/// `WFI`/`WFE` trap. +pub const SCTLR: u64 = SCTLR_RES1 | 1 << 0 | 1 << 2 | 1 << 3 | 1 << 4 | 1 << 7 | 1 << 8 | 1 << 12; + +/// `SCTLR_EL1` with the MMU and caches off — [`SCTLR`] less `M`, `C`, `I`, +/// `SA` and `SA0` — the only other value the entry writes: what a CPU entered at EL1 under firmware's tables is put through +/// before its translation registers change. +pub const SCTLR_MMU_OFF: u64 = SCTLR_RES1 | 1 << 7 | 1 << 8; + +const SCTLR_RES1: u64 = 1 << 29 | 1 << 28 | 1 << 23 | 1 << 22 | 1 << 20 | 1 << 11; + +/// `TCR_EL1` but for `IPS`: 48-bit regions from both tables (`T0SZ` = `T1SZ` = +/// 16), 4 KiB granules (`TG0` = 0, `TG1` = 2), walks inner-shareable and +/// write-back write-allocate cacheable, 8-bit ASIDs. +pub const TCR: u64 = 16 | 1 << 8 | 1 << 10 | 3 << 12 | 16 << 16 | 1 << 24 | 1 << 26 | 3 << 28 | 2 << 30; + +/// `TCR_EL1.IPS`'s position: the output address size, taken from +/// `ID_AA64MMFR0_EL1.PARange` (the same encoding), because a larger `IPS` +/// than the CPU implements is a reserved value. +pub const TCR_IPS_SHIFT: u64 = 32; + +/// `CPACR_EL1`: `FPEN` = 0, so FP and SIMD trap at EL1 and EL0. The kernel is +/// built soft-float and uses neither; user mode's are stage 7's. +pub const CPACR: u64 = 0; + +/// `HCR_EL2` when entered at EL2: `RW`, so EL1 is AArch64, and nothing else — +/// no stage-2 translation, no trap, `E2H` clear. +pub const HCR_EL2: u64 = 1 << 31; + +/// `CNTHCTL_EL2` when entered at EL2: `EL1PCTEN` and `EL1PCEN`, so EL1 reads the +/// physical counter and programs its timer without trapping. +pub const CNTHCTL_EL2: u64 = 1 << 1 | 1 << 0; + +/// `CPTR_EL2` when entered at EL2 (`E2H` clear): its RES1 bits (13, 12, 9:0) +/// and `TFP` clear, so FP is `CPACR_EL1`'s decision alone. +pub const CPTR_EL2: u64 = 0x33FF; + +/// `SPSR_EL2` for the drop: EL1 on `SP_EL1` (`M` = 0b0101), `D`, `A`, `I`, `F` masked. +pub const SPSR_EL2_TO_EL1: u64 = 0x3C5; + +/// The exception level the loader entered the kernel at, as the entry recorded it. +pub static ENTRY_EL: AtomicU64 = AtomicU64::new(0); + +/// One EL1 system register, read. +macro_rules! read { + ($reg:literal) => {{ + let value: u64; + // SAFETY: reads an EL1 system register, which EL1 may always read. + unsafe { core::arch::asm!(concat!("mrs {}, ", $reg), out(reg) value, options(nomem, nostack, preserves_flags)) }; + value + }}; +} + +/// `TCR_EL1` whole, as this CPU's physical address range makes it. +pub fn tcr() -> u64 { + TCR | (read!("id_aa64mmfr0_el1") & 0xF) << TCR_IPS_SHIFT +} + +/// Read every EL1 register the declaration names back, and refuse a CPU that +/// holds anything else; then say what it holds. +pub fn check() { + let declared = [ + ("SCTLR_EL1", read!("sctlr_el1"), SCTLR), + ("TCR_EL1", read!("tcr_el1"), tcr()), + ("MAIR_EL1", read!("mair_el1"), MAIR), + ("CPACR_EL1", read!("cpacr_el1"), CPACR), + ]; + for (name, live, value) in declared { + assert_eq!(live, value, "control registers: {name} holds {live:#x}, and the declaration says {value:#x}"); + } + // What the drop from EL2 left, or the EL1 entry kept: EL1, on `SP_EL1`. + let (el, spsel) = (read!("CurrentEL") >> 2 & 3, read!("SPSel") & 1); + assert_eq!((el, spsel), (1, 1), "control registers: running at EL{el} on SP_EL{spsel}, not EL1 on SP_EL1"); + let el = ENTRY_EL.load(Ordering::Relaxed); + log!( + "control registers: SCTLR_EL1={SCTLR:#x} TCR_EL1={:#x} MAIR_EL1={:#x} CPACR_EL1={CPACR:#x}, \ + as declared; entered at EL{el}{}", + tcr(), + MAIR, + if el == 2 { ", HCR_EL2 read back as declared, CNTHCTL_EL2/CPTR_EL2 written, and dropped to EL1" } else { "" }, + ); +} diff --git a/kernel/src/arch/aarch64/cpu.rs b/kernel/src/arch/aarch64/cpu.rs new file mode 100644 index 00000000000..cfe942ddd0c --- /dev/null +++ b/kernel/src/arch/aarch64/cpu.rs @@ -0,0 +1,118 @@ +//! The CPU's own registers and instructions, as generic code names them. + +use core::arch::asm; + +/// The CPU's free-running counter: the generic timer's virtual count, +/// `CNTVCT_EL0`. The `ISB` keeps the read from being taken early, ahead of +/// the code it times (Arm ARM K.a, D12.2.2). +#[inline] +pub fn counter() -> u64 { + let count: u64; + // SAFETY: reads a counter EL1 may always read; the `ISB` touches nothing. + unsafe { asm!("isb", "mrs {}, cntvct_el0", out(reg) count, options(nomem, nostack, preserves_flags)) }; + count +} + +/// The counter's frequency as firmware states it in `CNTFRQ_EL0`, which the +/// Arm ARM makes firmware's duty to program. Zero states nothing. +pub fn stated_counter_hz() -> Option { + let hz: u64; + // SAFETY: reads a register EL1 may always read. + unsafe { asm!("mrs {}, cntfrq_el0", out(reg) hz, options(nomem, nostack, preserves_flags)) }; + (hz != 0).then_some(hz) +} + +/// This function's caller's frame pointer: `x29`, which `-Cforce-frame-pointers=yes` makes one. +#[inline(always)] +pub fn frame_pointer() -> u64 { + let fp: u64; + // SAFETY: register-to-register move only. + unsafe { asm!("mov {}, x29", out(reg) fp, options(nomem, nostack, preserves_flags)) }; + fp +} + +/// The stack pointer this code is running on. +#[inline(always)] +pub fn stack_pointer() -> u64 { + let sp: u64; + // SAFETY: register-to-register move only. + unsafe { asm!("mov {}, sp", out(reg) sp, options(nomem, nostack, preserves_flags)) }; + sp +} + +/// Raise the architecture's undefined-instruction exception here: `UDF`, whose +/// synchronous exception the vectors catch as the kernel's own fault. +pub fn undefined_instruction() { + // SAFETY: `UDF` reads and writes nothing and raises an exception the + // installed vectors report. + unsafe { asm!("udf #0", options(nomem, nostack)) }; +} + +/// The thread pointer this CPU is running with: `TPIDR_EL0`, which user TLS is addressed from. +#[inline] +pub fn thread_pointer() -> u64 { + let tp: u64; + // SAFETY: reads a register EL1 may always read. + unsafe { asm!("mrs {}, tpidr_el0", out(reg) tp, options(nomem, nostack, preserves_flags)) }; + tp +} + +/// Unmask interrupts on this CPU: `DAIF.I` and `DAIF.F`. +pub fn enable_interrupts() { + // SAFETY: writes two `DAIF` bits; a compiler barrier, so no access moves across it. + unsafe { asm!("msr daifclr, #3", options(nostack)) }; +} + +/// Mask interrupts on this CPU. +pub fn disable_interrupts() { + // SAFETY: writes two `DAIF` bits; a compiler barrier, so no access moves across it. + unsafe { asm!("msr daifset, #3", options(nostack)) }; +} + +/// Whether this CPU takes interrupts: `DAIF.I` clear. +pub fn interrupts_enabled() -> bool { + let daif: u64; + // SAFETY: reads `DAIF`. + unsafe { asm!("mrs {}, daif", out(reg) daif, options(nomem, nostack, preserves_flags)) }; + daif & (1 << 7) == 0 +} + +/// Stop this CPU for good: interrupts masked, then `WFI` forever. A wake +/// that arrives anyway lands back in the loop. +pub fn halt() -> ! { + disable_interrupts(); + loop { + // SAFETY: waits for an event; touches no memory. + unsafe { asm!("wfi", options(nomem, nostack, preserves_flags)) }; + } +} + +/// Leave the current stack for good and run `func` on the one ending at `top`, +/// with a zeroed frame chain so a panic there backtraces instead of walking off +/// the top. +/// # Safety +/// Nothing on the current stack is live past this call, and `top` is the end of +/// a 16-byte-aligned stack this CPU owns. +pub unsafe fn run_on_stack(top: u64, func: extern "C" fn() -> !) -> ! { + // SAFETY: the caller's contract. + unsafe { + asm!( + "mov sp, {sp}", + "mov x29, xzr", + "mov x30, xzr", + "br {func}", + sp = in(reg) top, + func = in(reg) func as *const () as usize, + options(noreturn), + ); + } +} + +/// This CPU's hardware identity: `MPIDR_EL1`'s four affinity fields, packed +/// into 32 bits (`Aff3:Aff2:Aff1:Aff0`), readable before any per-CPU state exists. +pub fn hardware_id() -> u32 { + let mpidr: u64; + // SAFETY: reads an ID register. + unsafe { asm!("mrs {}, mpidr_el1", out(reg) mpidr, options(nomem, nostack, preserves_flags)) }; + ((mpidr & 0xFF_FFFF) | ((mpidr >> 8) & 0xFF00_0000)) as u32 +} diff --git a/kernel/src/arch/aarch64/entropy.rs b/kernel/src/arch/aarch64/entropy.rs new file mode 100644 index 00000000000..fb564800f1b --- /dev/null +++ b/kernel/src/arch/aarch64/entropy.rs @@ -0,0 +1,43 @@ +//! The CPU's own random source: `RNDR`, where `ID_AA64ISAR0_EL1.RNDR` says +//! there is one. A machine without it (QEMU under HVF is one) draws from +//! virtio-rng instead, which the port's stage 6 brings up. + +/// How often a caller may ask before taking "no data" as the answer: `RNDR` +/// reports a transient failure through `NZCV`, as `RDRAND` does through `CF`. +pub const ATTEMPTS: u32 = 10; + +fn has_rndr() -> bool { + let isar0: u64; + // SAFETY: reads an ID register; touches nothing. + unsafe { + core::arch::asm!("mrs {}, id_aa64isar0_el1", out(reg) isar0, options(nomem, nostack, preserves_flags)) + }; + (isar0 >> 60) & 0xF != 0 +} + +/// Whether this CPU can draw at all, or why not. +pub fn available() -> Result<(), &'static str> { + if has_rndr() { + Ok(()) + } else { + Err("ID_AA64ISAR0_EL1.RNDR is zero, so this CPU has no RNDR, and virtio-rng is the port's stage 6") + } +} + +/// One drawn `u64`, or `None` when the source had nothing to give; never waits. +pub fn draw() -> Option { + let value: u64; + let failed: u64; + // SAFETY: `RNDR` (`S3_3_C2_C4_0`) reads a random number, setting `Z` on + // failure; `available` said the register exists before any caller draws. + unsafe { + core::arch::asm!( + "mrs {value}, s3_3_c2_c4_0", + "cset {failed}, eq", + value = out(reg) value, + failed = out(reg) failed, + options(nomem, nostack), + ) + }; + (failed == 0).then_some(value) +} diff --git a/kernel/src/arch/aarch64/entry.rs b/kernel/src/arch/aarch64/entry.rs new file mode 100644 index 00000000000..8207fc9d14b --- /dev/null +++ b/kernel/src/arch/aarch64/entry.rs @@ -0,0 +1,27 @@ +//! Where a new context starts: the frame `switch` restores and the +//! trampolines to user mode and to a kernel thread, the port's stages 4 and 7. + + +pub(crate) extern "C" fn process_start() { + owed!("user mode", "stage 7") +} + +pub(crate) extern "C" fn thread_start() { + owed!("user mode", "stage 7") +} + +pub(crate) extern "C" fn kernel_start() { + owed!("kernel threads", "stage 4") +} + +/// # Safety +/// `top` is the end of a fresh kernel stack nothing else references. +pub unsafe fn initial_frame( + _top: u64, + _trampoline: unsafe extern "C" fn(), + _user_entry: u64, + _user_sp: u64, + _arg: u64, +) -> u64 { + owed!("kernel threads", "stage 4") +} diff --git a/kernel/src/arch/aarch64/fpu.rs b/kernel/src/arch/aarch64/fpu.rs new file mode 100644 index 00000000000..1613f2f8d5b --- /dev/null +++ b/kernel/src/arch/aarch64/fpu.rs @@ -0,0 +1,2 @@ +//! User FP/SIMD state: `CPACR_EL1.FPEN` traps it at EL1 and EL0 until the +//! port's stage 7 saves and restores it. The kernel itself never uses it. diff --git a/kernel/src/arch/aarch64/hw.rs b/kernel/src/arch/aarch64/hw.rs new file mode 100644 index 00000000000..472b8b6a4b4 --- /dev/null +++ b/kernel/src/arch/aarch64/hw.rs @@ -0,0 +1,82 @@ +//! `KernelHw` — the kernel's side of the scheduler-core hardware boundary: +//! the generic timer, SGIs, `WFI` and the context switch, the port's stage 4. + +use toyos_sched::cpu::RunToken; +use toyos_sched::hw::{CpuId, Hw, Kicker, Machine, Nanos, TraceEvent}; +use toyos_sched::task::{TaskAccounting, TaskKey}; + +use crate::sched::payload::KernelPayload; + +/// The one instance; zero-sized, holds no per-CPU state. +pub static HW: KernelHw = KernelHw; + +pub struct KernelHw; + +/// The scheduler clock, in raw nanoseconds. +pub fn now_ns() -> u64 { + HW.now().0 +} + +impl Kicker for KernelHw { + fn kick(&self, _target: CpuId) { + owed!("the interrupt controller", "stage 4") + } +} + +impl Machine for KernelHw { + type IrqGuard = crate::arch::IrqGuard; + + fn now(&self) -> Nanos { + Nanos(crate::clock::nanos_since_boot()) + } + + fn set_timer(&self, _deadline: Nanos) { + owed!("the timer", "stage 4") + } + + fn stop_timer(&self) { + owed!("the timer", "stage 4") + } + + fn irq_guard(&self) -> crate::arch::IrqGuard { + crate::arch::IrqGuard::close() + } + + fn halt(&self) { + owed!("the interrupt controller", "stage 4") + } + + fn need_resched(&self, _cpu: CpuId) { + owed!("the interrupt controller", "stage 4") + } + + fn trace(&self, ev: TraceEvent) { + crate::trace::record(ev); + } +} + +impl Hw for KernelHw { + type Payload = KernelPayload; + + unsafe fn switch(&self, _token: RunToken) { + owed!("the context switch", "stage 4") + } + + fn release(&self, _key: TaskKey, _payload: KernelPayload, _acct: TaskAccounting) { + owed!("the context switch", "stage 4") + } +} + +/// What every kernel crash says about the machine's contexts: until stage 4 +/// switches any, only the memory facts. Never owed: a crash report that +/// panicked would bury the crash. +pub fn report_contexts(sp: u64, _subject: Option) { + crate::log!(" Contexts: the boot CPU crashed at sp={sp:#018x}; no context has been switched"); + crate::mm::report_on_crash(); +} + +/// AMD's `SYSRET` erratum has no AArch64 counterpart; the probe is x86-64's. +#[cfg(feature = "boot-actuators")] +pub fn sysret_ss_probe(_parkable: &crate::scheduler::Parkable) { + crate::log!("sysret-ss: AArch64 has no SYSRET, so there is no SS to probe"); +} diff --git a/kernel/src/arch/aarch64/iommu_unit.rs b/kernel/src/arch/aarch64/iommu_unit.rs new file mode 100644 index 00000000000..3593845081f --- /dev/null +++ b/kernel/src/arch/aarch64/iommu_unit.rs @@ -0,0 +1,81 @@ +//! The IOMMU: an SMMUv3 the IORT names, the port's stage 6. Until then no +//! unit exists, and [`init`] says so; a domain is refused rather than built. + +use crate::drivers::pci::PciDevice; +use crate::log; + +pub fn init(_rsdp_addr: u64, _devices: &[PciDevice]) { + log!("IOMMU: the SMMUv3 is the port's stage 6; no device is translated this boot"); +} + +pub mod domain { + use crate::iommu::{DomainId, IommuError, Iova, StreamId}; + + /// No unit is driven, so there is no domain to give. + pub fn create() -> Result { + Err(IommuError::NoUnit) + } + + pub fn map(_id: DomainId, _phys: u64, _bytes: u64) -> Result { + unreachable!("no domain exists: `create` refuses every one") + } + + pub fn map_at(_id: DomainId, _at: Iova, _phys: u64, _bytes: u64) -> Result<(), IommuError> { + unreachable!("no domain exists: `create` refuses every one") + } + + pub fn reserve(_id: DomainId, _bytes: u64) -> Result { + unreachable!("no domain exists: `create` refuses every one") + } + + pub fn place(_id: DomainId, _at: Iova, _phys: u64, _bytes: u64) -> Result { + unreachable!("no domain exists: `create` refuses every one") + } + + pub fn unmap(_id: DomainId, _at: Iova, _bytes: u64) -> Result<(), IommuError> { + unreachable!("no domain exists: `create` refuses every one") + } + + pub fn attach(_stream: StreamId, _id: DomainId) { + unreachable!("no domain exists: `create` refuses every one") + } +} + +pub mod interrupt { + use crate::iommu::{Refused, StreamId}; + + pub struct Msi { + pub address: u32, + pub data: u32, + } + + pub struct Pin { + pub low: u32, + pub high: u32, + } + + /// No unit, so nothing remaps: callers deliver directly. + pub fn is_armed() -> bool { + false + } + + pub fn msi(_source: StreamId, _vector: u8, _dest: u32) -> Result { + unreachable!("no interrupt remapping without an IOMMU unit, and `is_armed` said so") + } + + pub fn pin(_apic_id: u8, _vector: u8, _dest: u32, _level: bool) -> Result { + unreachable!("no interrupt remapping without an IOMMU unit, and `is_armed` said so") + } +} + +pub mod fault { + use crate::iommu::StreamId; + + pub fn user_owned(_stream: StreamId, _slot: Option) { + owed!("SMMUv3 fault reporting", "stage 6") + } + + pub fn service() { + owed!("SMMUv3 fault reporting", "stage 6") + } +} diff --git a/kernel/src/arch/aarch64/irqchip.rs b/kernel/src/arch/aarch64/irqchip.rs new file mode 100644 index 00000000000..3edc03a567a --- /dev/null +++ b/kernel/src/arch/aarch64/irqchip.rs @@ -0,0 +1,31 @@ +//! The interrupt controller: a GICv3 — distributor, one redistributor per CPU +//! and an ITS for MSIs — the port's stage 4. Until then no interrupt is +//! delivered, and every way of raising one is owed. + + +/// Wake `cpu` so it runs a scheduler pass: an SGI. +pub fn kick_cpu(_cpu: u32) { + owed!("the interrupt controller", "stage 4") +} + +pub fn kick_all_but_self() { + owed!("the interrupt controller", "stage 4") +} + +/// Raise `vector` on this CPU. +pub fn send_self(_vector: u8) { + owed!("the interrupt controller", "stage 4") +} + +/// A pseudo-NMI to `cpu`. +pub fn send_nmi(_cpu: u32) { + owed!("the interrupt controller", "stage 4") +} + +/// This CPU's timer, armed to fire within `nanos`. +pub fn arm_within(_nanos: u64) { + owed!("the timer", "stage 4") +} + +/// Nothing to stop: before stage 5 the boot CPU is the only one running. +pub fn stop_other_cpus() {} diff --git a/kernel/src/arch/aarch64/keyboard_controller.rs b/kernel/src/arch/aarch64/keyboard_controller.rs new file mode 100644 index 00000000000..ef6462509b8 --- /dev/null +++ b/kernel/src/arch/aarch64/keyboard_controller.rs @@ -0,0 +1,18 @@ +//! The platform's own keyboard controller. An Arm machine has none: its +//! keyboards are USB, and the panic panel's key poll reads nothing here. + +/// No byte ever waits. +pub fn poll_byte() -> Option<(u8, bool)> { + None +} + +/// Nothing to decide about a controller that is not there. +pub fn verdict_due() -> bool { + false +} + +/// Nothing to service. +pub fn service() {} + +/// Nothing to report. +pub fn report_line() {} diff --git a/kernel/src/arch/aarch64/mod.rs b/kernel/src/arch/aarch64/mod.rs new file mode 100644 index 00000000000..1ac0e061e99 --- /dev/null +++ b/kernel/src/arch/aarch64/mod.rs @@ -0,0 +1,128 @@ +#![warn(clippy::undocumented_unsafe_blocks)] +//! AArch64: the Arm A-profile at EL1, found and described through ACPI. +//! +//! Every `unsafe` block here carries a one-line `SAFETY:` comment, enforced by the lint above. +//! +//! **What exists and what is owed.** The boot reaches its console: the entry +//! drops from EL2, applies the control-register declaration, turns on the +//! loader's tables and installs the exception vectors; the PL011 is found +//! through SPCR. Everything the kernel does after the console — the interrupt +//! controller, the timer, its own page tables, other CPUs, user mode — is +//! owed by a stage of the port (`issues/kernel/toyos-runs-on-arm64.md`), and +//! each item that stands for it here is an [`owed!`] that panics naming it. A kernel that reaches +//! one stops loudly on its panel; none of them returns a guess. + +/// Stands for work the port owes: panics naming what and which stage of the +/// track owns it (`issues/kernel/toyos-runs-on-arm64.md`), or that none does yet. +macro_rules! owed { + ($what:literal, $stage:literal) => { + panic!(concat!("aarch64: ", $what, ": owed by ", $stage)) + }; +} + +pub mod barrier; +pub mod boot; +pub mod cache; +pub mod console_uart; +pub mod control_regs; +pub mod cpu; +pub mod entropy; +pub mod entry; +pub mod fpu; +pub mod hw; +pub mod iommu_unit; +pub mod irqchip; +pub mod keyboard_controller; +pub mod paging; +pub mod percpu; +pub mod pio; +pub mod pmu; +pub mod rtc; +pub mod smp; +pub mod syscall; +pub mod tlb; +pub mod trap; +pub mod watchdog; + +/// The machine every program image this kernel loads must be built for. +pub const ELF_MACHINE: toyos_elf::Machine = toyos_elf::Machine::Aarch64; + +/// A message-signalled interrupt's address and data for `vector` on CPU +/// `dest`. On AArch64 the doorbell is an ITS's `GITS_TRANSLATER`, one per ITS +/// the MADT names, and the data is an event the ITS maps: stage 4 builds both. +pub fn msi_message(_dest: u32, _vector: u8) -> (u32, u32) { + owed!("an MSI doorbell (the GICv3 ITS)", "stage 4") +} + +/// Interrupts masked on this CPU for as long as the guard lives, and then put +/// back as they were — restored, not enabled — so a guard nests inside a region +/// that is already masked. `DAIF`'s `I` and `F` are what it closes; `D` and `A` +/// are the boot's and it leaves them alone. +/// +/// Both edges are compiler barriers (no `nomem`): a memory access written +/// inside the region is emitted inside it. +#[must_use = "dropping the guard reopens interrupts"] +pub struct IrqGuard { + daif: u64, + // Same-CPU only: keeps this guard `!Send + !Sync`. + _not_send_sync: core::marker::PhantomData<*mut ()>, +} + +impl IrqGuard { + pub fn close() -> Self { + let daif: u64; + // SAFETY: reads `DAIF` and sets its `I` and `F` bits; touches nothing else. + unsafe { + core::arch::asm!("mrs {saved}, daif", "msr daifset, #3", saved = out(reg) daif); + } + Self { daif, _not_send_sync: core::marker::PhantomData } + } + + /// The mask captured and interrupts left as they are: what the + /// `log-unbracketed-reserve` actuator stages a log reservation with. + #[cfg(feature = "boot-actuators")] + pub fn unclosed() -> Self { + let daif: u64; + // SAFETY: reads `DAIF` and writes nothing. + unsafe { + core::arch::asm!("mrs {saved}, daif", saved = out(reg) daif); + } + Self { daif, _not_send_sync: core::marker::PhantomData } + } +} + +impl Drop for IrqGuard { + fn drop(&mut self) { + // SAFETY: the word `close` read out of `DAIF` on this CPU (the guard is + // `!Send`), restored whole. + unsafe { + core::arch::asm!("msr daif, {saved}", saved = in(reg) self.daif); + } + } +} + +/// Adds one to `counter`, atomic against an interrupt on this CPU, and answers the value before the add. +/// # Safety: `counter` is written by no other CPU; `guard` covers the shard selection that owns it. +#[inline(always)] +pub unsafe fn percpu_fetch_add( + counter: &core::sync::atomic::AtomicU64, + _guard: &IrqGuard, +) -> u64 { + let previous = counter.load(core::sync::atomic::Ordering::Relaxed); + // Under `log-shared-reservation`, open the window the guard closes, so a + // nested record can land between the load and the store. + if crate::actuator::log_shared_reservation() && crate::log::nested::inject() { + // SAFETY: each writes `DAIF.I` and touches no memory. + unsafe { + core::arch::asm!("msr daifclr, #2"); + for _ in 0..256 { + core::hint::spin_loop(); + } + core::arch::asm!("msr daifset, #2"); + } + } + // A load and a store, not an atomic add: the guard masks the only other + // writer this CPU has, and no other CPU writes the counter. + counter.store(previous + 1, core::sync::atomic::Ordering::Relaxed); + previous +} diff --git a/kernel/src/arch/aarch64/paging.rs b/kernel/src/arch/aarch64/paging.rs new file mode 100644 index 00000000000..fe72ef6d093 --- /dev/null +++ b/kernel/src/arch/aarch64/paging.rs @@ -0,0 +1,194 @@ +//! Page tables. The boot runs on the loader's: one L0 table that `TTBR0_EL1` +//! walks for the identity view and `TTBR1_EL1` for the view at `PHYS_OFFSET`, +//! 2 MiB blocks whose `AttrIndx` names [`super::control_regs`]'s `MAIR_EL1`. +//! The kernel's own tables — the direct map, MMIO windows, and a user +//! address space per process with an ASID — are the port's stage 4, so an +//! [`AddressSpace`] cannot exist yet: the type is uninhabited, and every +//! method on it is a match on nothing. + +use core::convert::Infallible; + +use crate::mm::UserAddr; +pub use crate::mm::policy::{CachePolicy, MmioPolicy, Prot, WindowProt}; +use crate::sync::Lock; +use crate::vma::{Occupancy, Region, RegionKind}; +use crate::MemoryMapEntry; + +/// A translation table base: what `TTBR0_EL1` is loaded with for a space. +#[derive(Clone, Copy)] +pub struct Root(Infallible); + +impl Root { + pub fn phys(self) -> u64 { + match self.0 {} + } + + /// # Safety + /// The underlying page tables must be valid and live. + pub unsafe fn activate(self) { + match self.0 {} + } +} + +/// A process's address space: its regions and the tables that map them. +pub struct AddressSpace { + never: Infallible, +} + +impl AddressSpace { + pub fn new_user() -> Option { + owed!("a user address space", "stage 4") + } + + pub fn root(&self) -> Root { + match self.never {} + } + + pub fn map_range(&mut self, _vaddr: UserAddr, _phys: u64, _size: u64, _prot: Prot, _cache: CachePolicy) { + match self.never {} + } + + pub fn remap(&mut self, _vaddr: UserAddr, _phys: u64, _prot: Prot) { + match self.never {} + } + + pub fn map_window(&mut self, _vaddr: UserAddr, _phys: u64, _prot: &WindowProt) { + match self.never {} + } + + pub fn map_window_if_absent(&mut self, _vaddr: UserAddr, _phys: u64, _prot: &WindowProt) -> bool { + match self.never {} + } + + pub fn unmap(&mut self, _vaddr: UserAddr) { + match self.never {} + } + + pub fn translate(&self, _vaddr: UserAddr) -> Option { + match self.never {} + } + + pub fn alloc_region(&mut self, _size: u64, _kind: RegionKind) -> Option { + match self.never {} + } + + pub fn alloc_and_map( + &mut self, + _phys: u64, + _size: u64, + _prot: Prot, + _cache: CachePolicy, + ) -> Option<(UserAddr, u64)> { + match self.never {} + } + + pub fn free_and_unmap(&mut self, _addr: UserAddr) -> Option { + match self.never {} + } + + pub fn insert_region(&mut self, _addr: UserAddr, _region: Region) { + match self.never {} + } + + pub fn find_region(&self, _addr: UserAddr) -> Option<(UserAddr, &Region)> { + match self.never {} + } + + pub fn occupancy(&self, _addr: UserAddr, _size: u64) -> Occupancy { + match self.never {} + } + + pub fn overlapping_regions( + &self, + _start: UserAddr, + _end: UserAddr, + ) -> impl Iterator { + match self.never {} + #[allow(unreachable_code)] + core::iter::empty() + } + + pub fn direct_map_policy(&self, _phys: u64) -> Option { + match self.never {} + } + + pub fn user_policy(&self, _addr: UserAddr) -> Option { + match self.never {} + } + + pub fn guard_4k(&mut self, _phys: u64) { + match self.never {} + } +} + +pub fn kernel() -> &'static alloc::sync::Arc> { + owed!("the kernel's page tables", "stage 4") +} + +pub fn kernel_root() -> Root { + owed!("the kernel's page tables", "stage 4") +} + +pub fn activate_kernel() { + owed!("the kernel's page tables", "stage 4") +} + +pub fn map_mmio(_phys: u64, _size: u64, _policy: MmioPolicy) -> crate::mm::Mmio { + owed!("the kernel's page tables", "stage 4") +} + +pub(crate) fn init(_memory_map: &[MemoryMapEntry]) { + owed!("the kernel's page tables", "stage 4") +} + +/// `PAR_EL1` after the MMU translated `addr` for an EL1 read in the tables +/// this CPU runs on now (`AT S1E1R`; Arm ARM K.a, C6.2.x and D24.2.131): +/// bit 0 set is a fault, and otherwise bits 63:56 are the memory type. +fn translate_read(addr: u64) -> u64 { + let par: u64; + // SAFETY: `AT` translates without accessing memory and reports through + // `PAR_EL1`; the `ISB` makes the result the one read back. + unsafe { + core::arch::asm!( + "at s1e1r, {addr}", + "isb", + "mrs {par}, par_el1", + addr = in(reg) addr, + par = out(reg) par, + options(nostack, preserves_flags), + ); + } + par +} + +/// Whether `addr` translates for a read in the tables this CPU runs on now: +/// the MMU's own answer, so broken tables become "not present" and never a +/// fault here. +pub fn present_in_current_tables(addr: u64) -> bool { + translate_read(addr) & 1 == 0 +} + +/// Whether the scanout is write-combining at both of its addresses already: +/// on AArch64 that is Normal non-cacheable, whose stores gather, and the +/// loader maps it so. Asked of the MMU page by page rather than assumed, and +/// nothing is retyped: the type the loader wrote is the final one. +pub fn boot_map_write_combining(phys: u64, size: u64) -> bool { + [phys, crate::mm::PHYS_OFFSET + phys].iter().all(|&base| { + (0..size.div_ceil(4096)).all(|page| { + let par = translate_read(base + page * 4096); + // Outer Normal non-cacheable, and an inner nibble that says the same: + // 0b0100 as `MAIR_EL1` spells it, or 0b0000, which is how a CPU may + // report an inner Non-cacheable attribute in `PAR_EL1` (HVF on Apple + // silicon does), and which `MAIR_EL1` itself never holds for Normal. + let (outer, inner) = ((par >> 60) & 0xF, (par >> 56) & 0xF); + par & 1 == 0 && outer == 0b0100 && (inner == 0b0100 || inner == 0) + }) + }) +} + +/// What memory type the scanout is mapped with, for the GPU's report. +pub fn scanout_memory_type(_addr: u64, _size: u64) -> impl core::fmt::Display { + owed!("the kernel's page tables", "stage 4"); + #[allow(unreachable_code)] + "" +} diff --git a/kernel/src/arch/aarch64/percpu.rs b/kernel/src/arch/aarch64/percpu.rs new file mode 100644 index 00000000000..d9f26a7df18 --- /dev/null +++ b/kernel/src/arch/aarch64/percpu.rs @@ -0,0 +1,118 @@ +//! Per-CPU state. On AArch64 it is reached through `TPIDR_EL1`, and the block +//! it names is built by the port's stage 5 (other CPUs) on top of stage 4's +//! exception entry; until then every accessor is owed. The boot never reaches +//! one: the log and the panic path read [`crate::log::PERCPU_READY`] first. + +use crate::process::{Pid, Tid}; + +/// Per-CPU fault state machine for the escalation policy on nested faults. +#[repr(u8)] +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum CpuFaultState { + Normal = 0, + PageFault = 1, + Fatal = 2, + Panic = 3, +} + +pub fn cpu_id() -> u32 { + owed!("per-CPU state", "stage 4") +} + +pub fn current_tid() -> Option { + owed!("per-CPU state", "stage 4") +} + +pub fn set_current_tid(_tid: Option) { + owed!("per-CPU state", "stage 4") +} + +pub fn current_pid() -> Option { + owed!("per-CPU state", "stage 4") +} + +pub fn set_current_pid(_pid: Option) { + owed!("per-CPU state", "stage 4") +} + +/// # Safety +/// `top` is the top of the kernel stack the next entry from user mode lands on. +pub unsafe fn set_kernel_stack(_top: u64) { + owed!("per-CPU state", "stage 4") +} + +/// # Safety +/// Read on the CPU whose entry stacks they are. +pub unsafe fn entry_stacks() -> (u64, u64) { + owed!("per-CPU state", "stage 4") +} + +pub fn idle_stack_top() -> u64 { + owed!("per-CPU state", "stage 4") +} + +#[cfg(feature = "test-actuators")] +pub fn idle_guard_byte() -> u64 { + owed!("per-CPU state", "stage 4") +} + +#[cfg(feature = "test-actuators")] +pub fn idle_stack_size() -> usize { + owed!("per-CPU state", "stage 4") +} + +#[cfg(feature = "test-actuators")] +pub fn idle_stack_high_water() -> usize { + owed!("per-CPU state", "stage 4") +} + +pub fn in_syscall() -> bool { + owed!("per-CPU state", "stage 4") +} + +pub fn syscall_num() -> u64 { + owed!("per-CPU state", "stage 4") +} + +pub fn swap_fault_state(_new: CpuFaultState) -> CpuFaultState { + owed!("per-CPU state", "stage 4") +} + +/// This CPU's shard, its identity, and one sequence number out of that shard. +pub fn reserve_log_slot( + _guard: &crate::arch::IrqGuard, +) -> (*const crate::log::Shard, u64, u32, u32, u32) { + owed!("per-CPU state", "stage 4") +} + +pub fn irq_counts_here(_first: usize, _second: usize) -> (u64, u64) { + owed!("per-CPU state", "stage 4") +} + +pub fn preempt_count() -> u32 { + owed!("per-CPU state", "stage 4") +} + +pub fn set_preempt_count(_value: u32) { + owed!("per-CPU state", "stage 4") +} + +pub fn preempt_count_up() { + owed!("per-CPU state", "stage 4") +} + +pub fn preempt_count_down() { + owed!("per-CPU state", "stage 4") +} + +pub fn resched_owed() -> bool { + owed!("per-CPU state", "stage 4") +} + +pub fn set_resched_owed(_owed: bool) { + owed!("per-CPU state", "stage 4") +} + +pub fn faulting() -> bool { + owed!("per-CPU state", "stage 4") +} diff --git a/kernel/src/arch/aarch64/pio.rs b/kernel/src/arch/aarch64/pio.rs new file mode 100644 index 00000000000..b2886833d47 --- /dev/null +++ b/kernel/src/arch/aarch64/pio.rs @@ -0,0 +1,15 @@ +//! The I/O port space: AArch64 has none. + +/// Whether this architecture has an I/O port space at all. Firmware tables +/// that name a port are only honoured where it does. +pub const EXISTS: bool = false; + +/// Never called: every caller checks [`EXISTS`] first. +pub unsafe fn outb(_port: u16, _value: u8) { + unreachable!("AArch64 has no I/O port space") +} + +/// Never called: every caller checks [`EXISTS`] first. +pub unsafe fn outw(_port: u16, _value: u16) { + unreachable!("AArch64 has no I/O port space") +} diff --git a/kernel/src/arch/aarch64/pmu.rs b/kernel/src/arch/aarch64/pmu.rs new file mode 100644 index 00000000000..f9697ad93a7 --- /dev/null +++ b/kernel/src/arch/aarch64/pmu.rs @@ -0,0 +1,20 @@ +//! The performance monitor's overflow, as the hard-lockup detector's sample +//! source. On AArch64 that is `PMCCNTR_EL0` overflowing into a pseudo-NMI — +//! an interrupt the GICv3 delivers at a priority `DAIF.I` does not mask — +//! which needs the interrupt controller of the port's stage 4. + + +/// Start this CPU's counter overflowing into an NMI every `period` counter ticks. +pub fn arm(_period: u64) -> bool { + owed!("the PMU overflow NMI", "stage 4") +} + +/// Whether this CPU's counter is the reason this NMI arrived. +pub fn overflowed() -> bool { + owed!("the PMU overflow NMI", "stage 4") +} + +/// After an overflow's sample: clear it and reload. +pub fn rearm(_period: u64) { + owed!("the PMU overflow NMI", "stage 4") +} diff --git a/kernel/src/arch/aarch64/rtc.rs b/kernel/src/arch/aarch64/rtc.rs new file mode 100644 index 00000000000..9a0c49c0ad3 --- /dev/null +++ b/kernel/src/arch/aarch64/rtc.rs @@ -0,0 +1,20 @@ +//! The wall clock at boot: on an ACPI Arm machine, the UEFI runtime's +//! `GetTime` or a PL031 the tables name, the port's stage 6. + +use core::fmt; + +use toyos_wallclock::Civil; + +/// Why this machine did not say what time it is. None yet: the read is owed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RtcFault {} + +impl fmt::Display for RtcFault { + fn fmt(&self, _f: &mut fmt::Formatter<'_>) -> fmt::Result { + match *self {} + } +} + +pub fn read(_century_reg: Option) -> Result { + owed!("the wall clock", "no stage yet") +} diff --git a/kernel/src/arch/aarch64/smp.rs b/kernel/src/arch/aarch64/smp.rs new file mode 100644 index 00000000000..f52708238ed --- /dev/null +++ b/kernel/src/arch/aarch64/smp.rs @@ -0,0 +1,18 @@ +//! Other CPUs: PSCI `CPU_ON` for each GICC the MADT names, the port's stage 5. +//! Until then the boot CPU is the only one running, and the machine is never +//! released to the scheduler. + + +/// CPUs running: the boot CPU alone. +pub fn cpu_count() -> u32 { + 1 +} + +pub fn set_ready() { + owed!("other CPUs", "stage 5") +} + +/// Never: the boot ends before the machine is released. +pub fn is_ready() -> bool { + false +} diff --git a/kernel/src/arch/aarch64/syscall.rs b/kernel/src/arch/aarch64/syscall.rs new file mode 100644 index 00000000000..667d1f82fcb --- /dev/null +++ b/kernel/src/arch/aarch64/syscall.rs @@ -0,0 +1,12 @@ +//! The system-call gate: `SVC` from EL0 through the vectors' lower-EL entry, +//! the port's stage 7. + + +/// A syscall entered this CPU; x86-64's NMI gate counts it, and AArch64 has no +/// such gate. +pub fn note_entry() {} + +/// The actuator that storms NMIs at the syscall window. +pub fn window_storm() { + owed!("the syscall gate", "stage 7") +} diff --git a/kernel/src/arch/aarch64/tlb.rs b/kernel/src/arch/aarch64/tlb.rs new file mode 100644 index 00000000000..dae8edfbe64 --- /dev/null +++ b/kernel/src/arch/aarch64/tlb.rs @@ -0,0 +1,33 @@ +//! TLB invalidation across CPUs. AArch64 broadcasts it in hardware (`TLBI +//! …IS`), so the shootdown this interface stands for is a local instruction +//! plus `DSB ISH` once the kernel owns its page tables, the port's stage 4. + + +pub use crate::invalidation::Origin; + +pub fn log_census() { + owed!("TLB invalidation", "stage 4") +} + +pub fn shootdown(_origin: Origin) { + owed!("TLB invalidation", "stage 4") +} + +pub fn poll() { + owed!("TLB invalidation", "stage 4") +} + +#[cfg(feature = "boot-actuators")] +pub fn bench() { + owed!("TLB invalidation", "stage 4") +} + +#[cfg(feature = "test-actuators")] +pub fn debug_arm_ack_delay(_nanos: u64) -> u64 { + owed!("TLB invalidation", "stage 4") +} + +#[cfg(feature = "test-actuators")] +pub fn debug_disarm_ack_delay() -> u64 { + owed!("TLB invalidation", "stage 4") +} diff --git a/kernel/src/arch/aarch64/trap.rs b/kernel/src/arch/aarch64/trap.rs new file mode 100644 index 00000000000..24683360a39 --- /dev/null +++ b/kernel/src/arch/aarch64/trap.rs @@ -0,0 +1,204 @@ +//! Exceptions: the vector table `VBAR_EL1` names, and what the kernel says +//! when one is taken. +//! +//! Every exception is fatal until the port's stage 4 gives interrupts and +//! user faults somewhere to go: the entry saves the interrupted context into a +//! [`Frame`] on the stack it was taken on, and [`exception`] reports it and +//! panics. A table that does not reach [`exception`] — misaligned, or never +//! installed — is what the aarch64 boot test's fault arm catches, because +//! then nothing reports at all. + +use core::fmt; + +use crate::log; + +/// The interrupted context, as the entry stores it: `x0`–`x30`, the stack +/// pointer before the exception, then the four system registers that say +/// what happened. +#[repr(C)] +pub struct Frame { + pub x: [u64; 31], + pub sp: u64, + pub elr: u64, + pub spsr: u64, + pub esr: u64, + pub far: u64, +} + +const FRAME_BYTES: usize = core::mem::size_of::(); +const _: () = assert!(FRAME_BYTES == 288 && FRAME_BYTES.is_multiple_of(16)); + +/// Which of the table's sixteen entries was taken: Arm ARM K.a, D1.3.1, +/// Table D1-7 — four groups by where the exception came from, and in each the +/// synchronous, IRQ, FIQ and SError entries. +#[derive(Clone, Copy)] +struct Entry(u64); + +impl fmt::Display for Entry { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + let from = ["EL1 on SP_EL0", "EL1 on SP_EL1", "EL0 in AArch64", "EL0 in AArch32"]; + let kind = ["synchronous", "IRQ", "FIQ", "SError"]; + write!(f, "{} from {}", kind[(self.0 & 3) as usize], from[(self.0 >> 2) as usize & 3]) + } +} + +/// `ESR_ELx.EC`, the exception class, named for the report: Arm ARM K.a, +/// D24.2.40, the classes this kernel can take at EL1. +fn class_name(esr: u64) -> &'static str { + match esr >> 26 { + 0x00 => "unknown reason (an undefined instruction)", + 0x01 => "WFI or WFE trapped", + 0x07 => "SIMD or floating point trapped", + 0x0E => "illegal execution state", + 0x15 => "SVC from AArch64", + 0x18 => "system register access trapped", + 0x20 => "instruction abort from a lower EL", + 0x21 => "instruction abort", + 0x22 => "PC alignment fault", + 0x24 => "data abort from a lower EL", + 0x25 => "data abort", + 0x26 => "SP alignment fault", + 0x2F => "SError", + 0x3C => "BRK", + _ => "an exception class this kernel does not name", + } +} + +/// The Rust half of every vector entry: report the context and panic. The +/// panic handler's early branch puts the report on the console and the panel. +extern "C" fn exception(frame: &Frame, entry: u64) -> ! { + let entry = Entry(entry); + log!( + "KERNEL PANIC: {entry}: {} (ESR={:#010x})", + class_name(frame.esr), + frame.esr + ); + log!(" elr={:#018x} far={:#018x} spsr={:#010x}", frame.elr, frame.far, frame.spsr); + for pair in 0..15 { + log!( + " x{:<2}={:#018x} x{:<2}={:#018x}", + pair * 2, + frame.x[pair * 2], + pair * 2 + 1, + frame.x[pair * 2 + 1] + ); + } + log!(" x30={:#018x} sp ={:#018x}", frame.x[30], frame.sp); + crate::symbols::kernel_backtrace(frame.x[29], 20); + panic!("{entry}: {} at {:#x}", class_name(frame.esr), frame.elr); +} + +// The table: sixteen entries of 0x80 bytes, 2 KiB aligned (`VBAR_EL1` bits +// 10:0 are RES0). Each entry makes room for a `Frame`, saves x0/x1, names +// itself in x1 and branches to the common save, which fills in the rest and +// calls `exception` with the frame in x0. +core::arch::global_asm!( + ".macro toyos_vector n", + ".balign 0x80", + "sub sp, sp, #{frame}", + "stp x0, x1, [sp, #0]", + "mov x1, #\\n", + "b 1f", + ".endm", + ".pushsection .text.toyos_vectors, \"ax\"", + ".balign 0x800", + ".global toyos_vectors", + "toyos_vectors:", + "toyos_vector 0", "toyos_vector 1", "toyos_vector 2", "toyos_vector 3", + "toyos_vector 4", "toyos_vector 5", "toyos_vector 6", "toyos_vector 7", + "toyos_vector 8", "toyos_vector 9", "toyos_vector 10", "toyos_vector 11", + "toyos_vector 12", "toyos_vector 13", "toyos_vector 14", "toyos_vector 15", + "1:", + "stp x2, x3, [sp, #16]", + "stp x4, x5, [sp, #32]", + "stp x6, x7, [sp, #48]", + "stp x8, x9, [sp, #64]", + "stp x10, x11, [sp, #80]", + "stp x12, x13, [sp, #96]", + "stp x14, x15, [sp, #112]", + "stp x16, x17, [sp, #128]", + "stp x18, x19, [sp, #144]", + "stp x20, x21, [sp, #160]", + "stp x22, x23, [sp, #176]", + "stp x24, x25, [sp, #192]", + "stp x26, x27, [sp, #208]", + "stp x28, x29, [sp, #224]", + "add x2, sp, #{frame}", + "stp x30, x2, [sp, #240]", + "mrs x2, elr_el1", + "mrs x3, spsr_el1", + "stp x2, x3, [sp, #256]", + "mrs x2, esr_el1", + "mrs x3, far_el1", + "stp x2, x3, [sp, #272]", + "mov x0, sp", + "bl {exception}", + "brk #0", + ".popsection", + frame = const FRAME_BYTES, + exception = sym exception, +); + +/// Point `VBAR_EL1` at the table. Called by the entry before anything that can fault. +pub fn install() { + // SAFETY: the table is 2 KiB aligned, lives in the kernel image for the + // life of the machine, and every entry ends in `exception`, which never + // returns. The `ISB` makes the new base the one the next exception uses. + unsafe { + core::arch::asm!( + "adrp {t}, toyos_vectors", + "add {t}, {t}, :lo12:toyos_vectors", + "msr vbar_el1, {t}", + "isb", + t = out(reg) _, + options(nostack, preserves_flags), + ); + } +} + +/// Interrupt identifiers the generic drivers program. Inert until stage 4 +/// assigns them to GIC interrupts: every path that would deliver one +/// ([`super::msi_message`], [`super::irqchip::send_self`]) is owed. +#[repr(u8)] +enum Vector { + LogNest = 1, + Hda, + VirtioSound, +} + +pub const HDA_VECTOR: u8 = Vector::Hda as u8; +pub const VIRTIO_SOUND_VECTOR: u8 = Vector::VirtioSound as u8; +pub const LOG_NEST_VECTOR: u8 = Vector::LogNest as u8; + +/// The crash report for a panic, from the frame pointer the panic handler stood on. +pub(crate) fn report_panic(message: &core::panic::PanicInfo, frame: u64) { + crate::alert!("PANIC: {}", message); + log!(" Backtrace:"); + crate::symbols::kernel_backtrace(frame, 20); +} + +/// Whether the interrupted context an `SPSR` describes could have taken an +/// interrupt: `SPSR.I` clear. +pub(crate) const fn frame_interrupts_enabled(spsr: u64) -> bool { + spsr & (1 << 7) == 0 +} + +/// Nothing to report: the vectors run on the stack they interrupted. +pub(crate) fn report_fault_stack() {} + +pub(crate) fn try_recover_from_panic() -> ! { + owed!("recovering a syscall's panic", "stage 7") +} + +pub fn kernel_exit_to_user_check() { + owed!("the return to user mode", "stage 7") +} + +pub(crate) fn log_unclaimed() { + owed!("the interrupt controller", "stage 4") +} + +#[cfg(feature = "test-actuators")] +pub(crate) fn provoke_double_fault() -> ! { + owed!("a fault on the exception stack", "stage 4") +} diff --git a/kernel/src/arch/aarch64/watchdog.rs b/kernel/src/arch/aarch64/watchdog.rs new file mode 100644 index 00000000000..628608537fc --- /dev/null +++ b/kernel/src/arch/aarch64/watchdog.rs @@ -0,0 +1,16 @@ +//! The platform watchdog: on an ACPI Arm machine, an SBSA generic watchdog +//! the GTDT names, the port's stage 6. + +use crate::drivers::pci::PciDevice; + +pub fn init(_devices: &[PciDevice]) { + owed!("the platform watchdog", "no stage yet") +} + +pub fn feed(_now: u64) { + owed!("the platform watchdog", "no stage yet") +} + +pub fn disarm() { + owed!("the platform watchdog", "no stage yet") +} diff --git a/kernel/src/arch/mod.rs b/kernel/src/arch/mod.rs index 854092238c9..c89f3178ead 100644 --- a/kernel/src/arch/mod.rs +++ b/kernel/src/arch/mod.rs @@ -1,99 +1,18 @@ -#![warn(clippy::undocumented_unsafe_blocks)] //! The machine, and the only part of this kernel that knows which one. //! -//! Every `unsafe` block here carries a one-line `SAFETY:` comment, enforced by the lint above. -//! [`percpu`] owns every `gs:` access; nothing outside this directory writes one. +//! One implementation per architecture, chosen here at compile time and +//! nowhere else: generic code names `crate::arch::…` and never an +//! architecture, and a `cfg` on `target_arch` appears in this file and inside +//! the implementation it selects. Every item one implementation exports, the +//! other exports with the same meaning — a missing one is a compile error in +//! the architecture that lacks it, not a fallback. -pub mod apic; -pub mod control_regs; -pub mod cpu; -pub mod entry; -pub mod fpu; -pub mod idt; -pub mod mtrr; -pub mod pat; -pub mod percpu; -pub mod smp; -pub mod syscall; -pub mod tlb; +#[cfg(target_arch = "x86_64")] +mod x86_64; +#[cfg(target_arch = "x86_64")] +pub use x86_64::*; -/// One log reservation and its publication, atomic against an interrupt on this CPU. TF is always clear in Ring 0, so this guard leaves it alone. -#[must_use = "dropping the log commit guard reopens interrupts and single-step traps"] -pub(crate) struct LogCommitGuard { - rflags: u64, - // Same-CPU only: keeps this guard `!Send + !Sync`. - _not_send_sync: core::marker::PhantomData<*mut ()>, -} - -impl LogCommitGuard { - pub fn close() -> Self { - let rflags: u64; - // SAFETY: pushfq/pop is balanced; cli touches only RFLAGS — one uninterruptible read-and-clear of IF. - unsafe { - // No `nomem`: the clobber keeps shard selection and publication on the closed side. - core::arch::asm!( - "pushfq", - "pop {saved}", - saved = out(reg) rflags, - ); - // Skips `cli` under the `log-unbracketed-reserve` actuator, to stage that defect. - if crate::actuator::log_unbracketed_reserve() { - return Self { rflags, _not_send_sync: core::marker::PhantomData }; - } - core::arch::asm!("cli"); - } - Self { rflags, _not_send_sync: core::marker::PhantomData } - } -} - -impl Drop for LogCommitGuard { - fn drop(&mut self) { - // SAFETY: `close`'s argument, restored on the CPU that captured it. - unsafe { - // No `nomem`: the slot store must stay before this reopens IF. - core::arch::asm!( - "push {saved}", - "popfq", - saved = in(reg) self.rflags, - ); - } - } -} - -/// Adds one to `counter`, atomic against an interrupt on this CPU, and answers the value before the add. -/// # Safety: `counter` is written by no other CPU; `guard` covers the shard selection that owns it. -#[inline(always)] -pub unsafe fn percpu_fetch_add( - counter: &core::sync::atomic::AtomicU64, - _guard: &LogCommitGuard, -) -> u64 { - // Under `log-shared-reservation`, stage a load/store race instead of the `xadd` below. - if crate::actuator::log_shared_reservation() { - let previous = counter.load(core::sync::atomic::Ordering::Relaxed); - if crate::log::nested::inject() { - // SAFETY: `sti`/`cli` each write one `RFLAGS` bit and touch no memory. - unsafe { - core::arch::asm!("sti"); - for _ in 0..256 { - core::hint::spin_loop(); - } - core::arch::asm!("cli"); - } - } - counter.store(previous + 1, core::sync::atomic::Ordering::Relaxed); - return previous; - } - - let previous: u64; - // Not `AtomicU64::fetch_add`: its locked xadd is costly under QEMU TCG emulation. - // SAFETY: `counter.as_ptr()` is live; unlocked `xadd` retires whole, atomic against an interrupt here. - unsafe { - // No `preserves_flags`: `xadd` changes arithmetic flags. - core::arch::asm!( - "xadd [{ptr}], {out}", - ptr = in(reg) counter.as_ptr(), - out = inout(reg) 1u64 => previous, - ); - } - previous -} +#[cfg(target_arch = "aarch64")] +mod aarch64; +#[cfg(target_arch = "aarch64")] +pub use aarch64::*; diff --git a/kernel/src/arch/apic.rs b/kernel/src/arch/x86_64/apic.rs similarity index 66% rename from kernel/src/arch/apic.rs rename to kernel/src/arch/x86_64/apic.rs index 103a6ea9614..d6b2b703d99 100644 --- a/kernel/src/arch/apic.rs +++ b/kernel/src/arch/x86_64/apic.rs @@ -2,7 +2,7 @@ use core::sync::atomic::{AtomicBool, AtomicU32, Ordering}; use super::{cpu, percpu}; use crate::log; -use crate::time::{Budget, Delay, Duration, Floor}; +use crate::time::{Delay, Duration, Floor}; /// The local APIC registers and MSRs this file may name. // Every variant is an architectural local-APIC register touching no memory or control transfer, so none can make `Reg::write`'s unsafe wrmsr unsound. @@ -40,6 +40,18 @@ impl Reg { pub const TIMER_VECTOR: u8 = 0x20; +/// Where a device writes a message-signalled interrupt: the local APIC's +/// message window (SDM Vol. 3A §11.11.1). The one spelling of it — the +/// compatibility format below, VT-d's remappable format and VT-d's own fault +/// event all start here. +pub const MSI_DOORBELL: u32 = 0xFEE0_0000; + +/// The compatibility-format message that raises `vector` on the CPU whose APIC +/// ID is `dest`: the destination in address bits 19:12, the vector in the data. +pub fn msi_message(dest: u32, vector: u8) -> (u32, u32) { + (MSI_DOORBELL | (dest << 12), vector as u32) +} + /// Calibrated LAPIC timer ticks per 10ms (computed on BSP, reused by APs). static TIMER_TICKS: AtomicU32 = AtomicU32::new(0); @@ -172,100 +184,13 @@ pub fn arm_perf_nmi() { Reg::LvtPmc.write(0b100 << 8); } -// Time /system/bin/logd gets to durably write the panic report before halt; a Budget (not Bound) because expiry degrades gracefully instead of panicking. -const LOG_FILE_DRAIN: Budget = Budget::of( - Duration::from_millis(500), - "the report reaches the panel and not /log", -); - -// Read by tests/toyos.rs — keep in sync or its drift check fails. -const LOG_DRAIN_EXPIRED: &str = "the report did not reach /log"; - -/// Whether `/log` still owes this boot the report. -// True only before durable_ns passes `want` — logd publishes it after fsync returns, never before. -fn owed(want: u64) -> bool { - crate::log::user::durable_ns() < want -} - -/// Give `/system/bin/logd` a chance to put this report on the stick before the machine stops. -// The panic path never writes /log directly — every lock a write needs may already be held by the panicking thread itself. -fn wait_for_log_file() { - // Skip when serial exists: panic_flush already got the report off the box, and waiting here would only delay the pager. - if crate::drivers::serial::has_console() { - return; - } - // INVARIANT: nothing below runs before both facts this wait rests on hold. - // The deadline is read off the calibrated clock, so an uncalibrated one - // makes it unreachable; and what the wait is owed by is `logd`, which only - // a released machine can run. On a boot that crashes before either — the - // one this whole path exists for on a machine with no serial port — waiting - // costs the seal and buys nothing, so it is skipped and not shortened. - if !crate::clock::calibrated() || !crate::arch::smp::is_ready() { - return; - } - // Sampled once — a sibling still logging on its way down must not be able to push this deadline out indefinitely. - let want = crate::log::read::newest_committed_at_ns(); - if !owed(want) { - return; - } - // Wake siblings first: one may be halted in `sti; hlt` with no timer armed to wake it otherwise. - kick_all_but_self(); - let deadline = crate::clock::now() + LOG_FILE_DRAIN.duration(); - while owed(want) { - if crate::clock::now() >= deadline { - // /log has failed to answer, so fold this into the still-unpainted panel snapshot — the panel is the only channel left. - crate::log!( - "panic: {LOG_DRAIN_EXPIRED} in {}ns; the panel is the only copy", - LOG_FILE_DRAIN.nanos() - ); - crate::drivers::panic_console::refresh_capture(); - return; - } - core::hint::spin_loop(); - } -} - -/// Halt all CPUs: send the halt IPI, flush pending log output, then hold this -/// machine's panel until a key retires the reboot bound or the bound returns it -/// to firmware. -// panic_flush bypasses the log-ring and serial locks — after the halt IPI a wedged holder never releases them, so taking them normally could deadlock. -pub fn halt_all_cpus() -> ! { - // Before the wait and the panel: from here this machine holds a report for - // whoever is in front of it, with `IF` clear and under a bound of its own — - // which to a hard-lockup sample or a deadline poll is indistinguishable from - // a wedge, and is the opposite of one. Both bounds, because a `WEDGED` - // record either of them sealed would replace the report this path exists to - // deliver. - crate::hardlockup::stand_down(); - crate::deadline::stand_down(); - wait_for_log_file(); - // Under the same condition as the wait above: before the machine is - // released no sibling has been sent its `SIPI`, so this addresses CPUs that - // are still waiting for one rather than CPUs that need halting. +/// Send every other CPU the halt IPI, where the machine has released any: before +/// that no sibling has been sent its `SIPI`, so there are only CPUs still waiting +/// for one rather than CPUs that need halting. +pub fn stop_other_cpus() { if X2APIC_ENABLED.load(Ordering::Relaxed) && crate::arch::smp::is_ready() { Reg::Icr.write(0x000C_0000 | 0xFD); } - let bound = crate::panic_reboot::arm(true); - // Folded into the still-unpainted capture only where the panel is this - // boot's only account of itself, exactly as `wait_for_log_file`'s own line - // is: a refresh re-freezes the ring, and `screen_late_panic` reads the - // panel for a record written *after* `capture()` to prove the paint comes - // from the frozen snapshot. A machine with a console gets the arm line on it. - if !crate::drivers::serial::has_console() { - crate::drivers::panic_console::refresh_capture(); - } - // Render before the flush: it can't fail the proven serial channel, and a serial line then proves the paint already finished. - let painted = crate::drivers::panic_console::render(); - // SAFETY: sound only once nothing else will run — the halt IPI is already out. - unsafe { crate::drivers::serial::panic_flush(); } - // Must follow the flush — it's the deepest stack this path reaches. No-op off IST1. - percpu::ist1_report(); - // page_forever runs strictly after the flush: it is an unbounded loop and may only run once the serial report is out. - // Only the CPU that painted watches the bound; the rest halt below, since two CPUs polling port 0x60 would split every scancode. - if painted { - crate::drivers::panic_console::page_forever(bound); - } - super::cpu::halt(); } /// Calibrate the LAPIC timer on the BSP (requires HPET); does not start it. diff --git a/kernel/src/arch/x86_64/barrier.rs b/kernel/src/arch/x86_64/barrier.rs new file mode 100644 index 00000000000..44da0d49a0a --- /dev/null +++ b/kernel/src/arch/x86_64/barrier.rs @@ -0,0 +1,51 @@ +//! The orderings a device's view of memory needs, as x86-64 gives them. +//! +//! Every function here is the contract both architectures implement; on x86-64 +//! each is a compiler barrier and no instruction. TSO keeps a CPU's +//! write-back stores in program order and its loads in program order (SDM +//! Vol. 3A §9.2.2), and a store to an uncacheable register is not reordered +//! with an older store — so what a device sees follows program order once the +//! compiler is held to it. Write-combining memory is the exception, and its +//! one user orders it with its own `sfence`. + +use core::sync::atomic::{compiler_fence, Ordering}; + +/// Every store to memory this CPU made before this call is visible to a +/// device's DMA reads before any store it makes after it — a descriptor before +/// the index that publishes it. +#[inline(always)] +pub fn dma_wmb() { + compiler_fence(Ordering::Release); +} + +/// Every load of device-written memory this CPU makes after this call sees at +/// least what the loads before it saw — a completion's index before the entry +/// it counts. +#[inline(always)] +pub fn dma_rmb() { + compiler_fence(Ordering::Acquire); +} + +/// What an MMIO store is preceded by: [`dma_wmb`], so a register write that +/// starts a device's work comes after the memory that work reads. +#[inline(always)] +pub fn before_mmio_write() { + dma_wmb(); +} + +/// What an MMIO load is followed by: [`dma_rmb`], so memory read after a +/// status register is at least as new as the status. +#[inline(always)] +pub fn after_mmio_read() { + dma_rmb(); +} + +/// Every store this CPU made to the scanout reaches the display before this +/// returns: the scanout is write-combining, and its stores can sit in a buffer +/// with nothing to evict them. `SFENCE` (SDM Vol. 3A §11.3.1) is the only way +/// to drain one. +#[inline(always)] +pub fn scanout_flush() { + // SAFETY: `SFENCE` touches no memory or register. + unsafe { core::arch::asm!("sfence", options(nostack, preserves_flags)) }; +} diff --git a/kernel/src/arch/x86_64/boot.rs b/kernel/src/arch/x86_64/boot.rs new file mode 100644 index 00000000000..b2b3cf98401 --- /dev/null +++ b/kernel/src/arch/x86_64/boot.rs @@ -0,0 +1,141 @@ +//! The x86-64 steps of the boot: the entry the loader jumps to, and what +//! `kernel_main` asks of this architecture at the points where one differs +//! from another. + +use toyos_abi::boot::{KernelArgs, MemoryMapEntry}; + +use super::{apic, control_regs, idt, ioapic, pat, percpu}; +use crate::drivers::acpi::{self, MadtInfo}; +use crate::log; +use crate::mm::Region; + +/// Entry point: the bootloader jumps here at `PHYS_OFFSET` with `rdi = &KernelArgs`, +/// switches to the kernel's own stack, and calls `kernel_main`. +/// # Safety +/// Only the bootloader may call this, fresh from firmware, with `rdi` holding a live [`KernelArgs`]. +#[unsafe(naked)] +#[no_mangle] +pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { + core::arch::naked_asm!( + "mov rax, [rdi + 16]", // kernel_memory_addr + "add rax, [rdi + 32]", // + kernel_stack_addr + "add rax, [rdi + 40]", // + kernel_stack_size + "movabs rbx, {phys_offset}", + "add rax, rbx", + "mov rsp, rax", + "call {kernel_main}", + phys_offset = const crate::PHYS_OFFSET, + kernel_main = sym crate::kernel_main, + ); +} + +/// Before the panel is armed. +/// +/// **Before the panel and not after it**: the loader maps the scanout +/// uncacheable, and `panic_console::arm`'s own record is the panel's first +/// paint, so there is no window between arming the panel and painting +/// through it in which to establish a memory type. The write alone, and it +/// logs nothing: the read-back is `pat::check`, in [`after_console`], where a +/// refusal has a channel to reach. +/// The ACPI tables this architecture decodes: the MADT for its CPUs and I/O +/// APICs, the FADT for reset, soft-off and the century register, the HPET for +/// the clock, the MCFG for ECAM and the DMAR for the IOMMU. +pub const ACPI_TABLES: &[&[u8; 4]] = &[b"APIC", b"FACP", b"HPET", b"MCFG", b"DMAR"]; + +pub fn before_panel() { + pat::init(); +} + +/// Once the console and the boot parameter exist. +pub fn after_console(_args: &KernelArgs, _maps: &[MemoryMapEntry]) { + // The IDT loads inside `interrupts`, much later: a fault here would reach + // firmware's handlers, so the actuator is refused by name instead. + if crate::actuator::test_early_fault() { + panic!("test-early-fault: x86-64 has no vectors of its own this early"); + } + // After actuator::init, whose table the `control-regs-bench` probe inside + // this call reads. `pat::init` restored the `CR0` it found, so a firmware + // `CD` — which would make every mapping uncacheable whatever the PAT + // says — ends here. + control_regs::init_cr0(0); + + // The read-back `pat::init` owes, on a boot that now has three channels to + // carry a refusal. + pat::check(); + + log!("PAT: IA32_PAT={:#018x}, entry {} = {}", + pat::msr(), pat::WC_ENTRY, pat::entry_name(pat::WC_ENTRY)); +} + +/// Physical memory only this architecture's boot uses, kept from the +/// allocator: the AP trampoline page. +pub fn reserved() -> Region { + Region { start: 0x8000, end: 0x9000 } +} + +/// What the boot learns bringing interrupts up and hands later steps. +pub struct Platform { + madt: MadtInfo, +} + +/// Interrupt delivery, this CPU's per-CPU block and the syscall gate. +pub fn interrupts(rsdp_addr: u64) -> Platform { + // `init_bsp` loads the IDT partway through, as early as this CPU's `gs:` + // allows: a fault in any later phase then diagnoses instead of stopping in + // a handler the firmware left behind. + let madt = acpi::parse_madt(rsdp_addr).expect("ACPI: MADT not found"); + // Off the same tables as the MADT, and before the IDT below makes a panic + // reportable: a panic that can be reported but not ended leaves the machine + // holding its panel for a hand that may not be in the room. + acpi::init_reset(rsdp_addr); + apic::init(); + percpu::init_bsp(apic::id()); + ioapic::init(&madt); + idt::enable_interrupts(); + super::syscall::init(); + Platform { madt } +} + +/// The clock: the TSC, calibrated against the HPET, and the CMOS wall clock. +pub fn clock(args: &KernelArgs) { + // HPET clock — enables profiling for everything from here on + let hpet_base = acpi::find_hpet_base(args.rsdp_addr) + .expect("ACPI: HPET not found"); + super::hpet::calibrate_counter(hpet_base); + // Century register and time zone both come from ACPI/firmware, not the RTC's own registers. + let century_reg = match acpi::rtc_century_register(args.rsdp_addr) { + Ok(reg) => reg, + Err(e) => { + log!("ACPI: the FADT is unreadable ({e:?}), so where the RTC keeps its century is unknown too"); + None + } + }; + crate::clock::init_wall(century_reg, args.rtc_utc_offset()); +} + +/// The per-CPU timer, once the clock converts its bound. +pub fn timer() { + apic::init_timer(); +} + +/// The platform's own devices that are not PCI functions. +pub fn platform_devices(rsdp_addr: u64) { + super::i8042::init(rsdp_addr); +} + +/// Every other CPU, running. +pub fn start_other_cpus(platform: &Platform, args: &KernelArgs) { + super::smp::boot_aps(&platform.madt, args.boot_pml4_addr); +} + +/// The interrupt-controller selftests an actuator asks for, once the timer ticks. +#[cfg(feature = "boot-actuators")] +pub fn interrupt_selftests() { + // Needs interrupts on and the timer already ticking: its last assertion is that the interrupt after the spurious one arrives. + if crate::actuator::lapic_spurious_selftest() { + idt::spurious::selftest(); + } + if crate::actuator::unclaimed_vector_selftest() { + idt::unclaimed::selftest(); + } +} diff --git a/kernel/src/arch/x86_64/cache.rs b/kernel/src/arch/x86_64/cache.rs new file mode 100644 index 00000000000..c3f23d10601 --- /dev/null +++ b/kernel/src/arch/x86_64/cache.rs @@ -0,0 +1,33 @@ +//! Writing memory back out of the caches, for the one reader that is not a +//! CPU: DRAM across a reset. + +/// `CLFLUSH`'s line: `CPUID.01H:EBX[15:8] × 8`, which every x86-64 part +/// reports as 64. +const LINE: u64 = 64; + +/// Every line of `[at, at + len)` written back to memory and invalidated, in +/// this CPU's caches and every other CPU's — `CLFLUSH` is coherent across the +/// machine — before this returns. +/// +/// A reset invalidates the caches without writing them back, so a byte that +/// must outlive one has to reach DRAM first. +pub fn write_back(at: u64, len: usize) { + let mut line = at & !(LINE - 1); + while line < at + len as u64 { + // SAFETY: `CLFLUSH` writes back and invalidates the line containing the + // address and touches nothing else; the caller names memory it owns, + // and the instruction faults on nothing a mapped canonical address can + // be. Not privileged, and present on every x86-64 part. + unsafe { + core::arch::asm!( + "clflush [{addr}]", + addr = in(reg) line as *const u8, + options(nostack, preserves_flags), + ); + } + line += LINE; + } + // SAFETY: `SFENCE` orders those writebacks ahead of whatever ends this + // machine; it touches no memory or register. + unsafe { core::arch::asm!("sfence", options(nostack, preserves_flags)) }; +} diff --git a/kernel/src/arch/x86_64/console_uart.rs b/kernel/src/arch/x86_64/console_uart.rs new file mode 100644 index 00000000000..cdc1b0d66cc --- /dev/null +++ b/kernel/src/arch/x86_64/console_uart.rs @@ -0,0 +1,71 @@ +//! The console UART: the 16550 at COM1's I/O ports, where every PC this +//! kernel boots puts one if it has one. + +use super::cpu::{inb, outb}; +use crate::log; + +const PORT: u16 = 0x3f8; // COM1 + +/// Line status register bits: a received byte waits, and the transmitter +/// holding register is empty. +const LSR: u16 = PORT + 5; +const LSR_DATA_READY: u8 = 0x01; +const LSR_THR_EMPTY: u8 = 0x20; + +/// Program the 16550 and answer whether it is there: hardware with no SuperIO +/// reads 0xFF on every access, indistinguishable from a ready UART, so a +/// loopback probe latches the answer once. +// Every register is `PORT + n`; the identity op keeps that pattern uniform +// across all eight lines instead of special-casing the data register. +#[allow(clippy::identity_op)] +pub fn init(_rsdp_addr: u64) -> bool { + // SAFETY: `outb`/`inb` require the caller to own the port and the byte; + // every port here is `PORT + n` for `n` in 0..=4, inside COM1's own + // register block, and the writes are the 16550's documented init sequence. + // Order matters: DLAB must precede the divisor writes and loopback mode + // must precede the probe, or the sequence misprograms the chip. + let loopback = unsafe { + outb(PORT + 1, 0x00); // Disable all interrupts + outb(PORT + 3, 0x80); // Enable DLAB (set baud rate divisor) + outb(PORT + 0, 0x03); // Set divisor to 3 (lo byte) 38400 baud + outb(PORT + 1, 0x00); // (hi byte) + outb(PORT + 3, 0x03); // 8 bits, no parity, one stop bit + outb(PORT + 2, 0xC7); // Enable FIFO, clear them, with 14-byte threshold + outb(PORT + 4, 0x0B); // IRQs enabled, RTS/DSR set + outb(PORT + 4, 0x1E); // Set in loopback mode, test the serial chip + outb(PORT + 0, 0xAE); // Test serial chip (send byte 0xAE and check if serial returns same byte) + let seen = inb(PORT + 0); + outb(PORT + 4, 0x0F); // Normal operation mode + seen + }; + // Logs the raw byte, not just the verdict: distinguishes "no SuperIO" + // (0xFF) from a wrong response and a right chip at the wrong port. + log!( + "serial: 16550 loopback read {:#04x} ({})", + loopback, + if loopback == 0xAE { "present" } else { "absent or wrong port" } + ); + loopback == 0xAE +} + +/// Whether a received byte waits. +pub fn rx_ready() -> bool { + inb(LSR) & LSR_DATA_READY != 0 +} + +/// The received byte; only after [`rx_ready`] said one waits. +pub fn read_byte() -> u8 { + inb(PORT) +} + +/// Whether the transmitter will take a byte. +pub fn tx_ready() -> bool { + inb(LSR) & LSR_THR_EMPTY != 0 +} + +/// Put one byte in the transmitter; only after [`tx_ready`], or the byte may be lost. +pub fn write_byte(byte: u8) { + // SAFETY: `outb` requires ownership of the port and the byte; `PORT` is + // COM1's own data register, and the byte is console output only. + unsafe { outb(PORT, byte) }; +} diff --git a/kernel/src/arch/control_regs.rs b/kernel/src/arch/x86_64/control_regs.rs similarity index 100% rename from kernel/src/arch/control_regs.rs rename to kernel/src/arch/x86_64/control_regs.rs diff --git a/kernel/src/arch/cpu.rs b/kernel/src/arch/x86_64/cpu.rs similarity index 70% rename from kernel/src/arch/cpu.rs rename to kernel/src/arch/x86_64/cpu.rs index 7fafa16a443..2353b0f5868 100644 --- a/kernel/src/arch/cpu.rs +++ b/kernel/src/arch/x86_64/cpu.rs @@ -77,7 +77,7 @@ pub fn rdrand() -> Option { } #[inline] -pub fn read_rsp() -> u64 { +pub fn stack_pointer() -> u64 { let rsp: u64; // SAFETY: register-to-register mov into the declared output; nostack holds because rsp is read, not used. unsafe { @@ -107,7 +107,7 @@ pub fn df_witness(site: &str) { } // SAFETY: the observation is already made; only core::fmt and the log follow, which must not run with DF set. unsafe { asm!("cld", options(nomem, nostack)) }; - crate::hw::report_contexts(read_rsp(), None); + crate::hw::report_contexts(stack_pointer(), None); panic!( "DF WITNESS: cpu{} reached {site} with the direction flag set. \ `compiler_builtins::mem::memmove`'s overlapping-copy path holds it across \ @@ -356,3 +356,129 @@ pub fn io_wait() { // SAFETY: port 0x80 is the unused POST diagnostic port, so this commands nothing. unsafe { outb(0x80, 0) }; } + +/// The CPU's free-running counter: the TSC, which counts from reset. +#[inline] +pub fn counter() -> u64 { + rdtsc() +} + +/// This function's caller's frame pointer: `rbp`, which `-Cforce-frame-pointers=yes` makes one. +#[inline(always)] +pub fn frame_pointer() -> u64 { + let rbp: u64; + // SAFETY: register-to-register mov only. + unsafe { asm!("mov {}, rbp", out(reg) rbp, options(nomem, nostack, preserves_flags)) }; + rbp +} + +/// Raise the architecture's undefined-instruction exception here: `ud2`, whose `#UD` the IDT catches as the kernel's own fault. +pub fn undefined_instruction() { + // SAFETY: ud2 reads and writes nothing and raises #UD, caught by the installed IDT. + unsafe { asm!("ud2", options(nomem, nostack)) }; +} + +/// The TSC's frequency in hertz as CPUID *states* it, for the one caller that +/// may run before the clock is calibrated — the panic path, which has to bound a wait on a +/// machine that never reached the HPET. Nothing calibrates against it and no +/// third source is guessed at: a CPU that states neither leaf answers `None` +/// and its caller says so rather than inventing a rate. +pub fn stated_counter_hz() -> Option { + let max_leaf = cpuid(0, 0).0; + tsc_hz_from( + (max_leaf >= 0x15).then(|| cpuid(0x15, 0)), + (max_leaf >= 0x16).then(|| cpuid(0x16, 0)), + ) +} + +/// SDM Vol. 2A, CPUID leaf 15H: EAX is the denominator and EBX the numerator of +/// the core crystal's ratio to the TSC, ECX the crystal's hertz — any of the +/// three reading zero means the leaf states nothing. Leaf 16H's EAX is the +/// processor base frequency in MHz, which an invariant TSC counts at. +const fn tsc_hz_from( + leaf15: Option<(u32, u32, u32, u32)>, + leaf16: Option<(u32, u32, u32, u32)>, +) -> Option { + if let Some((denominator, numerator, crystal_hz, _)) = leaf15 { + if denominator != 0 && numerator != 0 && crystal_hz != 0 { + return Some(crystal_hz as u64 * numerator as u64 / denominator as u64); + } + } + if let Some((base_mhz, _, _, _)) = leaf16 { + if base_mhz != 0 { + return Some(base_mhz as u64 * 1_000_000); + } + } + None +} + +const _: () = { + // The ratio, then the fall-through to the base frequency when the crystal + // is not enumerated, then the CPU that states neither. + assert!(matches!(tsc_hz_from(Some((2, 4, 25_000_000, 0)), None), Some(50_000_000))); + assert!(matches!(tsc_hz_from(Some((0, 0, 0, 0)), Some((2_400, 0, 0, 0))), Some(2_400_000_000))); + assert!(tsc_hz_from(None, Some((0, 0, 0, 0))).is_none()); + assert!(tsc_hz_from(None, None).is_none()); +}; + + +/// The thread pointer this CPU is running with: the FS base, which user TLS is addressed from. +#[inline] +pub fn thread_pointer() -> u64 { + read_fs_base() +} + +/// Leave the current stack for good and run `func` on the one ending at `top`, +/// with a zeroed frame chain so a panic there backtraces instead of walking off +/// the top. +/// # Safety +/// Nothing on the current stack is live past this call, and `top` is the end of +/// a stack this CPU owns. +pub unsafe fn run_on_stack(top: u64, func: extern "C" fn() -> !) -> ! { + // SAFETY: the caller's contract; `push` leaves `rsp` where a function entry + // expects it. + unsafe { + asm!( + "mov rsp, {sp}", + "xor ebp, ebp", + "push rbp", + "jmp {func}", + sp = in(reg) top, + func = in(reg) func as *const () as usize, + options(noreturn), + ); + } +} + +/// `df-witness-mutate`'s staging: set `DF` one instruction before the reader +/// that must refuse it. Inlined, so `DF` crosses no `ret` to get there. +#[cfg(feature = "df-witness-mutate")] +#[inline(always)] +pub fn df_witness_mutate() { + // SAFETY: a build that exists to stage the defect, and the reader after it + // panics before any `rep movs` can run. Nothing runs in between, so no + // string op ever executes with it set. + unsafe { asm!("std", options(nomem, nostack)) }; +} + +/// This CPU's APIC id, from CPUID: the id the hardware gives it, readable +/// before any per-CPU state exists. +/// +/// Not `rdmsr(IA32_X2APIC_APICID)`: that MSR is `#GP` before `irqchip::init_ap` +/// has run, and a panic an AP takes before then must not fault inside the +/// reentry guard. +pub fn hardware_id() -> u32 { + let (max_leaf, _, _, _) = cpuid(0, 0); + for leaf in [0x1F, 0x0B] { + if max_leaf >= leaf { + let (_, ebx, _, edx) = cpuid(leaf, 0); + // SDM Vol. 2A, CPUID leaf 0BH: EBX[15:0] == 0 means unimplemented, + // not id 0, so the leaf-1 fallback below must still run. + if ebx & 0xFFFF != 0 { + return edx; + } + } + } + let (_, ebx, _, _) = cpuid(1, 0); + ebx >> 24 +} diff --git a/kernel/src/arch/x86_64/entropy.rs b/kernel/src/arch/x86_64/entropy.rs new file mode 100644 index 00000000000..9f8ca1149a0 --- /dev/null +++ b/kernel/src/arch/x86_64/entropy.rs @@ -0,0 +1,20 @@ +//! The CPU's own random source: `RDRAND`. + +use super::cpu; + +/// How often a caller may ask before taking "no data" as the answer. +pub const ATTEMPTS: u32 = cpu::RDRAND_ATTEMPTS; + +/// Whether this CPU can draw at all, or why not. +pub fn available() -> Result<(), &'static str> { + if cpu::has_rdrand() { + Ok(()) + } else { + Err("CPUID.01H:ECX[30] is clear, so this CPU has no RDRAND") + } +} + +/// One drawn `u64`, or `None` when the source had nothing to give; never waits. +pub fn draw() -> Option { + cpu::rdrand() +} diff --git a/kernel/src/arch/entry.rs b/kernel/src/arch/x86_64/entry.rs similarity index 58% rename from kernel/src/arch/entry.rs rename to kernel/src/arch/x86_64/entry.rs index 2335c239bfe..e2187ed1fc4 100644 --- a/kernel/src/arch/entry.rs +++ b/kernel/src/arch/x86_64/entry.rs @@ -159,6 +159,105 @@ macro_rules! restore_user_state { }; } -pub(crate) use { - initial_user_state, restore_user_state, ring3_naked_asm, ring3_trampoline_asm, save_user_state, -}; +pub(crate) use {restore_user_state, ring3_naked_asm, save_user_state}; + +/// Entry point for new processes, reached through `context_switch`'s `ret`. r12 = entry point, r13 = user stack pointer. +// State loads after `unlock`, not before: earlier, registers hold the previous tenant's kernel context. +#[unsafe(naked)] +pub(crate) extern "C" fn process_start() { + ring3_trampoline_asm!( + "push r12", + "push r13", + "call {unlock}", + "pop r13", + "pop r12", + initial_user_state!(), + "push {user_ss}", + "push r13", // RSP: user stack + "push 0x202", // RFLAGS: IF=1 + "push {user_cs}", + "push r12", // RIP: entry point + "iretq", + unlock = sym crate::sched::driver::trampoline_entry, + user_ss = const crate::arch::percpu::USER_DS, + user_cs = const crate::arch::percpu::USER_CS, + ); +} + +/// Entry point for new threads. r14 carries the argument, which lands in rdi. +#[unsafe(naked)] +pub(crate) extern "C" fn thread_start() { + ring3_trampoline_asm!( + "push r12", + "push r13", + "push r14", + "call {unlock}", + "pop r14", + "pop r13", + "pop r12", + initial_user_state!(), + "mov rdi, r14", + "sub r13, 8", // ABI: RSP must be 16n+8 at function entry + "push {user_ss}", + "push r13", + "push 0x202", + "push {user_cs}", + "push r12", + "iretq", + unlock = sym crate::sched::driver::trampoline_entry, + user_ss = const crate::arch::percpu::USER_DS, + user_cs = const crate::arch::percpu::USER_CS, + ); +} + +/// Entry point for a kernel thread: r12 = body, r14 = argument. Never reaches Ring 3. +// The `sti` is load-bearing: `alloc_kernel_stack` leaves `IF` clear, and `trampoline_entry` requires it clear on entry. +#[unsafe(naked)] +pub(crate) extern "C" fn kernel_start() { + core::arch::naked_asm!( + "call {unlock}", + "sti", + "mov rdi, r14", + "call r12", + "call {returned}", + unlock = sym crate::sched::driver::trampoline_entry, + returned = sym kernel_thread_returned, + ); +} + +/// What [`kernel_start`] calls when a kernel thread's body returns: panics rather than halting silently. +extern "C" fn kernel_thread_returned() -> ! { + panic!("a kernel thread's body returned; nothing runs on this stack now"); +} + + +/// Lay out, just below `top`, the frame `context_switch` restores a new +/// context from, and answer the stack pointer that names it: `trampoline` is +/// where its `ret` lands, with the entry, the stack and the argument where the +/// trampolines read them (`r12`, `r13`, `r14`). +/// # Safety +/// `top` is the end of a fresh kernel stack nothing else references, at least +/// 64 bytes deep. +pub unsafe fn initial_frame( + top: u64, + trampoline: unsafe extern "C" fn(), + user_entry: u64, + user_sp: u64, + arg: u64, +) -> u64 { + // Layout must match context_switch's pop sequence: pushfq, rbp..r15, return address. + let frame = (top - 8 * 8) as *mut u64; + // SAFETY: the eight writes cover `[frame, frame + 64)`, the top 64 bytes of + // the stack the caller owns. + unsafe { + *frame.add(0) = 0; // r15 + *frame.add(1) = arg; // r14 + *frame.add(2) = user_sp; // r13 + *frame.add(3) = user_entry; // r12 + *frame.add(4) = 0; // rbx + *frame.add(5) = 0; // rbp + *frame.add(6) = 0x002; // RFLAGS (IF=0, AC=0) + *frame.add(7) = trampoline as usize as u64; // return address + } + frame as u64 +} diff --git a/kernel/src/arch/fpu.rs b/kernel/src/arch/x86_64/fpu.rs similarity index 100% rename from kernel/src/arch/fpu.rs rename to kernel/src/arch/x86_64/fpu.rs diff --git a/kernel/src/arch/x86_64/hpet.rs b/kernel/src/arch/x86_64/hpet.rs new file mode 100644 index 00000000000..5a49311540a --- /dev/null +++ b/kernel/src/arch/x86_64/hpet.rs @@ -0,0 +1,97 @@ +//! The TSC, calibrated against the HPET: x86-64's free-running counter has no +//! stated rate every part can be trusted for, so the boot measures it. + +use crate::log; +use crate::mm::paging::MmioPolicy; +use crate::time::{Delay, Duration}; + +use super::cpu; + +const HPET_CAP: u64 = 0x000; +const HPET_CFG: u64 = 0x010; +const HPET_COUNTER: u64 = 0x0F0; + +/// Measure the TSC against the HPET at `hpet_base` and start the clock on it. +pub fn calibrate_counter(hpet_base: u64) { + let hpet = crate::mm::paging::map_mmio(hpet_base, 0x1000, MmioPolicy::Uncacheable); + + let cap = hpet.read_u64(HPET_CAP); + let hpet_period_fs = cap >> 32; + assert!(hpet_period_fs > 0, "HPET: invalid counter period"); + + let cfg = hpet.read_u64(HPET_CFG); + hpet.write_u64(HPET_CFG, cfg | 1); + + const CALIBRATION: Delay = Delay::to_measure( + Duration::from_millis(50), + "TSC ticks counted against the HPET; longer is a better ratio and boot time is what it costs", + ); + let calibration_ns = CALIBRATION.nanos(); + let calibration_hpet_ticks = calibration_ns * 1_000_000 / hpet_period_fs; + + let hpet_start = hpet.read_u64(HPET_COUNTER); + let tsc_start = cpu::counter(); + let hpet_target = hpet_start + calibration_hpet_ticks; + log!( + "clock: HPET at {:#x} enabled, period={}fs, counter reads {}, calibrating over {} ticks", + hpet_base, + hpet_period_fs, + hpet_start, + calibration_hpet_ticks, + ); + + // A main counter that does not advance would spin here forever, and this is + // the boot's last wait before it has a clock: the only unit available to + // bound it is the TSC's own, so the budget is the calibration converted at + // a frequency no x86-64 part reaches, which makes it an over-estimate of + // the cycles the calibration can legitimately take on any machine. + const TSC_CEILING_HZ: u64 = 10_000_000_000; + // Times two, so a machine merely slower than the ceiling is not refused for it. + let stall_budget_cycles = 2 * calibration_ns * (TSC_CEILING_HZ / 1_000_000_000); + while hpet.read_u64(HPET_COUNTER) < hpet_target { + assert!( + cpu::counter().wrapping_sub(tsc_start) <= stall_budget_cycles, + "clock: the HPET main counter at {:#x} did not reach {} in {} TSC cycles (it started \ + at {} and reads {}), so this machine offers no clock to calibrate against", + hpet_base, + hpet_target, + stall_budget_cycles, + hpet_start, + hpet.read_u64(HPET_COUNTER), + ); + } + let tsc_end = cpu::counter(); + let hpet_end = hpet.read_u64(HPET_COUNTER); + + let hpet_elapsed_fs = (hpet_end - hpet_start) as u128 * hpet_period_fs as u128; + let tsc_delta = tsc_end - tsc_start; + let tsc_period_fs = (hpet_elapsed_fs / tsc_delta as u128) as u64; + + crate::clock::set_counter(tsc_start, tsc_period_fs); + + let tsc_freq_mhz = 1_000_000_000_000_000u64 / tsc_period_fs / 1_000_000; + log!("TSC: {}MHz (period={}fs, calibrated over {}ms)", tsc_freq_mhz, tsc_period_fs, calibration_ns / 1_000_000); + + // **The one cross-source check this machine offers.** Everything else the + // kernel times is derived from the measurement just taken, so it could only + // agree with itself; CPUID 15H/16H is the part's own statement of the same + // frequency, arrived at by neither the HPET nor this counting loop, and the + // parts-per-million between the two is what a metal profile can hold a + // ceiling against. + let measured_hz = 1_000_000_000_000_000u64 / tsc_period_fs; + match cpu::stated_counter_hz() { + Some(stated) => { + let apart = measured_hz.abs_diff(stated); + log!( + "clock: TSC measured {measured_hz}Hz against the HPET, CPUID states {stated}Hz, \ + {}ppm apart", + apart * 1_000_000 / stated, + ); + } + None => log!( + "clock: TSC measured {measured_hz}Hz against the HPET; CPUID leaves 15H and 16H \ + stating no frequency, so nothing independent confirms it" + ), + } +} + diff --git a/kernel/src/hw.rs b/kernel/src/arch/x86_64/hw.rs similarity index 92% rename from kernel/src/hw.rs rename to kernel/src/arch/x86_64/hw.rs index 21fd5c78fcf..902433dee72 100644 --- a/kernel/src/hw.rs +++ b/kernel/src/arch/x86_64/hw.rs @@ -13,7 +13,7 @@ use toyos_sched::hw::{CpuId, Hw, Kicker, Machine, Nanos, TraceEvent}; use toyos_sched::task::{TaskAccounting, TaskKey}; use crate::arch::{apic, cpu, percpu}; -use crate::sched::driver::context_switch; +use super::switch::context_switch; use crate::sched::payload::{KernelCtx, KernelPayload}; /// The one instance; zero-sized, holds no per-CPU state. @@ -26,35 +26,6 @@ pub fn now_ns() -> u64 { HW.now().0 } -/// RAII interrupt gate: restores the caller's `IF` rather than setting it, so nesting inside an already-closed region is safe. -#[must_use = "the interrupt gate closes when the guard drops"] -pub struct IrqGuard { - rflags: u64, -} - -impl IrqGuard { - pub fn close() -> Self { - let rflags: u64; - // SAFETY: touches only RFLAGS and one pushed-and-popped stack slot; `cli` cannot fail in - // Ring 0 — reading the flags and closing them must be one uninterruptible sequence, which - // two safe calls could not guarantee. - unsafe { - asm!("pushfq", "pop {}", "cli", out(reg) rflags, options(nomem)); - } - Self { rflags } - } -} - -impl Drop for IrqGuard { - fn drop(&mut self) { - // SAFETY: `rflags` is the word this guard's own `close` read out of `RFLAGS` on this CPU — - // restoring it is not `sti`, and `arch::cpu` has no safe primitive for that. - unsafe { - asm!("push {}", "popfq", in(reg) self.rflags, options(nomem)); - } - } -} - impl Kicker for KernelHw { fn kick(&self, target: CpuId) { apic::kick_cpu(target.0); @@ -62,7 +33,7 @@ impl Kicker for KernelHw { } impl Machine for KernelHw { - type IrqGuard = IrqGuard; + type IrqGuard = crate::arch::IrqGuard; fn now(&self) -> Nanos { Nanos(crate::clock::nanos_since_boot()) @@ -79,8 +50,8 @@ impl Machine for KernelHw { apic::stop_timer(); } - fn irq_guard(&self) -> IrqGuard { - IrqGuard::close() + fn irq_guard(&self) -> crate::arch::IrqGuard { + crate::arch::IrqGuard::close() } fn halt(&self) { @@ -187,15 +158,7 @@ pub fn report_contexts(rsp: u64, subject: Option) { ), } } - if let Some((used, of)) = crate::sched::driver::stack_high_water() { - crate::log!(" Task kernel stacks: deepest {used} of {of} bytes"); - } - if let Some((sweeps, records, overflowed)) = crate::mm::sweep_stats() { - crate::log!( - " Heap sweeps: {sweeps} run, {records} live bands on the last walk{}", - if overflowed { ", and the page table filled — the walk is incomplete" } else { "" }, - ); - } + crate::mm::report_on_crash(); } /// Panics before the wild `ret` would restore register state that makes the failure unnameable. @@ -319,7 +282,7 @@ fn switch_witness_capture(ctx: &KernelCtx, token: &RunToken, rsp: /// Compares the frame about to be popped against the one [`check_switch_frame`] validated. /// # Safety -/// Must run from [`crate::sched::driver::context_switch`], with `rsp` equal to the live stack +/// Must run from [`super::switch::context_switch`], with `rsp` equal to the live stack /// pointer and this CPU's shadow already filled by [`switch_witness_capture`]. #[cfg(feature = "switch-witness")] pub(crate) unsafe extern "C" fn switch_witness_verify(rsp: u64) { @@ -484,13 +447,13 @@ impl Hw for KernelHw { #[cfg(feature = "boot-actuators")] crate::heartbeat::note_dispatch(); percpu::set_kernel_stack(incoming.kernel_stack_top); - incoming.cr3.activate(); + incoming.root.activate(); cpu::write_fs_base(incoming.fs_base); } // idle's stack top is per-CPU, unknowable at boot-time init, so it is read here instead. None => { percpu::set_kernel_stack(percpu::idle_stack_top()); - incoming.cr3.activate(); + incoming.root.activate(); } } RUNNING_CTX[percpu::cpu_id() as usize] diff --git a/kernel/src/drivers/i8042/mod.rs b/kernel/src/arch/x86_64/i8042/mod.rs similarity index 99% rename from kernel/src/drivers/i8042/mod.rs rename to kernel/src/arch/x86_64/i8042/mod.rs index c7c890777be..44abd8fc377 100644 --- a/kernel/src/drivers/i8042/mod.rs +++ b/kernel/src/arch/x86_64/i8042/mod.rs @@ -488,7 +488,7 @@ fn buffer_full(status: u8) -> bool { /// Rust half of the pin-interrupt handler. Read the module doc before adding /// anything to it. pub extern "sysv64" fn handler() { - crate::irq_census::irq_took!(I8042); + crate::arch::percpu::irq_took!(I8042); let timestamp = crate::clock::nanos_since_boot(); // No compare-exchange: this handler cannot nest, so there's no second writer. let first = FIRST_IRQ_NS.load(Ordering::Relaxed) == 0; @@ -607,7 +607,7 @@ pub fn service() { && is_irq_cpu() { SPLIT_RESCUED.store(true, Ordering::Relaxed); - let _irq = crate::hw::IrqGuard::close(); + let _irq = crate::arch::IrqGuard::close(); handler_poll(); } if has_bytes() { @@ -1043,7 +1043,7 @@ fn aux_command(bytes: &[u8], deadline: u64) -> bool { fn aux_reenable() { AUX_RESET_PENDING.store(false, Ordering::Relaxed); let ok = { - let _irq = crate::hw::IrqGuard::close(); + let _irq = crate::arch::IrqGuard::close(); let budget = deadline(ms(AUX_REENABLE)); // Masking the line doesn't stop the device: port 1 is disabled so a // stray keystroke mid-handshake can't be consumed as the aux ack. diff --git a/kernel/src/drivers/i8042/tally.rs b/kernel/src/arch/x86_64/i8042/tally.rs similarity index 100% rename from kernel/src/drivers/i8042/tally.rs rename to kernel/src/arch/x86_64/i8042/tally.rs diff --git a/kernel/src/arch/idt/device_irq.rs b/kernel/src/arch/x86_64/idt/device_irq.rs similarity index 100% rename from kernel/src/arch/idt/device_irq.rs rename to kernel/src/arch/x86_64/idt/device_irq.rs diff --git a/kernel/src/arch/idt/dma_fault.rs b/kernel/src/arch/x86_64/idt/dma_fault.rs similarity index 90% rename from kernel/src/arch/idt/dma_fault.rs rename to kernel/src/arch/x86_64/idt/dma_fault.rs index ee1e54c72aa..e4ccf7b0cfe 100644 --- a/kernel/src/arch/idt/dma_fault.rs +++ b/kernel/src/arch/x86_64/idt/dma_fault.rs @@ -3,7 +3,7 @@ use super::device_irq::device_irq_entry; // Reports the IOMMU blocking a device, not device work finished: the one wake // it owes is a claim holder's, which `pcidev::note_fault` posts itself. extern "sysv64" fn dma_fault_handler() { - crate::irq_census::irq_took!(DmaFault); + crate::arch::percpu::irq_took!(DmaFault); crate::iommu::fault_interrupt(); } diff --git a/kernel/src/arch/idt/exceptions.rs b/kernel/src/arch/x86_64/idt/exceptions.rs similarity index 95% rename from kernel/src/arch/idt/exceptions.rs rename to kernel/src/arch/x86_64/idt/exceptions.rs index e7d81edb463..7c095df69c0 100644 --- a/kernel/src/arch/idt/exceptions.rs +++ b/kernel/src/arch/x86_64/idt/exceptions.rs @@ -1,31 +1,13 @@ -use crate::arch::{apic, cpu, syscall, percpu}; +use crate::arch::{cpu, percpu}; +use crate::syscall; use crate::arch::percpu::CpuFaultState; use crate::{alert, log, mm, process, scheduler, symbols}; +use crate::symbols::kernel_backtrace; use toyos_userbound::{blame, Blame, Faulted, Ring}; use super::{Vector, TrapFrame, PF_PRESENT, PF_WRITE, PF_INSTRUCTION_FETCH}; -/// Walk RBP chain for kernel backtrace with symbol resolution. -pub(crate) fn kernel_backtrace(start_rbp: u64, max_frames: usize) { - let mut rbp = start_rbp; - for _ in 0..max_frames { - if rbp == 0 || !rbp.is_multiple_of(8) || !mm::is_kernel_addr(rbp) { break; } - // SAFETY: `rbp` is checked non-zero, 8-aligned and a kernel address, so - // both reads land in the direct map, mapped for the life of the machine. - // - // Not `read_volatile` like `safe_read_kernel`, whose double-fault path - // reads memory another CPU may still be writing: this walks the - // faulting thread's own frame chain from its handler. - let saved_rbp = unsafe { *(rbp as *const u64) }; - // SAFETY: same as above, for the return address one word up. - let return_addr = unsafe { *((rbp + 8) as *const u64) }; - if return_addr == 0 || !mm::is_kernel_addr(return_addr) { break; } - symbols::resolve_kernel_return(return_addr); - rbp = saved_rbp; - } -} - /// Walk RBP chain for user backtrace through page tables. Takes no pid: this /// always backtraces the process running on this CPU. fn user_backtrace(start_rbp: u64, pml4: *const u64, max_frames: usize) { @@ -366,7 +348,7 @@ pub(crate) fn recover_or_halt(blame: Blame) -> ! { } // Kernel fault on the thread's behalf — may hold locks, use try_lock path. Blame::ProcessThroughKernel => try_recover_from_panic(), - Blame::Kernel => apic::halt_all_cpus(), + Blame::Kernel => crate::panic::halt_all_cpus(), } } @@ -470,7 +452,7 @@ pub(super) fn double_fault_handler(frame: &TrapFrame) -> ! { addr += 8; } - apic::halt_all_cpus(); + crate::panic::halt_all_cpus(); } /// #MC halts whichever ring faulted rather than killing a process: there is @@ -481,7 +463,7 @@ pub(super) fn machine_check_handler(frame: &TrapFrame) -> ! { log!("MACHINE CHECK on CPU {}", percpu::cpu_id()); let ctx = ExceptionContext { frame, cr2: 0 }; crash_report(&CrashInfo::Exception(&ctx)); - apic::halt_all_cpus(); + crate::panic::halt_all_cpus(); } /// Returns if the fault was resolved (page mapped in); diverges if fatal. @@ -575,7 +557,7 @@ fn fatal_exception(ctx: &ExceptionContext) -> ! { crate::panic::forget(); syscall::kill_process(-1); } - apic::halt_all_cpus(); + crate::panic::halt_all_cpus(); } crash_report(&CrashInfo::Exception(ctx)); diff --git a/kernel/src/arch/idt/hda.rs b/kernel/src/arch/x86_64/idt/hda.rs similarity index 90% rename from kernel/src/arch/idt/hda.rs rename to kernel/src/arch/x86_64/idt/hda.rs index 56344884d4c..579bebedfc6 100644 --- a/kernel/src/arch/idt/hda.rs +++ b/kernel/src/arch/x86_64/idt/hda.rs @@ -2,7 +2,7 @@ use super::device_irq::device_irq_entry; // Lock-free, heap-free: may interrupt a CPU holding the controller lock (preemption disabled, not interrupts). extern "sysv64" fn hda_handler() { - crate::irq_census::irq_took!(Hda); + crate::arch::percpu::irq_took!(Hda); crate::drivers::hda::isr_complete(); crate::arch::apic::eoi(); } diff --git a/kernel/src/arch/idt/i8042.rs b/kernel/src/arch/x86_64/idt/i8042.rs similarity index 67% rename from kernel/src/arch/idt/i8042.rs rename to kernel/src/arch/x86_64/idt/i8042.rs index cc05ed01f4b..e6e223ef52f 100644 --- a/kernel/src/arch/idt/i8042.rs +++ b/kernel/src/arch/x86_64/idt/i8042.rs @@ -2,5 +2,5 @@ use super::device_irq::device_irq_entry; device_irq_entry! { /// The kernel's only reader of port 0x60 for both PS/2 lines. - pub(super) fn i8042_entry => crate::drivers::i8042::handler + pub(super) fn i8042_entry => crate::arch::i8042::handler } diff --git a/kernel/src/arch/idt/log_nest.rs b/kernel/src/arch/x86_64/idt/log_nest.rs similarity index 100% rename from kernel/src/arch/idt/log_nest.rs rename to kernel/src/arch/x86_64/idt/log_nest.rs diff --git a/kernel/src/arch/idt/mod.rs b/kernel/src/arch/x86_64/idt/mod.rs similarity index 92% rename from kernel/src/arch/idt/mod.rs rename to kernel/src/arch/x86_64/idt/mod.rs index 1e4417a0b84..1bc2744e6ca 100644 --- a/kernel/src/arch/idt/mod.rs +++ b/kernel/src/arch/x86_64/idt/mod.rs @@ -519,3 +519,40 @@ fn install_actuator_gates(idt: &mut Idt) { pub fn enable_interrupts() { cpu::enable_interrupts(); } + +pub(crate) use exceptions::try_recover_from_panic; + +/// The crash report for a panic, from the frame pointer the panic handler stood on. +pub(crate) fn report_panic(message: &core::panic::PanicInfo, frame: u64) { + exceptions::crash_report(&exceptions::CrashInfo::Panic { message, rbp: frame }); +} + +/// Whether the interrupted context a trap frame's flags word describes could +/// have taken an interrupt: `RFLAGS.IF`. +pub(crate) const fn frame_interrupts_enabled(rflags: u64) -> bool { + rflags & (1 << 9) != 0 +} + +/// A real `#DF`, not simulated: pushing to a non-canonical `rsp` raises `#SS`, +/// and delivering that needs another push to the same `rsp` — the `#DF` +/// condition. Non-canonical rather than unmapped: on a bigger machine an +/// unmapped address can fall inside the direct map and simply get written to. +/// Only `#DF` has an IST, so every fault on the way there lands on this same +/// unusable stack. `SYS_DEBUG`'s alone. +#[cfg(feature = "test-actuators")] +pub(crate) fn provoke_double_fault() -> ! { + // SAFETY: unsound by design, like `SYS_DEBUG`'s null read; the #DF this + // raises never returns here. + unsafe { + core::arch::asm!( + "mov rsp, {bad}", + "push 0", + bad = in(reg) 0x0000_8000_0000_0000u64, + options(noreturn), + ); + } +} + +pub(crate) use unclaimed::log_vectors as log_unclaimed; +/// How much of the double-fault stack the crash report used, once the report is out. +pub(crate) use super::percpu::ist1_report as report_fault_stack; diff --git a/kernel/src/arch/idt/nmi.rs b/kernel/src/arch/x86_64/idt/nmi.rs similarity index 96% rename from kernel/src/arch/idt/nmi.rs rename to kernel/src/arch/x86_64/idt/nmi.rs index 51e0f7cea4a..e334c99b0dc 100644 --- a/kernel/src/arch/idt/nmi.rs +++ b/kernel/src/arch/x86_64/idt/nmi.rs @@ -92,11 +92,11 @@ pub(super) extern "sysv64" fn nmi_entry() { /// Loads all four words in both builds so the observer and shipping handler share one frame layout. extern "sysv64" fn note(rip: u64, cs: u64, rsp: u64, rflags: u64) { - crate::irq_census::irq_took!(Nmi); + crate::arch::percpu::irq_took!(Nmi); #[cfg(not(feature = "boot-actuators"))] let _ = cs; #[cfg(feature = "boot-actuators")] - crate::nmi_gate::observe(rip, cs, rsp); + crate::arch::nmi_gate::observe(rip, cs, rsp); crate::sched::dump::note_nmi(rip); // After the probe's store and before the nested-NMI staging: a hard lockup // ends the machine from here, so the sibling asking where this CPU is still @@ -104,7 +104,7 @@ extern "sysv64" fn note(rip: u64, cs: u64, rsp: u64, rflags: u64) { // sealing a record. crate::hardlockup::sample(rip, rsp, rflags); #[cfg(feature = "boot-actuators")] - crate::nmi_gate::stage_nested_if_armed(); + crate::arch::nmi_gate::stage_nested_if_armed(); } /// A second NMI on a stack the first is still standing on. @@ -119,5 +119,5 @@ extern "sysv64" fn nested_nmi(rip: u64, rsp: u64) -> ! { serial(b" rsp="); crate::drivers::serial::panic_raw_hex(rsp); serial(b"\n[nmi] the outer handler's frame is gone; the machine stops here.\n"); - crate::arch::apic::halt_all_cpus() + crate::panic::halt_all_cpus() } diff --git a/kernel/src/arch/idt/spurious.rs b/kernel/src/arch/x86_64/idt/spurious.rs similarity index 98% rename from kernel/src/arch/idt/spurious.rs rename to kernel/src/arch/x86_64/idt/spurious.rs index e536dfcb074..06fdd73d23a 100644 --- a/kernel/src/arch/idt/spurious.rs +++ b/kernel/src/arch/x86_64/idt/spurious.rs @@ -52,7 +52,7 @@ pub(super) extern "sysv64" fn spurious_entry() { /// Counts one delivery and acknowledges it if the ISR bit shows it needed one. extern "sysv64" fn took() { - crate::irq_census::irq_took!(Spurious); + crate::arch::percpu::irq_took!(Spurious); if apic::in_service(SPURIOUS_VECTOR) { apic::eoi(); } diff --git a/kernel/src/arch/idt/timer.rs b/kernel/src/arch/x86_64/idt/timer.rs similarity index 95% rename from kernel/src/arch/idt/timer.rs rename to kernel/src/arch/x86_64/idt/timer.rs index 372d6b8b5ba..574e4a68713 100644 --- a/kernel/src/arch/idt/timer.rs +++ b/kernel/src/arch/x86_64/idt/timer.rs @@ -82,15 +82,15 @@ pub(super) extern "sysv64" fn timer_entry() { armed_ticks = const crate::arch::percpu::OFF_LAST_ARMED_TICKS, need_resched = const crate::arch::percpu::OFF_NEED_RESCHED, ring0_fires = const crate::arch::percpu::OFF_RING0_TIMER_FIRES, - irq_total = const crate::irq_census::slot_offset(crate::irq_census::TOTAL), - irq_timer = const crate::irq_census::slot_offset( + irq_total = const crate::arch::percpu::irq_slot_offset(crate::irq_census::TOTAL), + irq_timer = const crate::arch::percpu::irq_slot_offset( 1 + crate::irq_census::Source::Timer as usize ), ); } extern "sysv64" fn timer_handler() { - crate::irq_census::irq_took!(Timer); + crate::arch::percpu::irq_took!(Timer); // Before anything that can take a lock: a CPU running userland is the other // half of the coverage the Ring 0 branch above gives a CPU holding one. crate::deadline::poll(); diff --git a/kernel/src/arch/idt/tlb.rs b/kernel/src/arch/x86_64/idt/tlb.rs similarity index 97% rename from kernel/src/arch/idt/tlb.rs rename to kernel/src/arch/x86_64/idt/tlb.rs index 7eaee1207da..fa5a147aa24 100644 --- a/kernel/src/arch/idt/tlb.rs +++ b/kernel/src/arch/x86_64/idt/tlb.rs @@ -51,6 +51,6 @@ pub(super) extern "sysv64" fn tlb_flush_entry() { } fn flush() { - crate::irq_census::irq_took!(Tlb); + crate::arch::percpu::irq_took!(Tlb); crate::arch::tlb::serve_ipi(); } diff --git a/kernel/src/arch/idt/unclaimed.rs b/kernel/src/arch/x86_64/idt/unclaimed.rs similarity index 99% rename from kernel/src/arch/idt/unclaimed.rs rename to kernel/src/arch/x86_64/idt/unclaimed.rs index 875043d293f..9ca9f4b6dcf 100644 --- a/kernel/src/arch/idt/unclaimed.rs +++ b/kernel/src/arch/x86_64/idt/unclaimed.rs @@ -57,7 +57,7 @@ pub(super) extern "sysv64" fn unclaimed_entry() { /// Counts the delivery, remembers the vector, and EOIs only if the ISR needs one. extern "sysv64" fn took() { - crate::irq_census::irq_took!(Unclaimed); + crate::arch::percpu::irq_took!(Unclaimed); match apic::in_service_highest() { Some(vector) => { TAKEN[(vector >> 6) as usize].fetch_or(1 << (vector & 63), Ordering::Relaxed); diff --git a/kernel/src/arch/idt/user_dev.rs b/kernel/src/arch/x86_64/idt/user_dev.rs similarity index 97% rename from kernel/src/arch/idt/user_dev.rs rename to kernel/src/arch/x86_64/idt/user_dev.rs index d3feb90804c..18142f94df5 100644 --- a/kernel/src/arch/idt/user_dev.rs +++ b/kernel/src/arch/x86_64/idt/user_dev.rs @@ -13,7 +13,7 @@ use super::device_irq::device_irq_entry; use crate::irq_ring::IrqSource; fn took(slot: usize) { - crate::irq_census::irq_took!(UserDev); + crate::arch::percpu::irq_took!(UserDev); crate::pcidev::isr(slot); crate::irq_ring::isr_publish(IrqSource::UserDev, crate::clock::nanos_since_boot()); // Force resched now, so `drain_irqs` turns the record into a wake before diff --git a/kernel/src/arch/idt/virtio_sound.rs b/kernel/src/arch/x86_64/idt/virtio_sound.rs similarity index 91% rename from kernel/src/arch/idt/virtio_sound.rs rename to kernel/src/arch/x86_64/idt/virtio_sound.rs index b6f388b841c..98060824d62 100644 --- a/kernel/src/arch/idt/virtio_sound.rs +++ b/kernel/src/arch/x86_64/idt/virtio_sound.rs @@ -2,7 +2,7 @@ use super::device_irq::device_irq_entry; // Lock-free and heap-free: may interrupt a CPU holding the controller lock, which disables preemption but not interrupts. extern "sysv64" fn virtio_sound_handler() { - crate::irq_census::irq_took!(Sound); + crate::arch::percpu::irq_took!(Sound); crate::drivers::virtio_sound::isr_complete(); crate::arch::apic::eoi(); } diff --git a/kernel/src/arch/idt/xhci.rs b/kernel/src/arch/x86_64/idt/xhci.rs similarity index 93% rename from kernel/src/arch/idt/xhci.rs rename to kernel/src/arch/x86_64/idt/xhci.rs index baef2d74905..6ecce39df01 100644 --- a/kernel/src/arch/idt/xhci.rs +++ b/kernel/src/arch/x86_64/idt/xhci.rs @@ -3,7 +3,7 @@ use crate::irq_ring::IrqSource; // Lock-free and heap-free — the event ring is polled by drivers::xhci::poll_if_pending, never here. extern "sysv64" fn xhci_handler() { - crate::irq_census::irq_took!(Xhci); + crate::arch::percpu::irq_took!(Xhci); let timestamp = crate::clock::nanos_since_boot(); crate::irq_ring::isr_publish(IrqSource::Xhci, timestamp); // Forces a scheduler entry on IRQ return so drain_irqs polls now, not at the next quantum tick. diff --git a/kernel/src/drivers/ioapic.rs b/kernel/src/arch/x86_64/ioapic.rs similarity index 99% rename from kernel/src/drivers/ioapic.rs rename to kernel/src/arch/x86_64/ioapic.rs index be7a818f618..fdfd4aa9de1 100644 --- a/kernel/src/drivers/ioapic.rs +++ b/kernel/src/arch/x86_64/ioapic.rs @@ -16,7 +16,7 @@ use crate::mm::paging::MmioPolicy; use crate::log; use crate::mm::Mmio; use crate::sync::Lock; -use super::acpi::MadtInfo; +use crate::drivers::acpi::MadtInfo; const IOREGSEL: u64 = 0x00; const IOWIN: u64 = 0x10; diff --git a/kernel/src/arch/x86_64/mod.rs b/kernel/src/arch/x86_64/mod.rs new file mode 100644 index 00000000000..e44661ba2fb --- /dev/null +++ b/kernel/src/arch/x86_64/mod.rs @@ -0,0 +1,139 @@ +#![warn(clippy::undocumented_unsafe_blocks)] +//! x86-64: the PC this kernel was first written for. +//! +//! Every `unsafe` block here carries a one-line `SAFETY:` comment, enforced by the lint above. +//! [`percpu`] owns every `gs:` access; nothing outside this directory writes one. +//! +//! The PC platform's own devices live here too, because no other architecture +//! has them: the i8042, the I/O APIC, the CMOS RTC, the chipset's TCO +//! watchdog and VT-d. Generic code reaches each through the concept it serves +//! (`keyboard_controller`, `watchdog`, `iommu_unit`, …), never by its name. + +pub mod apic; +pub mod barrier; +pub mod boot; +pub mod cache; +pub mod control_regs; +pub mod console_uart; +pub mod cpu; +pub mod entropy; +pub mod entry; +pub mod fpu; +pub mod hpet; +pub mod hw; +pub mod i8042; +pub mod idt; +pub mod ioapic; +pub mod mtrr; +#[cfg(feature = "boot-actuators")] +pub mod nmi_gate; +pub mod paging; +pub mod pat; +pub mod percpu; +pub mod pio; +pub mod pmu; +pub mod rtc; +pub mod smp; +pub mod switch; +pub mod syscall; +pub mod tlb; +pub mod vtd; +pub mod watchdog; + +pub use apic as irqchip; +pub use i8042 as keyboard_controller; +pub use idt as trap; +pub use vtd as iommu_unit; + +pub use apic::{msi_message, MSI_DOORBELL}; + +/// The machine every program image this kernel loads must be built for. +pub const ELF_MACHINE: toyos_elf::Machine = toyos_elf::Machine::X86_64; + +/// Interrupts masked on this CPU for as long as the guard lives, and then put +/// back as they were — restored, not enabled — so a guard nests inside a region +/// that is already masked. The one way this kernel masks and restores: the +/// scheduler's pass, a log record's reservation and publication, and the +/// console backend each hold one. `TF` is always clear in Ring 0, so the guard +/// leaves it alone. +/// +/// Both edges are compiler barriers (no `nomem`): a memory access written +/// inside the region is emitted inside it. +#[must_use = "dropping the guard reopens interrupts"] +pub struct IrqGuard { + rflags: u64, + // Same-CPU only: keeps this guard `!Send + !Sync`. + _not_send_sync: core::marker::PhantomData<*mut ()>, +} + +impl IrqGuard { + pub fn close() -> Self { + let rflags: u64; + // SAFETY: pushfq/pop is balanced and cli touches only RFLAGS — one + // uninterruptible read-and-clear of IF. + unsafe { + core::arch::asm!("pushfq", "pop {saved}", "cli", saved = out(reg) rflags); + } + Self { rflags, _not_send_sync: core::marker::PhantomData } + } + + /// The flags captured and interrupts left as they are: what the + /// `log-unbracketed-reserve` actuator stages a log reservation with. + #[cfg(feature = "boot-actuators")] + pub fn unclosed() -> Self { + let rflags: u64; + // SAFETY: pushfq/pop is balanced and writes no RFLAGS bit. + unsafe { + core::arch::asm!("pushfq", "pop {saved}", saved = out(reg) rflags); + } + Self { rflags, _not_send_sync: core::marker::PhantomData } + } +} + +impl Drop for IrqGuard { + fn drop(&mut self) { + // SAFETY: the word `close` read out of RFLAGS on this CPU (the guard is + // `!Send`), restored whole. + unsafe { + core::arch::asm!("push {saved}", "popfq", saved = in(reg) self.rflags); + } + } +} + +/// Adds one to `counter`, atomic against an interrupt on this CPU, and answers the value before the add. +/// # Safety: `counter` is written by no other CPU; `guard` covers the shard selection that owns it. +#[inline(always)] +pub unsafe fn percpu_fetch_add( + counter: &core::sync::atomic::AtomicU64, + _guard: &IrqGuard, +) -> u64 { + // Under `log-shared-reservation`, stage a load/store race instead of the `xadd` below. + if crate::actuator::log_shared_reservation() { + let previous = counter.load(core::sync::atomic::Ordering::Relaxed); + if crate::log::nested::inject() { + // SAFETY: `sti`/`cli` each write one `RFLAGS` bit and touch no memory. + unsafe { + core::arch::asm!("sti"); + for _ in 0..256 { + core::hint::spin_loop(); + } + core::arch::asm!("cli"); + } + } + counter.store(previous + 1, core::sync::atomic::Ordering::Relaxed); + return previous; + } + + let previous: u64; + // Not `AtomicU64::fetch_add`: its locked xadd is costly under QEMU TCG emulation. + // SAFETY: `counter.as_ptr()` is live; unlocked `xadd` retires whole, atomic against an interrupt here. + unsafe { + // No `preserves_flags`: `xadd` changes arithmetic flags. + core::arch::asm!( + "xadd [{ptr}], {out}", + ptr = in(reg) counter.as_ptr(), + out = inout(reg) 1u64 => previous, + ); + } + previous +} diff --git a/kernel/src/arch/mtrr.rs b/kernel/src/arch/x86_64/mtrr.rs similarity index 98% rename from kernel/src/arch/mtrr.rs rename to kernel/src/arch/x86_64/mtrr.rs index fd33c9eaf3e..7d25220f4ce 100644 --- a/kernel/src/arch/mtrr.rs +++ b/kernel/src/arch/x86_64/mtrr.rs @@ -1,7 +1,7 @@ //! What memory type firmware gave a physical range. //! //! Read-only: firmware owns these registers, the kernel programs none. A -//! mapping with [`CachePolicy::DeferToMtrr`](crate::mm::paging::CachePolicy) +//! mapping with [`CachePolicy::Normal`](crate::mm::paging::CachePolicy) //! selects PAT entry 0 (WB), so what this module reports is the effective //! type; the exception is [`effective_under_wc`], where WC outvotes the MTRR //! instead of deferring to it. diff --git a/kernel/src/nmi_gate.rs b/kernel/src/arch/x86_64/nmi_gate.rs similarity index 99% rename from kernel/src/nmi_gate.rs rename to kernel/src/arch/x86_64/nmi_gate.rs index b41e55b9784..42b47045fde 100644 --- a/kernel/src/nmi_gate.rs +++ b/kernel/src/arch/x86_64/nmi_gate.rs @@ -28,6 +28,7 @@ //! itself with [`hold::EXPIRED`] in its word, and [`note_syscall`] is where it //! says so. +use crate::log; use core::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use crate::arch::{apic, percpu}; diff --git a/kernel/src/mm/paging.rs b/kernel/src/arch/x86_64/paging.rs similarity index 89% rename from kernel/src/mm/paging.rs rename to kernel/src/arch/x86_64/paging.rs index b2a494c77c2..803c5991352 100644 --- a/kernel/src/mm/paging.rs +++ b/kernel/src/arch/x86_64/paging.rs @@ -6,6 +6,7 @@ // No mapping here is global, which is what makes a single-address // invalidation (INVPCID or INVLPG) complete. +use crate::log; use alloc::boxed::Box; use alloc::collections::BTreeMap; use alloc::vec::Vec; @@ -13,11 +14,12 @@ use crate::hasher::HashMap; use toyos_pcid::{Alloc, Pcid, PcidPool}; -use super::{UserAddr, PAGE_2M}; +use crate::mm::{UserAddr, PAGE_2M}; +pub use crate::mm::policy::{CachePolicy, MmioPolicy, Prot, WindowProt}; use crate::arch::control_regs::PcidActive; use crate::arch::cpu::Invpcid; use crate::sync::Lock; -use crate::vma::{self, Region, RegionKind}; +use crate::vma::{self, Occupancy, Region, RegionKind}; use crate::MemoryMapEntry; const PAGE_PRESENT: u64 = 1 << 0; @@ -39,22 +41,6 @@ const ADDR_MASK_2M: u64 = 0x000F_FFFF_FFE0_0000; /// Every upper-level table entry's flags: present, writable, user. const TABLE_FLAGS: u64 = PAGE_PRESENT | PAGE_WRITE | PAGE_USER; -/// 4 KiB pages in one 2 MiB page. -const PAGES_PER_2M: usize = (PAGE_2M / 4096) as usize; - -/// What a user mapping may be used for: no variant is both writable and -/// executable, and every variant implies read since `PAGE_USER` grants it -/// unconditionally. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum Prot { - /// Read-only: neither writable nor executable. - Read, - /// Data: readable and writable, never executable. - ReadWrite, - /// Code. Never writable. - ReadExec, -} - impl Prot { /// The permission bits a leaf entry carries; address and cache policy stay the caller's. fn leaf_bits(self) -> u64 { @@ -67,45 +53,14 @@ impl Prot { } } -/// What each 4 KiB page of a 2 MiB window may be used for: split because -/// `toyos-ld` can align a window across the end of `.text` and start of `.data`. -pub struct WindowProt([Prot; PAGES_PER_2M]); - -impl WindowProt { - /// A window whose pages all say the same thing. - pub const fn uniform(prot: Prot) -> Self { - Self([prot; PAGES_PER_2M]) - } - - /// Sets the 4 KiB page `offset` bytes in; an out-of-window offset panics. - pub fn set(&mut self, offset: u64, prot: Prot) { - self.0[(offset / 4096) as usize] = prot; - } - - /// The one protection every page carries, or `None` where they disagree. - fn agreed(&self) -> Option { - let first = self.0[0]; - self.0.iter().all(|&p| p == first).then_some(first) - } -} - -/// Which PAT entry a 2 MiB mapping selects, out of the three this kernel -/// ever writes. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum CachePolicy { - /// PAT entry 0 (WB); the range's actual type is the MTRR's (SDM Vol. 3A Table 11-7). - DeferToMtrr, - /// PAT entry [`pat::UC_ENTRY`](crate::arch::pat::UC_ENTRY): UC under - /// every MTRR type (Table 11-7), whatever firmware set or forgot. - Uncacheable, - /// PAT entry [`pat::WC_ENTRY`](crate::arch::pat::WC_ENTRY). - WriteCombining, -} - +/// Which PAT entry a 2 MiB mapping selects: `Normal` is entry 0 (WB), whose +/// type for a range is the MTRR's (SDM Vol. 3A Table 11-7); `Uncacheable` is +/// [`pat::UC_ENTRY`](crate::arch::pat::UC_ENTRY), UC under every MTRR type; +/// `WriteCombining` is [`pat::WC_ENTRY`](crate::arch::pat::WC_ENTRY). impl CachePolicy { fn pde_bits(self) -> u64 { match self { - Self::DeferToMtrr => 0, + Self::Normal => 0, Self::Uncacheable => PAGE_CACHE_DISABLE | PAGE_WRITE_THROUGH, Self::WriteCombining => PAGE_PAT_2M, } @@ -114,7 +69,7 @@ impl CachePolicy { /// Any other combination is an entry this code never wrote. fn from_pde(pde: u64) -> Self { match (pde & PAGE_PAT_2M != 0, pde & (PAGE_CACHE_DISABLE | PAGE_WRITE_THROUGH)) { - (false, 0) => Self::DeferToMtrr, + (false, 0) => Self::Normal, (true, 0) => Self::WriteCombining, (false, low) if low == PAGE_CACHE_DISABLE | PAGE_WRITE_THROUGH => Self::Uncacheable, _ => panic!( @@ -135,46 +90,26 @@ const _: () = assert!( "Uncacheable sets PCD and PWT and leaves the PAT bit clear, which is entry 3", ); -/// What an MMIO window may select — never PAT entry 0: device registers -/// deferred to firmware's MTRR coverage were cacheable wherever an MTRR was -/// missing, and this type removes that as a possibility. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum MmioPolicy { - /// Registers. - Uncacheable, - /// The scanout alone. - WriteCombining, -} - -impl MmioPolicy { - fn cache(self) -> CachePolicy { - match self { - Self::Uncacheable => CachePolicy::Uncacheable, - Self::WriteCombining => CachePolicy::WriteCombining, - } - } -} - /// A 4KB-aligned page of 512 entries, matching the hardware page table format. #[repr(C, align(4096))] struct PageTablePage([u64; 512]); impl PageTablePage { fn phys(&self) -> u64 { - super::DirectMap::phys_of(self) + crate::mm::DirectMap::phys_of(self) } /// # Safety /// `phys` must be a `PageTablePage` this module built and linked in; /// the returned reference must not outlive its address space. unsafe fn from_phys<'a>(phys: u64) -> &'a PageTablePage { - &*super::DirectMap::from_phys(phys).as_ptr::() + &*crate::mm::DirectMap::from_phys(phys).as_ptr::() } /// # Safety /// Same as [`from_phys`], plus exclusivity: hold the only live reference. unsafe fn from_phys_mut<'a>(phys: u64) -> &'a mut PageTablePage { - &mut *super::DirectMap::from_phys(phys).as_mut_ptr::() + &mut *crate::mm::DirectMap::from_phys(phys).as_mut_ptr::() } fn child(&self, index: usize) -> Option<&PageTablePage> { @@ -340,6 +275,9 @@ pub fn flush_tlb_all() { #[derive(Clone, Copy)] pub struct Cr3(u64); +/// The address space a CPU runs in, as the architecture names its root: CR3. +pub type Root = Cr3; + impl Cr3 { pub fn current() -> Self { Self(crate::arch::cpu::read_cr3()) @@ -426,7 +364,7 @@ pub struct AddressSpace { root: Box, children: Vec>, /// Physical data pages mapped into user space, keyed by physical address. Freed on drop. - pages: HashMap, + pages: HashMap, /// All virtual memory regions, keyed by start address. regions: BTreeMap, /// Owned for this space's life: dropping the space returns a user tag, so two @@ -434,18 +372,6 @@ pub struct AddressSpace { pcid: PcidHandle, } -/// Needed because a *placed* mapping (`sys_mmap`'s FIXED arm) skips -/// `find_gap`'s implicit check. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum Occupancy { - /// Nothing is registered over any part of it. - Free, - /// One region covers it end for end, and that region is all it runs into. - Whole, - /// Part of a region, several regions, or one that merely starts here. - Partial, -} - fn align_up_2m(v: u64) -> u64 { (v + PAGE_2M - 1) & !(PAGE_2M - 1) } @@ -473,7 +399,7 @@ impl AddressSpace { }) } - pub fn cr3(&self) -> Cr3 { + pub fn root(&self) -> Cr3 { Cr3(self.root.phys() | self.pcid.value() as u64) } @@ -530,7 +456,7 @@ impl AddressSpace { ); let pd_idx = indices(va).2; - let target = self.cr3(); + let target = self.root(); let pd = self.ensure_table(va, TABLE_FLAGS); pd.write_pde(pd_idx, va, phys | prot.leaf_bits() | PAGE_SIZE_BIT) .discharge(target); @@ -555,15 +481,15 @@ impl AddressSpace { ); let mut table = Box::new(PageTablePage([0; 512])); - for (i, &page_prot) in prot.0.iter().enumerate() { - // No cache bits: `DeferToMtrr` is the zero pattern at both granularities. + for (i, page_prot) in prot.pages().enumerate() { + // No cache bits: `Normal` is the zero pattern at both granularities. table.init_entry(i, (phys + i as u64 * 4096) | page_prot.leaf_bits()); } let table_phys = table.phys(); self.children.push(table); let pd_idx = indices(va).2; - let target = self.cr3(); + let target = self.root(); let pd = self.ensure_table(va, TABLE_FLAGS); // No `Prot` here: `NX` would make the whole window non-executable // whatever the leaves say. @@ -599,7 +525,7 @@ impl AddressSpace { ); let (pml4_idx, pdpt_idx, pd_idx) = indices(va); - let target = self.cr3(); + let target = self.root(); if let Some(pdpt) = self.root.child_mut(pml4_idx) { if let Some(pd) = pdpt.child_mut(pdpt_idx) { @@ -627,7 +553,7 @@ impl AddressSpace { /// Checked here, not at the callers: a user space shallow-copies the /// kernel PML4 half, so a kernel address would otherwise walk to a writable kernel page. - pub fn translate(&self, vaddr: UserAddr) -> Option { + pub fn translate(&self, vaddr: UserAddr) -> Option { let va = vaddr.raw(); if !toyos_userbound::is_user_addr(va) { return None; @@ -646,11 +572,11 @@ impl AddressSpace { if pte & PAGE_PRESENT == 0 { return None; } - return Some(super::DirectMap::from_phys((pte & ADDR_MASK) + (va & 0xFFF))); + return Some(crate::mm::DirectMap::from_phys((pte & ADDR_MASK) + (va & 0xFFF))); } let page_phys = pde & ADDR_MASK_2M; let offset = va & (PAGE_2M - 1); - Some(super::DirectMap::from_phys(page_phys + offset)) + Some(crate::mm::DirectMap::from_phys(page_phys + offset)) } /// Find a free gap of at least `size` bytes (2MB-aligned), searching top-down. @@ -769,7 +695,7 @@ impl AddressSpace { } /// Private: not safe to use until every CPU is told — the free fn [`map_mmio`] is the whole operation. - fn map_mmio(&mut self, phys: u64, size: u64, cache: CachePolicy) -> super::Mmio { + fn map_mmio(&mut self, phys: u64, size: u64, cache: CachePolicy) -> crate::mm::Mmio { let start = phys & !(PAGE_2M - 1); let end = (phys + size + PAGE_2M - 1) & !(PAGE_2M - 1); let mut cur = start; @@ -777,7 +703,7 @@ impl AddressSpace { self.map_2m(cur, PAGE_PRESENT | PAGE_WRITE | cache.pde_bits()); cur += PAGE_2M; } - super::Mmio::new(super::DirectMap::from_phys(phys), size) + crate::mm::Mmio::new(crate::mm::DirectMap::from_phys(phys), size) } /// Read from the table rather than remembered, or `None` if unmapped. @@ -788,7 +714,7 @@ impl AddressSpace { return None; } if pde & PAGE_SIZE_BIT == 0 { - // A split table: leaves have cache bits clear (`DeferToMtrr`), + // A split table: leaves have cache bits clear (`Normal`), // asserted since the two granularities put the PAT bit at different offsets. let pte = self.root.child(pml4_idx)?.child(pdpt_idx)?.child(pd_idx)? [((virt >> 12) & 0x1FF) as usize]; @@ -797,13 +723,13 @@ impl AddressSpace { "policy_at: the 4 KiB entry {pte:#x} at {virt:#x} selects a PAT entry \ outside 0", ); - return Some(CachePolicy::DeferToMtrr); + return Some(CachePolicy::Normal); } Some(CachePolicy::from_pde(pde)) } pub fn direct_map_policy(&self, phys: u64) -> Option { - self.policy_at(super::DirectMap::from_phys(phys).as_ptr::() as u64) + self.policy_at(crate::mm::DirectMap::from_phys(phys).as_ptr::() as u64) } pub fn user_policy(&self, addr: UserAddr) -> Option { @@ -814,7 +740,7 @@ impl AddressSpace { /// handing the enclosing 2 MiB page back to the PMM would reissue memory with a hole. pub fn guard_4k(&mut self, phys: u64) { assert!(phys & 0xFFF == 0, "guard_4k: phys {phys:#x} not 4 KiB-aligned"); - let virt = super::DirectMap::from_phys(phys).as_ptr::() as u64; + let virt = crate::mm::DirectMap::from_phys(phys).as_ptr::() as u64; let (pml4_idx, pdpt_idx, pd_idx) = indices(virt); let pd_phys = { let pdpt = self.root.child(pml4_idx).expect("guard_4k: no PDPT over the direct map"); @@ -865,7 +791,7 @@ impl AddressSpace { /// address, so an MMIO window's target is pre-mapped by the time its /// driver asks; a page `guard_4k` already split must not reach here. fn map_2m(&mut self, phys: u64, flags: u64) { - let virt = super::DirectMap::from_phys(phys).as_ptr::() as u64; + let virt = crate::mm::DirectMap::from_phys(phys).as_ptr::() as u64; let pd_idx = indices(virt).2; let pd = self.ensure_table(virt, flags); let entry = phys | flags | PAGE_SIZE_BIT; @@ -874,7 +800,7 @@ impl AddressSpace { existing & PAGE_PRESENT == 0 || existing & !(PAGE_ACCESSED | PAGE_DIRTY) == entry || (existing & PAGE_SIZE_BIT != 0 - && CachePolicy::from_pde(existing) == CachePolicy::DeferToMtrr), + && CachePolicy::from_pde(existing) == CachePolicy::Normal), "map_2m: {phys:#x} is mapped {existing:#x} and cannot also be {entry:#x}" ); // Neither caller wants the single address: [`map_mmio`] flushes every @@ -887,7 +813,7 @@ impl AddressSpace { fn ensure_table(&mut self, va: u64, flags: u64) -> &mut PageTablePage { let flags = flags & TABLE_FLAGS; let (pml4_idx, pdpt_idx, _) = indices(va); - let target = self.cr3(); + let target = self.root(); if self.root[pml4_idx] & PAGE_PRESENT == 0 { let child = Box::new(PageTablePage([0; 512])); @@ -940,7 +866,7 @@ pub fn kernel() -> &'static alloc::sync::Arc> { } /// Kernel CR3. Lock-free — safe to call from panic context. -pub fn kernel_cr3() -> Cr3 { +pub fn kernel_root() -> Cr3 { Cr3(KERNEL_CR3.load(core::sync::atomic::Ordering::Relaxed)) } @@ -951,7 +877,7 @@ pub fn kernel_cr3() -> Cr3 { pub fn activate_kernel() { // SAFETY: `KERNEL_CR3` names the boot-built tables, mapping the code and // stack this call returns onto. - unsafe { kernel_cr3().activate() }; + unsafe { kernel_root().activate() }; } /// [`activate_kernel`] for a CPU without `CR4.PCIDE` yet: `activate` sets the @@ -959,13 +885,13 @@ pub fn activate_kernel() { /// reaches before setting it on itself. pub fn load_kernel_flush() { // SAFETY: same as `activate_kernel` above. - unsafe { kernel_cr3().load_flush() }; + unsafe { kernel_root().load_flush() }; } /// Free function (not a method): the lock and the shootdown are separate /// statements. Not optional — `map_2m` may change memory type under a /// sibling's stale entry, which is SDM Vol. 3A §11.12.4 undefined behaviour. -pub fn map_mmio(phys: u64, size: u64, policy: MmioPolicy) -> super::Mmio { +pub fn map_mmio(phys: u64, size: u64, policy: MmioPolicy) -> crate::mm::Mmio { let mmio = kernel().lock().map_mmio(phys, size, policy.cache()); crate::arch::tlb::shootdown(crate::arch::tlb::Origin::Mmio); // Read back off the table and logged beside firmware's MTRR verdict: the @@ -984,12 +910,12 @@ pub fn map_mmio(phys: u64, size: u64, policy: MmioPolicy) -> super::Mmio { /// Take the 4 KiB page holding `addr` out of the kernel direct map; `addr`'s /// page must be owned by the caller forever (see [`AddressSpace::guard_4k`]). pub fn guard_kernel_page(addr: u64) { - assert!(super::is_kernel_addr(addr), "guard_kernel_page: {addr:#x} is not a kernel address"); - kernel().lock().guard_4k(super::DirectMap::phys_of(addr as *const u8)); + assert!(crate::mm::is_kernel_addr(addr), "guard_kernel_page: {addr:#x} is not a kernel address"); + kernel().lock().guard_4k(crate::mm::DirectMap::phys_of(addr as *const u8)); } /// Build kernel page tables: map all physical memory in the high half using 2MB large pages. -pub(super) fn init(memory_map: &[MemoryMapEntry]) { +pub(crate) fn init(memory_map: &[MemoryMapEntry]) { let mut max_addr: u64 = MIN_PHYS_MAP; for entry in memory_map { if entry.end > max_addr { @@ -1012,7 +938,7 @@ pub(super) fn init(memory_map: &[MemoryMapEntry]) { addr += PAGE_2M; } - let cr3 = kernel.cr3(); + let cr3 = kernel.root(); KERNEL_CR3.store(cr3.0, core::sync::atomic::Ordering::Release); // Leaked, and `Release` after the space is built: see [`KERNEL`]. let published: &'static alloc::sync::Arc> = Box::leak(Box::new( @@ -1038,10 +964,10 @@ fn has(entry: u64, flag: u64) -> u8 { } } -/// The *currently loaded* CR3, not `kernel_cr3()` (a panic can run on a user +/// The *currently loaded* CR3, not `kernel_root()` (a panic can run on a user /// space); lock-free and silent, for the panic path to prove a mapping /// before writing through it. -pub fn present_in_current_cr3(addr: u64) -> bool { +pub fn present_in_current_tables(addr: u64) -> bool { // SAFETY: `Cr3::current().phys()` names the table this CPU runs under // right now, so it can't be freed meanwhile. No lock: each entry read is // one aligned `u64`, atomic at the hardware level, so no read is torn. @@ -1082,7 +1008,7 @@ pub fn present_in_current_cr3(addr: u64) -> bool { /// map into and [`map_mmio`] is the way. pub fn boot_map_write_combining(phys: u64, size: u64) -> bool { for pass in [Pass::Prove, Pass::Switch] { - for base in [phys, super::PHYS_OFFSET + phys] { + for base in [phys, crate::mm::PHYS_OFFSET + phys] { let mut at = base & !(PAGE_2M - 1); let end = base.saturating_add(size); while at < end { @@ -1107,7 +1033,7 @@ enum Pass { /// Whether the current tables hold a 2 MiB leaf for `addr`, and on /// [`Pass::Switch`] gives that leaf the write-combining type. fn write_combine_leaf(addr: u64, pass: Pass) -> bool { - // SAFETY: sound as `present_in_current_cr3`'s walk — `Cr3::current()` names + // SAFETY: sound as `present_in_current_tables`'s walk — `Cr3::current()` names // the table this CPU runs under and every step is taken through a // `PAGE_PRESENT` entry. Exclusive because this runs on the BSP before any // AP exists and before `mm::init` publishes a space of its own. @@ -1139,7 +1065,7 @@ fn write_combine_leaf(addr: u64, pass: Pass) -> bool { /// Dump page table entries for an address. Lock-free for crash safety. pub fn debug_page_walk(addr: u64) { let cr3 = Cr3::current(); - // SAFETY: same argument as `present_in_current_cr3` above. + // SAFETY: same argument as `present_in_current_tables` above. let pml4 = unsafe { PageTablePage::from_phys(cr3.phys()) }; let (pml4_idx, pdpt_idx, pd_idx) = indices(addr); let pt_idx = ((addr >> 12) & 0x1FF) as usize; @@ -1216,3 +1142,22 @@ pub fn debug_page_walk(addr: u64) { } log!(" -> 4KB page at {:#x}", pte & ADDR_MASK); } + +/// What a write-combining scanout at `[addr, addr + size)` actually is, as the +/// GOP driver reports it: the effective type under PAT entry WC, the MTRRs' +/// type for the range, and the entry. +pub fn scanout_memory_type(addr: u64, size: u64) -> impl core::fmt::Display { + struct Report(crate::arch::mtrr::Effective); + impl core::fmt::Display for Report { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + write!( + f, + "{} (MTRR {}, PAT entry {})", + crate::arch::mtrr::effective_under_wc(&self.0).map_or("unknown", |t| t.name()), + self.0.name(), + crate::arch::pat::WC_ENTRY, + ) + } + } + Report(crate::arch::mtrr::range_type(addr, size)) +} diff --git a/kernel/src/arch/pat.rs b/kernel/src/arch/x86_64/pat.rs similarity index 100% rename from kernel/src/arch/pat.rs rename to kernel/src/arch/x86_64/pat.rs diff --git a/kernel/src/arch/percpu.rs b/kernel/src/arch/x86_64/percpu.rs similarity index 90% rename from kernel/src/arch/percpu.rs rename to kernel/src/arch/x86_64/percpu.rs index a030d860a04..e042c7f17a1 100644 --- a/kernel/src/arch/percpu.rs +++ b/kernel/src/arch/x86_64/percpu.rs @@ -383,7 +383,7 @@ fn alloc_percpu(cpu_id: u32) -> *mut PerCpu { // Published before the CPU it belongs to runs an instruction — no window where the census misses it. crate::irq_census::publish(cpu_id, percpu.irq_counts.as_ptr()); #[cfg(feature = "boot-actuators")] - crate::nmi_gate::publish(cpu_id, &raw const percpu.nmi_hold, &raw const percpu.user_rsp); + crate::arch::nmi_gate::publish(cpu_id, &raw const percpu.nmi_hold, &raw const percpu.user_rsp); ptr } @@ -405,9 +405,9 @@ fn alloc_log_shard(cpu_id: u32) -> u64 { } /// This CPU's shard, its identity, and one sequence number out of that shard. -/// The `xadd` has no `lock` prefix, sound only while the live [`crate::arch::LogCommitGuard`] proves ownership; the four reads are one `asm!` block, not four [`gs`] calls, so the absent `nomem` keeps shard selection inside the guard's barrier. +/// The `xadd` has no `lock` prefix, sound only while the live [`crate::arch::IrqGuard`] proves ownership; the four reads are one `asm!` block, not four [`gs`] calls, so the absent `nomem` keeps shard selection inside the guard's barrier. pub fn reserve_log_slot( - guard: &crate::arch::LogCommitGuard, + guard: &crate::arch::IrqGuard, ) -> (*const log::Shard, u64, u32, u32, u32) { let shard: u64; let seq: u64; @@ -547,7 +547,7 @@ fn ist1_top() -> Option { /// Report how much of the double fault stack the crash report used, straight to the UART — bypassing the log ring, which is drained and may itself be corrupt. pub fn ist1_report() { let Some(top) = ist1_top() else { return }; - let rsp = cpu::read_rsp(); + let rsp = cpu::stack_pointer(); let stack_bottom = top - IST_STACK_SIZE as u64; if rsp < stack_bottom || rsp > top { return; @@ -590,9 +590,9 @@ pub fn init_bsp(lapic_id: u32) { // SAFETY: `alloc_percpu` just returned a live, initialised `PerCpu` with no other reference until the `wrmsr` below. let percpu = unsafe { &mut *ptr }; - percpu.kernel_rsp = cpu::read_rsp(); + percpu.kernel_rsp = cpu::stack_pointer(); // SAFETY: `Tss` is `repr(C, packed)`; `rsp0` may be unaligned. - unsafe { core::ptr::write_unaligned(&raw mut percpu.tss.rsp0, cpu::read_rsp()); } + unsafe { core::ptr::write_unaligned(&raw mut percpu.tss.rsp0, cpu::stack_pointer()); } alloc_idle_stack(percpu); alloc_ist_stacks(percpu); @@ -804,3 +804,97 @@ pub fn swap_fault_state(new: CpuFaultState) -> CpuFaultState { pub fn set_fault_state(new: CpuFaultState) { gs::write_u8::(new as u8); } + +/// The `gs:` displacement of interrupt counter `index` in this CPU's block. +pub const fn irq_slot_offset(index: usize) -> u32 { + OFF_IRQ_COUNTS + (index as u32) * 8 +} + +/// Records one delivery of `$source` as two lock-free `add`s to this CPU's own gs: slots. +/// A macro, not a function: the two offsets must be asm immediates, not const-generic values an optimiser could relax. +macro_rules! irq_took { + ($source:ident) => {{ + // SAFETY: both slots are this CPU's own counter block per `arch::percpu`, and the caller is an interrupt handler, so `GS_BASE` already points at this CPU's `PerCpu`. + unsafe { + ::core::arch::asm!( + "add qword ptr gs:[{total}], 1", + "add qword ptr gs:[{source}], 1", + total = const $crate::arch::percpu::irq_slot_offset($crate::irq_census::TOTAL), + source = const $crate::arch::percpu::irq_slot_offset( + 1 + $crate::irq_census::Source::$source as usize + ), + // no `nomem` because both instructions write; no `preserves_flags` because `add` clobbers flags. + options(nostack), + ); + } + }}; +} + +pub(crate) use irq_took; + +/// Two of this CPU's interrupt counters, read straight off `gs:` with one load +/// each — the form a CPU inside an NMI may use. +pub fn irq_counts_here(first: usize, second: usize) -> (u64, u64) { + let a: u64; + let b: u64; + // SAFETY: both slots are this CPU's own counter block, and `GS_BASE` points + // at the running CPU's `PerCpu` in every context this is read from; both + // indices are below `irq_census::SLOTS`, asserted by the callers' constants. + unsafe { + core::arch::asm!( + "mov {a}, qword ptr gs:[{first}]", + "mov {b}, qword ptr gs:[{second}]", + a = out(reg) a, + b = out(reg) b, + first = in(reg) u64::from(irq_slot_offset(first)), + second = in(reg) u64::from(irq_slot_offset(second)), + options(nostack, readonly, preserves_flags), + ); + } + (a, b) +} + +/// This CPU's preempt count: the per-CPU word `crate::preempt` keeps, read and +/// written only through the five functions below. +#[inline] +pub fn preempt_count() -> u32 { + gs::read_u32::() +} + +#[inline] +pub fn set_preempt_count(value: u32) { + gs::write_u32::(value); +} + +/// One increment, atomic against an interrupt on this CPU. +#[inline] +pub fn preempt_count_up() { + gs::lock_inc_u32::(); +} + +/// One decrement, atomic against an interrupt on this CPU. +#[inline] +pub fn preempt_count_down() { + gs::lock_dec_u32::(); +} + +/// Whether this CPU owes a reschedule. +#[inline] +pub fn resched_owed() -> bool { + gs::read_u8::() != 0 +} + +#[inline] +pub fn set_resched_owed(owed: bool) { + if owed { + gs::write_u8_imm::(); + } else { + gs::write_u8_imm::(); + } +} + +/// Whether this CPU is inside a fault or panic ([`CpuFaultState`] not `Normal`). +#[inline] +pub fn faulting() -> bool { + gs::read_u8::() != 0 +} diff --git a/kernel/src/arch/x86_64/pio.rs b/kernel/src/arch/x86_64/pio.rs new file mode 100644 index 00000000000..5a29539bbb2 --- /dev/null +++ b/kernel/src/arch/x86_64/pio.rs @@ -0,0 +1,7 @@ +//! The I/O port space: x86-64 has one, reached by `in` and `out`. + +/// Whether this architecture has an I/O port space at all. Firmware tables +/// that name a port are only honoured where it does. +pub const EXISTS: bool = true; + +pub use super::cpu::{outb, outw}; diff --git a/kernel/src/arch/x86_64/pmu.rs b/kernel/src/arch/x86_64/pmu.rs new file mode 100644 index 00000000000..31dd45c4110 --- /dev/null +++ b/kernel/src/arch/x86_64/pmu.rs @@ -0,0 +1,109 @@ +//! The performance-monitoring counter that samples a CPU with an NMI: fixed +//! counter 2, counting the reference clock, overflowing into the local APIC's +//! performance-counter LVT (SDM Vol. 3B §20.2.2 for the counters, Vol. 3A +//! §12.5.1 for the LVT, Vol. 2A CPUID leaf 0AH for whether there are any). +//! `crate::hardlockup` decides what a sample means; this is how one arrives. +//! +//! Every path here runs from an NMI or on the CPU it arms: no lock, no log. + +use core::sync::atomic::{AtomicU64, Ordering::Relaxed}; + +use super::{apic, cpu}; + +/// The architectural performance-monitoring MSRs this file writes, SDM Vol. 3B +/// §20.2.2 and Vol. 4 Table 2-2. Nothing else in this kernel programs the PMU, +/// which is why each control register below is declared whole and written +/// whole rather than read, modified and written back. +const IA32_FIXED_CTR2: u32 = 0x30B; +const IA32_FIXED_CTR_CTRL: u32 = 0x38D; +const IA32_PERF_GLOBAL_STATUS: u32 = 0x38E; +const IA32_PERF_GLOBAL_CTRL: u32 = 0x38F; +const IA32_PERF_GLOBAL_OVF_CTRL: u32 = 0x390; + +/// Fixed counter 2's nibble of `IA32_FIXED_CTR_CTRL` is bits 11:8: enable in +/// ring 0 (bit 8) and ring 3 (bit 9), no AnyThread (bit 10), and PMI on +/// overflow (bit 11). Counting in both rings, because a CPU that stops taking +/// interrupts in either is the same defect. +const FIXED_CTR2_ARMED: u64 = 0b1011 << 8; + +/// The fixed counters live in the high half of `IA32_PERF_GLOBAL_CTRL` and of +/// the status and overflow-clear registers beside it, so counter 2 is bit 34. +const GLOBAL_FIXED_CTR2: u64 = 1 << 34; + +/// The counter's width as CPUID states it, as a mask: bits above it may not be +/// written back. +static WIDTH_MASK: AtomicU64 = AtomicU64::new(0); + +/// The counter's width, or `None` on a CPU with no architectural performance +/// monitoring — which is every QEMU TCG guest. +/// +/// SDM Vol. 2A, CPUID leaf 0AH: EAX[7:0] is the version, and version 2 is where +/// the fixed-function counters and `IA32_PERF_GLOBAL_CTRL` appear; EDX[4:0] is +/// how many fixed counters there are and EDX[12:5] how wide they are. Fixed +/// counter 2 needs three of them. +fn architectural_pmu() -> Option { + if cpu::cpuid(0, 0).0 < 0x0A { + return None; + } + let (eax, _, _, edx) = cpu::cpuid(0x0A, 0); + let version = eax & 0xff; + let counters = edx & 0x1f; + let width = (edx >> 5) & 0xff; + if version < 2 || counters < 3 || width == 0 || width > 64 { + return None; + } + Some(width) +} + +const fn mask_of(width: u32) -> u64 { + match width >= 64 { + true => u64::MAX, + false => (1u64 << width) - 1, + } +} + +/// Start this CPU's counter overflowing into an NMI every `period` reference +/// cycles — which are TSC ticks — or answer `false` for a CPU that has none. +pub fn arm(period: u64) -> bool { + let Some(width) = architectural_pmu() else { return false }; + WIDTH_MASK.store(mask_of(width), Relaxed); + // Stopped, then set up, then started: a counter enabled while its control + // register is half written can overflow into an LVT that is not armed yet. + write_msr(IA32_PERF_GLOBAL_CTRL, 0); + write_msr(IA32_FIXED_CTR_CTRL, FIXED_CTR2_ARMED); + write_msr(IA32_PERF_GLOBAL_OVF_CTRL, GLOBAL_FIXED_CTR2); + reload(period); + apic::arm_perf_nmi(); + write_msr(IA32_PERF_GLOBAL_CTRL, GLOBAL_FIXED_CTR2); + true +} + +/// Whether this CPU's counter is the reason this NMI arrived: +/// `IA32_PERF_GLOBAL_STATUS` bit 34 is its overflow. +pub fn overflowed() -> bool { + cpu::rdmsr(IA32_PERF_GLOBAL_STATUS) & GLOBAL_FIXED_CTR2 != 0 +} + +/// After an overflow's sample: clear it, reload, and unmask the LVT hardware +/// masked on delivery. +pub fn rearm(period: u64) { + write_msr(IA32_PERF_GLOBAL_OVF_CTRL, GLOBAL_FIXED_CTR2); + reload(period); + apic::arm_perf_nmi(); +} + +/// Set the counter one period below its own overflow. +fn reload(period: u64) { + let mask = WIDTH_MASK.load(Relaxed); + // Masked to the width CPUID stated: a fixed counter refuses a write of the + // bits above it, and the negative count is what makes the overflow land a + // period from here. + write_msr(IA32_FIXED_CTR2, 0u64.wrapping_sub(period) & mask); +} + +fn write_msr(msr: u32, value: u64) { + // SAFETY: every MSR here is an architectural performance-monitoring counter + // or its control register, enumerated by CPUID leaf 0AH before this file + // writes any of them, and each value is that register's own field encoding. + unsafe { cpu::wrmsr(msr, value) }; +} diff --git a/kernel/src/rtc.rs b/kernel/src/arch/x86_64/rtc.rs similarity index 100% rename from kernel/src/rtc.rs rename to kernel/src/arch/x86_64/rtc.rs diff --git a/kernel/src/arch/smp.rs b/kernel/src/arch/x86_64/smp.rs similarity index 100% rename from kernel/src/arch/smp.rs rename to kernel/src/arch/x86_64/smp.rs diff --git a/kernel/src/arch/x86_64/switch.rs b/kernel/src/arch/x86_64/switch.rs new file mode 100644 index 00000000000..4795ec86928 --- /dev/null +++ b/kernel/src/arch/x86_64/switch.rs @@ -0,0 +1,57 @@ +//! `context_switch`: the callee-saved registers a context stands on, pushed +//! onto the outgoing stack and popped off the incoming one. + +use core::arch::naked_asm; + +/// The outgoing half of [`context_switch`]. A macro, not inlined twice, so both builds share one instruction sequence. +macro_rules! switch_save { + () => { + "pushfq + push rbp + push rbx + push r12 + push r13 + push r14 + push r15 + mov [rdi], rsp + mov rsp, rsi" + }; +} + +/// The incoming half: the seven words a resumed context stands on, ending in `ret`. +macro_rules! switch_restore { + () => { + "pop r15 + pop r14 + pop r13 + pop r12 + pop rbx + pop rbp + popfq + ret" + }; +} + +/// Callee-saved register save/restore. +#[cfg(not(feature = "switch-witness"))] +#[unsafe(naked)] +pub(crate) unsafe extern "C" fn context_switch(old_rsp: *mut u64, new_rsp: u64) { + naked_asm!(switch_save!(), switch_restore!()); +} + +/// The same switch with [`super::hw::switch_witness_verify`] between the stack move and the first `pop`; never fired. +/// +/// Placed after `mov rsp, rsi`, reading the incoming frame through the register the machine will use. Sound +/// to `call`: the return lands inside the incoming task's own stack, and every register `verify` may clobber +/// is caller-saved and already dead here. +#[cfg(feature = "switch-witness")] +#[unsafe(naked)] +pub(crate) unsafe extern "C" fn context_switch(old_rsp: *mut u64, new_rsp: u64) { + naked_asm!( + switch_save!(), + "mov rdi, rsp", + "call {verify}", + switch_restore!(), + verify = sym super::hw::switch_witness_verify, + ); +} diff --git a/kernel/src/arch/syscall/gate.rs b/kernel/src/arch/x86_64/syscall.rs similarity index 91% rename from kernel/src/arch/syscall/gate.rs rename to kernel/src/arch/x86_64/syscall.rs index 54d0fce5ff4..9c21bdbe37c 100644 --- a/kernel/src/arch/syscall/gate.rs +++ b/kernel/src/arch/x86_64/syscall.rs @@ -1,6 +1,6 @@ //! Where Ring 3 enters, and what the CPU is told to do when it does. //! -//! `STAR` names the selectors, `LSTAR` is the one address `syscall` can reach, and `FMASK` masks the `RFLAGS` bits a Ring 3 thread may not hand the kernel; [`super::dispatch`] is the first code that interprets the syscall number. +//! `STAR` names the selectors, `LSTAR` is the one address `syscall` can reach, and `FMASK` masks the `RFLAGS` bits a Ring 3 thread may not hand the kernel; [`crate::syscall::dispatch`] is the first code that interprets the syscall number. use crate::arch::cpu; use crate::arch::entry::{restore_user_state, ring3_naked_asm, save_user_state, Ring3Entry}; @@ -8,7 +8,7 @@ use crate::arch::percpu; #[cfg(feature = "boot-actuators")] use crate::arch::smp::asm_label_addr; -use super::dispatch::syscall_dispatch; +use crate::syscall::dispatch::syscall_dispatch; // `IA32_EFER.SCE` is `arch::control_regs`'s bit, decided in one place, not read back here. const MSR_STAR: u32 = 0xC000_0081; @@ -161,13 +161,13 @@ extern "sysv64" fn syscall_entry() { #[cfg(feature = "boot-actuators")] nmi_hold = const percpu::OFF_NMI_HOLD, #[cfg(feature = "boot-actuators")] - asked = const crate::nmi_gate::hold::ASKED, + asked = const crate::arch::nmi_gate::hold::ASKED, #[cfg(feature = "boot-actuators")] - held = const crate::nmi_gate::hold::HELD, + held = const crate::arch::nmi_gate::hold::HELD, #[cfg(feature = "boot-actuators")] - spin = const crate::nmi_gate::hold::SPIN, + spin = const crate::arch::nmi_gate::hold::SPIN, #[cfg(feature = "boot-actuators")] - expired = const crate::nmi_gate::hold::EXPIRED, + expired = const crate::arch::nmi_gate::hold::EXPIRED, ); } @@ -182,3 +182,16 @@ extern "sysv64" fn syscall_handler(num: u64, a1: u64, a2: u64, _: u64, a3: u64, percpu::leave_syscall(); out } + +/// Counts one syscall for `nmi_gate`, whatever it turns out to be; every +/// `syscall_dispatch` calls this first. +#[cfg(feature = "boot-actuators")] +pub fn note_entry() { + super::nmi_gate::note_syscall(); +} + +/// `syscall-window-nmi`'s storm on the sibling spinning in `syscall`, from the idle loop. +#[cfg(feature = "boot-actuators")] +pub fn window_storm() { + super::nmi_gate::storm(); +} diff --git a/kernel/src/arch/tlb.rs b/kernel/src/arch/x86_64/tlb.rs similarity index 90% rename from kernel/src/arch/tlb.rs rename to kernel/src/arch/x86_64/tlb.rs index 3a1870f37aa..1a4228c57bc 100644 --- a/kernel/src/arch/tlb.rs +++ b/kernel/src/arch/x86_64/tlb.rs @@ -19,31 +19,7 @@ use super::{apic, percpu, smp}; static SHOOTDOWN: Shootdown = Shootdown::new(); -/// Which path issued a shootdown, so the census names who pays: `Dlopen` (a -/// `Shared` window or rollback unmap), `Pcid` (pool reclaim), `Mmio`, `Unmap` -/// (`Unmapped::drop`), `Pipe`, `Staged` (the ack-delay actuator), `Bench` -/// ([`bench`]'s own, so a measured shootdown is never counted as one a path in -/// this kernel needed). -#[derive(Clone, Copy)] -#[repr(usize)] -pub enum Origin { - Dlopen, - Pcid, - Mmio, - Unmap, - Pipe, - #[cfg_attr(not(feature = "test-actuators"), allow(dead_code))] - Staged, - #[cfg_attr(not(feature = "boot-actuators"), allow(dead_code))] - Bench, -} - -impl Origin { - const COUNT: usize = 7; - /// Order matches the variants; `tests/toyos.rs`'s `irq_census_conservation` reads the line back. - const NAMES: [&'static str; Self::COUNT] = - ["dlopen", "pcid", "mmio", "unmap", "pipe", "staged", "bench"]; -} +pub use crate::invalidation::Origin; /// Issuer-side census; `irq_census`'s `tlb` column is the receiver side, and a /// delivery the two disagree on is an uncounted issuing path. @@ -84,12 +60,16 @@ pub fn log_census() { /// Set above xHCI's `CALL_AFTER_BREAK`, the longest a disk call spins with `IF` /// clear once its transport has broken, so no legitimate wait trips it; that -/// constant's own assertion holds the order. -pub(crate) const ACK_TIMEOUT: Tripwire = Tripwire::absurd( +/// assertion below holds the order. +const ACK_TIMEOUT: Tripwire = Tripwire::absurd( Duration::from_secs(5), "above the longest IF-clear device spin a target can be inside", ); +// A disk call spins with interrupts off, so one that outlasted this tripwire +// would panic another CPU over a device. +const _: () = assert!(crate::drivers::xhci::CALL_AFTER_BREAK.nanos() < ACK_TIMEOUT.nanos()); + /// Spins between deadline checks; `nanos_since_boot`'s 128-bit divide is too /// costly to call on every iteration. const SPINS_PER_DEADLINE_CHECK: u32 = 1024; @@ -155,7 +135,7 @@ pub fn bench() { ); } -/// Never logs: `drivers::serial`'s lock under `save_and_cli` would deadlock a +/// Never logs: `drivers::serial`'s lock under its `IrqGuard` would deadlock a /// target that cannot answer while blocked on it. fn wait_for(me: usize, cpu: u32, generation: Generation) { let mut spins = 0u32; diff --git a/kernel/src/iommu/vtd/dmar.rs b/kernel/src/arch/x86_64/vtd/dmar.rs similarity index 100% rename from kernel/src/iommu/vtd/dmar.rs rename to kernel/src/arch/x86_64/vtd/dmar.rs diff --git a/kernel/src/iommu/vtd/domain.rs b/kernel/src/arch/x86_64/vtd/domain.rs similarity index 99% rename from kernel/src/iommu/vtd/domain.rs rename to kernel/src/arch/x86_64/vtd/domain.rs index df90f2cbeaa..ab9aaddd951 100644 --- a/kernel/src/iommu/vtd/domain.rs +++ b/kernel/src/arch/x86_64/vtd/domain.rs @@ -15,6 +15,7 @@ //! //! Lock order here is `DOMAINS`, `REMAP`, `UNITS`, `TABLES`, never the reverse. +use crate::log; use alloc::vec::Vec; use crate::iommu::{AddressWidth, DomainId, IommuError, Iova, StreamId}; diff --git a/kernel/src/iommu/vtd/fault.rs b/kernel/src/arch/x86_64/vtd/fault.rs similarity index 98% rename from kernel/src/iommu/vtd/fault.rs rename to kernel/src/arch/x86_64/vtd/fault.rs index 1037d5c8e2e..c0188b87c1e 100644 --- a/kernel/src/iommu/vtd/fault.rs +++ b/kernel/src/arch/x86_64/vtd/fault.rs @@ -15,6 +15,7 @@ //! machine goes on, because one process's bug taking the machine down is the //! thing moving a driver out of the kernel was for. +use crate::log; use core::sync::atomic::{AtomicU32, AtomicU64, Ordering}; use crate::drivers::pci::{self, PciDevice}; @@ -25,10 +26,6 @@ use super::{ FECTL_REG, FEADDR_REG, FEDATA_REG, FEUADDR_REG, FSTS_REG, MAX_UNITS, REGISTER_WINDOW, }; -// LAPIC destination every device MSI in this kernel targets. -// The unit's own fault event is generated by the remapping hardware itself, so interrupt remapping never blocks it. -const MSG_ADDR: u32 = 0xFEE0_0000; - // Write-1-to-clear bits; the rest of FSTS is read-only status. const FSTS_WRITE_ONE_TO_CLEAR: u32 = 0x7F; // FSTS.PPF: a fault recording register has its F bit set. @@ -195,7 +192,9 @@ pub fn arm(index: usize, regs: Mmio, found: Records, vector: u8) { UNITS[index].regs.store(DirectMap::phys_of(regs.addr() as *const u8), Ordering::Release); regs.write_u32(FEDATA_REG, vector as u32); - regs.write_u32(FEADDR_REG, MSG_ADDR); + // Generated by the remapping hardware itself, so interrupt remapping never + // blocks it: a compatibility message at the doorbell, destination 0. + regs.write_u32(FEADDR_REG, crate::arch::MSI_DOORBELL); regs.write_u32(FEUADDR_REG, 0); regs.write_u32(FECTL_REG, 0); } @@ -247,7 +246,7 @@ pub fn service() { // that would be one process's bug taking the whole machine down, which // is the thing moving a driver out was for. crate::drivers::panic_console::capture(); - crate::arch::apic::halt_all_cpus(); + crate::panic::halt_all_cpus(); } crate::arch::apic::eoi(); } diff --git a/kernel/src/iommu/vtd/interrupt.rs b/kernel/src/arch/x86_64/vtd/interrupt.rs similarity index 99% rename from kernel/src/iommu/vtd/interrupt.rs rename to kernel/src/arch/x86_64/vtd/interrupt.rs index cfe4b99d647..8d86b74c860 100644 --- a/kernel/src/iommu/vtd/interrupt.rs +++ b/kernel/src/arch/x86_64/vtd/interrupt.rs @@ -28,6 +28,7 @@ //! This module's lock is taken before `UNITS` and before `TABLES`, never after //! either. +use crate::log; use alloc::vec::Vec; use crate::iommu::{Refused, StreamId}; @@ -58,7 +59,6 @@ const VERIFY_SOURCE_ID: u64 = 1 << 18; const NARROW_DESTINATION_SHIFT: u64 = 8; const NARROW_DESTINATIONS: u32 = 0xFF; -const MESSAGE_BASE: u32 = 0xFEE0_0000; const MESSAGE_REMAPPABLE: u32 = 1 << 4; const MESSAGE_SUBHANDLE_VALID: u32 = 1 << 3; const PIN_REMAPPABLE: u32 = 1 << 16; @@ -183,7 +183,7 @@ fn allocate(source: StreamId, vector: u8, dest: u32, level: bool) -> Result Result { let index = allocate(source, vector, dest, false)? as u32; Ok(Msi { - address: MESSAGE_BASE + address: crate::arch::MSI_DOORBELL | ((index & 0x7FFF) << 5) | MESSAGE_REMAPPABLE | MESSAGE_SUBHANDLE_VALID diff --git a/kernel/src/iommu/vtd/mod.rs b/kernel/src/arch/x86_64/vtd/mod.rs similarity index 99% rename from kernel/src/iommu/vtd/mod.rs rename to kernel/src/arch/x86_64/vtd/mod.rs index 2d8f63ecb1b..aa589e0e576 100644 --- a/kernel/src/iommu/vtd/mod.rs +++ b/kernel/src/arch/x86_64/vtd/mod.rs @@ -15,6 +15,7 @@ pub mod interrupt; mod queue; mod table; +use crate::log; use alloc::vec::Vec; use crate::drivers::acpi::TableError; @@ -259,7 +260,7 @@ fn remappable(ready: &[(Unit, Plan)], described: usize) -> Option { ); return None; } - let apics = crate::drivers::ioapic::ids(); + let apics = crate::arch::ioapic::ids(); if !interrupt::apics_are_named(&apics) { log!( "iommu: firmware named a requester id for only some of this machine's {} I/O APICs, \ diff --git a/kernel/src/iommu/vtd/queue.rs b/kernel/src/arch/x86_64/vtd/queue.rs similarity index 100% rename from kernel/src/iommu/vtd/queue.rs rename to kernel/src/arch/x86_64/vtd/queue.rs diff --git a/kernel/src/iommu/vtd/table.rs b/kernel/src/arch/x86_64/vtd/table.rs similarity index 100% rename from kernel/src/iommu/vtd/table.rs rename to kernel/src/arch/x86_64/vtd/table.rs diff --git a/kernel/src/drivers/watchdog.rs b/kernel/src/arch/x86_64/watchdog.rs similarity index 100% rename from kernel/src/drivers/watchdog.rs rename to kernel/src/arch/x86_64/watchdog.rs diff --git a/kernel/src/blackbox.rs b/kernel/src/blackbox.rs index a9bf2c67274..134e66a3c3c 100644 --- a/kernel/src/blackbox.rs +++ b/kernel/src/blackbox.rs @@ -238,32 +238,9 @@ fn with_page(write: impl FnOnce(&mut [u8; BYTES], u64, toyos_blackbox::Identity) flush(at); } -/// Write the page out of this CPU's caches, and every other CPU's. -/// -/// **A reset does not write dirty lines back.** INIT and RESET invalidate the -/// caches without flushing them, so a page sealed into write-back memory and -/// then reset over is a page whose bytes never reached DRAM — the one failure -/// this mechanism cannot survive, and it looks exactly like a seal that never -/// happened. The section number that states it is left out rather than cited -/// wrong. `CLFLUSH` is coherent across every CPU, so one caller's flush is the -/// whole machine's. +/// Write the page out of every CPU's caches: a reset does not write dirty +/// lines back, and a page that never reached DRAM reads, from the next boot, +/// exactly like a seal that never happened. fn flush(at: u64) { - let mut line = 0u64; - while line < BYTES as u64 { - // SAFETY: `CLFLUSH` writes back and invalidates the line containing the - // address and touches nothing else; the address is inside the page - // `arm` took, and the instruction faults on nothing a canonical address - // can be. Not privileged, and present on every x86-64 part. - unsafe { - core::arch::asm!( - "clflush [{addr}]", - addr = in(reg) (at + line) as *const u8, - options(nostack, preserves_flags), - ); - } - line += toyos_blackbox::CACHE_LINE as u64; - } - // SAFETY: `SFENCE` orders those writebacks ahead of whatever ends this - // machine; it touches no memory or register. - unsafe { core::arch::asm!("sfence", options(nostack, preserves_flags)) }; + crate::arch::cache::write_back(at, BYTES); } diff --git a/kernel/src/clock.rs b/kernel/src/clock.rs index 98e649b5ce6..66fa4bbba19 100644 --- a/kernel/src/clock.rs +++ b/kernel/src/clock.rs @@ -1,103 +1,23 @@ -//! The machine's clocks: monotonic since boot, calibrated from the HPET at -//! boot and read off the TSC after; and wall-clock, read from the CMOS RTC -//! exactly once — a CMOS read can block for up to a second — in -//! [`init_wall`], and answered after as that reading plus [`nanos_since_boot`]. +//! The machine's clocks: monotonic since boot, read off the CPU's free-running +//! counter at the period the architecture's boot gives [`set_counter`] (on +//! x86-64, measured against the HPET); and wall-clock, read from the +//! architecture's RTC exactly once — a CMOS read can block for up to a +//! second — in [`init_wall`], and answered after as that reading plus +//! [`nanos_since_boot`]. use core::sync::atomic::{AtomicBool, AtomicI64, AtomicU64, Ordering::{Acquire, Relaxed, Release}}; -use crate::mm::paging::MmioPolicy; use crate::arch::cpu; -use crate::time::{Delay, Duration, Instant}; - -const HPET_CAP: u64 = 0x000; -const HPET_CFG: u64 = 0x010; -const HPET_COUNTER: u64 = 0x0F0; +use crate::time::Instant; static TSC_BOOT: AtomicU64 = AtomicU64::new(0); static TSC_PERIOD_FS: AtomicU64 = AtomicU64::new(0); -pub fn init(hpet_base: u64) { - let hpet = crate::mm::paging::map_mmio(hpet_base, 0x1000, MmioPolicy::Uncacheable); - - let cap = hpet.read_u64(HPET_CAP); - let hpet_period_fs = cap >> 32; - assert!(hpet_period_fs > 0, "HPET: invalid counter period"); - - let cfg = hpet.read_u64(HPET_CFG); - hpet.write_u64(HPET_CFG, cfg | 1); - - const CALIBRATION: Delay = Delay::to_measure( - Duration::from_millis(50), - "TSC ticks counted against the HPET; longer is a better ratio and boot time is what it costs", - ); - let calibration_ns = CALIBRATION.nanos(); - let calibration_hpet_ticks = calibration_ns * 1_000_000 / hpet_period_fs; - - let hpet_start = hpet.read_u64(HPET_COUNTER); - let tsc_start = cpu::rdtsc(); - let hpet_target = hpet_start + calibration_hpet_ticks; - log!( - "clock: HPET at {:#x} enabled, period={}fs, counter reads {}, calibrating over {} ticks", - hpet_base, - hpet_period_fs, - hpet_start, - calibration_hpet_ticks, - ); - - // A main counter that does not advance would spin here forever, and this is - // the boot's last wait before it has a clock: the only unit available to - // bound it is the TSC's own, so the budget is the calibration converted at - // a frequency no x86-64 part reaches, which makes it an over-estimate of - // the cycles the calibration can legitimately take on any machine. - const TSC_CEILING_HZ: u64 = 10_000_000_000; - // Times two, so a machine merely slower than the ceiling is not refused for it. - let stall_budget_cycles = 2 * calibration_ns * (TSC_CEILING_HZ / 1_000_000_000); - while hpet.read_u64(HPET_COUNTER) < hpet_target { - assert!( - cpu::rdtsc().wrapping_sub(tsc_start) <= stall_budget_cycles, - "clock: the HPET main counter at {:#x} did not reach {} in {} TSC cycles (it started \ - at {} and reads {}), so this machine offers no clock to calibrate against", - hpet_base, - hpet_target, - stall_budget_cycles, - hpet_start, - hpet.read_u64(HPET_COUNTER), - ); - } - let tsc_end = cpu::rdtsc(); - let hpet_end = hpet.read_u64(HPET_COUNTER); - - let hpet_elapsed_fs = (hpet_end - hpet_start) as u128 * hpet_period_fs as u128; - let tsc_delta = tsc_end - tsc_start; - let tsc_period_fs = (hpet_elapsed_fs / tsc_delta as u128) as u64; - - TSC_BOOT.store(tsc_start, Relaxed); - TSC_PERIOD_FS.store(tsc_period_fs, Relaxed); - - let tsc_freq_mhz = 1_000_000_000_000_000u64 / tsc_period_fs / 1_000_000; - log!("TSC: {}MHz (period={}fs, calibrated over {}ms)", tsc_freq_mhz, tsc_period_fs, calibration_ns / 1_000_000); - - // **The one cross-source check this machine offers.** Everything else the - // kernel times is derived from the measurement just taken, so it could only - // agree with itself; CPUID 15H/16H is the part's own statement of the same - // frequency, arrived at by neither the HPET nor this counting loop, and the - // parts-per-million between the two is what a metal profile can hold a - // ceiling against. - let measured_hz = 1_000_000_000_000_000u64 / tsc_period_fs; - match cpuid_tsc_hz() { - Some(stated) => { - let apart = measured_hz.abs_diff(stated); - log!( - "clock: TSC measured {measured_hz}Hz against the HPET, CPUID states {stated}Hz, \ - {}ppm apart", - apart * 1_000_000 / stated, - ); - } - None => log!( - "clock: TSC measured {measured_hz}Hz against the HPET; CPUID leaves 15H and 16H \ - stating no frequency, so nothing independent confirms it" - ), - } +/// Start the clock on the counter: its reading at boot and its measured period. +/// The architecture's boot calls this once, after it has the period. +pub fn set_counter(boot: u64, period_fs: u64) { + TSC_BOOT.store(boot, Relaxed); + TSC_PERIOD_FS.store(period_fs, Relaxed); } /// Whether [`nanos_since_boot`] measures anything yet; false before [`init`]. @@ -105,54 +25,12 @@ pub fn calibrated() -> bool { TSC_PERIOD_FS.load(Relaxed) != 0 } -/// The TSC's frequency in hertz as CPUID *states* it, for the one caller that -/// may run before [`init`] — the panic path, which has to bound a wait on a -/// machine that never reached the HPET. Nothing calibrates against it and no -/// third source is guessed at: a CPU that states neither leaf answers `None` -/// and its caller says so rather than inventing a rate. -pub fn cpuid_tsc_hz() -> Option { - let max_leaf = cpu::cpuid(0, 0).0; - tsc_hz_from( - (max_leaf >= 0x15).then(|| cpu::cpuid(0x15, 0)), - (max_leaf >= 0x16).then(|| cpu::cpuid(0x16, 0)), - ) -} - -/// SDM Vol. 2A, CPUID leaf 15H: EAX is the denominator and EBX the numerator of -/// the core crystal's ratio to the TSC, ECX the crystal's hertz — any of the -/// three reading zero means the leaf states nothing. Leaf 16H's EAX is the -/// processor base frequency in MHz, which an invariant TSC counts at. -const fn tsc_hz_from( - leaf15: Option<(u32, u32, u32, u32)>, - leaf16: Option<(u32, u32, u32, u32)>, -) -> Option { - if let Some((denominator, numerator, crystal_hz, _)) = leaf15 { - if denominator != 0 && numerator != 0 && crystal_hz != 0 { - return Some(crystal_hz as u64 * numerator as u64 / denominator as u64); - } - } - if let Some((base_mhz, _, _, _)) = leaf16 { - if base_mhz != 0 { - return Some(base_mhz as u64 * 1_000_000); - } - } - None -} - -const _: () = { - // The ratio, then the fall-through to the base frequency when the crystal - // is not enumerated, then the CPU that states neither. - assert!(matches!(tsc_hz_from(Some((2, 4, 25_000_000, 0)), None), Some(50_000_000))); - assert!(matches!(tsc_hz_from(Some((0, 0, 0, 0)), Some((2_400, 0, 0, 0))), Some(2_400_000_000))); - assert!(tsc_hz_from(None, Some((0, 0, 0, 0))).is_none()); - assert!(tsc_hz_from(None, None).is_none()); -}; /// Nanoseconds since boot; lock-free, no MMIO, and never panics — `log::emit` /// reads it from inside a bracket where panicking would reenter the log. /// Saturating, not wrapping: a trailing CPU reads as oldest, not lying newest after a 584-year wrap. pub fn nanos_since_boot() -> u64 { - let delta = cpu::rdtsc().saturating_sub(TSC_BOOT.load(Relaxed)); + let delta = cpu::counter().saturating_sub(TSC_BOOT.load(Relaxed)); let period_fs = TSC_PERIOD_FS.load(Relaxed); ((delta as u128 * period_fs as u128) / 1_000_000) as u64 } @@ -166,7 +44,7 @@ pub fn now() -> Instant { /// The [`cpu::rdtsc`] value `nanos` in the future, for a wait loop that must /// not call the nanosecond clock. pub fn tsc_deadline(nanos: u64) -> u64 { - cpu::rdtsc().saturating_add(tsc_ticks(nanos)) + cpu::counter().saturating_add(tsc_ticks(nanos)) } /// `nanos` as a count of TSC ticks: a span converted once and then compared @@ -203,7 +81,7 @@ pub fn settles(nanos: u64, ready: impl Fn() -> bool) -> bool { } let until = tsc_deadline(nanos); while !ready() { - if cpu::rdtsc() >= until { + if cpu::counter() >= until { return false; } core::hint::spin_loop(); @@ -225,7 +103,7 @@ pub fn init_wall(century_reg: Option, utc_offset_minutes: Option) { let utc_offset_minutes = if crate::actuator::rtc_zone_east() { Some(-120) } else { utc_offset_minutes }; - let civil = match crate::rtc::read(century_reg) { + let civil = match crate::arch::rtc::read(century_reg) { Ok(civil) => civil, Err(fault) => { log!("clock: this machine will not say what time it is — {fault}"); diff --git a/kernel/src/deadline.rs b/kernel/src/deadline.rs index 078d3293e6e..4c4bd89dece 100644 --- a/kernel/src/deadline.rs +++ b/kernel/src/deadline.rs @@ -28,7 +28,7 @@ //! that can name where the core is standing. Whichever fires takes the //! machine's one seal through [`claim_the_reset`]. //! - **A panic in progress**, which is not a gap but a stand-down: -//! `apic::halt_all_cpus` calls [`stand_down`] before it holds the panel, so a +//! `panic::halt_all_cpus` calls [`stand_down`] before it holds the panel, so a //! panic report is never replaced by an expiry. use core::sync::atomic::{AtomicBool, AtomicU64, AtomicU8, Ordering::Relaxed}; @@ -72,7 +72,7 @@ pub fn claim_the_reset() -> bool { /// Stand this bound down for the rest of the machine's life. /// -/// Called from `apic::halt_all_cpus` beside [`crate::hardlockup::stand_down`]: +/// Called from `panic::halt_all_cpus` beside [`crate::hardlockup::stand_down`]: /// from there this machine holds a panic report under a bound of its own, and an /// expiry would seal a `WEDGED` record over it. Disarms rather than latching a /// second flag, so [`poll`] stays one relaxed load. @@ -189,11 +189,11 @@ pub fn start() { /// pays a caller-saved prologue on every tick of every CPU armed or not, and /// that cost is the entry's rather than this function's. /// -/// `extern "sysv64"` because the Ring 0 half of the timer entry calls it from +/// `extern "C"` because the Ring 0 half of the timer entry calls it from /// naked assembly, where the ABI is written out rather than inferred. -pub extern "sysv64" fn poll() { +pub extern "C" fn poll() { let at = AT_TSC.load(Relaxed); - if at == 0 || crate::arch::cpu::rdtsc() < at { + if at == 0 || crate::arch::cpu::counter() < at { return; } expire() @@ -243,7 +243,7 @@ pub fn stage_a_wedge() -> ! { log!("{WEDGE_STAGED}: every CPU stops taking scheduler passes from here"); STAGED.store(true, Relaxed); // A core still asleep is not a core this control has wedged. - crate::arch::apic::kick_all_but_self(); + crate::arch::irqchip::kick_all_but_self(); this_cpu() } @@ -278,7 +278,7 @@ fn this_cpu() -> ! { // claims. crate::preempt::disable(); let arrived_awake = crate::arch::cpu::interrupts_enabled(); - crate::arch::apic::arm_within(toyos_sched::fair::QUANTUM_NS); + crate::arch::irqchip::arm_within(toyos_sched::fair::QUANTUM_NS); crate::arch::cpu::enable_interrupts(); log!( "wedge: cpu{} {}", diff --git a/kernel/src/drivers/acpi.rs b/kernel/src/drivers/acpi.rs index ac094425d58..7c4f04417a7 100644 --- a/kernel/src/drivers/acpi.rs +++ b/kernel/src/drivers/acpi.rs @@ -105,10 +105,8 @@ const SDT_OEM_ID: usize = 10; /// — so a row here is a table that checksummed, and the line that closes the /// list is the machine's answer to how many of them it has. pub fn inventory(rsdp_addr: u64) { - /// The five this kernel decodes: the MADT for its CPUs and IO APICs, the - /// FADT for reset, soft-off and the century register, the HPET for the - /// clock, the MCFG for ECAM and the DMAR for the IOMMU. - const READ: &[&[u8; 4]] = &[b"APIC", b"FACP", b"HPET", b"MCFG", b"DMAR"]; + // The tables this architecture decodes. + const READ: &[&[u8; 4]] = crate::arch::boot::ACPI_TABLES; let mut validated = 0usize; for signature in READ { @@ -188,6 +186,10 @@ pub fn init_power(rsdp_addr: u64) { return; }; let pm1a = pm1a as u16; + if pm1a != 0 && !crate::arch::pio::EXISTS { + log!("ACPI: FADT puts PM1a control at port {pm1a:#x}, and this machine has no I/O port space — no soft-off"); + return; + } // Prefer X_DSDT over DSDT; a revision claiming 2.0 doesn't prove the field is present, so the length is checked rather than trusting the revision alone. let dsdt_addr = toyos_acpi::dsdt_address(&fadt.0); @@ -271,6 +273,9 @@ pub fn init_reset(rsdp_addr: u64) { } }; match toyos_acpi::reset_register(&fadt.0) { + Reset::Port { port, .. } if !crate::arch::pio::EXISTS => { + log!("ACPI: the reset register is SystemIO {port:#x}, and this machine has no I/O port space — no reboot"); + } Reset::Port { port, value } => { RESET_PORT.store(port, Ordering::Relaxed); RESET_VALUE.store(value, Ordering::Relaxed); @@ -322,7 +327,7 @@ pub fn reset_now() -> ! { crate::arch::cpu::halt(); } // SAFETY: the port is non-zero only where `init_reset` decoded an 8-bit System I/O register, and the value is that register's. - unsafe { crate::arch::cpu::outb(port, RESET_VALUE.load(Ordering::Relaxed)) }; + unsafe { crate::arch::pio::outb(port, RESET_VALUE.load(Ordering::Relaxed)) }; crate::arch::cpu::halt(); } @@ -342,7 +347,7 @@ pub fn shutdown() -> ! { if pm1a != 0 { let val = (slp_typ << 10) | SLP_EN; // SAFETY: pm1a and slp_typ come only from the validated FADT parse via PM1A_CNT_PORT/SLP_TYPA, and the zero check above confirms that parse happened. - unsafe { crate::arch::cpu::outw(pm1a, val) }; + unsafe { crate::arch::pio::outw(pm1a, val) }; } crate::arch::cpu::halt(); @@ -378,7 +383,13 @@ pub fn parse_madt(rsdp_addr: u64) -> Option { } Ok(MadtEntry::IoApic(entry)) => io_apics.push(entry), Ok(MadtEntry::SourceOverride(entry)) => source_overrides.push(entry), - Ok(MadtEntry::Other(_)) => {} + // A GIC structure on a machine this walk reads APICs from is as + // foreign to it as a type it does not know. + Ok(MadtEntry::Gicc(_) + | MadtEntry::Gicd { .. } + | MadtEntry::Gicr { .. } + | MadtEntry::Its { .. } + | MadtEntry::Other(_)) => {} Err(halt) => { log!( "ACPI: MADT entry at +{} declares {} bytes of a {}-byte list — stopping", diff --git a/kernel/src/drivers/gop.rs b/kernel/src/drivers/gop.rs index 7c15c56c9d0..8c4ce15260e 100644 --- a/kernel/src/drivers/gop.rs +++ b/kernel/src/drivers/gop.rs @@ -2,7 +2,6 @@ use alloc::boxed::Box; use toyos_abi::syscall::SyscallError; -use crate::arch::{mtrr, pat}; use crate::mm::paging::{CachePolicy, MmioPolicy}; use crate::mm::{PAGE_2M, align_2m_checked, DirectMap}; use crate::gpu::{Gpu, GpuInfo}; @@ -61,7 +60,7 @@ pub fn init( width, height, stride, pixel_format, addr); // Reads the cache policy actually installed, not the one requested. - let mtrr = mtrr::range_type(addr, aligned_size); + let memory_type = crate::mm::paging::scanout_memory_type(addr, aligned_size); let installed = crate::mm::paging::kernel() .lock() .direct_map_policy(addr) @@ -70,10 +69,7 @@ pub fn init( installed == CachePolicy::WriteCombining, "GOP: the scanout is mapped {installed:?}" ); - log!("GOP: scanout memory type {} (MTRR {}, PAT entry {})", - mtrr::effective_under_wc(&mtrr).map_or("unknown", |t| t.name()), - mtrr.name(), - pat::WC_ENTRY); + log!("GOP: scanout memory type {memory_type}"); let cursor_pages = crate::mm::pmm::alloc_contiguous(1, crate::mm::pmm::Category::Framebuffer).expect("GOP: cursor alloc failed"); let cursor_phys = cursor_pages[0].direct_map().phys(); @@ -81,7 +77,7 @@ pub fn init( let cursor = Region { phys: DirectMap::from_phys(cursor_phys), size: PAGE_2M, - cache: CachePolicy::DeferToMtrr, + cache: CachePolicy::Normal, pages: None, }; core::mem::forget(cursor_pages); // lives forever (GPU is never torn down) diff --git a/kernel/src/drivers/hda.rs b/kernel/src/drivers/hda.rs index 5f041465459..daf8f95b3d6 100644 --- a/kernel/src/drivers/hda.rs +++ b/kernel/src/drivers/hda.rs @@ -436,7 +436,7 @@ pub fn init(devices: &[PciDevice]) { let pcm_region = Region { phys: crate::DirectMap::from_phys(pcm_view.host_phys()), size: crate::mm::PAGE_2M, - cache: CachePolicy::DeferToMtrr, + cache: CachePolicy::Normal, pages: None, }; @@ -644,7 +644,7 @@ fn reset_stream(stream: Mmio) -> bool { /// Arm the completion interrupt, or say why this machine has no HDA audio; a refusal, never a /// panic, over a peripheral. fn arm_interrupt(pci: &PciDevice) -> bool { - let vector = crate::arch::idt::HDA_VECTOR; + let vector = crate::arch::trap::HDA_VECTOR; if pci.enable_msix(vector).is_ok() || pci.enable_msi(vector) { return true; } diff --git a/kernel/src/drivers/mod.rs b/kernel/src/drivers/mod.rs index 22058591658..586cf961a4e 100644 --- a/kernel/src/drivers/mod.rs +++ b/kernel/src/drivers/mod.rs @@ -5,9 +5,8 @@ #![warn(clippy::undocumented_unsafe_blocks)] pub mod serial; +pub mod serial_lock; pub mod acpi; -pub mod i8042; -pub mod ioapic; pub mod pci; pub mod nvme; pub mod xhci; @@ -19,7 +18,6 @@ pub mod virtio_sound; pub mod gop; pub mod hda; pub mod panic_console; -pub mod watchdog; /// The pool every driver here allocates its DMA out of. pub use crate::mm::DmaPool; diff --git a/kernel/src/drivers/nvme.rs b/kernel/src/drivers/nvme.rs index 4a89a6e9a74..b8674aa532d 100644 --- a/kernel/src/drivers/nvme.rs +++ b/kernel/src/drivers/nvme.rs @@ -7,7 +7,7 @@ //! `read_sectors`/`write_sectors`; `admin` takes none and is bounded by //! [`COMMAND`] alone. A refusal is taken between commands, never inside one. -use core::sync::atomic::{fence, Ordering}; +use crate::arch::barrier; use toyos_untrusted::{Refused, Untrusted}; use crate::mm::Mmio; use super::pci::PciDevice; @@ -202,10 +202,10 @@ impl NvmeQueue { fn submit(&mut self, bar: &Mmio, cmd: SqEntry) { // Bounded by `sq_tail % QUEUE_DEPTH` against the page `init` allocated; - // the fence and doorbell below are what tell the device it happened. + // the doorbell below is what tells the device it happened, and as an + // `Mmio` write it is ordered after the entry. self.sq.write(self.sq_tail as usize * core::mem::size_of::(), cmd); self.sq_tail = (self.sq_tail + 1) % QUEUE_DEPTH as u16; - fence(Ordering::Release); bar.write_u32(self.sq_doorbell, self.sq_tail as u32); } @@ -238,6 +238,9 @@ impl NvmeQueue { if !answered { return Err(Unanswered::Silent); } + // The entry, and the data it completes, read after the phase tag that + // says they are there. + barrier::dma_rmb(); let cq: CqEntry = self.cq.read(at(self.cq_head)); let status = cq.status >> 1; let cid = Untrusted::new(cq.cid); diff --git a/kernel/src/drivers/panic_console/mod.rs b/kernel/src/drivers/panic_console/mod.rs index acbb887a4ca..ab60e9f5b1e 100644 --- a/kernel/src/drivers/panic_console/mod.rs +++ b/kernel/src/drivers/panic_console/mod.rs @@ -13,6 +13,7 @@ mod access; mod latch; +mod published; use core::cell::UnsafeCell; use core::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, AtomicUsize, Ordering}; @@ -81,11 +82,31 @@ impl Fb { height: 0, format: 0, }; + + /// The descriptor as the seqlock's words. + fn words(self) -> [u64; published::WORDS] { + [ + self.ptr as u64, + self.bytes, + u64::from(self.stride_px) << 32 | u64::from(self.width), + u64::from(self.height) << 32 | u64::from(self.format), + ] + } + + fn from_words([ptr, bytes, stride_width, height_format]: [u64; published::WORDS]) -> Self { + Self { + ptr: ptr as *mut u8, + bytes, + stride_px: (stride_width >> 32) as u32, + width: stride_width as u32, + height: (height_format >> 32) as u32, + format: height_format as u32, + } + } } struct FbCell(UnsafeCell); -// SAFETY: the panic path may take no lock; `FB` is published and read only -// under the `SEQ` seqlock, and `PENDING` has one writer at a time. +// SAFETY: the panic path may take no lock; `PENDING` has one writer at a time. unsafe impl Sync for FbCell {} /// A screenful-and-then-some of rendered log; which lines came from an `alert!` is a [`Level`] flag, never inferred from the text. @@ -206,11 +227,10 @@ struct RenderedCell(UnsafeCell); // or swap atomically. `PAINTING`, `CAPTURE`, and `CAPTURE_ACCESS` serialise the three cells. unsafe impl Sync for RenderedCell {} -/// Seqlock over `FB`; even means stable, odd means a publisher is inside. +/// The descriptor painters draw through, behind a seqlock (`published`). /// Not a `Lock`: its guard drop can dispatch the scheduler, forbidden here. -/// A publisher that dies mid-update leaves this odd forever, costing the screen and nothing else. -static SEQ: AtomicU32 = AtomicU32::new(0); -static FB: FbCell = FbCell(UnsafeCell::new(Fb::DETACHED)); +/// A publisher that dies mid-update leaves it changing forever, costing the screen and nothing else. +static FB: published::Published = published::Published::new(); /// Exactly one painter at a time, taken by every painter without exception. /// [`render`] and [`seal_wedge`] never release it; every other painter does, @@ -312,17 +332,10 @@ static PENDING: FbCell = FbCell(UnsafeCell::new(Fb::DETACHED)); static RAW_PHYS: AtomicU64 = AtomicU64::new(0); static RAW_SIZE: AtomicU64 = AtomicU64::new(0); -/// Boot-time and `set_resolution`-window only: publishers never race, so the load-then-store of `SEQ` needs no CAS. +/// Boot-time and `set_resolution`-window only: publishers never race. fn publish(fb: Fb) { forget_the_glass(); - let seq = SEQ.load(Ordering::Relaxed); - SEQ.store(seq.wrapping_add(1), Ordering::Relaxed); - // The release fence orders the descriptor store between the odd and even markers; a release RMW alone would not. - core::sync::atomic::fence(Ordering::Release); - // SAFETY: the seqlock's payload store; `SEQ` is odd across this write, so - // a concurrent `snapshot` discards what it reads. Publishers never race each other. - unsafe { *FB.0.get() = fb }; - SEQ.store(seq.wrapping_add(2), Ordering::Release); + FB.publish(fb.words()); } /// Stop painting until the next [`rearm`], for a window where the framebuffer may be freed and reallocated. @@ -349,24 +362,11 @@ pub fn disable() { detach(); } -/// Torn means unavailable, never a wild pointer: only two descriptors are -/// ever published, and every torn mixture is caught downstream by the null check or a zero field collapsing draws to no-ops. -/// A second valid descriptor would break that argument, leaving only the fences. +/// One publication whole, or nothing: a descriptor still changing is no +/// screen this instant (`published`, and `kernel-loom`'s `panic_console_publish`). fn snapshot() -> Option { - for _ in 0..4 { - let before = SEQ.load(Ordering::Acquire); - if before & 1 != 0 { - continue; - } - // SAFETY: the seqlock's payload read, the one place `FB` is read - // while a publisher may be inside it — sound on the `SEQ` comparison below, not exclusion. - let fb = unsafe { *FB.0.get() }; - core::sync::atomic::fence(Ordering::Acquire); - if SEQ.load(Ordering::Relaxed) == before { - return (!fb.ptr.is_null()).then_some(fb); - } - } - None + let fb = Fb::from_words(FB.snapshot()?); + (!fb.ptr.is_null()).then_some(fb) } /// Reject a descriptor that could turn a panic into a wild write. @@ -461,10 +461,7 @@ pub fn arm(args: &KernelArgs, maps: &[MemoryMapEntry]) { // loader hands the scanout over uncacheable; `pat::init` runs before this // function so that there is a write-combining entry to point its leaves at. let combining = reclaimed.is_none() - && mm::paging::boot_map_write_combining( - args.gop_framebuffer, - align_2m(args.gop_framebuffer_size as usize) as u64, - ); + && mm::paging::boot_map_write_combining(args.gop_framebuffer, args.gop_framebuffer_size); // **The panel is taken before anything above or below it can fail, and // this record is what proves the kernel entered.** A fault in the walk or @@ -745,9 +742,9 @@ pub fn hold_the_panel(mut bound: Bound) -> ! { /// anything that does not skip them. [`i8042::poll_byte`] is an `inb` — no /// lock, no MMIO. /// -/// [`i8042::poll_byte`]: crate::drivers::i8042::poll_byte +/// [`i8042::poll_byte`]: crate::arch::keyboard_controller::poll_byte fn read_key(keys: &mut KeyDecoder, bound: &mut Bound) -> Option { - let (byte, false) = crate::drivers::i8042::poll_byte()? else { + let (byte, false) = crate::arch::keyboard_controller::poll_byte()? else { return None; }; let outcome = keys.feed(byte); @@ -813,7 +810,7 @@ pub mod stall { return; } crate::log!("{HELD}"); - crate::arch::apic::halt_all_cpus(); + crate::panic::halt_all_cpus(); } } @@ -1148,7 +1145,7 @@ pub fn log_census() { /// Charge one paint to the census. fn spent(began: u64, pixels: u64) { - let ticks = crate::arch::cpu::rdtsc().saturating_sub(began); + let ticks = crate::arch::cpu::counter().saturating_sub(began); PAINTS.fetch_add(1, Ordering::Relaxed); PIXELS.fetch_add(pixels, Ordering::Relaxed); TICKS.fetch_add(ticks, Ordering::Relaxed); @@ -1163,7 +1160,7 @@ fn paint(fill: Fill, view: View, page: Page, watch: Watch, stop: impl Fn() -> bo return; } let Some((cols, grid_rows)) = geometry(&fb) else { return }; - let began = crate::arch::cpu::rdtsc(); + let began = crate::arch::cpu::counter(); let mut pixels = 0u64; let text = view.text; let (total, pages, per) = pagination(text, cols, grid_rows); @@ -1288,10 +1285,9 @@ fn panel_carries_report(fb: &Fb) -> bool { }) } -/// Put every store this module has made on the bus: the scanout is write-combining, and stores can sit in a buffer with nothing to evict them. +/// Put every store this module has made on the bus. fn flush_stores() { - // SAFETY: `SFENCE` (SDM Vol. 3A §11.3.1) is the only way to drain a write-combining buffer; it touches no memory or register. - unsafe { core::arch::asm!("sfence", options(nostack, preserves_flags)) }; + crate::arch::barrier::scanout_flush(); } /// `[page 2/4]` into the bottom row's cells; not decoration — the pager advances on a timer with no key to press. @@ -1335,14 +1331,14 @@ fn write_num(out: &mut [u8], v: usize) -> usize { } /// Whether the first and last framebuffer pages resolve in the *current* -/// CR3, not `kernel_cr3()`: a panic in syscall context runs on a user address space. +/// tables, not `kernel_root()`: a panic in syscall context runs on a user address space. /// Proves it rather than assuming it, so broken paging becomes no console, never a fault inside the panic handler. fn mapped(fb: &Fb) -> bool { let base = fb.ptr as u64; let Some(last) = base.checked_add(fb.bytes.saturating_sub(1)) else { return false; }; - mm::paging::present_in_current_cr3(base) && mm::paging::present_in_current_cr3(last) + mm::paging::present_in_current_tables(base) && mm::paging::present_in_current_tables(last) } /// `pixel_format` is 0 for RGB, 1 for BGR (`bootloader/src/main.rs`). diff --git a/kernel/src/drivers/panic_console/published.rs b/kernel/src/drivers/panic_console/published.rs new file mode 100644 index 00000000000..44678a9c1d2 --- /dev/null +++ b/kernel/src/drivers/panic_console/published.rs @@ -0,0 +1,77 @@ +//! The published framebuffer descriptor: a seqlock whose payload is atomic +//! words, so a reader racing the one publisher reads torn words it throws +//! away rather than racing a plain write (a data race, and undefined, in the +//! Rust memory model whatever the check after it concludes). +//! +//! The writer marks the sequence odd, fences `Release`, stores the words +//! `Relaxed` and marks it even with a `Release` store; the reader loads the +//! sequence `Acquire`, the words `Relaxed`, fences `Acquire` and reloads the +//! sequence. A reader whose two loads agree on an even value read words the +//! publication that value ends wrote, all of them. +//! No `crate::` references: `kernel-loom` compiles this file under `feature = "loom"`. + +#[cfg(not(feature = "loom"))] +use core::sync::atomic::{fence, AtomicU32, AtomicU64, Ordering}; + +#[cfg(feature = "loom")] +use loom::sync::atomic::{fence, AtomicU32, AtomicU64, Ordering}; + +/// Words in one descriptor. +pub const WORDS: usize = 4; + +/// Tries a snapshot makes before answering that the descriptor is changing. +const TRIES: usize = 4; + +pub struct Published { + seq: AtomicU32, + words: [AtomicU64; WORDS], +} + +impl Published { + // Must stay `const`: the kernel's is a `static`, and loom's atomics have no const constructor. + #[cfg(not(feature = "loom"))] + pub const fn new() -> Self { + Self { seq: AtomicU32::new(0), words: [const { AtomicU64::new(0) }; WORDS] } + } + + #[allow(clippy::new_without_default)] + #[cfg(feature = "loom")] + pub fn new() -> Self { + Self { seq: AtomicU32::new(0), words: core::array::from_fn(|_| AtomicU64::new(0)) } + } + + /// Replace the descriptor. One publisher at a time: the caller's own + /// exclusion (the boot sequence, a mode set's single-CPU window) is what + /// makes the load-then-store of `seq` sound. + pub fn publish(&self, words: [u64; WORDS]) { + let seq = self.seq.load(Ordering::Relaxed); + self.seq.store(seq.wrapping_add(1), Ordering::Relaxed); + // Orders the odd mark before every word below: a reader that sees a + // new word sees the mark, and throws the read away. + #[cfg(not(feature = "seqlock-writer-fence-off"))] + fence(Ordering::Release); + for (slot, word) in self.words.iter().zip(words) { + slot.store(word, Ordering::Relaxed); + } + self.seq.store(seq.wrapping_add(2), Ordering::Release); + } + + /// The descriptor, whole, or `None` when every try met a publication in + /// progress. + pub fn snapshot(&self) -> Option<[u64; WORDS]> { + for _ in 0..TRIES { + let before = self.seq.load(Ordering::Acquire); + if before & 1 != 0 { + continue; + } + let words = core::array::from_fn(|i| self.words[i].load(Ordering::Relaxed)); + // Orders the word loads before the recheck: a word a later + // publication wrote is one whose odd mark the recheck sees. + fence(Ordering::Acquire); + if self.seq.load(Ordering::Relaxed) == before { + return Some(words); + } + } + None + } +} diff --git a/kernel/src/drivers/pci.rs b/kernel/src/drivers/pci.rs index 224f3e3913f..0deb9bf9a4c 100644 --- a/kernel/src/drivers/pci.rs +++ b/kernel/src/drivers/pci.rs @@ -27,9 +27,8 @@ const INVALID_VENDOR: u16 = 0xFFFF; /// The one MSI-X table entry this kernel programs; a device's queues must point at it too. pub const MSIX_ENTRY: u16 = 0; -// Every device interrupt in this kernel targets this LAPIC address, so all land on cpu0. -const MSG_ADDR: u32 = 0xFEE0_0000; -// The same CPU, named as a destination rather than encoded in an address, for the unit to put in an entry. +// Every device interrupt in this kernel targets cpu0, named as a destination for +// the message and for the unit to put in an entry. const MSG_DEST: u32 = 0; /// No requester id: a bus/device/function is sixteen bits, so this is none of them. @@ -375,9 +374,8 @@ impl PciDevice { fn message(&self, vector: u8) -> Option<(u32, u32)> { let dest = if crate::actuator::iommu_dest_apic1() { 1 } else { MSG_DEST }; match crate::iommu::remap_msi(self.bus, self.dev, self.func, vector, dest) { - // Destination id at address bits 19:12, so the compatibility message - // names the same CPU the remappable one would. - crate::iommu::Delivery::Direct => Some((MSG_ADDR | (dest << 12), vector as u32)), + // The same CPU the remappable one would name. + crate::iommu::Delivery::Direct => Some(crate::arch::msi_message(dest, vector)), crate::iommu::Delivery::Remapped(m) => Some((m.address, m.data)), crate::iommu::Delivery::Refused(why) => { log!("PCI {:02x}:{:02x}.{}: not armed — {why}", self.bus, self.dev, self.func); diff --git a/kernel/src/drivers/serial.rs b/kernel/src/drivers/serial.rs index f40fca95f37..c0e440c31af 100644 --- a/kernel/src/drivers/serial.rs +++ b/kernel/src/drivers/serial.rs @@ -1,4 +1,6 @@ -//! The 16550 and the virtio-console, and the one lock that serialises them. +//! The console UART and the virtio-console, and the one lock that serialises +//! them. Where the UART is and what it is are the architecture's +//! (`arch::console_uart`). //! Every writer takes [`BackendGuard`] once per whole unit (a record, a //! userland `write`, a panic report) and holds it for that whole unit; that //! is the only source of line atomicity. Every unit taken under the guard is @@ -6,46 +8,19 @@ //! kernel lock formats here. use core::sync::atomic::{AtomicBool, Ordering}; -use crate::arch::cpu::{inb, outb}; -use crate::log; +use crate::arch::IrqGuard; +use super::serial_lock::{BackendLock, Held}; -const PORT: u16 = 0x3f8; // COM1 +use crate::arch::console_uart as uart; -// Latched once from `init`'s loopback probe: hardware with no SuperIO -// reads 0xFF on every access, indistinguishable from a ready UART. +// Latched once from `init`: the architecture's own answer about whether a UART +// is there, since one that is not may still read as ready. static UART_PRESENT: AtomicBool = AtomicBool::new(false); -// Every register is `PORT + n`; the identity op keeps that pattern uniform -// across all eight lines instead of special-casing the data register. -#[allow(clippy::identity_op)] -pub fn init() { - // SAFETY: `outb`/`inb` require the caller to own the port and the byte; - // every port here is `PORT + n` for `n` in 0..=4, inside COM1's own - // register block, and the writes are the 16550's documented init sequence. - // Order matters: DLAB must precede the divisor writes and loopback mode - // must precede the probe, or the sequence misprograms the chip. - let loopback = unsafe { - outb(PORT + 1, 0x00); // Disable all interrupts - outb(PORT + 3, 0x80); // Enable DLAB (set baud rate divisor) - outb(PORT + 0, 0x03); // Set divisor to 3 (lo byte) 38400 baud - outb(PORT + 1, 0x00); // (hi byte) - outb(PORT + 3, 0x03); // 8 bits, no parity, one stop bit - outb(PORT + 2, 0xC7); // Enable FIFO, clear them, with 14-byte threshold - outb(PORT + 4, 0x0B); // IRQs enabled, RTS/DSR set - outb(PORT + 4, 0x1E); // Set in loopback mode, test the serial chip - outb(PORT + 0, 0xAE); // Test serial chip (send byte 0xAE and check if serial returns same byte) - let seen = inb(PORT + 0); - UART_PRESENT.store(seen == 0xAE, Ordering::Relaxed); - outb(PORT + 4, 0x0F); // Normal operation mode - seen - }; - // Logs the raw byte, not just the verdict: distinguishes "no SuperIO" - // (0xFF) from a wrong response and a right chip at the wrong port. - log!( - "serial: 16550 loopback read {:#04x} ({})", - loopback, - if loopback == 0xAE { "present" } else { "absent or wrong port" } - ); +/// Find and program the console UART, off the firmware tables at `rsdp_addr` +/// where the architecture places it by them. +pub fn init(rsdp_addr: u64) { + UART_PRESENT.store(uart::init(rsdp_addr), Ordering::Relaxed); console_changed(); } @@ -63,7 +38,7 @@ pub fn has_console() -> bool { !matches!(backend(), Backend::None) } -/// Which channel a write goes to right now; virtio-console is preferred over a 16550. +/// Which channel a write goes to right now; virtio-console is preferred over the UART. #[derive(Clone, Copy, PartialEq, Eq)] #[repr(u8)] pub enum Backend { @@ -84,61 +59,27 @@ pub fn backend() -> Backend { } -static BACKEND_LOCKED: AtomicBool = AtomicBool::new(false); +static BACKEND: BackendLock = BackendLock::new(); /// Exclusive access to the serial backend; interrupts are off for as long as the guard lives. /// Same-CPU re-entry from an IRQ handler deadlocks the spin. pub struct BackendGuard { - rflags: SavedFlags, -} - -/// This CPU's own `RFLAGS`, captured by `pushfq`; the only value `popfq` may be given. -/// Not `Copy`/`Clone`: one CPU's state at one instant, not to be duplicated. -pub struct SavedFlags(u64); - -impl SavedFlags { - /// Restores the flags; `&self` because `Drop` cannot move a field out, and restoring twice is idempotent. - #[inline] - fn restore(&self) { - // SAFETY: `popfq` has no safe spelling; `self.0` came only from this - // CPU's own `pushfq` in `save_and_cli`, so no unintended bit reaches RFLAGS. - unsafe { - core::arch::asm!( - "push {}", - "popfq", - in(reg) self.0, - options(nomem), - ); - } - } + // Fields drop in order: the backend is released before interrupts reopen. + _held: Held<'static>, + _irq: IrqGuard, } impl BackendGuard { pub fn lock() -> Self { - let rflags = save_and_cli(); - while BACKEND_LOCKED - .compare_exchange_weak(false, true, Ordering::Acquire, Ordering::Relaxed) - .is_err() - { - while BACKEND_LOCKED.load(Ordering::Relaxed) { - core::hint::spin_loop(); - } - } - Self { rflags } + let irq = IrqGuard::close(); + Self { _held: BACKEND.lock(), _irq: irq } } /// Non-blocking acquire: `None` if another CPU already holds the backend. pub fn try_lock() -> Option { - let rflags = save_and_cli(); - if BACKEND_LOCKED - .compare_exchange(false, true, Ordering::Acquire, Ordering::Relaxed) - .is_ok() - { - Some(Self { rflags }) - } else { - rflags.restore(); - None - } + let irq = IrqGuard::close(); + let held = BACKEND.try_lock()?; + Some(Self { _held: held, _irq: irq }) } /// Writes raw bytes with no escape stripping; callers must pre-strip via [`write_console`]. @@ -154,47 +95,21 @@ impl BackendGuard { if super::virtio_console::is_ready() { super::virtio_console::has_data_locked() } else { - uart_present() && inb(PORT + 5) & 0x01 != 0 + uart_present() && uart::rx_ready() } } pub fn try_read_byte(&mut self) -> Option { if super::virtio_console::is_ready() { super::virtio_console::try_read_byte_locked() - } else if uart_present() && inb(PORT + 5) & 0x01 != 0 { - Some(inb(PORT)) + } else if uart_present() && uart::rx_ready() { + Some(uart::read_byte()) } else { None } } } -impl Drop for BackendGuard { - fn drop(&mut self) { - BACKEND_LOCKED.store(false, Ordering::Release); - self.rflags.restore(); - } -} - -/// This CPU's `RFLAGS`, captured with interrupts off in one instruction sequence: -/// the value is stale if anything runs between the read and `cli`. -#[inline] -fn save_and_cli() -> SavedFlags { - let rflags: u64; - // SAFETY: irreducible — `pushfq`/`cli` have no safe spelling; the asm reads - // RFLAGS and clears IF only, writes no memory, and touches no other register. - unsafe { - core::arch::asm!( - "pushfq", - "pop {}", - "cli", - out(reg) rflags, - options(nomem), - ); - } - SavedFlags(rflags) -} - pub fn has_data() -> bool { let g = BackendGuard::lock(); g.has_data() @@ -386,18 +301,16 @@ fn uart_write_bytes(bytes: &[u8]) { } for &b in bytes { for _ in 0..THRE_SPIN_LIMIT { - if inb(PORT + 5) & 0x20 != 0 { + if uart::tx_ready() { break; } core::hint::spin_loop(); } - // SAFETY: `outb` requires ownership of the port and the byte; `PORT` - // is COM1's own data register, and the byte is console output only. - unsafe { outb(PORT, b) }; + uart::write_byte(b); } } -/// Writes straight to the 16550, bypassing the ring, the lock and virtio-console: no allocation, bounded per byte. +/// Writes straight to the UART, bypassing the ring, the lock and virtio-console: no allocation, bounded per byte. pub fn panic_raw(bytes: &[u8]) { uart_write_bytes(bytes); } diff --git a/kernel/src/drivers/serial_lock.rs b/kernel/src/drivers/serial_lock.rs new file mode 100644 index 00000000000..6928a93ce80 --- /dev/null +++ b/kernel/src/drivers/serial_lock.rs @@ -0,0 +1,68 @@ +//! The console backend's one lock: a flag that only the [`Held`] a won +//! exchange returns ever clears, by its drop. A lost `try_lock` builds no +//! [`Held`], so it can never release the lock another CPU holds. +//! No `crate::` references: `kernel-loom` compiles this file directly under `feature = "loom"`. + +#[cfg(not(feature = "loom"))] +use core::sync::atomic::{AtomicBool, Ordering}; + +#[cfg(feature = "loom")] +use loom::sync::atomic::{AtomicBool, Ordering}; + +pub struct BackendLock { + locked: AtomicBool, +} + +/// The lock, held for as long as this lives. +pub struct Held<'a> { + lock: &'a BackendLock, +} + +impl BackendLock { + /// Must stay `const`: the backend's lock is a `static`. + #[cfg(not(feature = "loom"))] + pub const fn new() -> Self { + Self { locked: AtomicBool::new(false) } + } + + #[allow(clippy::new_without_default)] + #[cfg(feature = "loom")] + pub fn new() -> Self { + Self { locked: AtomicBool::new(false) } + } + + pub fn lock(&self) -> Held<'_> { + while self + .locked + .compare_exchange_weak(false, true, Ordering::Acquire, Ordering::Relaxed) + .is_err() + { + while self.locked.load(Ordering::Relaxed) { + core::hint::spin_loop(); + } + } + Held { lock: self } + } + + /// `None` if another holder has it. + pub fn try_lock(&self) -> Option> { + // Not `then_some`: its argument is built whether or not the exchange + // won, and a `Held` built on a loss drops, and its drop releases the + // lock another CPU holds. + let won = self.locked.compare_exchange(false, true, Ordering::Acquire, Ordering::Relaxed).is_ok(); + #[cfg(feature = "serial-try-lock-then-some")] + return won.then_some(Held { lock: self }); + #[cfg(not(feature = "serial-try-lock-then-some"))] + if won { + Some(Held { lock: self }) + } else { + None + } + } +} + +impl Drop for Held<'_> { + fn drop(&mut self) { + self.lock.locked.store(false, Ordering::Release); + } +} diff --git a/kernel/src/drivers/virtio.rs b/kernel/src/drivers/virtio.rs index 015eab1fe02..3587b459cad 100644 --- a/kernel/src/drivers/virtio.rs +++ b/kernel/src/drivers/virtio.rs @@ -1,4 +1,4 @@ -use core::sync::atomic::{fence, Ordering}; +use crate::arch::barrier; use toyos_untrusted::{Refused, Untrusted}; @@ -376,8 +376,9 @@ impl UsedRingConsumer<'_> { if used_idx == self.last_used_idx { return None; } - // Acquire: pairs with the device's release when it bumps the used idx after writing the element. - fence(Ordering::Acquire); + // The device writes the element before it bumps the idx (virtio 1.2 + // §2.7.8), so the element is read after the idx that counts it. + barrier::dma_rmb(); let slot = self.last_used_idx % self.size; let id: Untrusted = Untrusted::new(self.used.read(USED_RING_OFF + slot as usize * USED_ELEM_SIZE)); @@ -554,10 +555,13 @@ impl<'pool> Virtqueue<'pool> { let avail_idx: u16 = self.avail.read(AVAIL_IDX_OFF); self.avail.write(AVAIL_RING_OFF + (avail_idx % size) as usize * 2, first_desc); - fence(Ordering::Release); + // The descriptors and the ring entry before the idx that publishes them + // (virtio 1.2 §2.7.13). + barrier::dma_wmb(); self.avail.write(AVAIL_IDX_OFF, avail_idx.wrapping_add(1)); - fence(Ordering::Release); + // The idx before the notification: the notify is an `Mmio` write, which + // orders every earlier store before it. let notify_off = self.notify_offset as u64 * notify_multiplier as u64; notify_mmio.write_u16(notify_off, queue_index); @@ -581,7 +585,8 @@ impl<'pool> Virtqueue<'pool> { if used_idx == self.last_used_idx { return None; } - fence(Ordering::Acquire); + // The element after the idx that counts it, as in `UsedRingConsumer::poll`. + barrier::dma_rmb(); let slot = self.last_used_idx % self.size; let id = self.used_ring_id(slot); let len = self.used_ring_len(slot); @@ -637,7 +642,8 @@ impl<'pool> Virtqueue<'pool> { let slot = at % self.size; self.used.write(self.used_elem_at(slot), id); self.used.write(self.used_elem_at(slot) + 4, len); - fence(Ordering::Release); + // As a device does: the element before the idx. + barrier::dma_wmb(); self.used.write::(USED_IDX_OFF, at.wrapping_add(1)); } diff --git a/kernel/src/drivers/virtio_console.rs b/kernel/src/drivers/virtio_console.rs index 58ed6141c27..5c4cfd01685 100644 --- a/kernel/src/drivers/virtio_console.rs +++ b/kernel/src/drivers/virtio_console.rs @@ -1,7 +1,7 @@ //! VirtIO console: single-port (no MULTIPORT), replacing the 16550 UART as //! the kernel log channel after init. Uses queues 0 (RX) and 1 (TX). //! -//! RX is poll-driven: no UART IRQ handler is wired (see `arch/idt/mod.rs`). +//! RX is poll-driven: no UART IRQ handler is wired (see `arch/x86_64/idt/mod.rs`). use core::cell::UnsafeCell; use core::mem::MaybeUninit; diff --git a/kernel/src/drivers/virtio_gpu.rs b/kernel/src/drivers/virtio_gpu.rs index 9cba23e4e87..d81f73c42cb 100644 --- a/kernel/src/drivers/virtio_gpu.rs +++ b/kernel/src/drivers/virtio_gpu.rs @@ -238,8 +238,8 @@ struct GpuController { impl GpuController { /// Reads what the device wrote into the response buffer. fn answer(&self) -> T { - // Safe only because submit_and_wait's fence(Acquire) already ordered - // the device's write before this read. + // Safe only because `submit_and_wait`'s `barrier::dma_rmb` already + // ordered the device's write before this read. self.resp.read(0) } @@ -426,7 +426,7 @@ impl GpuController { Region { phys: crate::DirectMap::from_phys(phys), size: fb_aligned, - cache: CachePolicy::DeferToMtrr, + cache: CachePolicy::Normal, pages: Some(alloc::sync::Arc::new(Pages::new(pages))), } }); @@ -658,7 +658,7 @@ pub fn init(devices: &[PciDevice]) -> Option<(Box, GpuInfo)> { gpu.cursor = Region { phys: crate::DirectMap::from_phys(cursor_phys), size: PAGE_2M, - cache: CachePolicy::DeferToMtrr, + cache: CachePolicy::Normal, pages: Some(alloc::sync::Arc::new(Pages::new(cursor_pages))), }; let cursor_backing = AttachedBacking::take(space, cursor_phys, PAGE_2M) diff --git a/kernel/src/drivers/virtio_sound.rs b/kernel/src/drivers/virtio_sound.rs index dab81c8621f..befdb9b15d8 100644 --- a/kernel/src/drivers/virtio_sound.rs +++ b/kernel/src/drivers/virtio_sound.rs @@ -354,7 +354,7 @@ pub fn init(devices: &[PciDevice]) { let dma_region = Region { phys: crate::DirectMap::from_phys(shared.host_phys()), size: crate::mm::PAGE_2M, - cache: CachePolicy::DeferToMtrr, + cache: CachePolicy::Normal, pages: None, }; let multiplier = device.notify_off_multiplier(); @@ -464,7 +464,7 @@ fn build_chains( /// handler is the TX used ring's only consumer, so an unarmed device leaves /// every period in flight forever. fn arm_interrupt(pci: &PciDevice, device: &VirtioDevice) -> bool { - let vector = crate::arch::idt::VIRTIO_SOUND_VECTOR; + let vector = crate::arch::trap::VIRTIO_SOUND_VECTOR; if pci.enable_msix(vector).is_err() { log!( "virtio-sound: NOT INITIALISED at PCI {:02x}:{:02x}.{} — its MSI-X could not be \ diff --git a/kernel/src/drivers/xhci/hid.rs b/kernel/src/drivers/xhci/hid.rs index 79a616d6591..49d205fb544 100644 --- a/kernel/src/drivers/xhci/hid.rs +++ b/kernel/src/drivers/xhci/hid.rs @@ -1,4 +1,3 @@ -use core::sync::atomic::{fence, Ordering}; use crate::{keyboard, mouse}; use super::{Mmio, Trb, TrbRing, TRB_NORMAL}; @@ -103,7 +102,7 @@ impl HidDevice { trb.status = self.report_size; trb.control = TRB_NORMAL | (1 << 5); // IOC self.int_ring.enqueue(trb); - fence(Ordering::Release); + // An `Mmio` write: ordered after the TRB it announces. db_base.write_u32(self.slot_id as u64 * 4, self.int_ep_dci as u32); } } diff --git a/kernel/src/drivers/xhci/mod.rs b/kernel/src/drivers/xhci/mod.rs index f54545b9b01..f5c2efb8bea 100644 --- a/kernel/src/drivers/xhci/mod.rs +++ b/kernel/src/drivers/xhci/mod.rs @@ -14,7 +14,9 @@ pub use wait::msc::{storage_flush, storage_read, storage_write}; use alloc::vec::Vec; use core::fmt; use core::num::NonZeroU8; -use core::sync::atomic::{fence, AtomicU64, AtomicUsize, Ordering}; +use core::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; + +use crate::arch::barrier; use crate::mm::Mmio; use crate::mm::Dma; use crate::log; @@ -319,8 +321,6 @@ pub(crate) const CALL_AFTER_BREAK: crate::time::Budget = crate::time::Budget::of "every wait is clipped to where the rungs still ahead of it begin, so the last rung runs whatever was spent before it and the call ends here", ); -// The caller spins with interrupts off, so a call that outlasted the TLB-ack tripwire would panic another CPU over a device. -const _: () = assert!(CALL_AFTER_BREAK.nanos() < crate::arch::tlb::ACK_TIMEOUT.nanos()); // A disk whose device left under the port rung's reset is waited for no longer than the rungs from that reset on were given to spend on it. const _: () = assert!( @@ -444,9 +444,16 @@ impl TrbRing { } /// One TRB at ring index `at` — the ring's one writer. + /// + /// The control word last, behind a barrier: it carries the Cycle bit, and a + /// controller running this ring owns the TRB the moment the bit matches its + /// own (xHCI 1.2 §4.9.2), so the body must already be there. fn put(&self, at: usize, trb: Trb) { - // Volatile: the controller reads this ring concurrently, and the Cycle bit tells it the TRB is complete. - self.buf.write(at * core::mem::size_of::(), trb) + let off = at * core::mem::size_of::(); + self.buf.write(off + core::mem::offset_of!(Trb, param), trb.param); + self.buf.write(off + core::mem::offset_of!(Trb, status), trb.status); + barrier::dma_wmb(); + self.buf.write(off + core::mem::offset_of!(Trb, control), trb.control); } /// Where the controller should resume, with the cycle state it must expect; bit 0 carries the cycle since a TRB address is 16-byte aligned. @@ -667,7 +674,7 @@ pub(super) fn look_for(index: usize) -> Option { /// caller that waits for a device on a CPU it holds with `IF` clear: the /// interrupt a connect raises may be this CPU's, and it takes none. `kick` /// wakes every other CPU, since a halted one has stopped its own timer; the -/// kick is refused only before `apic::init`, and no other CPU has been started +/// kick is refused only before `irqchip::init`, and no other CPU has been started /// before it. pub(super) fn ports_wanted(kick: bool) { PORT_WORK_AT.store(crate::clock::nanos_since_boot().max(1), Ordering::Relaxed); @@ -677,7 +684,7 @@ pub(super) fn ports_wanted(kick: bool) { let me = crate::arch::percpu::cpu_id(); for cpu in 0..crate::arch::smp::cpu_count() { if cpu != me { - crate::arch::apic::kick_cpu(cpu); + crate::arch::irqchip::kick_cpu(cpu); } } } @@ -866,7 +873,7 @@ impl XhciController { /// Put a command on the ring and ring the doorbell, answering with the address the completion will name it by. fn submit_command(&mut self, trb: Trb) -> u64 { let at = self.cmd_ring.enqueue(trb); - fence(Ordering::Release); + // An `Mmio` write: ordered after the TRB it announces. self.db_base.write_u32(0, 0); at } @@ -874,11 +881,15 @@ impl XhciController { /// One event, or `None` while the controller has not published the next; every reader goes through here since the ring is one shared queue. fn next_event(&mut self) -> Option { // Volatile so the poll observes the Cycle bit flipping; racing the controller by design (xHCI 1.2 §4.9.2). - let event: Trb = - self.event_ring.read(self.event_head as usize * core::mem::size_of::()); - if ((event.control & 1) != 0) != self.event_phase { + // The control word alone first: the rest of the TRB is the controller's + // until the Cycle bit it carries says otherwise. + let at = self.event_head as usize * core::mem::size_of::(); + let control: u32 = self.event_ring.read(at + core::mem::offset_of!(Trb, control)); + if ((control & 1) != 0) != self.event_phase { return None; } + barrier::dma_rmb(); + let event: Trb = self.event_ring.read(at); if crate::actuator::usb_slow_device() && !self.slow_device_would_have_answered(&event) { return None; } @@ -1569,7 +1580,7 @@ impl XhciController { } fn ring_doorbell(&self, slot: u8, dci: u8) { - fence(Ordering::Release); + // An `Mmio` write: ordered after every TRB enqueued before it. self.db_base.write_u32(slot as u64 * 4, dci as u32); } } @@ -1713,7 +1724,7 @@ fn take_within(bound: u64) -> Option= until { + if crate::arch::cpu::counter() >= until { return None; } core::hint::spin_loop(); diff --git a/kernel/src/drivers/xhci/wait/mod.rs b/kernel/src/drivers/xhci/wait/mod.rs index 28c0b1d5b3c..1c81b549cd3 100644 --- a/kernel/src/drivers/xhci/wait/mod.rs +++ b/kernel/src/drivers/xhci/wait/mod.rs @@ -31,10 +31,8 @@ mod depth_probe { "io-depth: a disk transfer is being waited for at preempt depth {depth}, task {:?}", crate::arch::percpu::current_tid().map(|t| t.raw()) ); - let rbp: u64; - // SAFETY: reads the frame pointer; `kernel_backtrace` stops at the first unreadable frame. - unsafe { core::arch::asm!("mov {}, rbp", out(reg) rbp, options(nomem, nostack)) }; - crate::arch::idt::exceptions::kernel_backtrace(rbp, 20); + // `kernel_backtrace` stops at the first unreadable frame. + crate::symbols::kernel_backtrace(crate::arch::cpu::frame_pointer(), 20); } } diff --git a/kernel/src/drivers/xhci/wait/msc.rs b/kernel/src/drivers/xhci/wait/msc.rs index 22eb37bfa23..19bce936a87 100644 --- a/kernel/src/drivers/xhci/wait/msc.rs +++ b/kernel/src/drivers/xhci/wait/msc.rs @@ -719,7 +719,9 @@ pub(in crate::drivers::xhci) mod staged { pub fn arm(n: u8, fault: Fault, only: Option) { FAULT.store(fault as u8, Ordering::Relaxed); ONLY.store(only.map_or(ANY, u16::from), Ordering::Relaxed); - LEFT.store(n, Ordering::Relaxed); + // Last and `Release`: the disk that takes a fault may be on another CPU, + // and what it takes has to be what was staged with the count it saw. + LEFT.store(n, Ordering::Release); } /// Take back whatever was staged and never taken, and say how many: a @@ -777,14 +779,19 @@ pub(in crate::drivers::xhci) mod staged { /// The fault the command about to go out was staged with, if any. pub fn take(opcode: u8) -> Option { + // The count first and `Acquire`, pairing with `arm`'s `Release`: the + // opcode and the fault read after it are the ones staged with it. + let mut left = LEFT.load(Ordering::Acquire); + if left == 0 { + return None; + } let only = ONLY.load(Ordering::Relaxed); if only != ANY && only != u16::from(opcode) { return None; } - let mut left = LEFT.load(Ordering::Relaxed); loop { let less = left.checked_sub(1)?; - match LEFT.compare_exchange_weak(left, less, Ordering::Relaxed, Ordering::Relaxed) { + match LEFT.compare_exchange_weak(left, less, Ordering::Acquire, Ordering::Acquire) { Ok(_) => break, Err(now) => left = now, } @@ -888,7 +895,7 @@ fn block_witness_holds(dev: &MscDevice, entered: BlockWitness) { dev.block, at.wrapping_sub(entered.at) as i64, top.wrapping_sub(at) as i64, - crate::arch::cpu::read_rsp(), + crate::arch::cpu::stack_pointer(), ); } diff --git a/kernel/src/elf/index.rs b/kernel/src/elf/index.rs index e0a5e7c605c..2d807a0c144 100644 --- a/kernel/src/elf/index.rs +++ b/kernel/src/elf/index.rs @@ -7,6 +7,7 @@ use alloc::vec::Vec; use crate::mm::{KernelSlice, MAX_HEAP_ALLOC}; +use toyos_elf::rela::ExeRefusal; use toyos_elf::{Rela, RelaCounts, RelaTable, RelocKind}; /// Relocation entries the loader needs, grouped by what it does with them. @@ -39,26 +40,23 @@ impl ParsedRelaEntries { } } -/// Groups both relocation tables, returning `None` when any one group would not fit a single kernel allocation. -pub fn parse_rela_entries(rela_data: &[u8], jmprel_data: &[u8]) -> Option { +/// Groups both relocation tables, reserved exactly from what +/// `RelaCounts::for_executable` allows, or its refusal. +pub fn parse_rela_entries(rela_data: &[u8], jmprel_data: &[u8]) -> Result { let entries = || { - RelaTable::new(rela_data) + RelaTable::new(rela_data, crate::arch::ELF_MACHINE) .iter() - .chain(RelaTable::new(jmprel_data).iter()) + .chain(RelaTable::new(jmprel_data, crate::arch::ELF_MACHINE).iter()) }; let counts = RelaCounts::of(entries()); // Ceiling assumes the widest record type, since any one group could be the whole table. let widest = core::mem::size_of::<(u64, u32, i64)>(); - let kept = [RelocKind::Relative, RelocKind::GlobDat, RelocKind::Tpoff64, RelocKind::Tpoff32]; - if counts.max_of(&kept).checked_mul(widest).is_none_or(|b| b > MAX_HEAP_ALLOC) { - log!("ELF: {:?} will not fit one allocation", counts); - return None; - } + let reserve = counts.for_executable(widest, MAX_HEAP_ALLOC).inspect_err(|_| log!("ELF: {:?} refused", counts))?; let mut out = ParsedRelaEntries { - relative: Vec::with_capacity(counts.relative), - glob_dat: Vec::with_capacity(counts.bind), - tpoff64: Vec::with_capacity(counts.tpoff64), - tpoff32: Vec::with_capacity(counts.tpoff32), + relative: Vec::with_capacity(reserve.relative), + glob_dat: Vec::with_capacity(reserve.bind), + tpoff64: Vec::with_capacity(reserve.tpoff64), + tpoff32: Vec::with_capacity(reserve.tpoff32), }; for r in entries() { match r.kind { @@ -71,7 +69,7 @@ pub fn parse_rela_entries(rela_data: &[u8], jmprel_data: &[u8]) -> Option {} } } - Some(out) + Ok(out) } /// Pre-computed writes, sorted by offset so a page's share is one binary search away. diff --git a/kernel/src/elf/mod.rs b/kernel/src/elf/mod.rs index aee2de76c6e..928aaa7287e 100644 --- a/kernel/src/elf/mod.rs +++ b/kernel/src/elf/mod.rs @@ -35,7 +35,7 @@ const _: () = assert!(toyos_elf::MAX_TLS_ALIGN == PAGE_2M); /// [`Layout::parse`] plus the kernel's ceiling on section header table size, /// checked once here since not every caller can refuse a malformed file. pub fn parse_layout(data: &[u8]) -> Result { - let layout = Layout::parse(data).map_err(|e| e.as_str())?; + let layout = Layout::parse(data, crate::arch::ELF_MACHINE).map_err(|e| e.as_str())?; if layout .section_headers .is_some_and(|s| s.byte_len() > MAX_HEAP_ALLOC) @@ -509,5 +509,5 @@ fn table_entries(table: &Option) -> impl Iterator(offset, tpoff as u64) }; count64 += 1; } let mut count32 = 0u64; for (offset, sym, addend) in lib.typed_entries(RelocKind::Tpoff32, |r| &r.tpoff32) { - let tpoff = compute_tpoff(lib, sym, addend, lib_base_offset, total_memsz, tls_info); + let tpoff = compute_tpoff(lib, sym, addend, lib_base_offset, tls, tls_info); // SAFETY: see write_at's `# Safety`. unsafe { lib.write_at::(offset, tpoff as i32) }; count32 += 1; @@ -157,7 +157,7 @@ pub fn apply_tpoff_relocs( if count64 > 0 || count32 > 0 { log!( "dlopen: applied {} TPOFF64 + {} TPOFF32 relocs (base_offset={}, total_memsz={})", - count64, count32, lib_base_offset, total_memsz + count64, count32, lib_base_offset, tls.total_memsz() ); } } @@ -249,20 +249,20 @@ fn compute_tpoff( r_sym: u32, r_addend: i64, lib_base_offset: usize, - total_memsz: usize, + tls: toyos_elf::tls::Static, tls_info: &TlsModuleInfo, ) -> i64 { if r_sym == 0 { - return toyos_elf::tls::tpoff(lib_base_offset as u64, r_addend, total_memsz); + return tls.tpoff(lib_base_offset as u64, r_addend); } let symbols = lib.symbols(); if let Some(sym) = symbols.get(r_sym as usize).filter(|s| s.is_defined()) { - return toyos_elf::tls::tpoff(lib_base_offset as u64 + sym.value, r_addend, total_memsz); + return tls.tpoff(lib_base_offset as u64 + sym.value, r_addend); } let name = symbols.name(r_sym as usize); match defining_module(name, tls_info) { Some((module, sym_offset)) => { - toyos_elf::tls::tpoff(module.base_offset as u64 + sym_offset, r_addend, total_memsz) + tls.tpoff(module.base_offset as u64 + sym_offset, r_addend) } None => { log!("tpoff: unresolved TLS symbol: {}", name); diff --git a/kernel/src/hardlockup/mod.rs b/kernel/src/hardlockup/mod.rs index fd0ef484162..97cfca1e717 100644 --- a/kernel/src/hardlockup/mod.rs +++ b/kernel/src/hardlockup/mod.rs @@ -47,7 +47,7 @@ //! is not spinning on anything, and one that is woken takes the interrupt //! that wakes it. //! - **The span before `clock::init`,** which has no TSC period to convert a -//! bound with, and before `apic::init`, which has no LVT to arm. The same +//! bound with, and before `irqchip::init`, which has no LVT to arm. The same //! floor `crate::deadline` states. //! - **A CPU with `IF` set that no timer ever interrupts.** Nothing resets it, //! deliberately, and what keeps that from being a hole is @@ -57,9 +57,9 @@ //! timer's. use core::fmt; -use core::sync::atomic::{AtomicBool, AtomicU64, Ordering::Relaxed}; +use core::sync::atomic::{AtomicBool, AtomicU64, Ordering::{Acquire, Relaxed, Release}}; -use crate::arch::{apic, cpu, percpu, smp}; +use crate::arch::{cpu, percpu, pmu, smp, trap}; use crate::sched::MAX_CPUS; /// The negative control, in a file of its own because it says what it staged @@ -72,10 +72,6 @@ pub mod probe; /// line and nothing links the two crates (`src/bootlog.rs`). pub const LOCKED_UP: &str = "a cpu locked up with interrupts off"; -/// `RFLAGS.IF` in the frame the NMI pushed: whether the CPU this landed on -/// could have taken any other interrupt at that instant. -const RFLAGS_IF: u64 = 1 << 9; - /// How often an armed CPU samples itself, in nanoseconds of unhalted time. /// /// A second: the bound is measured in tens of them, so a sample period this @@ -99,10 +95,6 @@ static TICKS_PER_MS: AtomicU64 = AtomicU64::new(0); /// since fixed counter 2 counts the reference clock (SDM Vol. 3B §20.2.2). static PERIOD: AtomicU64 = AtomicU64::new(0); -/// The counter's width as CPUID states it, as a mask: bits above it may not be -/// written back. -static WIDTH_MASK: AtomicU64 = AtomicU64::new(0); - /// Whether the panic path has taken this machine. A panicked kernel holds its /// panel with `IF` clear for [`toyos_tco::PANIC_BOUND_MS`] and is not wedged — /// somebody is reading it — so the detector stands down rather than resetting a @@ -126,26 +118,6 @@ static ARMED_PMU: [AtomicBool; MAX_CPUS] = [const { AtomicBool::new(false) }; MA static SPIN_LOCK: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; static SPIN_AT: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; -/// The architectural performance-monitoring MSRs this file writes, SDM Vol. 3B -/// §20.2.2 and Vol. 4 Table 2-2. Nothing else in this kernel programs the PMU, -/// which is why each control register below is declared whole and written -/// whole rather than read, modified and written back. -const IA32_FIXED_CTR2: u32 = 0x30B; -const IA32_FIXED_CTR_CTRL: u32 = 0x38D; -const IA32_PERF_GLOBAL_STATUS: u32 = 0x38E; -const IA32_PERF_GLOBAL_CTRL: u32 = 0x38F; -const IA32_PERF_GLOBAL_OVF_CTRL: u32 = 0x390; - -/// Fixed counter 2's nibble of `IA32_FIXED_CTR_CTRL` is bits 11:8: enable in -/// ring 0 (bit 8) and ring 3 (bit 9), no AnyThread (bit 10), and PMI on -/// overflow (bit 11). Counting in both rings, because a CPU that stops taking -/// interrupts in either is the same defect. -const FIXED_CTR2_ARMED: u64 = 0b1011 << 8; - -/// The fixed counters live in the high half of `IA32_PERF_GLOBAL_CTRL` and of -/// the status and overflow-clear registers beside it, so counter 2 is bit 34. -const GLOBAL_FIXED_CTR2: u64 = 1 << 34; - /// Turn the bound this boot named into a bound on one CPU, and arm the BSP. /// /// Called from `deadline::start` with the deadline's own bound: one parameter @@ -222,23 +194,16 @@ pub fn arm_this_cpu() { // is still sampled by any NMI that reaches it, and a sample against an // unwritten baseline would read as a CPU that has been stuck since boot. PROGRESS[me].store(crate::irq_census::taken_here(), Relaxed); - STILL_SINCE[me].store(cpu::rdtsc(), Relaxed); - let Some(width) = architectural_pmu() else { return }; - WIDTH_MASK.store(mask_of(width), Relaxed); - // Stopped, then set up, then started: a counter enabled while its control - // register is half written can overflow into an LVT that is not armed yet. - write_msr(IA32_PERF_GLOBAL_CTRL, 0); - write_msr(IA32_FIXED_CTR_CTRL, FIXED_CTR2_ARMED); - write_msr(IA32_PERF_GLOBAL_OVF_CTRL, GLOBAL_FIXED_CTR2); - reload(); - apic::arm_perf_nmi(); - write_msr(IA32_PERF_GLOBAL_CTRL, GLOBAL_FIXED_CTR2); + STILL_SINCE[me].store(cpu::counter(), Relaxed); + if !pmu::arm(PERIOD.load(Relaxed)) { + return; + } ARMED_PMU[me].store(true, Relaxed); } /// Stand the detector down for the rest of this machine's life. /// -/// Called from `apic::halt_all_cpus`, which is every fatal path's one funnel: a +/// Called from `panic::halt_all_cpus`, which is every fatal path's one funnel: a /// panicked kernel pages its panel with `IF` clear, under a bound of its own, /// and a reader holding the machine open is not a machine to reset. One relaxed /// store, and the LVT then stays masked of its own accord — hardware masks it @@ -257,7 +222,7 @@ pub fn bound_ms() -> u64 { /// One sample of the CPU this NMI landed on: where it is, whether it has taken /// anything since the last one, and whether that has gone on too long. /// -/// Called from `arch::idt::nmi`'s `note` and nowhere else. Returns on every NMI +/// Called from `arch::trap::nmi`'s `note` and nowhere else. Returns on every NMI /// that is not this CPU's own overflow, so the diagnostic senders — the blocked /// task dump's probe, the syscall-window storm — cost one load and one compare. pub fn sample(rip: u64, rsp: u64, rflags: u64) { @@ -268,14 +233,14 @@ pub fn sample(rip: u64, rsp: u64, rflags: u64) { if me >= MAX_CPUS { return; } - let mine = ARMED_PMU[me].load(Relaxed) && overflowed(); + let mine = ARMED_PMU[me].load(Relaxed) && pmu::overflowed(); // A machine with no PMU has this bound only under the actuator that sends // the NMI the counter would have, which is how QEMU's guest reaches this // decision at all. if !mine && !crate::actuator::hard_lockup_probe() { return; } - let now = cpu::rdtsc(); + let now = cpu::counter(); AT_RIP[me].store(rip, Relaxed); AT_RSP[me].store(rsp, Relaxed); AT_RFLAGS[me].store(rflags, Relaxed); @@ -286,7 +251,7 @@ pub fn sample(rip: u64, rsp: u64, rflags: u64) { // `IF` set is the whole difference between this bound and the deadline's: a // CPU that can still take an interrupt is one the timer entry's poll // reaches, and this mechanism is not about it. - if moved || rflags & RFLAGS_IF != 0 { + if moved || trap::frame_interrupts_enabled(rflags) { STILL_SINCE[me].store(now, Relaxed); } else if now.wrapping_sub(STILL_SINCE[me].load(Relaxed)) >= BOUND_TSC.load(Relaxed) { locked_up(me, rip, rsp, now) @@ -294,12 +259,10 @@ pub fn sample(rip: u64, rsp: u64, rflags: u64) { // **After the decision and never before it.** Re-arming clears the mask // hardware set on delivery; leaving it set is what stops a second NMI // landing on the stack this one is still standing on while it seals, which - // `arch::idt::nmi`'s `nested_nmi` would answer by stopping the machine + // `arch::trap::nmi`'s `nested_nmi` would answer by stopping the machine // without resetting it. if mine { - write_msr(IA32_PERF_GLOBAL_OVF_CTRL, GLOBAL_FIXED_CTR2); - reload(); - apic::arm_perf_nmi(); + pmu::rearm(PERIOD.load(Relaxed)); } } @@ -354,8 +317,8 @@ pub fn spinning_on(lock: u64, at: &'static core::panic::Location<'static>) -> Sp let was = Spinning { lock: SPIN_LOCK[me].load(Relaxed), at: SPIN_AT[me].load(Relaxed) }; SPIN_AT[me].store(at as *const _ as u64, Relaxed); // Last, and what the reader tests: a non-zero lock means the site beside it - // was already stored. - SPIN_LOCK[me].store(lock, Relaxed); + // was already stored — `Release`, so that holds on another CPU's record. + SPIN_LOCK[me].store(lock, Release); was } @@ -365,59 +328,8 @@ pub fn spinning_on_nothing(was: Spinning) { let me = percpu::cpu_id() as usize; if me < MAX_CPUS { SPIN_AT[me].store(was.at, Relaxed); - SPIN_LOCK[me].store(was.lock, Relaxed); - } -} - -/// The counter's width, or `None` on a CPU with no architectural performance -/// monitoring — which is every QEMU TCG guest. -/// -/// SDM Vol. 2A, CPUID leaf 0AH: EAX[7:0] is the version, and version 2 is where -/// the fixed-function counters and `IA32_PERF_GLOBAL_CTRL` appear; EDX[4:0] is -/// how many fixed counters there are and EDX[12:5] how wide they are. Fixed -/// counter 2 needs three of them. -fn architectural_pmu() -> Option { - if cpu::cpuid(0, 0).0 < 0x0A { - return None; - } - let (eax, _, _, edx) = cpu::cpuid(0x0A, 0); - let version = eax & 0xff; - let counters = edx & 0x1f; - let width = (edx >> 5) & 0xff; - if version < 2 || counters < 3 || width == 0 || width > 64 { - return None; + SPIN_LOCK[me].store(was.lock, Release); } - Some(width) -} - -const fn mask_of(width: u32) -> u64 { - match width >= 64 { - true => u64::MAX, - false => (1u64 << width) - 1, - } -} - -/// Whether fixed counter 2 is the reason this NMI arrived, SDM Vol. 3B §20.2.2: -/// `IA32_PERF_GLOBAL_STATUS` bit 34 is its overflow. -fn overflowed() -> bool { - cpu::rdmsr(IA32_PERF_GLOBAL_STATUS) & GLOBAL_FIXED_CTR2 != 0 -} - -/// Set the counter one period below its own overflow. -fn reload() { - let period = PERIOD.load(Relaxed); - let mask = WIDTH_MASK.load(Relaxed); - // Masked to the width CPUID stated: a fixed counter refuses a write of the - // bits above it, and the negative count is what makes the overflow land a - // period from here. - write_msr(IA32_FIXED_CTR2, 0u64.wrapping_sub(period) & mask); -} - -fn write_msr(msr: u32, value: u64) { - // SAFETY: every MSR here is an architectural performance-monitoring counter - // or its control register, enumerated by CPUID leaf 0AH before this file - // writes any of them, and each value is that register's own field encoding. - unsafe { cpu::wrmsr(msr, value) }; } /// Milliseconds, from a TSC span. One `div`, and only on the path that has @@ -438,7 +350,9 @@ struct Waiting(usize); impl fmt::Display for Waiting { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - let lock = SPIN_LOCK[self.0].load(Relaxed); + // `Acquire`: pairs with the store that set it, so the site read below is + // at least the one stored before it. + let lock = SPIN_LOCK[self.0].load(Acquire); if lock == 0 { return Ok(()); } @@ -518,7 +432,7 @@ impl fmt::Display for Report { " cpu{cpu} irqs={irqs} (={} when sampled {} ago) if={} at {}{}", PROGRESS[cpu].load(Relaxed), Ms(now.wrapping_sub(at)), - u8::from(AT_RFLAGS[cpu].load(Relaxed) & RFLAGS_IF != 0), + u8::from(trap::frame_interrupts_enabled(AT_RFLAGS[cpu].load(Relaxed))), At(AT_RIP[cpu].load(Relaxed)), Waiting(cpu), )?; diff --git a/kernel/src/hardlockup/probe.rs b/kernel/src/hardlockup/probe.rs index a6761c367ae..87e6b23721b 100644 --- a/kernel/src/hardlockup/probe.rs +++ b/kernel/src/hardlockup/probe.rs @@ -25,7 +25,7 @@ use core::sync::atomic::{AtomicBool, AtomicU32, Ordering}; -use crate::arch::{apic, cpu, percpu, smp}; +use crate::arch::{irqchip, cpu, percpu, smp}; /// What the staged CPU says before it stops answering, and the witness a sealed /// record carries in its tail: a `WEDGED` page whose text does not hold this @@ -112,9 +112,9 @@ fn hold_and_sample(victim: usize) -> ! { SEND_EVERY_NS / 1_000_000, ); loop { - apic::send_nmi(victim as u32); + irqchip::send_nmi(victim as u32); let until = crate::clock::tsc_deadline(SEND_EVERY_NS); - while cpu::rdtsc() < until { + while cpu::counter() < until { core::hint::spin_loop(); } } @@ -136,7 +136,7 @@ fn go_deaf(me: usize, bound_ms: u64) -> ! { STAGE.store(DEAF, Ordering::Release); // Not an `IrqGuard`: nothing here re-enables them, which is the point. cpu::disable_interrupts(); - while cpu::rdtsc() < until { + while cpu::counter() < until { core::hint::spin_loop(); } // Never acquires. The detector's NMI is what ends this CPU, and its `rip` diff --git a/kernel/src/hasher.rs b/kernel/src/hasher.rs index 28b6b5818f9..e47153c9a74 100644 --- a/kernel/src/hasher.rs +++ b/kernel/src/hasher.rs @@ -1,5 +1,5 @@ //! The `BuildHasher` every kernel hash container is built on, seeded from -//! `RDRAND` before any container exists — `kernel/Cargo.toml` takes hashbrown +//! the CPU's own random source before any container exists — `kernel/Cargo.toml` takes hashbrown //! without `default-hasher`, so a container must name a hasher. //! //! **A container built before [`seed`], or on a constant, is the wrong answer @@ -12,7 +12,7 @@ use core::hash::{BuildHasher, Hasher}; use core::sync::atomic::{AtomicU64, Ordering}; -use crate::arch::cpu; +use crate::arch::entropy; /// `0` until [`seed`] runs, and a value it refuses to draw. static SEED: AtomicU64 = AtomicU64::new(0); @@ -24,19 +24,18 @@ pub const UNSEEDED: &str = /// Two seeds in one boot means a container was built against the first. const RESEEDED: &str = "kernel hasher: seed() ran twice in one boot"; -pub const NO_RDRAND: &str = - "kernel hasher: CPUID.01H:ECX[30] is clear, so this CPU has no RDRAND and the seed has no \ - source"; - pub const NO_ENTROPY: &str = - "kernel hasher: RDRAND gave no usable value, so the seed would be a constant on every boot"; + "kernel hasher: the CPU's random source gave no usable value, so the seed would be a \ + constant on every boot"; /// Called once, before any container. `0` and all-ones are what a failing -/// `RDRAND` leaves behind, so neither may become a seed. +/// source leaves behind, so neither may become a seed. pub fn seed() { - assert!(cpu::has_rdrand(), "{NO_RDRAND}"); - let drawn = (0..cpu::RDRAND_ATTEMPTS) - .filter_map(|_| cpu::rdrand()) + if let Err(why) = entropy::available() { + panic!("kernel hasher: {why}, and the seed has no other source"); + } + let drawn = (0..entropy::ATTEMPTS) + .filter_map(|_| entropy::draw()) .find(|&v| v != 0 && v != u64::MAX) .unwrap_or_else(|| panic!("{NO_ENTROPY}")); assert_eq!(SEED.swap(drawn, Ordering::Release), 0, "{RESEEDED}"); diff --git a/kernel/src/heartbeat.rs b/kernel/src/heartbeat.rs index 4a58bcab903..30d21b1920f 100644 --- a/kernel/src/heartbeat.rs +++ b/kernel/src/heartbeat.rs @@ -97,7 +97,7 @@ pub fn poll() { ); // Must not block: report_line uses try_lock and prints `rte=busy` rather than waiting. - crate::drivers::i8042::report_line(); + crate::arch::keyboard_controller::report_line(); for (cpu, &stamp) in stamps.iter().enumerate().take(cpus) { if mask & (1 << cpu) != 0 { diff --git a/kernel/src/inbox/mod.rs b/kernel/src/inbox/mod.rs index d99cb52ebad..d7c69cc7906 100644 --- a/kernel/src/inbox/mod.rs +++ b/kernel/src/inbox/mod.rs @@ -100,7 +100,7 @@ impl WatchFlags { pub const READABLE: Self = Self(toyos_abi::inbox::READABLE); pub const WRITABLE: Self = Self(toyos_abi::inbox::WRITABLE); /// Every bit `toyos_abi::inbox` defines for `Submission::op_flags`; - /// hand-copied and unchecked, for the reason `arch/syscall/vm.rs`'s + /// hand-copied and unchecked, for the reason `syscall/vm.rs`'s /// `MMAP_PROT_KNOWN` gives for all four of these masks. const KNOWN: u32 = Self::READABLE.0 | Self::WRITABLE.0; diff --git a/kernel/src/invalidation.rs b/kernel/src/invalidation.rs new file mode 100644 index 00000000000..e12d820a54c --- /dev/null +++ b/kernel/src/invalidation.rs @@ -0,0 +1,28 @@ +//! Why a machine-wide translation invalidation was asked for: the census every +//! architecture's shootdown (`arch::tlb`) counts its callers by. + +/// Which path issued a shootdown, so the census names who pays: `Dlopen` (a +/// `Shared` window or rollback unmap), `Pcid` (pool reclaim), `Mmio`, `Unmap` +/// (`Unmapped::drop`), `Pipe`, `Staged` (the ack-delay actuator), `Bench` +/// (`arch::tlb::bench`'s own, so a measured shootdown is never counted as one +/// a path in this kernel needed). +#[derive(Clone, Copy)] +#[repr(usize)] +pub enum Origin { + Dlopen, + Pcid, + Mmio, + Unmap, + Pipe, + #[cfg_attr(not(feature = "test-actuators"), allow(dead_code))] + Staged, + #[cfg_attr(not(feature = "boot-actuators"), allow(dead_code))] + Bench, +} + +impl Origin { + pub const COUNT: usize = 7; + /// Order matches the variants; `tests/toyos.rs`'s `irq_census_conservation` reads the line back. + pub const NAMES: [&'static str; Self::COUNT] = + ["dlopen", "pcid", "mmio", "unmap", "pipe", "staged", "bench"]; +} diff --git a/kernel/src/iommu/mod.rs b/kernel/src/iommu/mod.rs index 25d25e8c39b..cbcb8c22011 100644 --- a/kernel/src/iommu/mod.rs +++ b/kernel/src/iommu/mod.rs @@ -9,7 +9,7 @@ // CI runs kernel clippy with `-D warnings`, so an undocumented `unsafe` block anywhere in this module tree fails the build. #![warn(clippy::undocumented_unsafe_blocks)] -pub mod vtd; +use crate::arch::iommu_unit as unit; /// The address width a device's translations cover. /// @@ -37,21 +37,21 @@ pub struct StreamId(u32); impl StreamId { /// Named `pci`, not `new`: an SMMU StreamID is not always a bus/device/function triple. - pub(in crate::iommu) const fn pci(bus: u8, device: u8, function: u8) -> Self { + pub(crate) const fn pci(bus: u8, device: u8, function: u8) -> Self { Self(((bus as u32) << 8) | ((device as u32) << 3) | function as u32) } /// The bus half of the id. - pub(in crate::iommu) const fn bus(self) -> u8 { + pub(crate) const fn bus(self) -> u8 { (self.0 >> 8) as u8 } - pub(in crate::iommu) const fn devfn(self) -> u8 { + pub(crate) const fn devfn(self) -> u8 { (self.0 & 0xFF) as u8 } /// The 16-bit requester id a source-id check compares against; `pci` is the only constructor, so it always fits. - pub(in crate::iommu) const fn requester(self) -> u16 { + pub(crate) const fn requester(self) -> u16 { self.0 as u16 } } @@ -65,12 +65,12 @@ impl Iova { /// The domain every kernel driver that has not moved is still on maps a device address to the physical address it equals. /// /// The single site that policy is stated in, so the stage that moves the last driver deletes it and the compiler flags every site that assumed it. - pub(in crate::iommu) const fn identity(phys: u64) -> Self { + pub(crate) const fn identity(phys: u64) -> Self { Self(phys) } /// An address a domain's allocator handed out, which is nothing else's address. - pub(in crate::iommu) const fn translated(at: u64) -> Self { + pub(crate) const fn translated(at: u64) -> Self { Self(at) } @@ -84,12 +84,12 @@ impl Iova { pub struct DomainId(u16); impl DomainId { - pub(in crate::iommu) const fn new(id: u16) -> Self { + pub(crate) const fn new(id: u16) -> Self { assert!(id != 0); Self(id) } - pub(in crate::iommu) const fn raw(self) -> u16 { + pub(crate) const fn raw(self) -> u16 { self.0 } } @@ -160,13 +160,13 @@ impl DeviceSpace { /// machine out of domains are both answers its caller refuses the claim /// with rather than degrading past. pub fn own() -> Result { - vtd::domain::create().map(Self::Own) + unit::domain::create().map(Self::Own) } /// One of a device's own, or the machine's own with the reason. For a /// driver **in this kernel**, whose addresses are the kernel's either way. pub fn create() -> Self { - match vtd::domain::create() { + match unit::domain::create() { Ok(id) => Self::Own(id), Err(why) => { log!("iommu: no domain of its own for a device: {why}"); @@ -185,7 +185,7 @@ impl DeviceSpace { pub fn map(self, phys: u64, bytes: u64) -> Result { match self { Self::Untranslated => Ok(phys), - Self::Own(id) => vtd::domain::map(id, phys, bytes).map(Iova::raw), + Self::Own(id) => unit::domain::map(id, phys, bytes).map(Iova::raw), } } @@ -195,7 +195,7 @@ impl DeviceSpace { pub fn map_at(self, at: u64, phys: u64, bytes: u64) -> Result<(), IommuError> { match self { Self::Untranslated => panic!("iommu: an untranslated space was asked to place {phys:#x} at {at:#x}"), - Self::Own(id) => vtd::domain::map_at(id, Iova::translated(at), phys, bytes), + Self::Own(id) => unit::domain::map_at(id, Iova::translated(at), phys, bytes), } } @@ -204,7 +204,7 @@ impl DeviceSpace { pub fn reserve(self, bytes: u64) -> Result { match self { Self::Untranslated => panic!("iommu: an untranslated space was asked for room"), - Self::Own(id) => vtd::domain::reserve(id, bytes).map(Iova::raw), + Self::Own(id) => unit::domain::reserve(id, bytes).map(Iova::raw), } } @@ -214,7 +214,7 @@ impl DeviceSpace { pub fn place(self, at: u64, phys: u64, bytes: u64) -> Result<(), IommuError> { match self { Self::Untranslated => panic!("iommu: an untranslated space was asked to place {phys:#x} at {at:#x}"), - Self::Own(id) => vtd::domain::place(id, Iova::translated(at), phys, bytes).map(|_| ()), + Self::Own(id) => unit::domain::place(id, Iova::translated(at), phys, bytes).map(|_| ()), } } @@ -222,7 +222,7 @@ impl DeviceSpace { pub fn unmap(self, at: u64, bytes: u64) -> Result<(), IommuError> { match self { Self::Untranslated => Ok(()), - Self::Own(id) => vtd::domain::unmap(id, Iova::translated(at), bytes), + Self::Own(id) => unit::domain::unmap(id, Iova::translated(at), bytes), } } @@ -230,7 +230,7 @@ impl DeviceSpace { /// in place: the device is translating the moment this returns. pub fn attach(self, bus: u8, device: u8, function: u8) { if let Self::Own(id) = self { - vtd::domain::attach(StreamId::pci(bus, device, function), id); + unit::domain::attach(StreamId::pci(bus, device, function), id); } } } @@ -248,9 +248,9 @@ impl core::fmt::Display for StreamId { /// /// The device list must be the complete enumeration: enabling translation with an unenumerated device left off it can brick the machine's own boot disk. /// -/// Calls `vtd::init` directly rather than through a dispatch, because x86-64 has one backend and the dispatch is not yet a real seam. +/// Calls `unit::init` directly rather than through a dispatch, because x86-64 has one backend and the dispatch is not yet a real seam. pub fn init(rsdp_addr: u64, devices: &[crate::drivers::pci::PciDevice]) { - vtd::init(rsdp_addr, devices); + unit::init(rsdp_addr, devices); } /// How a source must address its interrupt. Not a yes/no: a caller that folded @@ -310,20 +310,20 @@ pub fn remap_msi( vector: u8, dest: u32, ) -> Delivery { - if !vtd::interrupt::is_armed() { + if !unit::interrupt::is_armed() { return Delivery::Direct; } - match vtd::interrupt::msi(StreamId::pci(bus, device, function), vector, dest) { + match unit::interrupt::msi(StreamId::pci(bus, device, function), vector, dest) { Ok(msi) => Delivery::Remapped(MsiMessage { address: msi.address, data: msi.data }), Err(why) => Delivery::Refused(why), } } pub fn remap_pin(apic_id: u8, vector: u8, dest: u32, level: bool) -> Delivery { - if !vtd::interrupt::is_armed() { + if !unit::interrupt::is_armed() { return Delivery::Direct; } - match vtd::interrupt::pin(apic_id, vector, dest, level) { + match unit::interrupt::pin(apic_id, vector, dest, level) { Ok(pin) => Delivery::Remapped(PinRedirect { low: pin.low, high: pin.high }), Err(why) => Delivery::Refused(why), } @@ -338,12 +338,12 @@ pub fn remap_pin(apic_id: u8, vector: u8, dest: u32, level: bool) -> Delivery) { - vtd::fault::user_owned(StreamId::pci(bus, device, function), slot); + unit::fault::user_owned(StreamId::pci(bus, device, function), slot); } /// Reached from the IDT gate the unit's own `FEDATA` names. /// /// Fires when a device has been told no, so what it reports is a bug in whoever owns that device, not in the IOMMU. pub fn fault_interrupt() { - vtd::fault::service(); + unit::fault::service(); } diff --git a/kernel/src/irq_census.rs b/kernel/src/irq_census.rs index 6a6305bfadc..29782cdcfc0 100644 --- a/kernel/src/irq_census.rs +++ b/kernel/src/irq_census.rs @@ -15,7 +15,7 @@ use crate::scheduler::MAX_CPUS; #[derive(Clone, Copy, PartialEq, Eq, Debug)] #[repr(usize)] pub enum Source { - /// Vector 0x20: shared by this CPU's LAPIC one-shot and every `apic::kick_cpu` IPI. + /// Vector 0x20: shared by this CPU's LAPIC one-shot and every `irqchip::kick_cpu` IPI. Timer, /// Vector 0x21, xHCI MSI-X (or MSI). Xhci, @@ -38,7 +38,7 @@ pub enum Source { /// Vector 0xFF, the local APIC's spurious vector. /// A non-zero count on a machine that staged nothing is an interrupt-routing defect. Spurious, - /// Every vector no `idt_vectors!` row claims — `arch::idt::unclaimed`. + /// Every vector no `idt_vectors!` row claims — `arch::trap::unclaimed`. /// A non-zero count a boot staged nothing for is a routing defect the gate kept off `#DF`. Unclaimed, } @@ -53,38 +53,14 @@ impl Source { ]; } -/// One `u64` per source plus the total; `percpu::OFF_IRQ_COUNTS` is where the block starts. +/// One `u64` per source plus the total, in each CPU's own per-CPU block. pub const SLOTS: usize = 1 + Source::COUNT; /// Index of the machine's own total inside a CPU's block. pub const TOTAL: usize = 0; -/// The `gs:` displacement of slot `index` in this CPU's block. -pub const fn slot_offset(index: usize) -> u32 { - percpu::OFF_IRQ_COUNTS + (index as u32) * 8 -} - -/// Records one delivery of `$source` as two lock-free `add`s to this CPU's own gs: slots. -/// A macro, not a function: the two offsets must be asm immediates, not const-generic values an optimiser could relax. -macro_rules! irq_took { - ($source:ident) => {{ - // SAFETY: both slots are this CPU's own counter block per `arch::percpu`, and the caller is an interrupt handler, so `GS_BASE` already points at this CPU's `PerCpu`. - unsafe { - ::core::arch::asm!( - "add qword ptr gs:[{total}], 1", - "add qword ptr gs:[{source}], 1", - total = const $crate::irq_census::slot_offset($crate::irq_census::TOTAL), - source = const $crate::irq_census::slot_offset( - 1 + $crate::irq_census::Source::$source as usize - ), - // no `nomem` because both instructions write; no `preserves_flags` because `add` clobbers flags. - options(nostack), - ); - } - }}; -} - -pub(crate) use irq_took; +// Counted where each is taken, by the architecture's handlers +// (`arch::percpu::irq_took!`), into this CPU's own block. /// Each CPU's counter-array address; only the array is published, so a reader never touches the rest of the block the owning CPU writes through raw pointers. static BLOCKS: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; @@ -148,26 +124,11 @@ pub fn taken_by(cpu: u32) -> Option { read(cpu).map(|counts| counts[TOTAL].saturating_sub(counts[1 + Source::Nmi as usize])) } -/// [`taken_by`] for the CPU asking, read straight off `gs:` — the one form a CPU -/// inside an NMI may use, since it needs neither the published pointer array nor -/// a bounds check on a `cpu_id` it is standing on. +/// [`taken_by`] for the CPU asking, read straight off its own block — the one +/// form a CPU inside an NMI may use, since it needs neither the published +/// pointer array nor a bounds check on a `cpu_id` it is standing on. pub fn taken_here() -> u64 { - let total: u64; - let nmis: u64; - // SAFETY: both slots are this CPU's own counter block per `arch::percpu`; - // `GS_BASE` points at the running CPU's `PerCpu` in every context this is - // read from. - unsafe { - core::arch::asm!( - "mov {total}, qword ptr gs:[{at}]", - "mov {nmis}, qword ptr gs:[{nmi_at}]", - total = out(reg) total, - nmis = out(reg) nmis, - at = const slot_offset(TOTAL), - nmi_at = const slot_offset(1 + Source::Nmi as usize), - options(nostack, readonly, preserves_flags), - ); - } + let (total, nmis) = percpu::irq_counts_here(TOTAL, 1 + Source::Nmi as usize); total.saturating_sub(nmis) } diff --git a/kernel/src/loader/mod.rs b/kernel/src/loader/mod.rs index 8e52d0afd83..044a0eb05cb 100644 --- a/kernel/src/loader/mod.rs +++ b/kernel/src/loader/mod.rs @@ -15,8 +15,9 @@ mod symbols; mod tls; pub use start::{build_child_handles, PendingHandles, SLOT_PAIR_LEN}; -pub(crate) use start::{alloc_kernel_stack, kernel_start, process_start, thread_start}; -pub use tls::{setup_combined_tls, setup_tls, DTV_INITIAL_CAPACITY}; +pub(crate) use start::alloc_kernel_stack; +pub(crate) use crate::arch::entry::{kernel_start, process_start, thread_start}; +pub use tls::{setup_combined_tls, setup_tls, DTV_INITIAL_CAPACITY, VARIANT as TLS_VARIANT}; pub(crate) use tls::rebase_block; use alloc::string::String; @@ -271,10 +272,13 @@ fn read_exe_tables( } None => Vec::new(), }; - let Some(relas) = elf::parse_rela_entries(&rela_data, &jmprel_data) else { - log!("spawn: {}: relocation tables do not fit one allocation", path); - return Err(SyscallError::ResourceExhausted); - }; + let relas = elf::parse_rela_entries(&rela_data, &jmprel_data).map_err(|refused| { + log!("spawn: {}: {}", path, refused.as_str()); + match refused { + toyos_elf::rela::ExeRefusal::TooLarge => SyscallError::ResourceExhausted, + toyos_elf::rela::ExeRefusal::TlsDescriptor => SyscallError::InvalidArgument, + } + })?; let dynstr = match dyn_info.strtab_table() { Some(t) => { @@ -337,7 +341,7 @@ fn rela_dyn_from_sections( let shdrs = table(backing, path, "e_shnum", sections.file_offset, sections.byte_len())?; let mut first = |off: u64| { let head = read_file_range(backing, off, toyos_elf::rela::ENTRY_SIZE); - toyos_elf::RelaTable::new(&head).get(0) + toyos_elf::RelaTable::new(&head, crate::arch::ELF_MACHINE).get(0) }; match SectionTable::new(&shdrs).rela_dyn(&mut first) { Some((off, size)) => table(backing, path, "SHT_RELA sh_size", off, size as usize), @@ -502,7 +506,7 @@ pub fn spawn( // `Prot::ReadWrite`, never executable: a fixed-address W+X stack is the // shape stack-smashing payloads target. pt.map_range(stack_vaddr, stack_pages.phys(), USER_STACK_SIZE as u64, - Prot::ReadWrite, CachePolicy::DeferToMtrr); + Prot::ReadWrite, CachePolicy::Normal); pt.insert_region(stack_vaddr, crate::vma::Region { size: USER_STACK_SIZE as u64, kind: crate::vma::RegionKind::Anonymous { prot: Prot::ReadWrite }, @@ -535,7 +539,7 @@ pub fn spawn( None => None, }; - let Some((tls_modules, tls_total_memsz, max_tls_align, next_tls_module_id)) = + let Some((tls_modules, tls, next_tls_module_id)) = tls::build_tls_layout(&loaded_libs.libs, &layout, exe_tls_template.as_ref()) else { log!("spawn: {}: the TLS modules do not fit one block", path); @@ -543,7 +547,7 @@ pub fn spawn( }; apply_tls_relocs(&exe, backing.as_ref(), &loaded_libs.libs, &tls_modules, - tls_total_memsz, &mut reloc_index); + tls, &mut reloc_index); reloc_index.finalize(); let reloc_index = if reloc_index.len() > 0 { @@ -553,11 +557,11 @@ pub fn spawn( None }; - log!("spawn: TLS {} modules, total_memsz={}", tls_modules.len(), tls_total_memsz); + log!("spawn: TLS {} modules, total_memsz={}", tls_modules.len(), tls.total_memsz()); let Some((tls_pages, fs_base)) = - tls::map_block(&child_pt, &tls_modules, tls_total_memsz, max_tls_align) + tls::map_block(&child_pt, &tls_modules, tls) else { - log!("spawn: {}: failed to allocate TLS ({} bytes)", path, tls_total_memsz); + log!("spawn: {}: failed to allocate TLS ({} bytes)", path, tls.total_memsz()); return Err(SyscallError::ResourceExhausted.into()); }; @@ -593,8 +597,7 @@ pub fn spawn( elf: ElfInfo { elf_alloc: exe_tls_template, tls_modules, - tls_total_memsz, - tls_max_align: max_tls_align, + tls, next_tls_module_id, dynamic_tls_blocks: alloc::collections::BTreeMap::new(), loaded_libs, @@ -657,8 +660,8 @@ pub fn spawn( drop(guard); let t3 = crate::clock::nanos_since_boot(); - log!("spawn: {} pid={} tid={} dst={} base={:#x} entry={:#x} cr3={:#x} symbols={}KiB (layout={}ms relocs={}ms deps={}ms tls={}ms total={}ms)", - path, pid, tid, dst.0, base, entry, child_pt.lock().cr3().phys(), sym_bytes / 1024, + log!("spawn: {} pid={} tid={} dst={} base={:#x} entry={:#x} root={:#x} symbols={}KiB (layout={}ms relocs={}ms deps={}ms tls={}ms total={}ms)", + path, pid, tid, dst.0, base, entry, child_pt.lock().root().phys(), sym_bytes / 1024, (t1 - t0) / 1_000_000, (t2 - t1) / 1_000_000, (t_deps - t2) / 1_000_000, (t_tls - t_deps) / 1_000_000, (t3 - t0) / 1_000_000); @@ -794,7 +797,7 @@ fn apply_tls_relocs( backing: &dyn crate::file_backing::FileBacking, loaded_libs: &[elf::LoadedLib], tls_modules: &[elf::TlsModule], - tls_total_memsz: usize, + tls: toyos_elf::tls::Static, reloc_index: &mut elf::RelocationIndex, ) { let tls_info = elf::TlsModuleInfo { libs: loaded_libs, modules: tls_modules }; @@ -803,7 +806,7 @@ fn apply_tls_relocs( let module = tls_modules.iter().find(|m| m.template == lib.tls_template); let base_offset = module.map_or(0, |m| m.base_offset); // Initial-exec: references to TLS in the static block. - elf::apply_tpoff_relocs(lib, base_offset, tls_total_memsz, &tls_info); + elf::apply_tpoff_relocs(lib, base_offset, tls, &tls_info); // General-dynamic: this lib's own TLS, reached through the DTV. if let Some(m) = module { elf::apply_dtpmod_relocs(lib, m.module_id, &tls_info); @@ -816,12 +819,12 @@ fn apply_tls_relocs( .map_or(0, |m| m.base_offset); for &(r_offset, r_sym, r_addend) in &exe.relas.tpoff64 { let tpoff = exe_tpoff(exe, backing, r_sym, r_addend, exe_base_offset, - tls_total_memsz, &tls_info); + tls, &tls_info); reloc_index.add_u64(r_offset, tpoff as u64); } for &(r_offset, r_sym, r_addend) in &exe.relas.tpoff32 { let tpoff = exe_tpoff(exe, backing, r_sym, r_addend, exe_base_offset, - tls_total_memsz, &tls_info); + tls, &tls_info); reloc_index.add_i32(r_offset, tpoff as i32); } } @@ -837,10 +840,10 @@ fn exe_tpoff( r_sym: u32, r_addend: i64, exe_base_offset: usize, - total_memsz: usize, + tls: toyos_elf::tls::Static, tls_info: &elf::TlsModuleInfo, ) -> i64 { - let unnamed = toyos_elf::tls::tpoff(exe_base_offset as u64, r_addend, total_memsz); + let unnamed = tls.tpoff(exe_base_offset as u64, r_addend); if r_sym == 0 { return unnamed; } @@ -848,7 +851,7 @@ fn exe_tpoff( return unnamed; }; if sym.is_defined() { - return toyos_elf::tls::tpoff(exe_base_offset as u64 + sym.value, r_addend, total_memsz); + return tls.tpoff(exe_base_offset as u64 + sym.value, r_addend); } let name = toyos_elf::cstr(&exe.dynstr, sym.name as u64); @@ -857,7 +860,7 @@ fn exe_tpoff( // rather than guessed at with base_offset 0. match elf::defining_module(name, tls_info) { Some((module, sym_offset)) => { - toyos_elf::tls::tpoff(module.base_offset as u64 + sym_offset, r_addend, total_memsz) + tls.tpoff(module.base_offset as u64 + sym_offset, r_addend) } None => { log!("tpoff: unresolved exe TLS symbol: {}", name); diff --git a/kernel/src/loader/start.rs b/kernel/src/loader/start.rs index 8cc176aa76f..57686d5b93a 100644 --- a/kernel/src/loader/start.rs +++ b/kernel/src/loader/start.rs @@ -1,10 +1,9 @@ //! Loads a built process onto a CPU and builds the handle table it starts -//! with. The two trampolines below are the loader's only per-architecture -//! code; everything else here is architecture-neutral. +//! with. The frame a new stack starts from and the trampolines it returns into +//! are the architecture's (`arch::entry`). use alloc::vec::Vec; -use crate::arch::entry::{initial_user_state, ring3_trampoline_asm}; use crate::object::{HandleTable, Refusal}; use crate::process::{ process_data, Endowments, OwnedAlloc, ENDOW_ENTRY_LEN, KERNEL_STACK_SIZE, @@ -27,90 +26,9 @@ pub(crate) fn alloc_kernel_stack( let alloc = OwnedAlloc::new(KERNEL_STACK_SIZE, 4096)?; scheduler::write_stack_canary(&alloc); let top = alloc.ptr() as u64 + KERNEL_STACK_SIZE as u64; - // Layout must match context_switch's pop sequence: pushfq, rbp..r15, return address. - let frame = (top - 8 * 8) as *mut u64; - // SAFETY: `alloc` is fresh and exclusively owned; the eight writes cover - // `[frame, frame + 64)`, the top 64 bytes of that allocation. - unsafe { - *frame.add(0) = 0; // r15 - *frame.add(1) = arg; // r14 - *frame.add(2) = user_sp; // r13 - *frame.add(3) = user_entry; // r12 - *frame.add(4) = 0; // rbx - *frame.add(5) = 0; // rbp - *frame.add(6) = 0x002; // RFLAGS (IF=0, AC=0) - *frame.add(7) = trampoline as usize as u64; // return address - } - Some((alloc, frame as u64)) -} - -/// Entry point for new processes, reached through `context_switch`'s `ret`. r12 = entry point, r13 = user stack pointer. -// State loads after `unlock`, not before: earlier, registers hold the previous tenant's kernel context. -#[unsafe(naked)] -pub(crate) extern "C" fn process_start() { - ring3_trampoline_asm!( - "push r12", - "push r13", - "call {unlock}", - "pop r13", - "pop r12", - initial_user_state!(), - "push {user_ss}", - "push r13", // RSP: user stack - "push 0x202", // RFLAGS: IF=1 - "push {user_cs}", - "push r12", // RIP: entry point - "iretq", - unlock = sym crate::sched::driver::trampoline_entry, - user_ss = const crate::arch::percpu::USER_DS, - user_cs = const crate::arch::percpu::USER_CS, - ); -} - -/// Entry point for new threads. r14 carries the argument, which lands in rdi. -#[unsafe(naked)] -pub(crate) extern "C" fn thread_start() { - ring3_trampoline_asm!( - "push r12", - "push r13", - "push r14", - "call {unlock}", - "pop r14", - "pop r13", - "pop r12", - initial_user_state!(), - "mov rdi, r14", - "sub r13, 8", // ABI: RSP must be 16n+8 at function entry - "push {user_ss}", - "push r13", - "push 0x202", - "push {user_cs}", - "push r12", - "iretq", - unlock = sym crate::sched::driver::trampoline_entry, - user_ss = const crate::arch::percpu::USER_DS, - user_cs = const crate::arch::percpu::USER_CS, - ); -} - -/// Entry point for a kernel thread: r12 = body, r14 = argument. Never reaches Ring 3. -// The `sti` is load-bearing: `alloc_kernel_stack` leaves `IF` clear, and `trampoline_entry` requires it clear on entry. -#[unsafe(naked)] -pub(crate) extern "C" fn kernel_start() { - core::arch::naked_asm!( - "call {unlock}", - "sti", - "mov rdi, r14", - "call r12", - "call {returned}", - unlock = sym crate::sched::driver::trampoline_entry, - returned = sym kernel_thread_returned, - ); -} - -/// What [`kernel_start`] calls when a kernel thread's body returns: panics rather than halting silently. -extern "C" fn kernel_thread_returned() -> ! { - panic!("a kernel thread's body returned; nothing runs on this stack now"); + // SAFETY: `alloc` is fresh and exclusively owned, and `top` is its end. + let frame = unsafe { crate::arch::entry::initial_frame(top, trampoline, user_entry, user_sp, arg) }; + Some((alloc, frame)) } /// The last path component, truncated to what a process entry can hold. diff --git a/kernel/src/loader/tls.rs b/kernel/src/loader/tls.rs index cf25b920c23..fb01f221452 100644 --- a/kernel/src/loader/tls.rs +++ b/kernel/src/loader/tls.rs @@ -1,13 +1,18 @@ -//! A thread's TLS block: x86-64 variant II with the DTV in front of the data, built holding -//! physical addresses that `rebase_block` shifts once the block is mapped. The layout -//! arithmetic is `toyos_elf::tls`; this is the allocation, the template copies and the DTV. +//! A thread's TLS block, in this machine's psABI variant with the DTV in front of it, built +//! holding physical addresses that `rebase_block` shifts once the block is mapped. The layout +//! arithmetic is `toyos_elf::tls`; this is the allocation, the template copies, the TCB and the DTV. use crate::elf::TlsModule; use crate::mm::KernelSlice; use crate::process::{OwnedAlloc, PageAlloc}; use crate::DirectMap; +use toyos_elf::tls::{Static, Variant}; use toyos_elf::Layout; +/// This machine's TLS layout. +pub const VARIANT: Variant = Variant::of(crate::arch::ELF_MACHINE); + +/// Variant II's TCB; variant I's is the gap `toyos_elf::tls` leaves below the data. const TCB_SIZE: usize = 64; /// Module entries a thread's DTV can hold; `SYS_TLS_ALLOC_BLOCK` refuses a module id above it: there is nowhere to record the answer. pub const DTV_INITIAL_CAPACITY: usize = 64; @@ -23,6 +28,7 @@ pub fn setup_tls( tls_memsz: usize, tls_align: usize, ) -> Option<(PageAlloc, u64)> { + let tls = Static::new(VARIANT, tls_memsz, tls_align, tls_align)?; setup_combined_tls( &[TlsModule { template: tls_template, @@ -31,24 +37,13 @@ pub fn setup_tls( module_id: 1, is_static: true, }], - tls_memsz, - tls_align, + tls, ) } /// One thread's TLS block for every static module; `None` when no allocation holds the layout. -pub fn setup_combined_tls( - modules: &[TlsModule], - total_memsz: usize, - tls_align: usize, -) -> Option<(PageAlloc, u64)> { - let plan = toyos_elf::tls::plan( - total_memsz, - tls_align, - TCB_SIZE, - DTV_BYTES, - crate::mm::PAGE_2M as usize, - )?; +pub fn setup_combined_tls(modules: &[TlsModule], tls: Static) -> Option<(PageAlloc, u64)> { + let plan = tls.plan(TCB_SIZE, DTV_BYTES, crate::mm::PAGE_2M as usize)?; let page_alloc = PageAlloc::new(plan.alloc_size, crate::mm::pmm::Category::InitTls)?; let block = page_alloc.ptr(); @@ -74,11 +69,18 @@ pub fn setup_combined_tls( let tp_user = block_phys + plan.tp_offset as u64; // SAFETY: the plan reserves `TCB_SIZE` bytes at `tp_offset` inside `alloc_size`. let tp_kernel = unsafe { block.add(plan.tp_offset) } as *mut u64; - // TP+0 is the psABI self-pointer, TP+8 the DTV pointer. - // SAFETY: two words of the `TCB_SIZE` reserved at `tp_kernel`; `block` is still unpublished. + // SAFETY: two words of the TCB the plan reserves at `tp_kernel` (`TCB_SIZE`, or + // variant I's gap of at least 16); `block` is still unpublished. unsafe { - *tp_kernel = tp_user; - *tp_kernel.add(1) = block_phys; + match VARIANT { + // TP+0 the psABI self-pointer, TP+8 the DTV pointer. + Variant::II => { + *tp_kernel = tp_user; + *tp_kernel.add(1) = block_phys; + } + // TP+0 the DTV pointer, TP+8 reserved (zeroed above). + Variant::I => *tp_kernel = block_phys, + } } let dtv = block as *mut u64; @@ -110,8 +112,13 @@ pub(crate) unsafe fn rebase_block(phys: u64, tp_offset: usize, fs_base: u64, reb unsafe { let block = DirectMap::from_phys(phys).as_mut_ptr::(); let tp = block.add(tp_offset) as *mut u64; - *tp = fs_base; - *tp.add(1) = (*tp.add(1) as i64 + rebase) as u64; + match VARIANT { + Variant::II => { + *tp = fs_base; + *tp.add(1) = (*tp.add(1) as i64 + rebase) as u64; + } + Variant::I => *tp = (*tp as i64 + rebase) as u64, + } let dtv = block as *mut u64; let dtv_len = *dtv.add(1) as usize; for i in 0..dtv_len { @@ -127,11 +134,10 @@ pub(crate) unsafe fn rebase_block(phys: u64, tp_offset: usize, fs_base: u64, reb pub fn map_block( child_pt: &crate::process::PageTables, modules: &[TlsModule], - total_memsz: usize, - max_align: usize, + tls: Static, ) -> Option<(crate::process::MappedPages, u64)> { - let (alloc, fs_base) = if total_memsz > 0 { - setup_combined_tls(modules, total_memsz, max_align)? + let (alloc, fs_base) = if tls.total_memsz() > 0 { + setup_combined_tls(modules, tls)? } else { setup_tls(None, 0, 1)? }; @@ -149,48 +155,38 @@ pub fn map_block( } /// One combined block for every startup module; `None` when they do not fit, since a missing module would mean relocations resolving against a block that is not there. +/// The executable's module goes where its linker resolved its own accesses: next to the thread +/// pointer, last in variant II and first in variant I. pub fn build_tls_layout( loaded_libs: &[crate::elf::LoadedLib], layout: &Layout, exe_tls_template: Option<&OwnedAlloc>, -) -> Option<(alloc::vec::Vec, usize, usize, u64)> { - let exe = layout.tls.filter(|t| t.memsz > 0); - let libs = loaded_libs.iter().filter(|lib| lib.tls_memsz > 0); +) -> Option<(alloc::vec::Vec, Static, u64)> { + // (template, memsz, align, module id). Module id 1 is the executable's; libraries start at 2. + let exe = layout.tls.filter(|t| t.memsz > 0).map(|tls| { + (exe_tls_template.map(|buf| buf.slice(tls.filesz as usize)), tls.memsz as usize, tls.align as usize, 1) + }); + let libs = loaded_libs + .iter() + .filter(|lib| lib.tls_memsz > 0) + .zip(2u64..) + .map(|(lib, id)| (lib.tls_template, lib.tls_memsz, lib.tls_align, id)); + let next_module_id = 2 + loaded_libs.iter().filter(|lib| lib.tls_memsz > 0).count() as u64; + let order: alloc::vec::Vec<_> = match VARIANT { + Variant::II => libs.chain(exe).collect(), + Variant::I => exe.into_iter().chain(libs).collect(), + }; - let mut modules = alloc::vec::Vec::with_capacity(loaded_libs.len() + 1); + let mut modules = alloc::vec::Vec::with_capacity(order.len()); let mut cursor = 0usize; let mut max_align = 1usize; - // Module id 1 is the executable's; libraries start at 2. - let mut next_module_id = 2u64; - - for lib in libs { - let (base_offset, next) = - toyos_elf::tls::place_module(cursor, lib.tls_memsz, lib.tls_align)?; - cursor = next; - max_align = max_align.max(lib.tls_align); - modules.push(TlsModule { - template: lib.tls_template, - memsz: lib.tls_memsz, - base_offset, - module_id: next_module_id, - is_static: true, - }); - next_module_id += 1; - } - - if let Some(tls) = exe { - let (base_offset, next) = - toyos_elf::tls::place_module(cursor, tls.memsz as usize, tls.align as usize)?; + for (template, memsz, align, module_id) in order.iter().copied() { + let (base_offset, next) = toyos_elf::tls::place_module(cursor, memsz, align)?; cursor = next; - max_align = max_align.max(tls.align as usize); - modules.push(TlsModule { - template: exe_tls_template.map(|buf| buf.slice(tls.filesz as usize)), - memsz: tls.memsz as usize, - base_offset, - module_id: 1, - is_static: true, - }); + max_align = max_align.max(align); + modules.push(TlsModule { template, memsz, base_offset, module_id, is_static: true }); } - - Some((modules, cursor, max_align, next_module_id)) + let first_align = order.first().map_or(1, |&(_, _, align, _)| align); + let tls = Static::new(VARIANT, cursor, max_align, first_align)?; + Some((modules, tls, next_module_id)) } diff --git a/kernel/src/log/mod.rs b/kernel/src/log/mod.rs index 6a6ab818d0d..749cd37afe3 100644 --- a/kernel/src/log/mod.rs +++ b/kernel/src/log/mod.rs @@ -215,7 +215,7 @@ struct Origin { /// the `xadd` is atomic against a same-CPU interrupt only, not against another /// CPU, so this holds only while the CPU keeps ownership of the shard across /// the whole bracket, since work stealing is enabled. -fn reserve(guard: &crate::arch::LogCommitGuard) -> (Origin, u64) { +fn reserve(guard: &crate::arch::IrqGuard) -> (Origin, u64) { if !PERCPU_READY.load(Ordering::Relaxed) { // SAFETY: nothing else is running, so this CPU owns the boot shard. let seq = unsafe { BOOT_SHARD.reserve(guard) }; @@ -250,7 +250,15 @@ pub fn emit(level: Level, args: core::fmt::Arguments) { record.len = message.len as u16; record.elided = message.elided.min(u16::MAX as usize) as u16; - let guard = crate::arch::LogCommitGuard::close(); + // `log-unbracketed-reserve` stages a reservation made with interrupts open. + #[cfg(feature = "boot-actuators")] + let guard = if crate::actuator::log_unbracketed_reserve() { + crate::arch::IrqGuard::unclosed() + } else { + crate::arch::IrqGuard::close() + }; + #[cfg(not(feature = "boot-actuators"))] + let guard = crate::arch::IrqGuard::close(); // Stamped inside the bracket: outside it, ordering by seq and by at_ns // could disagree. The NMI handler never logs and #MC halts rather than // returning, which is what closes the two paths IF/TF masking alone diff --git a/kernel/src/log/nested.rs b/kernel/src/log/nested.rs index 10b3ae8bcae..5e80f015d60 100644 --- a/kernel/src/log/nested.rs +++ b/kernel/src/log/nested.rs @@ -72,7 +72,7 @@ mod armed { return false; } OWED.store(true, Ordering::Relaxed); - crate::arch::apic::send_self(crate::arch::idt::LOG_NEST_VECTOR); + crate::arch::irqchip::send_self(crate::arch::trap::LOG_NEST_VECTOR); true } @@ -92,7 +92,7 @@ mod armed { return; } OWED.store(true, Ordering::Relaxed); - crate::arch::apic::send_self(crate::arch::idt::LOG_NEST_VECTOR); + crate::arch::irqchip::send_self(crate::arch::trap::LOG_NEST_VECTOR); for _ in 0..WINDOW { core::hint::spin_loop(); } diff --git a/kernel/src/log/shard.rs b/kernel/src/log/shard.rs index 0ac0da0b6b5..63e63e7e4c9 100644 --- a/kernel/src/log/shard.rs +++ b/kernel/src/log/shard.rs @@ -150,7 +150,7 @@ impl Shard { /// Take the next sequence number, on the CPU that owns this shard. /// # Safety /// Caller must be the owning CPU, and `guard` must stay live through the matching [`Shard::commit`]. - pub unsafe fn reserve(&self, guard: &crate::arch::LogCommitGuard) -> u64 { + pub unsafe fn reserve(&self, guard: &crate::arch::IrqGuard) -> u64 { crate::arch::percpu_fetch_add(&self.head, guard) } @@ -161,7 +161,7 @@ impl Shard { &self, seq: u64, record: &LogRecord, - _guard: &crate::arch::LogCommitGuard, + _guard: &crate::arch::IrqGuard, ) { debug_assert!( self.head().saturating_sub(seq) < SHARD_RECORDS as u64, diff --git a/kernel/src/main.rs b/kernel/src/main.rs index 97ffa155518..0aaf0760365 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -12,6 +12,7 @@ static DEBUG_WAIT: core::sync::atomic::AtomicBool = core::sync::atomic::AtomicBo pub use mm::{UserAddr, DirectMap, PHYS_OFFSET}; +mod invalidation; mod shootdown; mod sleeplock; mod smp_roster; @@ -45,8 +46,6 @@ mod usb_gate; mod nvme_gate; #[cfg(feature = "boot-actuators")] mod sched_gate; -#[cfg(feature = "boot-actuators")] -mod nmi_gate; mod block; mod durability; mod gpt; @@ -74,7 +73,6 @@ mod process; mod loader; mod scheduler; mod sched; -mod hw; mod iommu; mod preempt; mod irq_census; @@ -82,7 +80,6 @@ mod irq_ring; mod trace; mod time; mod clock; -mod rtc; mod watch; mod iod; @@ -95,6 +92,7 @@ mod pcidev; mod gpu; mod user_ptr; mod vma; +mod syscall; /// Nested generic forces a demangled symbol wider than the console grid, /// proving `screen_late_panic`'s renderer really wraps. @@ -116,8 +114,9 @@ mod late_panic { use crate::mm::paging::MmioPolicy; use alloc::boxed::Box; use alloc::sync::Arc; -use arch::{apic, cpu, idt, pat, percpu, smp, syscall}; -use drivers::{acpi, gop, i8042, ioapic, nvme, pci, serial, virtio_console, virtio_gpu, virtio_sound, xhci}; +use arch::{cpu, percpu, smp}; +pub(crate) use arch::hw; +use drivers::{acpi, gop, nvme, pci, serial, virtio_console, virtio_gpu, virtio_sound, xhci}; use toyos_abi::boot::{KernelArgs, MemoryMapEntry}; #[panic_handler] @@ -148,7 +147,7 @@ fn panic(info: &core::panic::PanicInfo) -> ! { alert!("EARLY PANIC: {}", info); // Before the capture, so the arm line is the panel's last one. let bound = panic_reboot::arm(true); - // Halts directly instead of via halt_all_cpus: idt::init hasn't run yet, so a renderer fault would triple-fault. + // Halts directly instead of via halt_all_cpus: the exception table is not loaded yet, so a renderer fault would find firmware's. drivers::panic_console::capture(); // SAFETY: no other writer can be mid-transmission — IF is clear here and every other CPU is about to halt. unsafe { drivers::serial::panic_flush(); } @@ -163,16 +162,10 @@ fn panic(info: &core::panic::PanicInfo) -> ! { if prev != percpu::CpuFaultState::Normal { // Escalate: reentry depth is zero here, so this landed on a fatal exception or page fault no handler was inside. panic::last_words("DOUBLE PANIC", Some(prev), info, true); - apic::halt_all_cpus(); + panic::halt_all_cpus(); } - let rbp: u64; - // SAFETY: register-to-register mov only (nomem, nostack); -Cforce-frame-pointers=yes makes rbp a real frame pointer. - unsafe { core::arch::asm!("mov {}, rbp", out(reg) rbp, options(nomem, nostack)); } - - arch::idt::exceptions::crash_report( - &arch::idt::exceptions::CrashInfo::Panic { message: info, rbp } - ); + arch::trap::report_panic(info, cpu::frame_pointer()); // Captures now: recovery below may re-enter a scheduler this panic left locked, so a later drain isn't guaranteed. drivers::panic_console::capture(); @@ -191,29 +184,10 @@ fn panic(info: &core::panic::PanicInfo) -> ! { depth.store(0, core::sync::atomic::Ordering::SeqCst); // Discarded here: a stale capture would blame this panic for the next fatal one. drivers::panic_console::discard_capture(); - arch::idt::exceptions::try_recover_from_panic(); + arch::trap::try_recover_from_panic(); } - apic::halt_all_cpus(); -} - -/// Entry point: the bootloader jumps here with `rdi = &KernelArgs`, switches to the kernel's own stack, calls `kernel_main`. -/// # Safety -/// Only the bootloader may call this, fresh from firmware, with `rdi` holding a live [`KernelArgs`]. -#[unsafe(naked)] -#[no_mangle] -pub unsafe extern "sysv64" fn _start(_kernel_args: &KernelArgs) -> ! { - core::arch::naked_asm!( - "mov rax, [rdi + 16]", // kernel_memory_addr - "add rax, [rdi + 32]", // + kernel_stack_addr - "add rax, [rdi + 40]", // + kernel_stack_size - "movabs rbx, {phys_offset}", - "add rax, rbx", - "mov rsp, rax", - "call {kernel_main}", - phys_offset = const PHYS_OFFSET, - kernel_main = sym kernel_main, - ); + panic::halt_all_cpus(); } fn register_gpu(driver: Box, info: gpu::GpuInfo) { @@ -266,7 +240,11 @@ fn report_log_destination() { } } -unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { +/// The architecture's entry calls this once, on the kernel's own stack, with +/// the loader's arguments. +/// # Safety +/// `kernel_args` is the loader's live [`KernelArgs`], and nothing has run before this. +pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { // Copied onto the kernel stack: the original lives on the UEFI stack, unreachable once mm::init drops the identity map. let kernel_args = *kernel_args; @@ -276,13 +254,7 @@ unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { entry_count, ); - // **Before the panel and not after it**: the loader maps the scanout - // uncacheable, and `panic_console::arm`'s own record is the panel's first - // paint, so there is no window between arming the panel and painting - // through it in which to establish a memory type. The write alone, and it - // logs nothing: the read-back is `pat::check`, below, where a refusal has - // a channel to reach. - pat::init(); + arch::boot::before_panel(); // Before serial::init: the screen may be the only surviving channel if serial::init itself faults. drivers::panic_console::arm(&kernel_args, maps); @@ -299,7 +271,7 @@ unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { ) }); - serial::init(); + serial::init(kernel_args.rsdp_addr); // After both channels exist, before the first actuator site. // cmdline_len==0 is checked first: an empty bootloader Vec has no backing allocation to point at. @@ -319,24 +291,13 @@ unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { actuator::init(cmdline); let root_image = rootfs::init(cmdline, &kernel_args, maps); - // Armed here so the next record — `PAT:` — reaches the console and the panel keeps the one before it. + // Armed here so the next record — the architecture's first — reaches the console and the panel keeps the one before it. #[cfg(feature = "boot-actuators")] if actuator::test_early_halt() { log::halt_before_the_next_repaint(); } - // After actuator::init, whose table the `control-regs-bench` probe inside - // this call reads. `pat::init` above restored the `CR0` it found, so a - // firmware `CD` — which would make every mapping uncacheable whatever the - // PAT says — ends here. - arch::control_regs::init_cr0(0); - - // The read-back `pat::init` owes, on a boot that now has three channels to - // carry a refusal. - pat::check(); - - log!("PAT: IA32_PAT={:#018x}, entry {} = {}", - pat::msr(), pat::WC_ENTRY, pat::entry_name(pat::WC_ENTRY)); + arch::boot::after_console(&kernel_args, maps); // percpu, the allocator and our own paging aren't up yet, so a fault here only reaches the early-panic branch. if actuator::test_early_panic() { @@ -401,7 +362,7 @@ unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { mm::Region { start: kernel_args.kernel_memory_addr, end: kernel_args.kernel_memory_addr + kernel_args.kernel_memory_size }, mm::Region { start: kernel_args.kernel_elf_addr, end: kernel_args.kernel_elf_addr + kernel_args.kernel_elf_size }, mm::Region { start: kernel_args.kernel_stack_addr, end: kernel_args.kernel_stack_addr + kernel_args.kernel_stack_size }, - mm::Region { start: 0x8000, end: 0x9000 }, // AP trampoline page + arch::boot::reserved(), // The loader's black-box page, which is ordinary `LoaderData` and so // memory the allocator would otherwise hand out. Empty on a boot whose // parameter line names none. @@ -427,39 +388,15 @@ unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { // reads off a refusal below is which tables the firmware published at all. acpi::inventory(kernel_args.rsdp_addr); - // `init_bsp` loads the IDT partway through, as early as this CPU's `gs:` - // allows: a fault in any later phase then diagnoses instead of stopping in - // a handler the firmware left behind. - let madt = acpi::parse_madt(kernel_args.rsdp_addr).expect("ACPI: MADT not found"); - // Off the same tables as the MADT, and before the IDT below makes a panic - // reportable: a panic that can be reported but not ended leaves the machine - // holding its panel for a hand that may not be in the room. - acpi::init_reset(kernel_args.rsdp_addr); - apic::init(); - percpu::init_bsp(apic::id()); - ioapic::init(&madt); - idt::enable_interrupts(); - syscall::init(); + let platform = arch::boot::interrupts(kernel_args.rsdp_addr); symbols::set_kernel_base(kernel_args.kernel_memory_addr); if !kernel_elf.is_empty() { symbols::load_kernel(kernel_elf, mm::PHYS_OFFSET + kernel_args.kernel_memory_addr); } - // HPET clock — enables profiling for everything from here on - let hpet_base = acpi::find_hpet_base(kernel_args.rsdp_addr) - .expect("ACPI: HPET not found"); - clock::init(hpet_base); - // Century register and time zone both come from ACPI/firmware, not the RTC's own registers. - let century_reg = match acpi::rtc_century_register(kernel_args.rsdp_addr) { - Ok(reg) => reg, - Err(e) => { - log!("ACPI: the FADT is unreadable ({e:?}), so where the RTC keeps its century is unknown too"); - None - } - }; - clock::init_wall(century_reg, kernel_args.rtc_utc_offset()); + arch::boot::clock(kernel_args); trace::enable(); - apic::init_timer(); + arch::boot::timer(); // After both halves of what it needs: a TSC period to convert its bound // with, and a timer whose every tick polls it. deadline::start(); @@ -485,17 +422,17 @@ unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { iommu::init(kernel_args.rsdp_addr, &pci_devices); // Before storage and everything under it: what it covers is the rest of this // boot, and a wedge down there is the reason to have one. - drivers::watchdog::init(&pci_devices); + arch::watchdog::init(&pci_devices); file_cache::init(); gpt::init(kernel_args); - i8042::init(kernel_args.rsdp_addr); + arch::boot::platform_devices(kernel_args.rsdp_addr); acpi::init_power(kernel_args.rsdp_addr); boot_phase!("peripherals ready", t_periph); let t_subsys = clock::nanos_since_boot(); - smp::boot_aps(&madt, kernel_args.boot_pml4_addr); + arch::boot::start_other_cpus(&platform, kernel_args); vfs::init(); process::init(); scheduler::init(); @@ -655,16 +592,8 @@ unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { drivers::virtio::used_selftest(); } - // Needs interrupts on and the timer already ticking: its last assertion is that the interrupt after the spurious one arrives. #[cfg(feature = "boot-actuators")] - if actuator::lapic_spurious_selftest() { - arch::idt::spurious::selftest(); - } - - #[cfg(feature = "boot-actuators")] - if actuator::unclaimed_vector_selftest() { - arch::idt::unclaimed::selftest(); - } + arch::boot::interrupt_selftests(); virtio_console::init(&pci_devices); @@ -706,7 +635,7 @@ unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { } report_log_destination(); - let complete_tsc = cpu::rdtsc(); + let complete_tsc = cpu::counter(); boot_phase!("complete", 0); report_power_on(kernel_args, complete_tsc); @@ -720,8 +649,7 @@ unsafe fn kernel_main(kernel_args: &KernelArgs) -> ! { // Same no-current-task window as above: blame is Kernel, so fatal_exception halts the machine. if actuator::test_kernel_fault() { - // SAFETY: ud2 reads and writes nothing (nomem, nostack) and raises #UD, caught by the already-installed IDT. - unsafe { core::arch::asm!("ud2", options(nomem, nostack)) }; + cpu::undefined_instruction(); } // Last thing before enter_idle_loop: nothing can run before it, and a klogd spawned earlier would idle through phases 5-7 with no drainer. diff --git a/kernel/src/mm/mmio.rs b/kernel/src/mm/mmio.rs index fe3d8f1e6fc..32bade8a15c 100644 --- a/kernel/src/mm/mmio.rs +++ b/kernel/src/mm/mmio.rs @@ -1,6 +1,28 @@ +//! A bounds-checked window over a device's registers, and the ordering every +//! access through it carries. +//! +//! **The contract, which is what a driver may rely on:** a `write_*` is ordered +//! after every store this CPU made to memory before it, as a device's DMA reads +//! observe them — the descriptor before the doorbell — and a `read_*` is +//! ordered before every load this CPU makes after it — the status before the +//! entry it reports. It is Linux's `writel`/`readl` and not their `_relaxed` +//! forms: `arch::barrier::before_mmio_write` and `after_mmio_read` supply it, +//! which is a compiler barrier on x86-64's TSO and a `dmb` on AArch64, where a +//! plain `fence` is inner-shareable and orders nothing a device sees. +//! A `write_*` orders prior *stores* only, as `writel` does: a doorbell that +//! hands entries back to the device after this CPU *read* them (an NVMe +//! completion queue head, xHCI's ERDP) is ordered after those loads only by +//! its value depending on them. +//! +//! A store to device memory is not a store to plain memory, and `volatile` +//! alone promises nothing about the two against each other: without the +//! barrier the compiler may move a descriptor write past the doorbell that +//! publishes it on either architecture. + use core::ptr::{read_volatile, write_volatile}; use super::DirectMap; +use crate::arch::barrier; /// Bounds-checked volatile window over device or kernel-owned memory. Copy, no ownership, no lifetime. #[derive(Clone, Copy)] @@ -11,11 +33,12 @@ pub struct Mmio { // SAFETY: the window's address is fixed for the machine's life and Mmio carries no lock, so Send costs nothing new. unsafe impl Send for Mmio {} -// SAFETY: every access goes through read_volatile/write_volatile below, which order correctly regardless of which CPU issues them. +// SAFETY: every access is one volatile load or store carrying the module's ordering +// contract on whichever CPU issues it; nothing here is shared state of its own. unsafe impl Sync for Mmio {} impl Mmio { - pub(super) fn new(base: DirectMap, size: u64) -> Self { + pub(crate) fn new(base: DirectMap, size: u64) -> Self { Self { base: base.as_mut_ptr(), size } } @@ -70,13 +93,17 @@ impl Mmio { pub fn read_u8(self, offset: u64) -> u8 { self.check(offset, 1); // SAFETY: check asserted the offset fits; read_volatile preserves the register's read side effect. - unsafe { read_volatile(self.base.add(offset as usize) as *const u8) } + let value = unsafe { read_volatile(self.base.add(offset as usize) as *const u8) }; + barrier::after_mmio_read(); + value } #[inline] pub fn write_u8(self, offset: u64, val: u8) { self.check(offset, 1); - // SAFETY: check asserted the offset fits; write_volatile preserves ordering against other MMIO accesses. + barrier::before_mmio_write(); + // SAFETY: check asserted the offset fits; write_volatile keeps the store, and the + // barrier above orders it after this CPU's earlier stores. unsafe { write_volatile(self.base.add(offset as usize), val) } } @@ -84,13 +111,17 @@ impl Mmio { pub fn read_u16(self, offset: u64) -> u16 { self.check(offset, 2); // SAFETY: check asserted the offset fits; read_volatile preserves the register's read side effect. - unsafe { read_volatile(self.base.add(offset as usize) as *const u16) } + let value = unsafe { read_volatile(self.base.add(offset as usize) as *const u16) }; + barrier::after_mmio_read(); + value } #[inline] pub fn write_u16(self, offset: u64, val: u16) { self.check(offset, 2); - // SAFETY: check asserted the offset fits; write_volatile preserves ordering against other MMIO accesses. + barrier::before_mmio_write(); + // SAFETY: check asserted the offset fits; write_volatile keeps the store, and the + // barrier above orders it after this CPU's earlier stores. unsafe { write_volatile(self.base.add(offset as usize) as *mut u16, val) } } @@ -98,13 +129,17 @@ impl Mmio { pub fn read_u32(self, offset: u64) -> u32 { self.check(offset, 4); // SAFETY: check asserted the offset fits; read_volatile preserves the register's read side effect. - unsafe { read_volatile(self.base.add(offset as usize) as *const u32) } + let value = unsafe { read_volatile(self.base.add(offset as usize) as *const u32) }; + barrier::after_mmio_read(); + value } #[inline] pub fn write_u32(self, offset: u64, val: u32) { self.check(offset, 4); - // SAFETY: check asserted the offset fits; write_volatile preserves ordering against other MMIO accesses. + barrier::before_mmio_write(); + // SAFETY: check asserted the offset fits; write_volatile keeps the store, and the + // barrier above orders it after this CPU's earlier stores. unsafe { write_volatile(self.base.add(offset as usize) as *mut u32, val) } } @@ -112,13 +147,17 @@ impl Mmio { pub fn read_u64(self, offset: u64) -> u64 { self.check(offset, 8); // SAFETY: check asserted the offset fits; read_volatile preserves the register's read side effect. - unsafe { read_volatile(self.base.add(offset as usize) as *const u64) } + let value = unsafe { read_volatile(self.base.add(offset as usize) as *const u64) }; + barrier::after_mmio_read(); + value } #[inline] pub fn write_u64(self, offset: u64, val: u64) { self.check(offset, 8); - // SAFETY: check asserted the offset fits; write_volatile preserves ordering against other MMIO accesses. + barrier::before_mmio_write(); + // SAFETY: check asserted the offset fits; write_volatile keeps the store, and the + // barrier above orders it after this CPU's earlier stores. unsafe { write_volatile(self.base.add(offset as usize) as *mut u64, val) } } } diff --git a/kernel/src/mm/mod.rs b/kernel/src/mm/mod.rs index ee4e5145411..af525445e6d 100644 --- a/kernel/src/mm/mod.rs +++ b/kernel/src/mm/mod.rs @@ -2,10 +2,13 @@ #![warn(clippy::undocumented_unsafe_blocks)] pub mod pmm; -pub mod paging; +/// The architecture's page tables: `crate::arch::paging`, named here because +/// every caller asks it as memory management. +pub use crate::arch::paging; mod alloc; mod dma; mod mmio; +pub mod policy; mod region; mod unmapped; @@ -18,8 +21,6 @@ pub use alloc::sweep as sweep_heap_bands; /// Isolates the sweep's lock hold from the sweep itself, to attribute contention correctly. #[cfg(feature = "heap-lockspin")] pub use alloc::hold_lock as hold_heap_lock; -/// Unconditional: `hw::report_contexts` runs on every crash and must build with no sweep present. -pub use alloc::sweep_stats; pub use dma::{Dma, DmaPool, Unaligned}; pub use mmio::Mmio; pub use region::{Allocation, KernelSlice}; @@ -155,3 +156,18 @@ pub fn init(memory_map: &[MemoryMapEntry], reserved: &[Region]) { paging::init(memory_map); alloc::init(); } + +/// The two memory facts every crash report ends its contexts with: how deep +/// any task's kernel stack went, and what the heap sweep last saw. +/// Unconditional: it runs on every crash and must build with no sweep present. +pub fn report_on_crash() { + if let Some((used, of)) = crate::sched::driver::stack_high_water() { + crate::log!(" Task kernel stacks: deepest {used} of {of} bytes"); + } + if let Some((sweeps, records, overflowed)) = alloc::sweep_stats() { + crate::log!( + " Heap sweeps: {sweeps} run, {records} live bands on the last walk{}", + if overflowed { ", and the page table filled — the walk is incomplete" } else { "" }, + ); + } +} diff --git a/kernel/src/mm/policy.rs b/kernel/src/mm/policy.rs new file mode 100644 index 00000000000..a95e65629dd --- /dev/null +++ b/kernel/src/mm/policy.rs @@ -0,0 +1,79 @@ +//! What a mapping is for and what memory type it gets, named by concept: each +//! architecture's page tables encode these (x86-64's PAT bits, AArch64's +//! `MAIR` indices), and nothing above them spells a bit. + +use super::PAGE_2M; + +/// 4 KiB pages in one 2 MiB page. +pub const PAGES_PER_2M: usize = (PAGE_2M / 4096) as usize; + +/// What a user mapping may be used for: no variant is both writable and +/// executable, and every variant implies read. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Prot { + /// Read-only: neither writable nor executable. + Read, + /// Data: readable and writable, never executable. + ReadWrite, + /// Code. Never writable. + ReadExec, +} + +/// What each 4 KiB page of a 2 MiB window may be used for: split because +/// `toyos-ld` can align a window across the end of `.text` and start of `.data`. +pub struct WindowProt([Prot; PAGES_PER_2M]); + +impl WindowProt { + /// A window whose pages all say the same thing. + pub const fn uniform(prot: Prot) -> Self { + Self([prot; PAGES_PER_2M]) + } + + /// Sets the 4 KiB page `offset` bytes in; an out-of-window offset panics. + pub fn set(&mut self, offset: u64, prot: Prot) { + self.0[(offset / 4096) as usize] = prot; + } + + /// The one protection every page carries, or `None` where they disagree. + pub(crate) fn agreed(&self) -> Option { + let first = self.0[0]; + self.0.iter().all(|&p| p == first).then_some(first) + } + + /// Every 4 KiB page's protection, in order. + pub(crate) fn pages(&self) -> impl Iterator + '_ { + self.0.iter().copied() + } +} + +/// The memory type a mapping gets, out of the three this kernel ever writes. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum CachePolicy { + /// Ordinary memory: RAM, write-back. + Normal, + /// Uncached and unbuffered, whatever the firmware set or forgot: device + /// registers. + Uncacheable, + /// Uncached with stores gathered: a scanout. + WriteCombining, +} + +/// What an MMIO window may select — never [`CachePolicy::Normal`]: a device +/// register mapped cacheable is one a speculative or combined access can +/// reach, and this type removes that as a possibility. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum MmioPolicy { + /// Registers. + Uncacheable, + /// The scanout alone. + WriteCombining, +} + +impl MmioPolicy { + pub fn cache(self) -> CachePolicy { + match self { + Self::Uncacheable => CachePolicy::Uncacheable, + Self::WriteCombining => CachePolicy::WriteCombining, + } + } +} diff --git a/kernel/src/object/ops.rs b/kernel/src/object/ops.rs index 3ceabceb821..29452bea925 100644 --- a/kernel/src/object/ops.rs +++ b/kernel/src/object/ops.rs @@ -70,7 +70,7 @@ pub fn install(table: &mut HandleTable, object: KObjectRef) -> Result Self { - Self { phys: DirectMap::from_phys(0), size: 0, cache: CachePolicy::DeferToMtrr, pages: None } + Self { phys: DirectMap::from_phys(0), size: 0, cache: CachePolicy::Normal, pages: None } } } @@ -87,7 +87,7 @@ impl SharedMemObject { Ok(Self::over(Region { phys, size: aligned as u64, - cache: CachePolicy::DeferToMtrr, + cache: CachePolicy::Normal, pages: Some(Arc::new(Pages(pages))), })) } @@ -102,7 +102,7 @@ impl SharedMemObject { /// a second device's domain would be one device reaching another's /// registers. pub fn ram(&self) -> Option<(u64, u64)> { - (self.region.pages.is_some() && self.region.cache == CachePolicy::DeferToMtrr) + (self.region.pages.is_some() && self.region.cache == CachePolicy::Normal) .then(|| (self.region.phys.phys(), self.region.size)) } @@ -139,7 +139,7 @@ impl SharedMemObject { // Logged only for a non-default policy: this process is the one // paying for it. Read back the installed policy, not the request, // so the line describes the mapping. - if self.region.cache != CachePolicy::DeferToMtrr { + if self.region.cache != CachePolicy::Normal { let installed = pt.lock().user_policy(addr).expect("shm: just mapped"); crate::log!( "shm: {:#x} mapped {:?} into pid {}", diff --git a/kernel/src/panic.rs b/kernel/src/panic.rs index a3820895781..635ba88326f 100644 --- a/kernel/src/panic.rs +++ b/kernel/src/panic.rs @@ -11,6 +11,7 @@ use core::sync::atomic::{AtomicU32, AtomicU64, AtomicU8, Ordering}; use crate::arch::cpu; use crate::arch::percpu::CpuFaultState; use crate::drivers::serial; +use crate::time::{Budget, Duration}; /// Slots in every per-CPU array here: an APIC id masked to six bits. const SLOTS: usize = 64; @@ -22,30 +23,10 @@ const SLOTS: usize = 64; /// `halt_all_cpus` halts the second CPU anyway. static PANIC_DEPTH: [AtomicU32; SLOTS] = [const { AtomicU32::new(0) }; SLOTS]; -/// This CPU's APIC id, from CPUID. -/// -/// Not `rdmsr(IA32_X2APIC_APICID)`: that MSR is `#GP` before `apic::init_ap` -/// has run, and a panic an AP takes before then must not fault inside the -/// reentry guard. -pub fn apic_id() -> u32 { - let (max_leaf, _, _, _) = cpu::cpuid(0, 0); - for leaf in [0x1F, 0x0B] { - if max_leaf >= leaf { - let (_, ebx, _, edx) = cpu::cpuid(leaf, 0); - // SDM Vol. 2A, CPUID leaf 0BH: EBX[15:0] == 0 means unimplemented, - // not id 0, so the leaf-1 fallback below must still run. - if ebx & 0xFFFF != 0 { - return edx; - } - } - } - let (_, ebx, _, _) = cpu::cpuid(1, 0); - ebx >> 24 -} - -/// This CPU's reentry depth. +/// This CPU's reentry depth, indexed by the id the hardware gives the CPU +/// (`arch::cpu::hardware_id`), which answers before any per-CPU state exists. pub fn depth_slot() -> &'static AtomicU32 { - &PANIC_DEPTH[apic_id() as usize & (SLOTS - 1)] + &PANIC_DEPTH[cpu::hardware_id() as usize & (SLOTS - 1)] } /// Path capture bound; overflow is cut from the front (see [`copy_tail`]). @@ -72,11 +53,12 @@ impl Kind { } } -/// One CPU's first unfinished crash. Every field is `Relaxed`: one CPU writes -/// its own slot and the same CPU reads it, with interrupts masked throughout, -/// and no other CPU ever looks. Lengths are stored after the bytes, so a slot -/// read mid-fill — an NMI panicking between the claim and the copy — reports -/// a short string rather than the previous crash's tail. +/// One CPU's first unfinished crash. One CPU writes its own slot and the same +/// CPU reads it, with interrupts masked throughout, and no other CPU ever looks. +/// Lengths are stored `Release` after the bytes and loaded `Acquire` before +/// them, so a slot read mid-fill — an NMI panicking between the claim and the +/// copy — reports a short string rather than the previous crash's tail; every +/// other field is `Relaxed`. struct Evidence { kind: AtomicU8, file_len: AtomicU8, @@ -117,7 +99,7 @@ impl Evidence { static FIRST: [Evidence; SLOTS] = [const { Evidence::new() }; SLOTS]; fn evidence() -> &'static Evidence { - &FIRST[apic_id() as usize & (SLOTS - 1)] + &FIRST[cpu::hardware_id() as usize & (SLOTS - 1)] } /// Claims this CPU's slot for the first crash; declines if one is already claimed. @@ -134,7 +116,7 @@ pub fn record_panic(info: &core::panic::PanicInfo) { if !claim(slot, Kind::Panic) { return; } - slot.apic.store(apic_id(), Ordering::Relaxed); + slot.apic.store(cpu::hardware_id(), Ordering::Relaxed); if let Some(location) = info.location() { copy_tail(&slot.file, &slot.file_len, &slot.file_cut, location.file().as_bytes()); slot.line.store(location.line(), Ordering::Relaxed); @@ -153,7 +135,7 @@ pub fn record_fault(name: &str, rip: u64, cr2: u64, error_code: u64) { if !claim(slot, Kind::Fault) { return; } - slot.apic.store(apic_id(), Ordering::Relaxed); + slot.apic.store(cpu::hardware_id(), Ordering::Relaxed); copy_head(&slot.msg, &slot.msg_len, name.as_bytes()); slot.rip.store(rip, Ordering::Relaxed); slot.cr2.store(cr2, Ordering::Relaxed); @@ -178,7 +160,9 @@ fn copy_tail(dst: &[AtomicU8], len: &AtomicU8, cut: &AtomicU8, src: &[u8]) { for (slot, &b) in dst.iter().zip(tail) { slot.store(b, Ordering::Relaxed); } - len.store(tail.len() as u8, Ordering::Relaxed); + // `Release`: the bytes before the length, for a reader that interrupts + // this copy. + len.store(tail.len() as u8, Ordering::Release); cut.store(u8::from(from > 0), Ordering::Relaxed); } @@ -191,12 +175,15 @@ fn copy_head(dst: &[AtomicU8], len: &AtomicU8, src: &[u8]) { for (slot, &b) in dst.iter().zip(src.get(..n).unwrap_or(&[])) { slot.store(b, Ordering::Relaxed); } - len.store(n as u8, Ordering::Relaxed); + // `Release`, as in [`copy_tail`]. + len.store(n as u8, Ordering::Release); } /// A slot's bytes as text, in the caller's own buffer. fn read<'a>(src: &[AtomicU8], len: &AtomicU8, out: &'a mut [u8]) -> &'a str { - let n = (len.load(Ordering::Relaxed) as usize).min(out.len()).min(src.len()); + // `Acquire`: pairs with the copy's `Release`, so the bytes read are at + // least the ones that length was stored after. + let n = (len.load(Ordering::Acquire) as usize).min(out.len()).min(src.len()); for (byte, slot) in out.iter_mut().zip(src.iter()) { *byte = slot.load(Ordering::Relaxed); } @@ -286,7 +273,7 @@ pub fn last_words( raw(b"\n!!! "); raw(header.as_bytes()); raw(b" !!! (apic "); - serial::panic_raw_dec(u64::from(apic_id())); + serial::panic_raw_dec(u64::from(cpu::hardware_id())); if let Some(prev) = prev { raw(b", the cpu was already in "); raw(state_name(prev).as_bytes()); @@ -350,3 +337,95 @@ pub fn last_words( ), } } + +/// Time `/system/bin/logd` gets to durably write the panic report before halt; +/// a `Budget` (not a `Bound`) because expiry degrades gracefully instead of panicking. +const LOG_FILE_DRAIN: Budget = Budget::of( + Duration::from_millis(500), + "the report reaches the panel and not /log", +); + +// Read by tests/toyos.rs — keep in sync or its drift check fails. +const LOG_DRAIN_EXPIRED: &str = "the report did not reach /log"; + +/// Whether `/log` still owes this boot the report. +// True only before durable_ns passes `want` — logd publishes it after fsync returns, never before. +fn log_file_owed(want: u64) -> bool { + crate::log::user::durable_ns() < want +} + +/// Give `/system/bin/logd` a chance to put this report on the stick before the machine stops. +// The panic path never writes /log directly — every lock a write needs may already be held by the panicking thread itself. +fn wait_for_log_file() { + // Skip when serial exists: panic_flush already got the report off the box, and waiting here would only delay the pager. + if serial::has_console() { + return; + } + // INVARIANT: nothing below runs before both facts this wait rests on hold. + // The deadline is read off the calibrated clock, so an uncalibrated one + // makes it unreachable; and what the wait is owed by is `logd`, which only + // a released machine can run. On a boot that crashes before either — the + // one this whole path exists for on a machine with no serial port — waiting + // costs the seal and buys nothing, so it is skipped and not shortened. + if !crate::clock::calibrated() || !crate::arch::smp::is_ready() { + return; + } + // Sampled once — a sibling still logging on its way down must not be able to push this deadline out indefinitely. + let want = crate::log::read::newest_committed_at_ns(); + if !log_file_owed(want) { + return; + } + // Wake siblings first: one may be halted waiting for an interrupt with no timer armed to wake it otherwise. + crate::arch::irqchip::kick_all_but_self(); + let deadline = crate::clock::now() + LOG_FILE_DRAIN.duration(); + while log_file_owed(want) { + if crate::clock::now() >= deadline { + // /log has failed to answer, so fold this into the still-unpainted panel snapshot — the panel is the only channel left. + crate::log!( + "panic: {LOG_DRAIN_EXPIRED} in {}ns; the panel is the only copy", + LOG_FILE_DRAIN.nanos() + ); + crate::drivers::panic_console::refresh_capture(); + return; + } + core::hint::spin_loop(); + } +} + +/// Halt all CPUs: stop the others, flush pending log output, then hold this +/// machine's panel until a key retires the reboot bound or the bound returns it +/// to firmware. Every fatal path's one funnel. +// panic_flush bypasses the log-ring and serial locks — once the others are stopped a wedged holder never releases them, so taking them normally could deadlock. +pub fn halt_all_cpus() -> ! { + // Before the wait and the panel: from here this machine holds a report for + // whoever is in front of it, with interrupts masked and under a bound of its + // own — which to a hard-lockup sample or a deadline poll is + // indistinguishable from a wedge, and is the opposite of one. Both bounds, + // because a `WEDGED` record either of them sealed would replace the report + // this path exists to deliver. + crate::hardlockup::stand_down(); + crate::deadline::stand_down(); + wait_for_log_file(); + crate::arch::irqchip::stop_other_cpus(); + let bound = crate::panic_reboot::arm(true); + // Folded into the still-unpainted capture only where the panel is this + // boot's only account of itself, exactly as `wait_for_log_file`'s own line + // is: a refresh re-freezes the ring, and `screen_late_panic` reads the + // panel for a record written *after* `capture()` to prove the paint comes + // from the frozen snapshot. A machine with a console gets the arm line on it. + if !serial::has_console() { + crate::drivers::panic_console::refresh_capture(); + } + // Render before the flush: it can't fail the proven serial channel, and a serial line then proves the paint already finished. + let painted = crate::drivers::panic_console::render(); + // SAFETY: sound only once nothing else will run — the other CPUs are already stopped. + unsafe { serial::panic_flush(); } + // Must follow the flush — it's the deepest stack this path reaches. + crate::arch::trap::report_fault_stack(); + // page_forever runs strictly after the flush: it is an unbounded loop and may only run once the serial report is out. + // Only the CPU that painted watches the bound; the rest halt below, since two CPUs polling one keyboard would split every key. + if painted { + crate::drivers::panic_console::page_forever(bound); + } + cpu::halt(); +} diff --git a/kernel/src/panic_reboot.rs b/kernel/src/panic_reboot.rs index 3662dbcf795..30ea48abd79 100644 --- a/kernel/src/panic_reboot.rs +++ b/kernel/src/panic_reboot.rs @@ -4,10 +4,11 @@ //! how a person at the machine says the panel is being read — and nobody //! pressing one inside the bound means nobody is there to read it. //! -//! **The bound is carried in TSC cycles, not nanoseconds.** A panic may land -//! before `clock::init`, where the calibrated clock reads zero for every -//! interval; the cycle count comes from the frequency CPUID states instead, and -//! both phases then compare the same `rdtsc` against the same unit. A machine +//! **The bound is carried in counter ticks, not nanoseconds** (`cpu::counter`: +//! the TSC, the generic timer's count). A panic may land before `clock::init`, +//! where the calibrated clock reads zero for every interval; the tick count +//! comes from the frequency the CPU states instead (CPUID, `CNTFRQ_EL0`), and +//! both phases then compare the same counter against the same unit. A machine //! that states no frequency and has no calibrated clock cannot time anything, //! and the arm line says so instead of resetting on a guess. //! @@ -40,7 +41,7 @@ const FAST_BOUND: Budget = Budget::of( /// Whether a reboot is armed on this panic, and when. #[derive(Clone, Copy)] pub enum Bound { - /// Reset the machine at this `rdtsc` reading. + /// Reset the machine at this `cpu::counter` reading. At(u64), /// Hold the panel: somebody is reading it, or nothing here could time a /// wait, or this machine has no reset register to write. @@ -57,7 +58,7 @@ impl Bound { /// calls this, and it is the only place that decides the reset has come due. pub fn check(self) { if let Self::At(cycles) = self { - if cpu::rdtsc() >= cycles { + if cpu::counter() >= cycles { reboot_now(); } } @@ -72,14 +73,14 @@ impl Bound { #[derive(Clone, Copy)] enum Source { Calibrated, - Cpuid, + Stated, } impl Source { fn named(self) -> &'static str { match self { Source::Calibrated => "the calibrated clock", - Source::Cpuid => "the TSC frequency CPUID states", + Source::Stated => "the counter frequency the CPU states", } } } @@ -90,15 +91,15 @@ impl Source { const ARMED: &str = "panic: rebooting"; const HELD: &str = "panic: holding this panel"; -/// The bound in `rdtsc` cycles from now, and which clock said so. +/// The bound in counter ticks from now, and which clock said so. fn deadline(bound: Budget) -> Option<(u64, Source)> { if crate::clock::calibrated() { return Some((crate::clock::tsc_deadline(bound.nanos()), Source::Calibrated)); } - let hz = crate::clock::cpuid_tsc_hz()?; + let hz = crate::arch::cpu::stated_counter_hz()?; // Nanoseconds first, so a bound under a second is not rounded to nothing. let cycles = (u128::from(bound.nanos()) * u128::from(hz) / 1_000_000_000) as u64; - Some((cpu::rdtsc().saturating_add(cycles), Source::Cpuid)) + Some((cpu::counter().saturating_add(cycles), Source::Stated)) } /// Arm the reboot and say so in one line — the panel's last, because the panic @@ -146,7 +147,7 @@ pub fn arm(on_the_record: bool) -> Bound { (None, _) => { if on_the_record { alert!( - "{HELD}: this CPU states no TSC frequency and none is calibrated, so \ + "{HELD}: this CPU states no counter frequency and none is calibrated, so \ nothing here can time a wait" ); } else { diff --git a/kernel/src/pcidev/mod.rs b/kernel/src/pcidev/mod.rs index a024913a2de..498c99ffa3e 100644 --- a/kernel/src/pcidev/mod.rs +++ b/kernel/src/pcidev/mod.rs @@ -168,7 +168,7 @@ static IRQ: [Interrupt; MAX_FUNCTIONS] = [const { Interrupt::new() }; MAX_FUNCTI /// One address space per slot, made on that slot's first claim and kept. /// -/// Kept because a domain id is never given back (`iommu/vtd/domain.rs`), so a +/// Kept because a domain id is never given back (`arch/x86_64/vtd/domain.rs`), so a /// domain per claim would let a process spawn and die its way through every id /// the units report. [`release`] empties it, so the next holder attaches to one /// that maps nothing. diff --git a/kernel/src/preempt.rs b/kernel/src/preempt.rs index bee08a94f08..8b13fe1e9aa 100644 --- a/kernel/src/preempt.rs +++ b/kernel/src/preempt.rs @@ -1,13 +1,14 @@ //! Deferred preemption: per-CPU `preempt_count` and `need_resched` words in -//! `arch::percpu::PerCpu`, touched only through `arch::percpu::gs`. `enable()` +//! the architecture's per-CPU block, touched only through `arch::percpu`'s +//! accessors. `enable()` //! calls `scheduler::do_preempt()` when the count drops to zero and //! `need_resched` is set; every accessor no-ops before `PERCPU_READY`. use core::sync::atomic::Ordering; -use crate::arch::percpu::{gs, OFF_FAULT_STATE, OFF_NEED_RESCHED, OFF_PREEMPT_COUNT}; +use crate::arch::percpu; -// Before `percpu::init_bsp`, `gs:[N]` reads low identity-mapped memory — corruption, not a fault. +// Before `percpu::init_bsp` the per-CPU block is not there to read — corruption, not a fault. #[inline] fn percpu_ready() -> bool { crate::log::PERCPU_READY.load(Ordering::Relaxed) @@ -16,38 +17,38 @@ fn percpu_ready() -> bool { #[inline] pub fn count() -> u32 { if !percpu_ready() { return 0; } - gs::read_u32::() + percpu::preempt_count() } /// Sets the raw preempt-depth word; `Hw::switch` uses this to swap in the incoming context's saved depth. #[inline] pub fn set_count(v: u32) { if !percpu_ready() { return; } - gs::write_u32::(v); + percpu::set_preempt_count(v); } #[inline] pub fn need_resched() -> bool { if !percpu_ready() { return false; } - gs::read_u8::() != 0 + percpu::resched_owed() } #[inline] pub fn set_need_resched() { if !percpu_ready() { return; } - gs::write_u8_imm::(); + percpu::set_resched_owed(true); } #[inline] pub fn clear_need_resched() { if !percpu_ready() { return; } - gs::write_u8_imm::(); + percpu::set_resched_owed(false); } #[inline] pub fn disable() { if !percpu_ready() { return; } - gs::lock_inc_u32::(); + percpu::preempt_count_up(); } /// Drops the count without polling `need_resched`, for a caller about to reschedule anyway (see `sched::driver::pass_block`). @@ -55,13 +56,13 @@ pub fn disable() { pub fn enable_no_resched() { if !percpu_ready() { return; } // The request stays set; the imminent reschedule serves it. - gs::lock_dec_u32::(); + percpu::preempt_count_down(); } #[inline] pub fn enable() { if !percpu_ready() { return; } - gs::lock_dec_u32::(); + percpu::preempt_count_down(); // `do_preempt` clears `need_resched` itself; clearing here would drop a request racing a nested schedule. if count() == 0 && need_resched() && !faulting() { crate::scheduler::do_preempt(); @@ -72,5 +73,5 @@ pub fn enable() { #[inline] fn faulting() -> bool { if !percpu_ready() { return false; } - gs::read_u8::() != 0 + percpu::faulting() } diff --git a/kernel/src/process.rs b/kernel/src/process.rs index 569cdd4eeec..cbba1e3a80a 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -49,7 +49,7 @@ pub fn vma_map( size: u64, prot: Prot, ) -> Option<(UserAddr, u64)> { - pt.lock().alloc_and_map(phys, size, prot, CachePolicy::DeferToMtrr) + pt.lock().alloc_and_map(phys, size, prot, CachePolicy::Normal) } @@ -436,8 +436,8 @@ impl PageFaultTrace { pub struct ElfInfo { pub elf_alloc: Option, pub tls_modules: Vec, - pub tls_total_memsz: usize, - pub tls_max_align: usize, + /// Every static module's TLS, as a thread's block is laid out from it. + pub tls: toyos_elf::tls::Static, /// Next module ID to assign on dlopen (1-based, exe=1). pub next_tls_module_id: u64, /// Dynamically allocated TLS blocks for dlopen'd modules, keyed by (Tid, module_id). @@ -462,8 +462,7 @@ impl ElfInfo { Self { elf_alloc: None, tls_modules: Vec::new(), - tls_total_memsz: 0, - tls_max_align: 0, + tls: toyos_elf::tls::Static::empty(crate::loader::TLS_VARIANT), next_tls_module_id: 1, dynamic_tls_blocks: alloc::collections::BTreeMap::new(), loaded_libs: Vec::new(), @@ -830,16 +829,16 @@ pub fn spawn_thread(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> Op .expect("spawn_thread: the spawning thread runs in an address space"); (addr_space, Arc::clone(&proc.process_data)) }; - let (tls_modules, tls_total_memsz, tls_max_align) = { + let (tls_modules, tls) = { let data = process_data_arc.lock(); - (data.elf.tls_modules.clone(), data.elf.tls_total_memsz, data.elf.tls_max_align) + (data.elf.tls_modules.clone(), data.elf.tls) }; // Phase 2: allocate TLS outside any lock. An empty module set still gets a DTV+TCB block via `setup_tls(None, 0, ..)`. let (tls_alloc, fs_base) = if !tls_modules.is_empty() { - setup_combined_tls(&tls_modules, tls_total_memsz, tls_max_align)? + setup_combined_tls(&tls_modules, tls)? } else { - setup_tls(None, 0, tls_max_align)? + setup_tls(None, 0, tls.max_align())? }; let (tls_alloc, fs_base) = { let addr_space = &parent_addr_space; @@ -955,7 +954,7 @@ fn teardown_resources( crate::irq_census::log_census(); // After the irq lines: the tlb conservation check reads deliveries first, issues second. crate::arch::tlb::log_census(); - crate::arch::idt::unclaimed::log_vectors(); + crate::arch::trap::log_unclaimed(); ops::close_all(&mut data.handles); data.elf.elf_alloc.take(); @@ -1536,7 +1535,7 @@ pub fn dump_crash_diagnostics(fault_addr: u64, rip: u64) { } dump_region("rip", rip); - let fs_base = crate::arch::cpu::read_fs_base(); + let fs_base = crate::arch::cpu::thread_pointer(); if fs_base != 0 { log!(" FS base: {:#x}", fs_base); if let Some(self_ptr) = read_user(fs_base) { diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 39fd939a4c4..cf15f01cacc 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -192,7 +192,7 @@ pub fn stop(stage: Stage) -> Record { let armed = watch::arm(&PROGRESS, 0, WaitClass::Other) .expect("quiesce::stop: the caller holds no task to park"); // The kick is the timer vector, whose return to Ring 3 is the gate. - crate::arch::apic::kick_all_but_self(); + crate::arch::irqchip::kick_all_but_self(); let cpus = crate::arch::smp::cpu_count(); let began = crate::clock::now(); diff --git a/kernel/src/sched/driver.rs b/kernel/src/sched/driver.rs index ab0f3e35135..d0a3a47c5a4 100644 --- a/kernel/src/sched/driver.rs +++ b/kernel/src/sched/driver.rs @@ -7,7 +7,6 @@ use alloc::boxed::Box; use alloc::sync::Arc; use alloc::vec::Vec; -use core::arch::{asm, naked_asm}; use core::cell::UnsafeCell; use core::ptr; use core::sync::atomic::{AtomicBool, AtomicPtr, AtomicU64, Ordering}; @@ -52,7 +51,7 @@ unsafe impl PreemptGuard for IrqOff {} /// Run `f` with interrupts masked, holding [`IrqOff`] for exactly that region. pub fn irq_off(f: impl FnOnce(&IrqOff) -> R) -> R { - let _guard = crate::hw::IrqGuard::close(); + let _guard = crate::arch::IrqGuard::close(); f(&IrqOff(())) } @@ -371,7 +370,7 @@ pub fn init() { fn idle_ctx() -> KernelCtx { KernelCtx { rsp: 0, - cr3: crate::mm::paging::kernel_cr3(), + root: crate::mm::paging::kernel_root(), fs_base: 0, kernel_stack_top: 0, id: None, @@ -409,11 +408,11 @@ pub fn spawn(new: NewTask) -> (ThreadSched, CpuId) { // A kernel thread's is the kernel address space, the one every CPU // sits in between user threads — why `idle_ctx` names the same `cr3`. // Nothing is released at teardown: this `Arc` clones a leaked, permanent kernel mapping. - let cr3 = new.address_space.lock().cr3(); + let root = new.address_space.lock().root(); let kernel_stack_top = new.kernel_stack.ptr() as u64 + KERNEL_STACK_SIZE as u64; let ctx = KernelCtx { rsp: new.entry_rsp, - cr3, + root, fs_base: new.fs_base, kernel_stack_top, id: Some(new.id), @@ -490,12 +489,7 @@ fn env(preempt: &PreemptOff) -> Env<'_, crate::hw::KernelHw, PreemptOff> { pub fn pass(dispose: Dispose) { // The witness's negative control: sets DF one instruction before the reader that must refuse it. #[cfg(feature = "df-witness-mutate")] - // SAFETY: a build that exists to stage the defect, and the reader below - // panics before any `rep movs` can run. Nothing runs in between, so no - // string op ever executes with it set. - unsafe { - core::arch::asm!("std", options(nomem, nostack)) - }; + crate::arch::cpu::df_witness_mutate(); #[cfg(feature = "df-witness")] crate::arch::cpu::df_witness("a scheduler pass"); // Read before this pass's own level goes on top of it. @@ -510,7 +504,7 @@ pub fn pass(dispose: Dispose) { // posts is in the run queue by the time the pass chooses. crate::object::drain_zero_handles(); let now = HW.now(); - crate::drivers::watchdog::feed(now.0); + crate::arch::watchdog::feed(now.0); #[cfg(feature = "heap-sweep")] maybe_sweep(now); #[cfg(feature = "pass-spin")] @@ -643,7 +637,7 @@ fn execute(action: Action) { || crate::irq_ring::any_pending_self() || !with_cpu(|c| c.mailbox_is_empty()) // The i8042 verdict needs a pass to notice its deadline; a quiet machine after boot runs none otherwise. - || crate::drivers::i8042::verdict_due() + || crate::arch::keyboard_controller::verdict_due() // No log condition here: a log to write means a runnable process, covered above. A pending // root-hub port needs a pass too — no interrupt is coming. || crate::drivers::xhci::port_work_pending(); @@ -664,7 +658,7 @@ fn drain_irqs(entered: super::dump::Entered) { #[cfg(feature = "boot-actuators")] crate::heartbeat::note_pass(); crate::drivers::xhci::poll_if_pending(); - crate::drivers::i8042::service(); + crate::arch::keyboard_controller::service(); // Here, not at the keystroke: the keystroke's decoding driver's guard is done by this point. super::dump::serve_request(entered); // A CPU cannot read a sibling's `CpuSched`, so the dump reaches every CPU @@ -692,23 +686,10 @@ pub fn enter_idle_loop() -> ! { percpu::set_current_pid(None); // SAFETY: `set_kernel_stack` requires the caller be the CPU its GS base belongs to — true here, on that CPU, after its base was set. unsafe { percpu::set_kernel_stack(percpu::idle_stack_top()) }; - // SAFETY: `kernel_cr3` is the space this function's own code and stack already run in, so the write cannot unmap what executes it. - unsafe { crate::mm::paging::kernel_cr3().activate() }; - let sp = percpu::idle_stack_top(); - // SAFETY: nothing on the outgoing stack is live past this — the function returns `!`, and `sp` is this CPU's own idle stack top. - unsafe { - asm!( - "mov rsp, {sp}", - // Zeroes the frame chain, so a panic here can backtrace instead of walking off the top of this stack; - // `push` also leaves `rsp` where a function entry expects it. - "xor ebp, ebp", - "push rbp", - "jmp {func}", - sp = in(reg) sp, - func = in(reg) idle_loop as *const () as usize, - options(noreturn), - ); - } + // SAFETY: `kernel_root` is the space this function's own code and stack already run in, so the write cannot unmap what executes it. + unsafe { crate::mm::paging::kernel_root().activate() }; + // SAFETY: nothing on the outgoing stack is live past this — the function returns `!`, and the stack is this CPU's own idle stack. + unsafe { crate::arch::cpu::run_on_stack(percpu::idle_stack_top(), idle_loop) } } extern "C" fn idle_loop() -> ! { @@ -723,7 +704,7 @@ extern "C" fn idle_loop() -> ! { // while the one under observation spins on `syscall` from Ring 3. #[cfg(feature = "boot-actuators")] if crate::actuator::syscall_window_nmi() { - crate::nmi_gate::storm(); + crate::arch::syscall::window_storm(); } // Here, not from a syscall: the panic handler recovers, not paints, when a userland // thread is current, and the idle loop has none. @@ -878,9 +859,9 @@ pub fn for_each_parked(mut f: impl FnMut(ParkedInfo)) -> bool { /// Tail of the first switch into a fresh task. No lock to release, no outgoing task to park: only the /// preempt-count bracket's other half is owed. -pub extern "sysv64" fn trampoline_entry() { +pub extern "C" fn trampoline_entry() { crate::preempt::enable_no_resched(); - crate::arch::idt::kernel_exit_to_user_check(); + crate::arch::trap::kernel_exit_to_user_check(); } const STACK_CANARY: u64 = 0xDEAD_BEEF_CAFE_BABE; @@ -963,7 +944,7 @@ fn check_stack_ownership(payload: &KernelPayload) { let top = bottom + KERNEL_STACK_SIZE as u64; // SAFETY: a pass runs on the CPU whose GS base is its own `PerCpu`. let (kernel_rsp, rsp0) = unsafe { percpu::entry_stacks() }; - let rsp = crate::arch::cpu::read_rsp(); + let rsp = crate::arch::cpu::stack_pointer(); if kernel_rsp == top && rsp0 == top && rsp <= top && rsp > bottom { return; } @@ -1004,55 +985,3 @@ fn stack_depth(payload: &KernelPayload) { DEEPEST.fetch_max(used, Ordering::Relaxed); } -/// The outgoing half of [`context_switch`]. A macro, not inlined twice, so both builds share one instruction sequence. -macro_rules! switch_save { - () => { - "pushfq - push rbp - push rbx - push r12 - push r13 - push r14 - push r15 - mov [rdi], rsp - mov rsp, rsi" - }; -} - -/// The incoming half: the seven words a resumed context stands on, ending in `ret`. -macro_rules! switch_restore { - () => { - "pop r15 - pop r14 - pop r13 - pop r12 - pop rbx - pop rbp - popfq - ret" - }; -} - -/// Callee-saved register save/restore. -#[cfg(not(feature = "switch-witness"))] -#[unsafe(naked)] -pub(crate) unsafe extern "C" fn context_switch(old_rsp: *mut u64, new_rsp: u64) { - naked_asm!(switch_save!(), switch_restore!()); -} - -/// The same switch with [`crate::hw::switch_witness_verify`] between the stack move and the first `pop`; never fired. -/// -/// Placed after `mov rsp, rsi`, reading the incoming frame through the register the machine will use. Sound -/// to `call`: the return lands inside the incoming task's own stack, and every register `verify` may clobber -/// is caller-saved and already dead here. -#[cfg(feature = "switch-witness")] -#[unsafe(naked)] -pub(crate) unsafe extern "C" fn context_switch(old_rsp: *mut u64, new_rsp: u64) { - naked_asm!( - switch_save!(), - "mov rdi, rsp", - "call {verify}", - switch_restore!(), - verify = sym crate::hw::switch_witness_verify, - ); -} diff --git a/kernel/src/sched/dump.rs b/kernel/src/sched/dump.rs index a1fc465d68a..10cdc57dab1 100644 --- a/kernel/src/sched/dump.rs +++ b/kernel/src/sched/dump.rs @@ -17,7 +17,7 @@ use core::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use super::dump_request::{DumpRequest, Left}; -use crate::arch::{apic, percpu, smp}; +use crate::arch::{irqchip, percpu, smp}; use crate::sched::payload::{SCHED_BLOCKED, SCHED_READY, SCHED_RUNNING}; use crate::time::{Budget, Duration, Floor}; @@ -61,7 +61,7 @@ const NMI_BUDGET: Budget = Budget::of( static REQUEST: DumpRequest = DumpRequest::new(); static OWES: [AtomicBool; MAX_CPUS] = [const { AtomicBool::new(false) }; MAX_CPUS]; -/// NMI handshake: the handler (`arch/idt/nmi.rs`) may not allocate, log, or +/// NMI handshake: the handler (`arch/x86_64/idt/nmi.rs`) may not allocate, log, or /// lock, so it only stores and clears; the asking CPU reads. static NMI_OWES: [AtomicBool; MAX_CPUS] = [const { AtomicBool::new(false) }; MAX_CPUS]; static NMI_RIP: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; @@ -270,7 +270,7 @@ fn report(_proof: &UnderNothing) { // Every flag set before any kick, so an instant answer can't race its own flag. for cpu in 0..cpus { if cpu != me { - apic::kick_cpu(cpu as u32); + irqchip::kick_cpu(cpu as u32); } } @@ -321,7 +321,7 @@ fn probe_silent(asked: &[bool; MAX_CPUS], cpus: usize) { #[allow(clippy::needless_range_loop)] for cpu in 0..cpus { if asked[cpu] { - apic::send_nmi(cpu as u32); + irqchip::send_nmi(cpu as u32); } } @@ -387,7 +387,7 @@ pub(super) fn deaf_window() { // Not an `IrqGuard`: this must unconditionally set IF on exit, and // panic recovery may already have left IF clear. crate::arch::cpu::disable_interrupts(); - while crate::arch::cpu::rdtsc() < until { + while crate::arch::cpu::counter() < until { core::hint::spin_loop(); } crate::arch::cpu::enable_interrupts(); @@ -406,7 +406,7 @@ pub(super) fn deaf_window() { } // Driven from here, not idle-loop iterations: cpu0 may halt between them. STAGE.store(ASKED, Ordering::Release); - apic::kick_cpu(victim as u32); + irqchip::kick_cpu(victim as u32); let deadline = crate::clock::nanos_since_boot().saturating_add(ACK_BUDGET_NS); while STAGE.load(Ordering::Acquire) != DEAF { if crate::clock::nanos_since_boot() >= deadline { @@ -567,7 +567,7 @@ pub mod staged { } } -/// Where this CPU was, for the NMI probe. Called only from `arch/idt/nmi.rs`. +/// Where this CPU was, for the NMI probe. Called only from `arch/x86_64/idt/nmi.rs`. /// Stores unconditionally: reading the flag first would race the requester that owns it. pub fn note_nmi(rip: u64) { let me = percpu::cpu_id() as usize; diff --git a/kernel/src/sched/payload.rs b/kernel/src/sched/payload.rs index 56af5f810e1..7e1cad4fabd 100644 --- a/kernel/src/sched/payload.rs +++ b/kernel/src/sched/payload.rs @@ -13,7 +13,7 @@ use toyos_sched::task::{SchedPayload, TaskAccounting, TaskShared, WaitClass}; use toyos_sched::park::WaitTicket; use crate::watch::Watch; -use crate::mm::paging::Cr3; +use crate::mm::paging::Root; use crate::scheduler::OperationSlot; use crate::process::{OwnedAlloc, PageTables, ProcessAccounting, TaskId}; use crate::symbols::SymbolTable; @@ -44,7 +44,7 @@ pub type RawTicket = WaitTicket; pub struct KernelCtx { /// Saved kernel stack pointer, written by the `context_switch` asm. pub rsp: u64, - pub cr3: Cr3, + pub root: Root, pub fs_base: u64, pub kernel_stack_top: u64, /// `None` is this CPU's idle context. diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index a3104c51840..c66bbd3bfc4 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -651,7 +651,7 @@ pub(crate) fn reap_poisoned() { pub fn schedule_no_return() -> ! { if in_schedule_self() { crate::log!("schedule_no_return: panicked inside a pass, cannot rejoin"); - crate::arch::apic::halt_all_cpus(); + crate::panic::halt_all_cpus(); } if percpu::current_tid().is_none() { enter_idle_loop(); diff --git a/kernel/src/symbols.rs b/kernel/src/symbols.rs index 23a1296ab9b..f6756cf16d6 100644 --- a/kernel/src/symbols.rs +++ b/kernel/src/symbols.rs @@ -236,3 +236,26 @@ fn log_user(syms: &SymbolTable, addr: u64, resolved: Option<(&str, u64)>) -> boo false } } + +/// Walk the frame-pointer chain from `start_fp`, resolving each return +/// address. Every architecture this kernel builds for lays a frame record out +/// the same way under `-Cforce-frame-pointers=yes`: the caller's frame pointer, +/// then the return address one word up. +pub(crate) fn kernel_backtrace(start_fp: u64, max_frames: usize) { + let mut fp = start_fp; + for _ in 0..max_frames { + if fp == 0 || !fp.is_multiple_of(8) || !crate::mm::is_kernel_addr(fp) { break; } + // SAFETY: `fp` is checked non-zero, 8-aligned and a kernel address, so + // both reads land in the direct map, mapped for the life of the machine. + // + // Not `read_volatile` like the double-fault path's reads of memory + // another CPU may still be writing: this walks the faulting thread's + // own frame chain from its handler. + let saved_fp = unsafe { *(fp as *const u64) }; + // SAFETY: same as above, for the return address one word up. + let return_addr = unsafe { *((fp + 8) as *const u64) }; + if return_addr == 0 || !crate::mm::is_kernel_addr(return_addr) { break; } + resolve_kernel_return(return_addr); + fp = saved_fp; + } +} diff --git a/kernel/src/arch/syscall/debug.rs b/kernel/src/syscall/debug.rs similarity index 100% rename from kernel/src/arch/syscall/debug.rs rename to kernel/src/syscall/debug.rs diff --git a/kernel/src/arch/syscall/device.rs b/kernel/src/syscall/device.rs similarity index 100% rename from kernel/src/arch/syscall/device.rs rename to kernel/src/syscall/device.rs diff --git a/kernel/src/arch/syscall/dispatch.rs b/kernel/src/syscall/dispatch.rs similarity index 96% rename from kernel/src/arch/syscall/dispatch.rs rename to kernel/src/syscall/dispatch.rs index 73d3158ed8d..a7e013b11f1 100644 --- a/kernel/src/arch/syscall/dispatch.rs +++ b/kernel/src/syscall/dispatch.rs @@ -12,9 +12,6 @@ use crate::object::{ops, KObjectRef}; use crate::user_ptr::SyscallContext; use crate::UserAddr; use crate::{device, process}; -// The macro, not the module: only `SYS_DEBUG`'s arms below spell `log!` unqualified. -#[cfg(feature = "test-actuators")] -use crate::log; use toyos_abi::handle::{RawHandle, Rights}; #[cfg(feature = "test-actuators")] @@ -100,10 +97,10 @@ retired_syscalls! { 96 => "SYS_SET_RT_PRIORITY", } -pub(super) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> u64 { - // Placed first so `nmi_gate` counts the call whatever it turns out to be. +pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> u64 { + // Placed first so the architecture counts the call whatever it turns out to be. #[cfg(feature = "boot-actuators")] - crate::nmi_gate::note_syscall(); + crate::arch::syscall::note_entry(); let t0 = crate::clock::nanos_since_boot(); process::with_current_data(|data| { @@ -550,23 +547,10 @@ pub(super) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> } // Unlike every other action, this costs the machine, not just the caller's // process: one call is already a permanent halt. - DA::FATAL_HALT => { log!("{}", FATAL_HALT_NONCE); crate::arch::apic::halt_all_cpus(); } - // A real #DF, not simulated: pushing to a non-canonical rsp raises #SS, - // and delivering that needs another push to the same rsp — the #DF condition. - // Non-canonical rather than unmapped: on a bigger machine an unmapped - // address can fall inside the direct map and simply get written to. - // Only #DF has an IST, so every fault on the way there lands on this same unusable stack. + DA::FATAL_HALT => { log!("{}", FATAL_HALT_NONCE); crate::panic::halt_all_cpus(); } DA::DOUBLE_FAULT => { log!("SYS_DEBUG: provoking a double fault"); - // SAFETY: unsound by design like NULL_READ; the #DF this raises never returns here. - unsafe { - core::arch::asm!( - "mov rsp, {bad}", - "push 0", - bad = in(reg) 0x0000_8000_0000_0000u64, - options(noreturn), - ); - } + crate::arch::trap::provoke_double_fault() } // Both sides of MAX_HEAP_ALLOC plus the alignment corner: the page-aligned // case pads past one page and must error, not panic inside the allocator's lock. diff --git a/kernel/src/arch/syscall/fs.rs b/kernel/src/syscall/fs.rs similarity index 99% rename from kernel/src/arch/syscall/fs.rs rename to kernel/src/syscall/fs.rs index 50d92dd59bc..90af32bc9f1 100644 --- a/kernel/src/arch/syscall/fs.rs +++ b/kernel/src/syscall/fs.rs @@ -6,7 +6,7 @@ use crate::object::ops; use crate::user_ptr::UserBytesMut; -use crate::{log, process, vfs}; +use crate::{process, vfs}; use toyos_abi::syscall::*; diff --git a/kernel/src/arch/syscall/handles.rs b/kernel/src/syscall/handles.rs similarity index 100% rename from kernel/src/arch/syscall/handles.rs rename to kernel/src/syscall/handles.rs diff --git a/kernel/src/arch/syscall/io.rs b/kernel/src/syscall/io.rs similarity index 98% rename from kernel/src/arch/syscall/io.rs rename to kernel/src/syscall/io.rs index d21632a4e99..20560cc0cdc 100644 --- a/kernel/src/arch/syscall/io.rs +++ b/kernel/src/syscall/io.rs @@ -12,7 +12,6 @@ use crate::time::{Cadence, Deadline, Duration}; use crate::user_ptr::{UserBytes, UserBytesMut}; use crate::{device, pipe, process}; -use crate::arch::cpu; use toyos_abi::handle::{RawHandle, Rights}; use toyos_abi::syscall::*; use toyos_sched::task::WaitClass; @@ -279,13 +278,13 @@ pub(super) fn sys_write_nonblock(h: RawHandle, buf: &UserBytes) -> u64 { pub(super) fn sys_random(out: &mut UserBytesMut) -> u64 { let mut i = 0; while i + 8 <= out.len() { - let Some(drawn) = cpu::rdrand() else { return SyscallError::Io.to_u64() }; + let Some(drawn) = crate::arch::entropy::draw() else { return SyscallError::Io.to_u64() }; out.write_at(i, &drawn.to_ne_bytes()); i += 8; } let remaining = out.len() - i; if remaining > 0 { - let Some(drawn) = cpu::rdrand() else { return SyscallError::Io.to_u64() }; + let Some(drawn) = crate::arch::entropy::draw() else { return SyscallError::Io.to_u64() }; out.write_at(i, &drawn.to_ne_bytes()[..remaining]); } 0 diff --git a/kernel/src/arch/syscall/ipc.rs b/kernel/src/syscall/ipc.rs similarity index 100% rename from kernel/src/arch/syscall/ipc.rs rename to kernel/src/syscall/ipc.rs diff --git a/kernel/src/arch/syscall/machine.rs b/kernel/src/syscall/machine.rs similarity index 99% rename from kernel/src/arch/syscall/machine.rs rename to kernel/src/syscall/machine.rs index 17bbb6109c3..b19ec54515f 100644 --- a/kernel/src/arch/syscall/machine.rs +++ b/kernel/src/syscall/machine.rs @@ -88,7 +88,7 @@ fn quiesce(last: &str) -> Result<(), SyscallError> { crate::usb_gate::sweep_under_load(); } // First: what follows outlasts a feed cadence, and no pass runs to feed again. - crate::drivers::watchdog::disarm(); + crate::arch::watchdog::disarm(); // **Before the sync, because the sync is a claim about a machine.** A // process that issues a `write` after `sync_all` returns has dirty pages // nothing will flush. The log's holders run on until `wait_for_durable` diff --git a/kernel/src/arch/syscall/mod.rs b/kernel/src/syscall/mod.rs similarity index 71% rename from kernel/src/arch/syscall/mod.rs rename to kernel/src/syscall/mod.rs index e642901a1f1..c46f91864e2 100644 --- a/kernel/src/arch/syscall/mod.rs +++ b/kernel/src/syscall/mod.rs @@ -1,12 +1,12 @@ -//! The syscall ABI: the entry gate, [`dispatch`]'s argument-decode boundary, -//! and the per-subsystem handlers. +//! The syscall ABI: [`dispatch`]'s argument-decode boundary and the +//! per-subsystem handlers. The entry that reaches it is the architecture's +//! (`arch::syscall`). #[cfg(feature = "test-actuators")] mod debug; mod device; -mod dispatch; +pub(crate) mod dispatch; mod fs; -mod gate; mod handles; mod io; mod ipc; @@ -14,10 +14,6 @@ mod machine; mod proc; mod vm; -#[cfg(feature = "boot-actuators")] -pub(crate) use gate::{entry_extent, hold_spin}; -pub use gate::init; - use toyos_abi::handle::RawHandle; use toyos_abi::syscall::SyscallError; diff --git a/kernel/src/arch/syscall/proc.rs b/kernel/src/syscall/proc.rs similarity index 99% rename from kernel/src/arch/syscall/proc.rs rename to kernel/src/syscall/proc.rs index 14f3f1003b0..8f3d825ddcf 100644 --- a/kernel/src/arch/syscall/proc.rs +++ b/kernel/src/syscall/proc.rs @@ -61,7 +61,7 @@ pub(super) fn sys_process_wait(h: RawHandle, flags: u64) -> u64 { // First, so the answer is the bit and not the handle's own refusal — which // for an unheld handle is the kill policy, so a call carrying both returns // instead of ending the caller. `WNOHANG` is the whole of this word, - // hand-copied and unchecked (`arch/syscall/vm.rs`'s `MMAP_PROT_KNOWN`). + // hand-copied and unchecked (`syscall/vm.rs`'s `MMAP_PROT_KNOWN`). if flags & !WNOHANG != 0 { return SyscallError::InvalidArgument.to_u64(); } diff --git a/kernel/src/arch/syscall/vm.rs b/kernel/src/syscall/vm.rs similarity index 98% rename from kernel/src/arch/syscall/vm.rs rename to kernel/src/syscall/vm.rs index 43b49647d2a..67788d95ef0 100644 --- a/kernel/src/arch/syscall/vm.rs +++ b/kernel/src/syscall/vm.rs @@ -7,10 +7,11 @@ //! shoots down and waits, and a sibling thread can be spinning on that same //! lock with `IF` clear. -use crate::mm::paging::{CachePolicy, Occupancy, Prot}; +use crate::mm::paging::{CachePolicy, Prot}; +use crate::vma::Occupancy; use crate::user_ptr::UserBytesMut; use crate::UserAddr; -use crate::{log, process, vfs}; +use crate::{process, vfs}; use toyos_abi::syscall::*; @@ -122,7 +123,7 @@ pub(super) fn sys_mmap(req_addr: u64, size: u64, prot: MmapProt, flags: MmapFlag pages.phys(), aligned as u64, mapping_prot, - CachePolicy::DeferToMtrr, + CachePolicy::Normal, ); } data.mmap_regions.push(process::MmapRegion { @@ -146,7 +147,7 @@ pub(super) fn sys_mmap(req_addr: u64, size: u64, prot: MmapProt, flags: MmapFlag let pt = process::current_address_space(); let vaddr = process::with_process_data(|data| { let placed = match &pages { - Some(pages) => pt.lock().alloc_and_map(pages.phys(), aligned as u64, mapping_prot, CachePolicy::DeferToMtrr).map(|(v, _)| v), + Some(pages) => pt.lock().alloc_and_map(pages.phys(), aligned as u64, mapping_prot, CachePolicy::Normal).map(|(v, _)| v), None => pt.lock().alloc_region(aligned as u64, crate::vma::RegionKind::Mapped), }; let Some(vaddr) = placed else { return Err(()) }; @@ -282,12 +283,12 @@ pub(super) fn sys_dlopen(ctx: &crate::user_ptr::SyscallContext, path: &str, init let data = data_arc.lock(); crate::elf::resolve_dlopen_relocs(&lib, &data.elf.loaded_libs); - if data.elf.tls_total_memsz > 0 { + if data.elf.tls.total_memsz() > 0 { let tls_info = crate::elf::TlsModuleInfo { libs: &data.elf.loaded_libs, modules: &data.elf.tls_modules, }; - crate::elf::apply_tpoff_relocs(&lib, 0, data.elf.tls_total_memsz, &tls_info); + crate::elf::apply_tpoff_relocs(&lib, 0, data.elf.tls, &tls_info); } // init_info layout: [init_array_vaddr, init_array_count], vaddr rebased to user_base. diff --git a/kernel/src/usb_gate.rs b/kernel/src/usb_gate.rs index 51717872174..159da5dc04a 100644 --- a/kernel/src/usb_gate.rs +++ b/kernel/src/usb_gate.rs @@ -458,7 +458,7 @@ pub fn sweep_under_load() { // than about this function. let interrupts_were_on = crate::arch::cpu::interrupts_enabled(); crate::preempt::disable(); - crate::arch::apic::arm_within(toyos_sched::fair::QUANTUM_NS); + crate::arch::irqchip::arm_within(toyos_sched::fair::QUANTUM_NS); crate::arch::cpu::enable_interrupts(); let mut buf = vec![0u8; WEDGE_CHUNK as usize * BLOCK]; let mut at = first; diff --git a/kernel/src/vma.rs b/kernel/src/vma.rs index 5cf35bd54f3..eb617ca8374 100644 --- a/kernel/src/vma.rs +++ b/kernel/src/vma.rs @@ -45,3 +45,15 @@ pub struct Region { /// For the demand-paged kinds, what a fault in this region installs. pub kind: RegionKind, } + +/// How a range meets the regions an address space registers. Needed because a +/// *placed* mapping (`sys_mmap`'s FIXED arm) skips `find_gap`'s implicit check. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Occupancy { + /// Nothing is registered over any part of it. + Free, + /// One region covers it end for end, and that region is all it runs into. + Whole, + /// Part of a region, several regions, or one that merely starts here. + Partial, +} diff --git a/licenses/Apache-2.0-OpenSSL.txt b/licenses/Apache-2.0-OpenSSL.txt new file mode 100644 index 00000000000..49cc83d2ee2 --- /dev/null +++ b/licenses/Apache-2.0-OpenSSL.txt @@ -0,0 +1,177 @@ + + Apache License + Version 2.0, January 2004 + https://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS diff --git a/rust b/rust index 80ea645f83b..c34ecdf0ab6 160000 --- a/rust +++ b/rust @@ -1 +1 @@ -Subproject commit 80ea645f83bf528b50ccac4f23235dc61dc7ea6b +Subproject commit c34ecdf0ab6114fd189ee3c683c896659956f3ca diff --git a/src/CLAUDE.md b/src/CLAUDE.md index d831ab24312..7d61612a283 100644 --- a/src/CLAUDE.md +++ b/src/CLAUDE.md @@ -20,7 +20,7 @@ Loads when you read a file under `src/` — the root cargo project, package name ## The host's locks and slots - **Sysroots are content-addressed** (`src/sysroot.rs`): one per key — the identity (`src/identity.rs`, so a comment is no change) of `toyos-abi/src`, `toyos/src`, `userland/libc/src` and their manifests, the std fork's `library/` and `src/bootstrap/`, and the compiler — at `rust/build/sysroots//`, made by whichever worktree first needs it and never written again. Every build compiles against its own key's, so two worktrees with different ABIs never refuse or wait for each other; the only shared step is the primary's compiler, which a sysroot build reads under the global lock in shared mode. A new key costs one std build of the three guest targets; `--worktree remove` sweeps the keys no worktree records. -- **The std fork is built per worktree, and nothing but the primary's own sync moves the primary's `rust/`.** A linked worktree's `rust/` becomes, on its first build, a git worktree of the primary's fork repository at the commit its tree pins — that is where the fork is edited, committed and pinned. A fork commit whose `compiler/` is not the one the primary's compiler was built from is refused by name: a compiler change lands, and the primary's sync and next build rebuild it. If that checkout later falls behind the commit its tree pins (a merge moved the pin), the build moves the checkout to it itself, fetching from the primary's repository first if it holds the commit, unless the checkout has local changes, which it refuses to move out from under. +- **The std fork is built per worktree, and nothing but the primary's own sync moves the primary's `rust/`.** A linked worktree's `rust/` becomes, on its first build, a git worktree of the primary's fork repository at the commit its tree pins — that is where the fork is edited, committed and pinned. A worktree whose fork `compiler/` differs from the one the primary built builds its own compiler, keyed by that source and placed beside the primary's without touching it. If that checkout later falls behind the commit its tree pins (a merge moved the pin), the build moves the checkout to it itself, fetching from the primary's repository first if it holds the commit, unless the checkout has local changes, which it refuses to move out from under. - `src/buildlock.rs` serialises the stateful phases in two scopes: `Global` (the primary's compiler and the rustup link — one directory in `.git/`, shared by every worktree) and `Worktree` (the crate-target cleans, and this worktree's std build). Only `./x.py` typed by hand in `rust/` escapes it. - **Never kill a build that has taken the global lock** — the kill removes the shell wrappers, not the bootstrap, which inherits the file descriptor and runs on regardless; a toolchain rebuild interrupted or unobserved this way can leave `stage2/bin` without a `cargo`. - **The host hands out guest slots and build slots** — `buildlock::guest_slot` (twelve across every worktree, one per task) and `buildlock::build_slot` (four), separate counts so a suite holding every guest slot can still compile. **The order is a constraint at every acquirer**: host slot → a sysroot key's lock → build lock → artifact. Every blocking lock names its holders and repeats itself every 30 s — a queue is never silence. `cargo test --test toyos-build -- --host-slots N --host-builds N` overrides, 0 turns either off. diff --git a/src/arch.rs b/src/arch.rs new file mode 100644 index 00000000000..72e4ac6df5a --- /dev/null +++ b/src/arch.rs @@ -0,0 +1,239 @@ +//! The machines ToyOS runs on. +//! +//! Every target triple, QEMU binary, firmware image, guest CPU and accelerator +//! the build system and the harness name is a function of one [`Arch`]. Neither +//! architecture is the reference: a new question about a machine is a new +//! method here that both variants answer, or it does not compile. + +use std::path::Path; + +/// One instruction-set architecture ToyOS builds for and boots. +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] +pub enum Arch { + X86_64, + Aarch64, +} + +/// How a guest's CPU is provided. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Accel { + /// Linux's hypervisor, on a host of the guest's own architecture. + Kvm, + /// macOS's Hypervisor.framework, on a host of the guest's own architecture. + Hvf, + /// QEMU's emulator: every other host. + Tcg, +} + +impl Accel { + /// The name `-accel` takes. + pub const fn name(self) -> &'static str { + match self { + Accel::Kvm => "kvm", + Accel::Hvf => "hvf", + Accel::Tcg => "tcg", + } + } + + /// Whether the guest runs on the host's own CPU rather than an emulated one. + pub const fn is_hardware(self) -> bool { + !matches!(self, Accel::Tcg) + } +} + +impl Arch { + pub const ALL: [Arch; 2] = [Arch::X86_64, Arch::Aarch64]; + + /// The architecture this build system runs on, when ToyOS runs there too. + pub const HOST: Option = if cfg!(target_arch = "x86_64") { + Some(Arch::X86_64) + } else if cfg!(target_arch = "aarch64") { + Some(Arch::Aarch64) + } else { + None + }; + + /// The name a command line spells it with, and Rust's `target_arch`. + pub const fn name(self) -> &'static str { + match self { + Arch::X86_64 => "x86_64", + Arch::Aarch64 => "aarch64", + } + } + + pub fn parse(name: &str) -> Result { + Arch::ALL.into_iter().find(|arch| arch.name() == name).ok_or_else(|| { + let known: Vec<&str> = Arch::ALL.iter().map(|a| a.name()).collect(); + format!("{name:?} is not an architecture ToyOS builds for; it builds for {}", known.join(" and ")) + }) + } + + /// ToyOS userland's target: the rust fork's. + pub const fn userland(self) -> &'static str { + match self { + Arch::X86_64 => "x86_64-unknown-toyos", + Arch::Aarch64 => "aarch64-unknown-toyos", + } + } + + /// The kernel's bare-metal target: one without hardware float, so kernel + /// code never touches the FP/SIMD registers that are the user's state. + pub const fn kernel(self) -> &'static str { + match self { + Arch::X86_64 => "x86_64-unknown-none", + Arch::Aarch64 => "aarch64-unknown-none-softfloat", + } + } + + /// The UEFI loader's target. + pub const fn loader(self) -> &'static str { + match self { + Arch::X86_64 => "x86_64-unknown-uefi", + Arch::Aarch64 => "aarch64-unknown-uefi", + } + } + + /// Whether this architecture's guest binaries link through toyos-ld, which + /// is frozen, rather than the rust-lld its targets name: x86-64's do until + /// its own switch to rust-lld. + pub const fn links_through_toyos_ld(self) -> bool { + match self { + Arch::X86_64 => true, + Arch::Aarch64 => false, + } + } + + /// Where on the ESP firmware looks for a removable medium's loader: UEFI + /// 2.11 §3.5.1.1 names one file per architecture. + pub const fn removable_loader(self) -> &'static str { + match self { + Arch::X86_64 => "EFI/BOOT/BOOTx64.EFI", + Arch::Aarch64 => "EFI/BOOT/BOOTAA64.EFI", + } + } + + /// The QEMU that emulates this machine. + pub const fn qemu(self) -> &'static str { + match self { + Arch::X86_64 => "qemu-system-x86_64", + Arch::Aarch64 => "qemu-system-aarch64", + } + } + + /// The firmware's code and variable-store images, relative to the + /// repository root, pinned and hashed in `NOTICE`. + pub const fn firmware(self) -> (&'static str, &'static str) { + match self { + Arch::X86_64 => ("ovmf/OVMF_CODE-pure-efi.fd", "ovmf/OVMF_VARS-pure-efi.fd"), + Arch::Aarch64 => ("aavmf/AAVMF_CODE.fd", "aavmf/AAVMF_VARS.fd"), + } + } + + /// The two `-drive` values that give a guest its firmware, from the + /// repository at `root`. The store is never written back: OVMF boots from a + /// read-only one, and AAVMF's `DEBUG` build asserts on one, so it writes a + /// snapshot QEMU discards. + pub fn pflash(self, root: &Path) -> [String; 2] { + let (code, vars) = self.firmware(); + let store = match self { + Arch::X86_64 => "readonly=on", + Arch::Aarch64 => "snapshot=on", + }; + [ + format!("if=pflash,format=raw,unit=0,file={},readonly=on", root.join(code).display()), + format!("if=pflash,format=raw,unit=1,file={},{store}", root.join(vars).display()), + ] + } + + /// QEMU's `-boot` for this machine's firmware, if it needs one. AAVMF + /// waits its platform boot timeout, five seconds, for a key before it boots + /// unless QEMU hands it a menu wait through fw_cfg, which only `menu=on` + /// does; `splash-time=0` makes that wait zero. + pub const fn boot(self) -> Option<&'static str> { + match self { + Arch::X86_64 => None, + Arch::Aarch64 => Some("menu=on,splash-time=0"), + } + } + + /// How this host provides a guest of this architecture: its own + /// hypervisor when the host is the same architecture and will open it, + /// and emulation otherwise. + /// + /// **Presence is not permission**, and `Path::exists` cannot tell the two + /// apart. A GitHub runner ships `/dev/kvm` as `crw-rw---- root:kvm` with the + /// build user outside the group, so a check on existence puts `-accel kvm` + /// on every boot and every boot dies on `failed to initialize kvm: + /// Permission denied` — a whole suite red for a reason no test names. + /// Opening it is the question QEMU is about to ask. + pub fn accel(self) -> Accel { + if Arch::HOST != Some(self) { + return Accel::Tcg; + } + if cfg!(target_os = "macos") { + return Accel::Hvf; + } + let opens = cfg!(target_os = "linux") + && std::fs::OpenOptions::new().read(true).write(true).open(Path::new("/dev/kvm")).is_ok(); + if opens { + Accel::Kvm + } else { + Accel::Tcg + } + } + + /// The CPU every guest of this architecture gets under `accel`. + /// + /// **One declaration, read by `cargo run` and by the harness both**, because + /// the two drifted: the harness gained `+smep` and the interactive path did + /// not, so the machine an owner looked at differed from the machine the + /// suite judged in exactly the dimension the suite had been changed for. + /// The emulated CPU carries the same features off a base model, because a + /// TCG guest that withholds one is a feature this tree stops exercising. + pub const fn cpu(self, accel: Accel) -> &'static str { + match (self, accel.is_hardware()) { + (Arch::X86_64, true) => "host,+rdrand,+smap,+fsgsbase,+x2apic,+smep", + (Arch::X86_64, false) => "qemu64,+rdrand,+smap,+fsgsbase,+x2apic,+smep", + (Arch::Aarch64, true) => "host", + (Arch::Aarch64, false) => "max", + } + } + + /// The machine's three targets: userland, kernel, loader. + pub const fn targets(self) -> [&'static str; 3] { + [self.userland(), self.kernel(), self.loader()] + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn every_name_parses_back_to_its_arch_and_nothing_else_does() { + for arch in Arch::ALL { + assert_eq!(Arch::parse(arch.name()), Ok(arch)); + } + let refusal = Arch::parse("riscv64").unwrap_err(); + assert!(refusal.contains("x86_64") && refusal.contains("aarch64"), "{refusal}"); + } + + #[test] + fn every_target_names_its_own_architecture() { + for arch in Arch::ALL { + for triple in arch.targets() { + assert!(triple.starts_with(&format!("{}-", arch.name())), "{triple}"); + } + assert!(arch.qemu().ends_with(arch.name())); + } + } + + #[test] + fn a_guest_of_another_architecture_is_emulated() { + for arch in Arch::ALL { + if Arch::HOST != Some(arch) { + assert_eq!(arch.accel(), Accel::Tcg, "{arch:?}"); + } + } + } +} diff --git a/src/bootlog.rs b/src/bootlog.rs index b29a0cf4417..069d6c1c167 100644 --- a/src/bootlog.rs +++ b/src/bootlog.rs @@ -11,7 +11,7 @@ use std::fmt; /// The word the kernel writes as it hands the machine back to the firmware, -/// in `kernel/src/arch/syscall/machine.rs`'s `quiesce`. +/// in `kernel/src/syscall/machine.rs`'s `quiesce`. pub const REBOOTING: &str = "Rebooting."; /// What `userland/test-runner` says when its job list runs past @@ -36,7 +36,7 @@ pub const WEDGE_STAGED: &str = "wedge: staged, and only the boot deadline ends t /// also in `kernel/src/deadline.rs`. /// /// **The one line that measures that control's own claim.** It arrives through -/// the shutdown syscall, and `arch::syscall::gate` masks `IF` for the whole of a +/// the shutdown syscall, and `arch::syscall` masks `IF` for the whole of a /// syscall — so a wedge that inherited its state leaves exactly one CPU per boot /// taking no interrupt at all, which is not a wedge but a hard lockup. A boot on /// which no CPU says this is a boot whose wedge never reached the CPU that asked @@ -254,7 +254,7 @@ pub fn lines_of(log: &str, name: &str) -> String { .collect() } -/// The AP bring-up record, in `kernel/src/arch/smp.rs`. A reader asks for the +/// The AP bring-up record, in `kernel/src/arch/x86_64/smp.rs`. A reader asks for the /// trailing ` online` as a separate word: the same head carries the failure. pub const AP_BRINGUP: &str = "SMP: AP cpu"; @@ -564,7 +564,7 @@ mod tests { let root = std::path::Path::new(env!("CARGO_MANIFEST_DIR")); for (file, needle) in [ ("kernel/src/process.rs", format!("log!(\"{EXIT}{{name}} pid=")), - ("kernel/src/arch/smp.rs", format!("log!(\"{AP_BRINGUP}")), + ("kernel/src/arch/x86_64/smp.rs", format!("log!(\"{AP_BRINGUP}")), ("kernel/src/process.rs", format!("THREAD_NAME_LEN: usize = {NAME_LEN}")), ("kernel/src/deadline.rs", format!("EXPIRED: &str = \"{DEADLINE_EXPIRED}\"")), ("kernel/src/deadline.rs", format!("WEDGE_STAGED: &str = \"{WEDGE_STAGED}\"")), diff --git a/src/build.rs b/src/build.rs index 9050f94db48..98a2439b442 100644 --- a/src/build.rs +++ b/src/build.rs @@ -10,6 +10,7 @@ use std::time::{Duration, Instant, UNIX_EPOCH}; use serde::Deserialize; +use crate::arch::Arch; use crate::assets; use crate::buildlock; use crate::flags; @@ -232,8 +233,8 @@ enum Clean { /// Crates with explicit paths (toyos-ld, toyos-cc) also have host builds /// that must survive: the host toyos-ld *is* the cross linker. Both are /// host-workspace members, so the directory this empties is the - /// workspace's `target/x86_64-unknown-toyos` — the guest halves of the two, - /// and nothing the host builds. + /// workspace's `target/` for every architecture — the + /// guest halves of the two, and nothing the host builds. ToyosOnly, } @@ -266,10 +267,12 @@ fn clean(root: &Path, crate_dir: &Path, kind: Clean, fingerprint: &str) { .status(); } Clean::ToyosOnly => { - let toyos_dir = target.join("x86_64-unknown-toyos"); - if toyos_dir.exists() { - eprintln!("external deps changed: cleaning {}", toyos_dir.display()); - fs::remove_dir_all(&toyos_dir).ok(); + for arch in Arch::ALL { + let toyos_dir = target.join(arch.userland()); + if toyos_dir.exists() { + eprintln!("external deps changed: cleaning {}", toyos_dir.display()); + fs::remove_dir_all(&toyos_dir).ok(); + } } } } @@ -426,6 +429,9 @@ pub const PROFILE: &str = "toyos"; #[derive(Clone)] struct GuestEnv { toolchain: PathBuf, + /// Whether that sysroot's compiler is the primary's, the one the hosted + /// rustc is built from (`src/compiler.rs`). + primary_compiler: bool, path: String, /// The public key the loader and `/system/bin/update` embed /// (`signing::KEY_ENV`): every guest build carries it, so no crate that @@ -439,6 +445,7 @@ impl GuestEnv { fn new(root: &Path, sysroot: &crate::sysroot::Sysroot) -> Self { Self { toolchain: sysroot.dir.clone(), + primary_compiler: sysroot.primary_compiler, path: toolchain::path_with_toyos_ld(root), image_key: crate::signing::key().public_hex(), floor_scope: crate::signing::key().floor_scope().word(), @@ -528,8 +535,14 @@ fn cargo_build( /// empty request; the harness passes `BootOptions::kernel_features` joined. /// Nothing between them may add a name — which is what `qemu::fold_inert` used /// to do to every boot in the suite. -fn kernel_key(features: &str) -> u64 { - key_hash(&[PROFILE, features]) +fn kernel_key(arch: Arch, features: &str) -> u64 { + key_hash(&[PROFILE, arch.name(), features]) +} + +/// The loader's build key: its profile, its architecture, and the key and +/// floor scope it embeds. +fn loader_key(arch: Arch, image_key: &str, floor_scope: &str) -> u64 { + key_hash(&[PROFILE, arch.name(), image_key, floor_scope]) } fn key_hash(parts: &[&str]) -> u64 { @@ -562,41 +575,55 @@ const OVERFLOW_CHECK_MARKER: &[u8] = b"attempt to add with overflow"; /// Refuse to build a kernel whose target has hardware float. /// -/// `arch::entry`'s bracket saves the user machine state at the ring transition -/// and nowhere else, which is sound only because kernel code cannot disturb it: -/// the FPU may be left dirty for a whole Ring 0 excursion because nothing in -/// Ring 0 reads or writes it. That rests on one line of the target spec — +/// The kernel's entry saves the user machine state at the ring transition and +/// nowhere else, which is sound only because kernel code cannot disturb it: +/// the FP/SIMD registers may be left dirty for a whole kernel excursion because +/// nothing in the kernel reads or writes them. That rests on the target spec — /// `RustcAbi::Softfloat` and `+soft-float` in -/// `rust/compiler/rustc_target/src/spec/targets/x86_64_unknown_none.rs` — and an -/// edit turning it off would make every bracket in the kernel insufficient -/// without changing a byte of `kernel/`. +/// `rust/compiler/rustc_target/src/spec/targets/x86_64_unknown_none.rs`, and +/// `aarch64_unknown_none_softfloat.rs`'s `-fp-armv8,-neon` — and an edit +/// turning it off would make every bracket in the kernel insufficient without +/// changing a byte of `kernel/`. /// -/// Asked of the compiler rather than of the manifest, and once per process: it -/// is a property of the toolchain rather than of any one image. -fn assert_kernel_is_softfloat(env: &GuestEnv) { - static CHECKED: std::sync::OnceLock<()> = std::sync::OnceLock::new(); - CHECKED.get_or_init(|| { - let out = Command::new("rustc") - .args(["--print", "cfg", "--target", "x86_64-unknown-none"]) - .env("RUSTUP_TOOLCHAIN", &env.toolchain) - .env("PATH", &env.path) - .env_remove("RUSTFLAGS") - .env_remove("RUSTC") - .output() - .expect("rustc --print cfg failed to launch"); - assert!(out.status.success(), "rustc --print cfg failed for the kernel target"); - let cfg = String::from_utf8_lossy(&out.stdout); - assert!( - cfg.lines().any(|l| l == r#"target_feature="x87""#), - "the kernel target no longer reports x87, so `arch::fpu`'s FXSAVE64 image is not \ - the state this machine has:\n{cfg}" - ); - assert!( - !cfg.lines().any(|l| l == r#"target_feature="sse""#), - "the kernel target has hardware float, so kernel code may now clobber the user \ - machine state between `arch::entry`'s save and its restore:\n{cfg}" - ); - }); +/// Asked of the compiler rather than of the manifest, and once per process and +/// architecture: it is a property of the toolchain rather than of any one image. +fn assert_kernel_is_softfloat(env: &GuestEnv, arch: Arch) { + static CHECKED: std::sync::Mutex> = std::sync::Mutex::new(BTreeSet::new()); + let mut checked = CHECKED.lock().expect("a softfloat check panicked"); + if checked.contains(&arch) { + return; + } + let out = Command::new("rustc") + .args(["--print", "cfg", "--target", arch.kernel()]) + .env("RUSTUP_TOOLCHAIN", &env.toolchain) + .env("PATH", &env.path) + .env_remove("RUSTFLAGS") + .env_remove("RUSTC") + .output() + .expect("rustc --print cfg failed to launch"); + assert!(out.status.success(), "rustc --print cfg failed for the kernel target"); + let cfg = String::from_utf8_lossy(&out.stdout); + let has = |feature: &str| cfg.lines().any(|l| l == format!(r#"target_feature="{feature}""#)); + match arch { + Arch::X86_64 => { + assert!( + has("x87"), + "the kernel target no longer reports x87, so `arch::fpu`'s FXSAVE64 image is \ + not the state this machine has:\n{cfg}" + ); + assert!( + !has("sse"), + "the kernel target has hardware float, so kernel code may now clobber the user \ + machine state between the entry's save and its restore:\n{cfg}" + ); + } + Arch::Aarch64 => assert!( + !has("neon") && !has("fp-armv8"), + "the kernel target has FP/SIMD, so kernel code may now clobber the user machine \ + state between the entry's save and its restore:\n{cfg}" + ), + } + checked.insert(arch); } /// Whether `haystack` contains `needle` as a contiguous subslice. @@ -723,72 +750,25 @@ fn build_and_assemble( env: &GuestEnv, extra_files: &[(String, Vec)], quiet: bool, + arch: Arch, ) -> Vec { - let userland_dir = root.join("userland"); - - let programs: Vec = config_crates(root, config) - .into_iter() - .filter(|c| matches!(c.built, Built::Member | Built::Standalone)) - .collect(); - for c in &programs { - assert!( - c.dir.join("Cargo.toml").exists(), - "Program '{}' crate not found at {}", - c.name, - c.dir.display() - ); - } - let workspace_packages: Vec<&str> = - programs.iter().filter(|c| c.built == Built::Member).map(|c| c.name.as_str()).collect(); - let mut root_files: Vec<(String, Vec)> = Vec::new(); - let ws_target = userland_dir.join(format!("target/x86_64-unknown-toyos/{PROFILE}")); - - // Build and read under one hold, exactly as `build_toyos_bins` does and for - // the same reason: a program's path is keyed on (crate, target, profile) - // alone, so every config in this run writes and reads the same - // `userland/target/.../toybox`. Cargo's own lock orders the two *builds* and - // says nothing about a read between them — `ioapic_topology` died on - // `Failed to read binary for toybox` while another worker's config was - // relinking it, and was green the moment it was re-run alone. - { - let _artifact = buildlock::artifact(root); - if !workspace_packages.is_empty() { - let mut extra: Vec<&str> = Vec::new(); - for pkg in &workspace_packages { - extra.push("-p"); - extra.push(pkg); - } - cargo_build( - &userland_dir, - "x86_64-unknown-toyos", - &extra, - env, - &[], - quiet, - ); - } - - for c in programs.iter().filter(|c| c.built == Built::Standalone) { - cargo_build(&c.dir, "x86_64-unknown-toyos", &c.features.args(), env, &[], quiet); - } - - for c in &programs { - let name = &c.name; - let binary = match c.built { - Built::Member => ws_target.join(name), - Built::Standalone => hostws::target_dir(root, &c.dir) - .join(format!("x86_64-unknown-toyos/{PROFILE}/{name}")), - Built::Kernel | Built::Bootloader => unreachable!("filtered out above"), - }; - let data = - fs::read(&binary).unwrap_or_else(|_| panic!("Failed to read binary for {name}")); - root_files.push((format!("bin/{name}"), data)); - } - root_files.push((toyos_manifest::PATH.to_string(), render_manifest(config))); - } + build_programs(root, config, env, quiet, arch, &mut root_files); + root_files.push((toyos_manifest::PATH.to_string(), render_manifest(config))); if config.hosted_rustc { + assert!( + arch == toolchain::HOSTED_ARCH, + "hosted-rustc is built to run on {}, and this image is for {}", + toolchain::HOSTED_ARCH.name(), + arch.name() + ); + assert!( + env.primary_compiler, + "hosted-rustc ships the primary checkout's hosted compiler, and this worktree builds with \ + a compiler of its own (src/compiler.rs): the image would carry a rustc that is not the \ + one its programs were built with" + ); collect_hosted_rustc(root, &env.toolchain, &mut root_files); } @@ -824,6 +804,100 @@ fn build_and_assemble( image::create_root_image(&root_files, &symlinks, quiet) } +/// The programs an architecture cannot build yet, each with why. Such a program +/// is left off that architecture's ROOT, said by name at build time; its row +/// goes when its reason does. +const NOT_YET_BUILT: &[(Arch, &str, &str)] = &[ + (Arch::Aarch64, "calc", TOOLKIT_FORKS), + (Arch::Aarch64, "snake", TOOLKIT_FORKS), + ( + Arch::Aarch64, + "doom", + "its C is compiled by toyos-cc, which no AArch64 build has run, and softbuffer's toyos \ + fork stops it first (issues/build/the-toolkit-forks-resolve-an-x86-only-toyos-window.md)", + ), +]; + +const TOOLKIT_FORKS: &str = "softbuffer's and winit's toyos forks resolve the published \ + toyos-window 0.2.0, whose framebuffer is x86-64 only \ + (issues/build/the-toolkit-forks-resolve-an-x86-only-toyos-window.md)"; + +/// Why `arch`'s userland leaves `program` out, if it does. +fn not_built_for(arch: Arch, program: &str) -> Option<&'static str> { + NOT_YET_BUILT.iter().find(|(a, name, _)| *a == arch && *name == program).map(|(_, _, why)| *why) +} + +/// Build `config`'s programs and init for `arch`, and add each to `root_files`. +fn build_programs( + root: &Path, + config: &SystemConfig, + env: &GuestEnv, + quiet: bool, + arch: Arch, + root_files: &mut Vec<(String, Vec)>, +) { + let userland_dir = root.join("userland"); + let target = arch.userland(); + + let programs: Vec = config_crates(root, config) + .into_iter() + .filter(|c| matches!(c.built, Built::Member | Built::Standalone)) + .filter(|c| match not_built_for(arch, &c.name) { + Some(why) => { + eprintln!("{}: not built for {}, and not on this ROOT: {why}", c.name, arch.name()); + false + } + None => true, + }) + .collect(); + for c in &programs { + assert!( + c.dir.join("Cargo.toml").exists(), + "Program '{}' crate not found at {}", + c.name, + c.dir.display() + ); + } + let workspace_packages: Vec<&str> = + programs.iter().filter(|c| c.built == Built::Member).map(|c| c.name.as_str()).collect(); + + let ws_target = userland_dir.join(format!("target/{target}/{PROFILE}")); + + // Build and read under one hold, exactly as `build_toyos_bins` does and for + // the same reason: a program's path is keyed on (crate, target, profile) + // alone, so every config in this run writes and reads the same + // `userland/target/.../toybox`. Cargo's own lock orders the two *builds* and + // says nothing about a read between them — `ioapic_topology` died on + // `Failed to read binary for toybox` while another worker's config was + // relinking it, and was green the moment it was re-run alone. + let _artifact = buildlock::artifact(root); + if !workspace_packages.is_empty() { + let mut extra: Vec<&str> = Vec::new(); + for pkg in &workspace_packages { + extra.push("-p"); + extra.push(pkg); + } + cargo_build(&userland_dir, target, &extra, env, &[], quiet); + } + + for c in programs.iter().filter(|c| c.built == Built::Standalone) { + cargo_build(&c.dir, target, &c.features.args(), env, &[], quiet); + } + + for c in &programs { + let name = &c.name; + let binary = match c.built { + Built::Member => ws_target.join(name), + Built::Standalone => { + hostws::target_dir(root, &c.dir).join(format!("{target}/{PROFILE}/{name}")) + } + Built::Kernel | Built::Bootloader => unreachable!("filtered out above"), + }; + let data = fs::read(&binary).unwrap_or_else(|_| panic!("Failed to read binary for {name}")); + root_files.push((format!("bin/{name}"), data)); + } +} + /// What `tests/common/qemu.rs` prefixes every binary it injects with. const HARNESS_PREFIXES: [&str; 2] = ["bin/test_rs_", "bin/test_c_"]; @@ -876,6 +950,7 @@ const CONFIG: &str = "system.toml"; /// Everything one image is built from: the config, the kernel's feature list, /// and the parameters that kernel is armed with. pub struct Plan { + pub arch: Arch, pub config: PathBuf, pub features: Vec, pub params: Vec, @@ -886,8 +961,9 @@ pub struct Plan { } impl Plan { - pub fn new(config: &Path, features: &[&str], params: &[&str]) -> Self { + pub fn new(arch: Arch, config: &Path, features: &[&str], params: &[&str]) -> Self { Self { + arch, config: config.to_path_buf(), features: features.iter().map(|f| (*f).to_string()).collect(), params: params.iter().map(|p| (*p).to_string()).collect(), @@ -914,7 +990,9 @@ pub fn plan_for(root: &Path, boot: &Boot, debug: bool, args: &[String]) -> Plan // anything waits on a lock. let features = kernel_features(root, debug, &feature, ¶m); check_params(root, ¶m); + let arch = arch_for(args); Plan { + arch, config: boot.config.clone(), features: features.split(',').filter(|f| !f.is_empty()).map(Into::into).collect(), params: param, @@ -923,6 +1001,19 @@ pub fn plan_for(root: &Path, boot: &Boot, debug: bool, args: &[String]) -> Plan } } +/// The architecture a `cargo run` command line names with `--arch`: x86-64 +/// when it names none, because that is the one whose userland boots to a +/// desktop until the port's userland stage lands. +pub fn arch_for(args: &[String]) -> Arch { + match flags::CARGO_RUN.value(args, &flags::ARCH) { + Some(name) => Arch::parse(name).unwrap_or_else(|why| { + eprintln!("Error: --arch {why}"); + std::process::exit(2); + }), + None => Arch::X86_64, + } +} + /// Which boot the image being built is for: the directory holding its config, /// and the artifact that directory's name gives it. /// @@ -964,6 +1055,16 @@ impl Boot { Ok(Self { config: dir.join(CONFIG), image: PathBuf::from(image), case }) } + /// The image `arch`'s build of this boot writes. x86-64 keeps the name + /// every flashing step already reads; every other architecture's carries its + /// name, so two architectures' builds of one config never share a file. + pub fn image_for(&self, arch: Arch) -> PathBuf { + match arch { + Arch::X86_64 => self.image.clone(), + Arch::Aarch64 => self.image.with_extension(format!("{}.img", arch.name())), + } + } + /// The three modes' directories are the repository's own, so a refusal from /// [`Boot::at`] on one of them is a broken checkout and not a command line. fn mode(root: &Path, dir: &Path) -> Self { @@ -1441,7 +1542,7 @@ fn assert_actuators_match_features(root: &Path, features: &str, kernel: &[u8]) { ); } -/// The labels `arch::syscall::gate` defines inside `syscall_entry` for +/// The labels `arch::syscall` defines inside `syscall_entry` for /// `nmi_gate`, which `toyos-ld` carries into `.strtab`. const ENTRY_LABELS: [&str; 3] = ["syscall_entry_hold_spin", "syscall_entry_hold_end", "syscall_entry_end"]; @@ -1456,15 +1557,15 @@ fn assert_entry_labels_match_features(features: &str, kernel: &[u8]) { kernel, TEST_KERNEL, &ENTRY_LABELS, - "labels `arch::syscall::gate` puts inside `syscall_entry`", + "labels `arch::syscall` puts inside `syscall_entry`", "They bound what `nmi_gate` holds and counts, and belong to a kernel built with \ `boot-actuators`.", ); } -/// `arch::syscall::gate::syscall_entry`'s v0-mangled path, less the crate +/// `arch::syscall::syscall_entry`'s v0-mangled path, less the crate /// disambiguator that stands in front of it. -const SYSCALL_ENTRY_SYMBOL: &str = "6kernel4arch7syscall4gate13syscall_entry"; +const SYSCALL_ENTRY_SYMBOL: &str = "6kernel4arch6x86_647syscall13syscall_entry"; /// `cld`, which `arch::entry::ring3_naked_asm` puts first in every Ring 0 entry. const CLD: u8 = 0xfc; @@ -1629,20 +1730,39 @@ fn assert_sched_check_matches_features(features: &str, kernel: &[u8]) { /// the kernel of every image this build system produces that gets it. The /// caller has already run `cargo_build` on the kernel crate and must hold /// [`buildlock::artifact`], since the stage below copies the shared cargo path. -fn stage_and_certify_kernel(root: &Path, features: &str, env: &GuestEnv) -> Vec { +/// Stage the loader `arch`'s build just wrote, under the key that names it. +/// The caller holds [`buildlock::artifact`], as [`stage_and_certify_kernel`]'s does. +fn stage_loader(root: &Path, arch: Arch, env: &GuestEnv) -> PathBuf { + stage_artifact( + root, + &root.join(format!("bootloader/target/{}/{PROFILE}/bootloader.efi", arch.loader())), + &format!("bootloader-{}.efi", arch.name()), + loader_key(arch, &env.image_key, env.floor_scope), + ) +} + +fn stage_and_certify_kernel(root: &Path, features: &str, env: &GuestEnv, arch: Arch) -> Vec { let staged = stage_artifact( root, - &root.join(format!("kernel/target/x86_64-unknown-none/{PROFILE}/kernel")), - "kernel", - kernel_key(features), + &root.join(format!("kernel/target/{}/{PROFILE}/kernel", arch.kernel())), + &format!("kernel-{}", arch.name()), + kernel_key(arch, features), ); let bytes = fs::read(&staged).expect("Failed to read staged kernel"); assert_overflow_checked("kernel", &bytes); assert_actuators_match_features(root, features, &bytes); - assert_entry_window_matches_features(features, &bytes); - assert_entry_labels_match_features(features, &bytes); + match arch { + Arch::X86_64 => { + assert_entry_window_matches_features(features, &bytes); + assert_entry_labels_match_features(features, &bytes); + } + // Taking an exception to EL1 sets `PSTATE.SP`, so its handler's first + // instruction already runs on `SP_EL1`: no instruction runs at EL1 on a + // user's stack, and there is no window to judge. + Arch::Aarch64 => {} + } assert_sched_check_matches_features(features, &bytes); - assert_kernel_is_softfloat(env); + assert_kernel_is_softfloat(env, arch); bytes } @@ -1657,7 +1777,7 @@ pub fn build( // combine with were refused before any of them. if boot.case { let bytes = build_test_image(root, plan, false, &[]); - let image_path = root.join(&boot.image); + let image_path = root.join(boot.image_for(plan.arch)); fs::write(&image_path, bytes) .unwrap_or_else(|e| panic!("write {}: {e}", image_path.display())); return image_path; @@ -1670,6 +1790,7 @@ pub fn build( // this one's size. let second = image::SecondSlot { root_bytes: 2 * root_bytes.len() as u64 }; let disk_bytes = image::create_boot_image( + plan.arch, &kernel_bytes, &bl_bytes, &root_bytes, @@ -1677,7 +1798,7 @@ pub fn build( image::Signing { key, version: plan.version }, Some(second), ); - let image_path = root.join(&boot.image); + let image_path = root.join(boot.image_for(plan.arch)); fs::write(&image_path, disk_bytes).expect("Failed to write image"); let nvme_path = root.join("target/nvme.img"); @@ -1722,6 +1843,7 @@ fn said_key(plan: &Plan) -> &'static crate::signing::Key { /// The kernel, the loader and ROOT a mode's image is made of. fn shipped_parts(root: &Path, boot: &Boot, rebuild_toolchain: bool, plan: &Plan) -> (Vec, Vec, Vec) { let kernel_features = plan.features.join(","); + let arch = plan.arch; // Before every build lock, which is the order `buildlock`'s header fixes. // What it bounds is the host: ten agents' builds spend the same fourteen @@ -1754,42 +1876,18 @@ fn shipped_parts(root: &Path, boot: &Boot, rebuild_toolchain: bool, plan: &Plan) extra.push("--features"); extra.push(&features); } - cargo_build( - &root.join("kernel"), - "x86_64-unknown-none", - &extra, - &env, - &[], - false, - ); + cargo_build(&root.join("kernel"), arch.kernel(), &extra, &env, &[], false); }) }; - { - cargo_build( - &root.join("bootloader"), - "x86_64-unknown-uefi", - &[], - &env, - &[], - false, - ); - } + cargo_build(&root.join("bootloader"), arch.loader(), &[], &env, &[], false); kernel_handle.join().expect("kernel build thread panicked"); ( - stage_and_certify_kernel(root, &kernel_features, &env), - stage_artifact( - root, - &root.join(format!( - "bootloader/target/x86_64-unknown-uefi/{PROFILE}/bootloader.efi" - )), - "bootloader.efi", - key_hash(&[PROFILE, &env.image_key, env.floor_scope]), - ), + stage_and_certify_kernel(root, &kernel_features, &env, arch), + stage_loader(root, arch, &env), ) }; - let root_bytes = - build_and_assemble(root, &config, &env, &[], false); + let root_bytes = build_and_assemble(root, &config, &env, &[], false, arch); let bl_bytes = fs::read(&bl_art).expect("Failed to read staged bootloader"); (kernel_bytes, bl_bytes, root_bytes) @@ -1881,14 +1979,16 @@ static KERNEL: Memo = Memo::new(); static BOOTLOADER: Memo = Memo::new(); static ROOT_IMAGE: Memo = Memo::new(); -/// What the ROOT image is a function of: the config naming the programs, and the -/// files the caller adds to it. Hashed whole: a key over the test binaries' +/// What the ROOT image is a function of: the config naming the programs, the +/// architecture and whether they are built at all, the key its +/// `/system/bin/update` embeds, and the files the caller adds to it. Hashed whole: a key over the test binaries' /// names and lengths would call two different builds of one binary the same /// image. -fn root_image_key(config_path: &Path, image_key: &str, extra_files: &[(String, Vec)]) -> u64 { +fn root_image_key(plan: &Plan, image_key: &str, extra_files: &[(String, Vec)]) -> u64 { use std::hash::{Hash, Hasher}; let mut h = std::collections::hash_map::DefaultHasher::new(); - config_path.hash(&mut h); + plan.config.hash(&mut h); + plan.arch.hash(&mut h); image_key.hash(&mut h); for (name, data) in extra_files { name.hash(&mut h); @@ -1912,6 +2012,7 @@ pub fn build_test_image( ) -> Vec { let parts = build_test_parts(root, plan, quiet, extra_files); image::create_boot_image( + plan.arch, &parts.kernel, &parts.bootloader, &parts.root, @@ -1961,12 +2062,13 @@ pub fn build_test_parts( || kernel_features.iter().eq(TEST_KERNEL.iter().copied()), "a boot asking for {kernel_params:?} must boot the test kernel, not {kernel_features:?}" ); - let kernel_key = kernel_key(&features); + let arch = plan.arch; + let kernel_key = kernel_key(arch, &features); // The loader and ROOT (whose `/system/bin/update` embeds it) are each a // function of the key this process signs with. let image_key = crate::signing::key().public_hex(); - let bl_key = key_hash(&[PROFILE, &image_key, crate::signing::key().floor_scope().word()]); - let root_image_key = root_image_key(config_path, &image_key, extra_files); + let bl_key = loader_key(arch, &image_key, crate::signing::key().floor_scope().word()); + let root_image_key = root_image_key(plan, &image_key, extra_files); // Nothing left to build, so nothing for the lock, the toolchain check or the // staleness sweep to protect. @@ -2013,38 +2115,18 @@ pub fn build_test_parts( kernel_extra.push("--features"); kernel_extra.push(&features); } - cargo_build( - &root.join("kernel"), - "x86_64-unknown-none", - &kernel_extra, - &env, - &[], - quiet, - ); - stage_and_certify_kernel(root, &features, &env) + cargo_build(&root.join("kernel"), arch.kernel(), &kernel_extra, &env, &[], quiet); + stage_and_certify_kernel(root, &features, &env, arch) }); let bl = BOOTLOADER.get_or_build(bl_key, || { - cargo_build( - &root.join("bootloader"), - "x86_64-unknown-uefi", - &[], - &env, - &[], - quiet, - ); - let staged = stage_artifact( - root, - &root.join(format!("bootloader/target/x86_64-unknown-uefi/{PROFILE}/bootloader.efi")), - "bootloader.efi", - bl_key, - ); - fs::read(&staged).expect("Failed to read staged bootloader") + cargo_build(&root.join("bootloader"), arch.loader(), &[], &env, &[], quiet); + fs::read(stage_loader(root, arch, &env)).expect("Failed to read staged bootloader") }); (kernel, bl) }; let root_bytes = ROOT_IMAGE.get_or_build(root_image_key, || { - build_and_assemble(root, &config, &env, extra_files, quiet) + build_and_assemble(root, &config, &env, extra_files, quiet, arch) }); drop(build_timer); @@ -2101,8 +2183,8 @@ pub fn https_fetch_host(root: &Path) -> PathBuf { /// Copy to `to` the binary the build leaves for userland workspace program /// `name`: the bytes a swap sends a running machine in place of the ones its /// image carries. Read under the artifact lock, as every image build reads it. -pub fn copy_guest_program(root: &Path, name: &str, to: &Path) -> Result<(), String> { - let from = root.join(format!("userland/target/x86_64-unknown-toyos/{PROFILE}/{name}")); +pub fn copy_guest_program(root: &Path, arch: Arch, name: &str, to: &Path) -> Result<(), String> { + let from = root.join(format!("userland/target/{}/{PROFILE}/{name}", arch.userland())); let _artifact = buildlock::artifact(root); fs::copy(&from, to) .map(|_| ()) @@ -2127,7 +2209,8 @@ fn host_judge(root: &Path, (dir, bin): Judge) -> PathBuf { /// target-directory scan keeps shipping a renamed or merged test from an artifact /// nothing in the tree can produce any more — into the ROOT image, into the test list, /// and over the name of whatever gets it next. -pub fn build_toyos_bins(root: &Path, crate_path: &Path, quiet: bool) -> Vec<(String, Vec)> { +pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) -> Vec<(String, Vec)> { + let target = arch.userland(); let _slot = buildlock::build_slot(root, "the test binaries"); let mut lock = buildlock::shared(root, "test binaries"); let sysroot = crate::toolchain::ensure(root, false, &mut lock); @@ -2173,9 +2256,9 @@ pub fn build_toyos_bins(root: &Path, crate_path: &Path, quiet: bool) -> Vec<(Str if !quiet { eprintln!("[build] Building cdylib subcrate: {lib_name}"); } - cargo_build(&sub_path, "x86_64-unknown-toyos", &[], &env, &[], quiet); + cargo_build(&sub_path, target, &[], &env, &[], quiet); - let lib_out = sub_path.join(format!("target/x86_64-unknown-toyos/{PROFILE}")); + let lib_out = sub_path.join(format!("target/{target}/{PROFILE}")); lib_search_dirs.push(lib_out.clone()); for so_entry in fs::read_dir(&lib_out).unwrap() { @@ -2200,16 +2283,9 @@ pub fn build_toyos_bins(root: &Path, crate_path: &Path, quiet: bool) -> Vec<(Str } else { vec![("RUSTFLAGS", link_flags.trim_end())] }; - cargo_build( - crate_path, - "x86_64-unknown-toyos", - &["--bins"], - &env, - &extra_env, - quiet, - ); + cargo_build(crate_path, target, &["--bins"], &env, &extra_env, quiet); - let bin_dir = crate_path.join(format!("target/x86_64-unknown-toyos/{PROFILE}")); + let bin_dir = crate_path.join(format!("target/{target}/{PROFILE}")); let bin_src = crate_path.join("src/bin"); if bin_src.exists() { for entry in fs::read_dir(&bin_src).unwrap() { @@ -2241,7 +2317,8 @@ pub fn build_toyos_bins(root: &Path, crate_path: &Path, quiet: bool) -> Vec<(Str /// lock its rebuild takes. fn collect_hosted_rustc(root: &Path, toolchain: &Path, root_files: &mut Vec<(String, Vec)>) { let _compiler = buildlock::compiler_shared(root, "reading the hosted rustc"); - let sysroot = toolchain::rust_dir(root).join("build/x86_64-unknown-toyos/stage2"); + let target = toolchain::HOSTED_ARCH.userland(); + let sysroot = toolchain::rust_dir(root).join(format!("build/{target}/stage2")); assert!( sysroot.exists(), "Hosted rustc sysroot missing: {}", @@ -2267,7 +2344,7 @@ fn collect_hosted_rustc(root: &Path, toolchain: &Path, root_files: &mut Vec<(Str } } - let backends = sysroot.join("lib/rustlib/x86_64-unknown-toyos/codegen-backends"); + let backends = sysroot.join(format!("lib/rustlib/{target}/codegen-backends")); if backends.exists() { for entry in fs::read_dir(&backends).into_iter().flatten().flatten() { let path = entry.path(); @@ -2275,20 +2352,20 @@ fn collect_hosted_rustc(root: &Path, toolchain: &Path, root_files: &mut Vec<(Str let name = path.file_name().unwrap().to_str().unwrap().to_string(); let data = fs::read(&path).unwrap(); root_files.push(( - format!("lib/rustlib/x86_64-unknown-toyos/codegen-backends/{name}"), + format!("lib/rustlib/{target}/codegen-backends/{name}"), data, )); } } } - let rlibs = toolchain.join("lib/rustlib/x86_64-unknown-toyos/lib"); + let rlibs = toolchain.join(format!("lib/rustlib/{target}/lib")); for entry in fs::read_dir(&rlibs).unwrap_or_else(|e| panic!("read {}: {e}", rlibs.display())) { let path = entry.unwrap_or_else(|e| panic!("read {}: {e}", rlibs.display())).path(); if path.extension().is_some_and(|e| e == "rlib" || e == "rmeta") { let name = path.file_name().unwrap().to_str().unwrap().to_string(); root_files.push(( - format!("lib/rustlib/x86_64-unknown-toyos/lib/{name}"), + format!("lib/rustlib/{target}/lib/{name}"), fs::read(&path).unwrap(), )); } @@ -2369,13 +2446,13 @@ mod tests { assert_eq!(shipping, "", "`cargo run` asks the kernel for {shipping:?}, not nothing"); let harness = <&[&str]>::default().join(","); assert_eq!( - kernel_key(&shipping), - kernel_key(&harness), + kernel_key(Arch::X86_64, &shipping), + kernel_key(Arch::X86_64, &harness), "a featureless boot and the shipping build stage different kernels" ); assert_ne!( - kernel_key(&shipping), - kernel_key("test-actuators"), + kernel_key(Arch::X86_64, &shipping), + kernel_key(Arch::X86_64, "test-actuators"), "the key ignores the features, so it cannot tell two kernels apart" ); } @@ -2616,6 +2693,14 @@ mod tests { // not in `TEST_SUITE_KERNEL_BUILDS`, so a full run pays nothing // for it and a boot storm asks for it by name. "sched-tripwire", + // Costs no kernel build, for `wake-fence-off`'s reason: turned on + // only by `kernel-loom`, to drop the panic console publisher's + // `Release` fence and prove `panic_console_publish` reds without it. + "seqlock-writer-fence-off", + // Costs no kernel build: turned on only by `kernel-loom`, to build + // the backend lock's `try_lock` with `then_some` and prove + // `serial_lock` reds. + "serial-try-lock-then-some", "shard-publish-relaxed", "shootdown-serve-relaxed", // The eighth loom control, and the first over a *contended* @@ -2741,6 +2826,22 @@ mod tests { ); } + /// **An architecture leaves out only what the shipped config builds**, and + /// each reason names the issue file that owns it. + #[test] + fn every_program_an_architecture_leaves_out_is_one_the_shipped_config_builds() { + let root = Path::new(env!("CARGO_MANIFEST_DIR")); + let programs = parse_config(&Boot::shipped(root).config).programs; + for (arch, name, why) in NOT_YET_BUILT { + assert!(programs.contains_key(*name), "{name} is left out for {arch:?} and the shipped config builds no such program"); + assert_eq!(not_built_for(*arch, name), Some(*why)); + let issue = why.split("issues/").nth(1).map(|rest| rest.split(')').next().unwrap_or(rest)); + let issue = issue.unwrap_or_else(|| panic!("{name}'s reason names no issue file: {why}")); + assert!(root.join("issues").join(issue).is_file(), "{name}'s reason cites issues/{issue}, which does not exist"); + } + assert_eq!(not_built_for(Arch::X86_64, "calc"), None, "x86-64 builds every program"); + } + /// No image this repository ships starts sshd. /// /// It listens on every interface and authenticates against a file that is @@ -3569,7 +3670,7 @@ mod tests { file } - const ENTRY_NAME: &str = "_RNvNtNtNtCs2TF9wDo3GXK_6kernel4arch7syscall4gate13syscall_entry"; + const ENTRY_NAME: &str = "_RNvNtNtNtCs2TF9wDo3GXK_6kernel4arch6x86_647syscall13syscall_entry"; fn entry_with(between: &[u8]) -> Vec { [&ENTRY_OPENS[..], between, &SWITCH[..], &[0x90; 32][..]].concat() diff --git a/src/buildlock.rs b/src/buildlock.rs index 153a7843aa0..fa9fc5cf07b 100644 --- a/src/buildlock.rs +++ b/src/buildlock.rs @@ -26,9 +26,11 @@ //! (`src/sysroot.rs`), which nothing rewrites. Only a sysroot being *made* //! reads the compiler, and it holds [`compiler_shared`] while it does. //! -//! A sysroot's own lock ([`sysroot_building`], [`sysroot_using`]) is per key, +//! A sysroot's own lock ([`keyed_building`], [`keyed_using`]) is per key, //! so two worktrees with different ABIs never meet in it, and two with the same -//! one build it once. +//! one build it once. A compiler a worktree's fork checkout names apart from the +//! primary's (`src/compiler.rs`) is locked the same way under its own key, and +//! neither it nor a sysroot built from it takes the global lock. //! //! [`integration`] is neither: one file of its own, exclusive-only, and held //! while this host's `main` moves rather than while anything builds. @@ -49,7 +51,8 @@ //! `[host-builds] waiting …` line was printed. //! //! **The order between them is a constraint, not a preference:** host slot -//! (guest or build) → a sysroot key's lock → the worktree build lock → the +//! (guest or build) → a compiler key's lock → a sysroot key's lock → the +//! worktree build lock → the //! global one → artifact. A build slot is taken before any build lock and never //! while one is held; a key's lock is taken with the worktree lock put down //! ([`Held::without_shared`]), because the key's builder takes the worktree lock @@ -352,40 +355,64 @@ pub fn build_slot(root: &Path, what: &str) -> Option { const BUILD_SLOT_DIR: &str = "build-slots"; -/// Make the sysroot `key` names: exclusive, and waited for by every other -/// process that wants the same key, which then finds it made. -pub fn sysroot_building(root: &Path, key: &str) -> Guard { - exclusive(&sysroot_lock_path(root, key), "sysroot lock", &format!("building sysroot {key}")) +/// A content-addressed product of the host, locked per key: a sysroot, or a +/// compiler a worktree's fork checkout names (`src/compiler.rs`). +#[derive(Clone, Copy)] +pub enum Keyed { + Sysroot, + Compiler, } -/// Compile against the sysroot `key` names: shared, so any number of builds use -/// it at once, a builder of it is waited for, and a sweep cannot remove it. -pub fn sysroot_using(root: &Path, key: &str) -> Guard { - let path = sysroot_lock_path(root, key); +impl Keyed { + fn dir(self) -> &'static str { + match self { + Keyed::Sysroot => "sysroots", + Keyed::Compiler => "compilers", + } + } + + fn name(self) -> &'static str { + match self { + Keyed::Sysroot => "sysroot", + Keyed::Compiler => "compiler", + } + } +} + +/// Make what `key` names: exclusive, and waited for by every other process that +/// wants the same key, which then finds it made. +pub fn keyed_building(root: &Path, kind: Keyed, key: &str) -> Guard { + let lock = format!("{} lock", kind.name()); + exclusive(&keyed_lock_path(root, kind, key), &lock, &format!("building {} {key}", kind.name())) +} + +/// Use what `key` names: shared, so any number of builds use it at once, a +/// builder of it is waited for, and a sweep cannot remove it. +pub fn keyed_using(root: &Path, kind: Keyed, key: &str) -> Guard { + let path = keyed_lock_path(root, kind, key); let file = open_lock_file(&path); if !try_lock(&file, LOCK_SH) { - let what = format!("using sysroot {key}"); + let lock = format!("{} lock", kind.name()); + let what = format!("using {} {key}", kind.name()); let holder = describe_holder(&path) .unwrap_or_else(|| "held, but the holder left no readable note".to_string()); - announce("sysroot lock", &what, &holder); - take_lock_announcing(&file, LOCK_SH, &path, "sysroot lock", &what); + announce(&lock, &what, &holder); + take_lock_announcing(&file, LOCK_SH, &path, &lock, &what); } Guard { file, records_holder: false } } -/// The sysroot `key` names, exclusively and only if nobody is making or using -/// it: what a sweep holds while it removes one. -pub fn sysroot_idle(root: &Path, key: &str) -> Option { - let file = open_lock_file(&sysroot_lock_path(root, key)); +/// What `key` names, exclusively and only if nobody is making or using it: what +/// a sweep holds while it removes one. +pub fn keyed_idle(root: &Path, kind: Keyed, key: &str) -> Option { + let file = open_lock_file(&keyed_lock_path(root, kind, key)); try_lock(&file, LOCK_EX).then_some(Guard { file, records_holder: false }) } -fn sysroot_lock_path(root: &Path, key: &str) -> PathBuf { - git_lock_dir(root).join(SYSROOT_DIR).join(key) +fn keyed_lock_path(root: &Path, kind: Keyed, key: &str) -> PathBuf { + git_lock_dir(root).join(kind.dir()).join(key) } -const SYSROOT_DIR: &str = "sysroots"; - fn slot_path(dir: &Path, index: usize) -> PathBuf { dir.join(format!("slot-{index}")) } @@ -873,7 +900,7 @@ mod tests { until_orphaned(); } "hold-sysroot-build" => { - let _building = sysroot_building(&root, "k1"); + let _building = keyed_building(&root, Keyed::Sysroot, "k1"); touch(&root.join("held")); appeared(&root.join("release"), Duration::from_secs(20)); note(&root, "built"); @@ -883,7 +910,7 @@ mod tests { note(&root, "landed"); } "want-sysroot" => { - let _using = sysroot_using(&root, "k1"); + let _using = keyed_using(&root, Keyed::Sysroot, "k1"); note(&root, "used"); } "want-slot" => { @@ -1148,8 +1175,8 @@ mod tests { let mut builder = child(&root, "hold-sysroot-build"); assert!(appeared(&root.join("held"), Duration::from_secs(20)), "the builder never started"); - assert!(sysroot_idle(&root, "k1").is_none(), "a sweep could remove a key being built"); - let other = sysroot_building(&root, "k2"); + assert!(keyed_idle(&root, Keyed::Sysroot, "k1").is_none(), "a sweep could remove a key being built"); + let other = keyed_building(&root, Keyed::Sysroot, "k2"); drop(other); let mut user = child(&root, "want-sysroot"); @@ -1162,10 +1189,10 @@ mod tests { assert!(user.wait().unwrap().success()); assert_eq!(fs::read_to_string(root.join("order.log")).unwrap(), "built\nused\n"); - let using = sysroot_using(&root, "k1"); - assert!(sysroot_idle(&root, "k1").is_none(), "a sweep could remove a key in use"); + let using = keyed_using(&root, Keyed::Sysroot, "k1"); + assert!(keyed_idle(&root, Keyed::Sysroot, "k1").is_none(), "a sweep could remove a key in use"); drop(using); - assert!(sysroot_idle(&root, "k1").is_some()); + assert!(keyed_idle(&root, Keyed::Sysroot, "k1").is_some()); } /// The whole point of a counting semaphore: the run past the budget waits. diff --git a/src/ci.rs b/src/ci.rs index 52ccf75082a..daa02852628 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -34,6 +34,7 @@ use std::io::{BufRead, BufReader, Write}; use std::path::Path; use std::process::Command; +use crate::arch::Arch; use crate::{flags, pr, release, sdkversion}; /// The checks `main`'s ruleset must require, as `gate-stage` reads them back: @@ -248,6 +249,13 @@ pub(crate) const CONTROLS: &[Control] = &[ red(KERNEL_LOOM, "lock-acquire-off", Some("ticket_lock"), &[ "try_lock_observes_the_previous_owners_writes ... FAILED", ]), + red(KERNEL_LOOM, "seqlock-writer-fence-off", Some("panic_console_publish"), &[ + "a_snapshot_is_one_publication_whole ... FAILED", + ]), + red(KERNEL_LOOM, "serial-try-lock-then-some", Some("serial_lock"), &[ + "a_lost_try_lock_leaves_the_lock_held ... FAILED", + "two_writers_never_overlap ... FAILED", + ]), red(KERNEL_LOOM, "poison-overwrite", Some("poison_set"), &[ "a_second_death_banks_beside_the_first ... FAILED", ]), @@ -411,8 +419,8 @@ fn run_control(root: &Path, control: &Control) -> Result { /// end, is a test that fills the host's disk one run at a time. /// /// Clippy needs none of the ToyOS toolchain the nightly alone builds — the -/// kernel and the bootloader lint against `x86_64-unknown-none` and -/// `x86_64-unknown-uefi`, targets any rustup installs, and userland carries no +/// kernel and the bootloader lint against every architecture's bare targets +/// ([`crate::clippy::BARE_TARGETS`]), which any rustup installs, and userland carries no /// clippy shape (`src/clippy.rs`). Userland and the SDK are tested against the /// host triple for the same reason. fn host(root: &Path) -> Vec { @@ -430,10 +438,10 @@ fn host(root: &Path) -> Vec { ]; steps.push(step("clippy and the bare targets", || { for args in [ - &["component", "add", "clippy"][..], - &["target", "add", "x86_64-unknown-none", "x86_64-unknown-uefi"], + vec!["component", "add", "clippy"], + [&["target", "add"][..], &crate::clippy::BARE_TARGETS].concat(), ] { - let status = Command::new("rustup").args(args).status().map_err(|e| e.to_string())?; + let status = Command::new("rustup").args(&args).status().map_err(|e| e.to_string())?; if !status.success() { return Err(format!("rustup {} exited {status}", args.join(" "))); } @@ -621,7 +629,9 @@ fn guest(root: &Path, suite: &[String]) -> Vec { // inherits this, and nothing here reads the environment concurrently with // the write. std::env::set_var("TMPDIR", tmp.path()); - let mut steps = vec![step("the instrument", || instrument(root))]; + // Every guest lane boots x86-64 guests: no hosted runner has been measured + // for an aarch64 one. + let mut steps = vec![step("the instrument", || instrument(root, Arch::X86_64))]; if steps.iter().all(|s| s.verdict.is_ok()) { steps.push(step("the toolchain", || release::install(root))); } @@ -671,16 +681,17 @@ fn verdicts(log: &str) -> String { /// The QEMU on `PATH` against `.github/qemu-version`, and whether `/dev/kvm` /// opens where it is present — the two things a guest verdict must be read /// against. -fn instrument(root: &Path) -> Result { +fn instrument(root: &Path, arch: Arch) -> Result { let want = declared_qemu_version(root).ok_or(".github/qemu-version declares no version")?; - let out = Command::new("qemu-system-x86_64") + let out = Command::new(arch.qemu()) .arg("--version") .output() - .map_err(|e| format!("qemu-system-x86_64: {e}"))?; + .map_err(|e| format!("{}: {e}", arch.qemu()))?; let said = String::from_utf8_lossy(&out.stdout).into_owned(); let have = parse_qemu_version(&said).ok_or_else(|| format!("QEMU said {said:?}"))?; let node = Path::new("/dev/kvm").exists(); - let accel = match (node, crate::kvm_usable()) { + let accelerated = arch.accel().is_hardware(); + let accel = match (node, accelerated) { (true, true) => "/dev/kvm opens", (true, false) => "/dev/kvm is present and does not open", (false, _) => "no /dev/kvm: emulated", @@ -702,7 +713,7 @@ fn instrument(root: &Path) -> Result { instrument moved" )); } - if node && !crate::kvm_usable() { + if node && !accelerated { return Err(format!("{line}: every boot would fall back to emulation in silence")); } Ok(line) @@ -733,9 +744,9 @@ fn parse_qemu_version(text: &str) -> Option { /// The line `cargo run` prints when this host is not the instrument the /// project's numbers were taken on, and nothing at all when it is. -pub fn qemu_version_note(root: &Path) -> Option { +pub fn qemu_version_note(root: &Path, arch: Arch) -> Option { let want = declared_qemu_version(root)?; - let out = Command::new("qemu-system-x86_64").arg("--version").output().ok()?; + let out = Command::new(arch.qemu()).arg("--version").output().ok()?; let have = parse_qemu_version(&String::from_utf8_lossy(&out.stdout))?; (have != want).then(|| { format!( diff --git a/src/clippy.rs b/src/clippy.rs index 5ea33719eda..8ed62340909 100644 --- a/src/clippy.rs +++ b/src/clippy.rs @@ -4,12 +4,15 @@ //! host` runs the same list as one of its steps, so the local command and //! the merge gate cannot verify different sets. //! -//! Userland is not here: `x86_64-unknown-toyos` is a custom target, and the -//! fork's `toyos` toolchain ships no clippy. +//! Userland is not here: its targets are the fork's own, and the fork's +//! `toyos` toolchain ships no clippy. The kernel and the bootloader are linted +//! for every architecture. use std::path::Path; use std::process::Command; +use crate::arch::Arch; + /// The pedantic/nursery lints adopted one at a time, each on a measured finding /// (`issues/build/clippy-stage-two-is-lints-one-at-a-time.md`). const ADOPTED: &[&str] = &[ @@ -46,22 +49,42 @@ const SHAPES: &[Shape] = &[ }, Shape { dir: "kernel", - before: &["--target", "x86_64-unknown-none"], + before: &["--target", Arch::X86_64.kernel()], + after: &["$ADOPTED", "-D", "warnings"], + }, + Shape { + dir: "kernel", + before: &["--target", Arch::X86_64.kernel(), "--features", "boot-actuators,test-actuators"], + after: &["$ADOPTED", "-D", "warnings"], + }, + Shape { + dir: "kernel", + before: &["--target", Arch::X86_64.kernel(), "--features", "boot-actuators"], after: &["$ADOPTED", "-D", "warnings"], }, Shape { dir: "kernel", - before: &["--target", "x86_64-unknown-none", "--features", "boot-actuators,test-actuators"], + before: &["--target", Arch::Aarch64.kernel()], after: &["$ADOPTED", "-D", "warnings"], }, Shape { dir: "kernel", - before: &["--target", "x86_64-unknown-none", "--features", "boot-actuators"], + before: &["--target", Arch::Aarch64.kernel(), "--features", "boot-actuators,test-actuators"], + after: &["$ADOPTED", "-D", "warnings"], + }, + Shape { + dir: "kernel", + before: &["--target", Arch::Aarch64.kernel(), "--features", "boot-actuators"], after: &["$ADOPTED", "-D", "warnings"], }, Shape { dir: "bootloader", - before: &["--target", "x86_64-unknown-uefi"], + before: &["--target", Arch::X86_64.loader()], + after: &["$ADOPTED", "-W", "clippy::undocumented_unsafe_blocks", "-D", "warnings"], + }, + Shape { + dir: "bootloader", + before: &["--target", Arch::Aarch64.loader()], after: &["$ADOPTED", "-W", "clippy::undocumented_unsafe_blocks", "-D", "warnings"], }, Shape { @@ -71,6 +94,11 @@ const SHAPES: &[Shape] = &[ }, ]; +/// The bare targets the kernel and bootloader shapes lint against, which +/// `rustup target add` installs: every architecture's. +pub const BARE_TARGETS: [&str; 4] = + [Arch::X86_64.kernel(), Arch::X86_64.loader(), Arch::Aarch64.kernel(), Arch::Aarch64.loader()]; + impl Shape { /// The command as a reader writes it, `$ADOPTED` unexpanded. fn line(&self) -> String { diff --git a/src/compiler.rs b/src/compiler.rs new file mode 100644 index 00000000000..4293edf567d --- /dev/null +++ b/src/compiler.rs @@ -0,0 +1,548 @@ +//! The compiler a worktree's sysroot is cloned from and compiled by: the +//! primary's, or one of its own, content-addressed. +//! +//! **Every worktree builds with the compiler its own fork checkout names.** The +//! primary's `stage2` is built from what the primary's `rust/compiler/` holds, +//! and [`record`] writes which that is. A linked worktree whose fork checkout +//! holds the same `compiler/` ([`source`]) compiles with that one. One whose +//! `compiler/` differs — a new target spec, a codegen change — gets its own: +//! built by bootstrap in its own fork checkout, under that checkout's +//! `build/toyos-compiler/`, and placed at `rust/build/compilers//`, where +//! the key ([`key`]) is the identity (`src/identity.rs`) of the checkout's +//! `compiler/`, `src/bootstrap/`, `src/tools/`, `src/stage0` and `Cargo.lock`, +//! the `src/llvm-project` commit, and [`RECIPE`]. Nothing writes that directory after its [`SOURCE`] file exists, +//! and two worktrees naming the same compiler share one copy. +//! +//! **A compiler of a worktree's own never touches what the others build with**: +//! not the primary's `stage2`, not its record, not the machine-global rustup +//! `toyos` link — a sysroot is named by its directory, never by a toolchain +//! name, so no link is made. The global lock is not taken either; nothing of +//! the primary's is read. +//! +//! Locks, in the one order every acquirer takes them: the key's +//! (`buildlock::keyed_*` with [`Keyed::Compiler`]), with this worktree's build +//! lock put down, held shared for as long as a sysroot is being made from it; +//! then, to build, this worktree's exclusively, because its fork build +//! directory is written. +//! +//! A compiler no worktree names any more is removed by [`sweep`], which +//! `--worktree remove` and every placement run: each build records the key it used in its +//! worktree's `target/`, and a key no registered worktree records, that nobody +//! is making or using, goes. +//! +//! What such a worktree's image cannot carry is the ToyOS-hosted rustc: that is +//! the primary's, built from the primary's `compiler/`, so `hosted-rustc` in a +//! worktree building with its own compiler is refused by name +//! (`src/build.rs`). + +use std::fs; +use std::path::{Path, PathBuf}; + +use crate::buildlock::{self, Guard, Held, Keyed}; +use crate::sysroot::{clone_tree, git_bytes, git_out, short, tree_identity, Restore}; +use crate::toolchain::{self, host_triple}; + +/// What changes how a key's sources become a compiler and is none of them: the +/// build below. Moving it moves every key. +const RECIPE: &str = "bootstrap stage 2 of compiler/rustc and library, profile compiler, host only, with rust-lld, host linker pinned; 3"; + +/// What a compiler's key is the identity of, in its fork checkout. +const KEYED: [&str; 5] = ["compiler", "src/bootstrap", "src/tools", "src/stage0", "Cargo.lock"]; + +/// The submodule a compiler is built against by commit: its LLVM, which +/// bootstrap takes prebuilt for that commit, so its content is never read. +const LLVM: &str = "src/llvm-project"; + +/// The file a finished compiler carries last, naming what it was built from. A +/// directory without it is a build that did not finish. +const SOURCE: &str = "SOURCE"; + +/// Where each build records the key of the compiler of its own it used, for +/// [`sweep`]. Absent while a worktree builds with the primary's. +const RECORD: &str = "target/toyos-compiler-key"; + +/// A compiler, held in use for as long as this lives. +pub struct Compiler { + /// Its toolchain directory: `bin/rustc`, `lib/`. + pub stage2: PathBuf, + /// The file naming what it was built from. + record: PathBuf, + /// Whether it is the primary's, the one the hosted rustc and the rustup + /// link are built from. + pub primary: bool, + _using: Option, +} + +impl Compiler { + /// The primary's `stage2`. + pub fn primary(rust_dir: &Path) -> Self { + Self { stage2: toolchain::stage2(rust_dir), record: primary_record(rust_dir), primary: true, _using: None } + } + + /// The compiler as a sysroot's key sees it: the source it was built from, + /// and the driver that build left, so a rebuild of the same source is a new + /// compiler too. + pub fn identity(&self) -> String { + let source = fs::read_to_string(&self.record).unwrap_or_else(|_| { + panic!( + "{} is missing, so no sysroot can say which compiler it was built with.\n\ + The primary checkout writes the primary's: run `cargo run -- --build-only` there once.", + self.record.display(), + ) + }); + let lib = self.stage2.join("lib"); + let driver = fs::read_dir(&lib) + .unwrap_or_else(|e| panic!("read {}: {e}", lib.display())) + .flatten() + .find(|e| e.file_name().to_string_lossy().starts_with("librustc_driver")) + .unwrap_or_else(|| panic!("{} holds no librustc_driver", lib.display())); + let meta = driver.metadata().unwrap_or_else(|e| panic!("stat the driver: {e}")); + let mtime = meta + .modified() + .ok() + .and_then(|t| t.duration_since(std::time::UNIX_EPOCH).ok()) + .map_or(0, |d| d.as_nanos()); + format!("{} {} {} {mtime}", source.trim(), driver.file_name().to_string_lossy(), meta.len()) + } +} + +/// The primary's record of which `compiler/` its `stage2` was built from. +fn primary_record(rust_dir: &Path) -> PathBuf { + rust_dir.join("build/toyos-compiler") +} + +/// Every compiler of a worktree's own on this host. +pub fn compilers_dir(rust_dir: &Path) -> PathBuf { + rust_dir.join("build/compilers") +} + +/// What `checkout`'s `compiler/` is: its commit's tree, and whatever the working +/// tree changes in it — an edit, or a file git does not track yet, which is +/// what a new target spec is before its commit. +pub fn source(checkout: &Path) -> String { + let tree = git_out(checkout, &["rev-parse", "HEAD:compiler"]); + let mut local = git_bytes(checkout, &["diff", "HEAD", "--", "compiler"]); + let untracked = git_bytes(checkout, &["ls-files", "-z", "--others", "--exclude-standard", "--", "compiler"]); + for name in untracked.split(|b| *b == 0).filter(|n| !n.is_empty()) { + let path = checkout.join(String::from_utf8_lossy(name).as_ref()); + local.extend_from_slice(name); + local.push(0); + local.extend(fs::read(&path).unwrap_or_else(|e| panic!("read {}: {e}", path.display()))); + local.push(0); + } + if local.is_empty() { + tree.trim().to_string() + } else { + format!("{} with local changes {}", tree.trim(), short(&local)) + } +} + +/// Record which compiler the primary's `stage2` is. The primary calls this +/// after a toolchain build, and when the record is missing — its compiler stamp +/// has just said `stage2` is built from what its `rust/` holds. +pub fn record(rust_dir: &Path) { + let at = primary_record(rust_dir); + let want = source(rust_dir); + if fs::read_to_string(&at).ok().as_deref() != Some(want.as_str()) { + fs::write(&at, &want).unwrap_or_else(|e| panic!("write {}: {e}", at.display())); + } +} + +/// The key of the compiler `fork`'s sources name: their content, so committing +/// what was built as local changes names the same compiler. +pub fn key(fork: &Path) -> String { + let parts = [RECIPE.to_string(), tree_identity(fork, &KEYED), llvm_commit(fork)]; + short(parts.join("\n\0\n").as_bytes()) +} + +/// The LLVM commit `fork` builds against: the one its index records and, where +/// the submodule is checked out, the one it has, which a local checkout of +/// another commit moves. +fn llvm_commit(fork: &Path) -> String { + let recorded = git_out(fork, &["ls-files", "--stage", "--", LLVM]); + let checkout = fork.join(LLVM); + let held = if checkout.join(".git").exists() { + git_out(&checkout, &["rev-parse", "HEAD"]) + } else { + String::new() + }; + format!("{} {}", recorded.trim(), held.trim()) +} + +/// The compiler `root`'s fork checkout at `fork` names: the primary's where its +/// `compiler/` is the one the primary's was built from, and otherwise its own, +/// built if nobody has built it, held in use for as long as the returned value +/// lives. +pub fn resolve(root: &Path, rust_dir: &Path, fork: &Path, lock: &mut Held) -> Compiler { + lock.without_shared(|| choose(root, rust_dir, fork, build_in_fork)) +} + +/// [`resolve`] with the build that makes a compiler's `stage2` passed in, so a +/// test can stand in for bootstrap: `build` compiles the fork checkout it is +/// given and returns the `stage2` it left there. +fn choose(root: &Path, rust_dir: &Path, fork: &Path, build: impl Fn(&Path) -> PathBuf) -> Compiler { + let recorded = root.join(RECORD); + if fork == rust_dir { + return Compiler::primary(rust_dir); + } + let record = primary_record(rust_dir); + let built_from = fs::read_to_string(&record).unwrap_or_else(|e| { + panic!( + "{} cannot be read ({e}), so nothing says which compiler the primary's stage2 is, \ + and no worktree can know whether it names that one.\n\ + The primary checkout writes it: run `cargo run -- --build-only` there once.", + record.display(), + ) + }); + if built_from.trim() == source(fork) { + let _ = fs::remove_file(&recorded); + return Compiler::primary(rust_dir); + } + let key = key(fork); + let dir = compilers_dir(rust_dir).join(&key); + fs::create_dir_all(recorded.parent().expect("a file under target/")).ok(); + fs::write(&recorded, &key).unwrap_or_else(|e| panic!("write {}: {e}", recorded.display())); + let mut placed = false; + let using = loop { + let using = buildlock::keyed_using(root, Keyed::Compiler, &key); + if dir.join(SOURCE).is_file() { + break using; + } + drop(using); + let _building = buildlock::keyed_building(root, Keyed::Compiler, &key); + if !dir.join(SOURCE).is_file() { + place(root, fork, &key, &dir, &build); + placed = true; + } + }; + // A compiler edit loop places one per edit, and the one this replaced is + // named by nobody now; the one in use is held, so the sweep leaves it. + if placed { + for gone in sweep(root, rust_dir) { + eprintln!("Removed compiler {}: no worktree names it", gone.display()); + } + } + Compiler { stage2: dir.join("stage2"), record: dir.join(SOURCE), primary: false, _using: Some(using) } +} + +/// Build the compiler `key` names from `fork` and put it at `dir`. The caller +/// holds the key's lock. +fn place(root: &Path, fork: &Path, key: &str, dir: &Path, build: &impl Fn(&Path) -> PathBuf) { + let what = format!("building compiler {key}"); + let _worktree = buildlock::worktree_exclusive(root, &what); + eprintln!("Building compiler {key} in {}: its compiler/ is not the one the primary's was built from", fork.display()); + let stage2 = build(fork); + let partial = dir.with_extension("partial"); + if partial.exists() { + fs::remove_dir_all(&partial).unwrap_or_else(|e| panic!("remove {}: {e}", partial.display())); + } + clone_tree(&stage2, &partial.join("stage2")); + // The sources the key named are the ones built, or this is not that key's. + let again = self::key(fork); + assert!( + again == key, + "the fork's compiler sources moved while compiler {key} was being built (they are now \ + {again}); nothing was kept, and the next build makes the one they name" + ); + fs::write(partial.join(SOURCE), format!("{key}\n")) + .unwrap_or_else(|e| panic!("write {}: {e}", partial.join(SOURCE).display())); + fs::rename(&partial, dir).unwrap_or_else(|e| panic!("rename {} -> {}: {e}", partial.display(), dir.display())); +} + +/// Bootstrap's build of the compiler in `fork`, into its own build directory, +/// and the `stage2` it made, with the cargo every toolchain directory carries. +fn build_in_fork(fork: &Path) -> PathBuf { + crate::ensure_submodule(fork, "library/backtrace"); + let host = host_triple(); + let build_dir = fork.join("build/toyos-compiler"); + fs::create_dir_all(&build_dir).unwrap_or_else(|e| panic!("create {}: {e}", build_dir.display())); + let config = build_dir.join("bootstrap.toml"); + fs::write(&config, config_text(&build_dir, &host)).unwrap_or_else(|e| panic!("write {}: {e}", config.display())); + let config = config.to_str().unwrap_or_else(|| panic!("{} is not UTF-8", config.display())); + // Bootstrap re-locks both lockfiles to this worktree's `toyos-abi` and + // `toyos`; the fork's own are put back, so the checkout stays clean and the + // key stays the one it was built for. + let _locks = (Restore::holding(&fork.join("Cargo.lock")), Restore::holding(&fork.join("library/Cargo.lock"))); + let args = ["build", "--stage", "2", "--config", config, "--warnings", "warn", "compiler/rustc", "library"]; + let (ok, log) = toolchain::x_build(fork, &args, "the compiler"); + toolchain::refuse_on_compile_error(&log, "the compiler"); + assert!(ok, "the compiler build in {} failed, and nothing in its output was a compile error", fork.display()); + let stage2 = build_dir.join(&host).join("stage2"); + assert!(stage2.join("bin/rustc").is_file(), "the compiler build left no {}", stage2.join("bin/rustc").display()); + toolchain::provision_toolchain_cargo(&stage2); + toolchain::assert_toolchain_is_honest(&stage2); + stage2 +} + +/// Bootstrap's configuration for a compiler of a worktree's own: the primary's +/// `profile` and options, for the host alone, since every guest target's +/// libraries are the sysroot's to build. +fn config_text(build_dir: &Path, host: &str) -> String { + format!( + r#"change-id = "ignore" +profile = "compiler" + +[build] +build-dir = "{build_dir}" +host = ["{host}"] +target = ["{host}"] + +[rust] +incremental = true +lld = true + +[target.{host}] +{pin} +"#, + build_dir = build_dir.display(), + pin = toolchain::HOST_LINKER_PIN, + ) +} + +/// The key of the compiler of its own `root`'s last build used, if it used one. +pub fn recorded_key(root: &Path) -> Option { + fs::read_to_string(root.join(RECORD)).ok().map(|k| k.trim().to_string()) +} + +/// Remove every compiler no registered worktree records and nobody is making +/// or using, and every half-built one nobody is making. Returns what went. +pub fn sweep(root: &Path, rust_dir: &Path) -> Vec { + let dir = compilers_dir(rust_dir); + let Ok(entries) = fs::read_dir(&dir) else { return Vec::new() }; + let named: std::collections::BTreeSet = git_out(root, &["worktree", "list", "--porcelain"]) + .lines() + .filter_map(|l| l.strip_prefix("worktree ")) + .filter_map(|w| recorded_key(Path::new(w))) + .collect(); + let mut removed = Vec::new(); + for entry in entries.flatten() { + let name = entry.file_name().to_string_lossy().into_owned(); + let (key, whole) = match name.split_once('.') { + Some((key, _)) => (key.to_string(), false), + None => (name.clone(), true), + }; + if whole && named.contains(&key) { + continue; + } + let Some(_idle) = buildlock::keyed_idle(root, Keyed::Compiler, &key) else { continue }; + let path = entry.path(); + fs::remove_dir_all(&path).unwrap_or_else(|e| panic!("remove {}: {e}", path.display())); + removed.push(path); + } + removed +} + +#[cfg(test)] +mod tests { + use std::cell::Cell; + + use toyos_tmpdir::TempDir; + use std::process::Command; + + use super::*; + + fn git(dir: &Path, args: &[&str]) -> String { + let out = Command::new("git") + .args(["-c", "commit.gpgsign=false", "-c", "user.email=t@t", "-c", "user.name=t"]) + .args(["-c", "protocol.file.allow=always", "-c", "init.defaultBranch=main"]) + .args(args) + .current_dir(dir) + .output() + .expect("run git"); + assert!(out.status.success(), "git {args:?} in {}: {}", dir.display(), String::from_utf8_lossy(&out.stderr)); + String::from_utf8(out.stdout).unwrap().trim().to_string() + } + + fn write(path: &Path, text: &str) { + fs::create_dir_all(path.parent().unwrap()).unwrap(); + fs::write(path, text).unwrap(); + } + + /// Every file under `dir` with its bytes, for "nothing here changed". + fn snapshot(dir: &Path) -> Vec<(PathBuf, Vec)> { + let mut out = Vec::new(); + let mut stack = vec![dir.to_path_buf()]; + while let Some(at) = stack.pop() { + for entry in fs::read_dir(&at).unwrap().flatten() { + let path = entry.path(); + if path.is_dir() { + stack.push(path); + } else { + out.push((path.clone(), fs::read(&path).unwrap())); + } + } + } + out.sort(); + out + } + + /// A primary whose `rust` pins fork commit `C0` and has built a compiler + /// from it, and three linked worktrees: `same` pins `C0`, `a` and `b` each + /// pin a commit whose `compiler/` is its own. + fn estate(scratch: &Path) -> (PathBuf, PathBuf, [PathBuf; 3]) { + let base = fs::canonicalize(scratch).unwrap(); + + let fork = base.join("fork-src"); + fs::create_dir_all(&fork).unwrap(); + git(&fork, &["init", "-q"]); + write(&fork.join("compiler/rustc_target/src/lib.rs"), "pub fn targets() {}\n"); + write(&fork.join("src/bootstrap/src/lib.rs"), "fn main() {}\n"); + write(&fork.join("src/stage0"), "compiler_version=beta\n"); + write(&fork.join("Cargo.lock"), "# lock\n"); + write(&fork.join("library/std/src/lib.rs"), "pub fn a() {}\n"); + write(&fork.join(".gitignore"), "/build\n"); + git(&fork, &["add", "-A"]); + git(&fork, &["commit", "-qm", "C0"]); + let c0 = git(&fork, &["rev-parse", "HEAD"]); + let mut pins = Vec::new(); + for spec in ["pub fn targets() { aarch64() }\n", "pub fn targets() { riscv() }\n"] { + git(&fork, &["checkout", "-q", &c0]); + write(&fork.join("compiler/rustc_target/src/lib.rs"), spec); + git(&fork, &["commit", "-qam", "a target"]); + pins.push(git(&fork, &["rev-parse", "HEAD"])); + } + git(&fork, &["checkout", "-q", &c0]); + + let primary = base.join("primary"); + fs::create_dir_all(&primary).unwrap(); + git(&primary, &["init", "-q"]); + write(&primary.join("README"), "x\n"); + git(&primary, &["submodule", "add", "-q", fork.to_str().unwrap(), "rust"]); + git(&primary, &["add", "-A"]); + git(&primary, &["commit", "-qm", "pins C0"]); + let rust_dir = primary.join("rust"); + write(&toolchain::stage2(&rust_dir).join("bin/rustc"), "the primary's rustc"); + write(&toolchain::stage2(&rust_dir).join("lib/librustc_driver-0.dylib"), "the primary's driver"); + record(&rust_dir); + + let mut linked = Vec::new(); + for (name, pin) in [("same", c0.as_str()), ("a", pins[0].as_str()), ("b", pins[1].as_str())] { + let wt = base.join(name); + git(&primary, &["worktree", "add", "-q", "-b", name, wt.to_str().unwrap()]); + let _ = fs::remove_dir(wt.join("rust")); + git(&rust_dir, &["worktree", "add", "-q", "--detach", wt.join("rust").to_str().unwrap(), pin]); + linked.push(wt); + } + (primary, rust_dir, linked.try_into().unwrap()) + } + + /// **Two worktrees with different compilers build side by side, and the + /// primary's toolchain is untouched by either.** Each gets its own compiler + /// at its own key, built once and found again; one that names the primary's + /// `compiler/` builds nothing and gets the primary's; the primary's `stage2`, + /// its record and the rustup `toyos` link are byte-for-byte what they were; + /// a sweep takes a compiler only once no worktree names it. + #[test] + fn worktrees_with_different_compilers_coexist_and_the_primary_s_is_untouched() { + let scratch = TempDir::new("compiler"); + let (primary, rust_dir, [same, a, b]) = estate(&scratch); + let before = snapshot(&rust_dir.join("build")); + let link = toolchain::rustup_link(); + let builds = Cell::new(0); + let fake = |fork: &Path| { + builds.set(builds.get() + 1); + let stage2 = fork.join("build/toyos-compiler/stage2"); + let spec = fs::read_to_string(fork.join("compiler/rustc_target/src/lib.rs")).unwrap(); + write(&stage2.join("bin/rustc"), &format!("a rustc knowing {spec}")); + write(&stage2.join("lib/librustc_driver-1.dylib"), &spec); + stage2 + }; + + let mine = choose(&same, &rust_dir, &same.join("rust"), fake); + assert!(mine.primary && mine.stage2 == toolchain::stage2(&rust_dir)); + assert_eq!(builds.get(), 0, "a worktree naming the primary's compiler built one"); + + let ca = choose(&a, &rust_dir, &a.join("rust"), fake); + let cb = choose(&b, &rust_dir, &b.join("rust"), fake); + assert_eq!(builds.get(), 2); + assert!(!ca.primary && !cb.primary); + assert_ne!(ca.stage2, cb.stage2, "two compilers were given one directory"); + assert!(ca.stage2.starts_with(compilers_dir(&rust_dir)) && cb.stage2.starts_with(compilers_dir(&rust_dir))); + assert!(fs::read_to_string(ca.stage2.join("bin/rustc")).unwrap().contains("aarch64")); + assert!(fs::read_to_string(cb.stage2.join("bin/rustc")).unwrap().contains("riscv")); + assert_ne!(ca.identity(), cb.identity()); + assert_ne!(ca.identity(), Compiler::primary(&rust_dir).identity()); + + // Found again, not rebuilt; and still both there. + let again = choose(&a, &rust_dir, &a.join("rust"), fake); + assert_eq!((again.stage2.clone(), builds.get()), (ca.stage2.clone(), 2)); + assert!(ca.stage2.join("bin/rustc").is_file() && cb.stage2.join("bin/rustc").is_file()); + + // An uncommitted file in `compiler/` is a new compiler too, and + // committing it is not another one. + let pinned = git(&a.join("rust"), &["rev-parse", "HEAD"]); + write(&a.join("rust/compiler/rustc_target/src/new_target.rs"), "pub fn t() {}\n"); + let ca2 = choose(&a, &rust_dir, &a.join("rust"), fake); + assert_ne!(ca2.stage2, ca.stage2, "an untracked target spec kept the old compiler"); + git(&a.join("rust"), &["add", "-A"]); + git(&a.join("rust"), &["commit", "-qm", "the target, committed"]); + let committed = choose(&a, &rust_dir, &a.join("rust"), fake); + assert_eq!((committed.stage2.clone(), builds.get()), (ca2.stage2.clone(), 3), "a commit rebuilt the compiler"); + git(&a.join("rust"), &["checkout", "-q", &pinned]); + + // The primary's own: nothing under its `build/` but `compilers/` moved. + let after: Vec<_> = snapshot(&rust_dir.join("build")) + .into_iter() + .filter(|(p, _)| !p.starts_with(compilers_dir(&rust_dir))) + .collect(); + assert_eq!(after, before, "the primary's stage2 or its record was written"); + assert_eq!(git(&rust_dir, &["rev-parse", "HEAD"]), git(&primary, &["rev-parse", "HEAD:rust"])); + assert_eq!(toolchain::rustup_link(), link, "the machine-global toyos link moved"); + + // A sweep takes the compiler nobody names, and only that one — and + // not while it is still in use, though nobody names it any more. + let orphan = ca2.stage2.parent().unwrap().to_path_buf(); + choose(&a, &rust_dir, &a.join("rust"), fake); + assert_eq!(sweep(&primary, &rust_dir), Vec::::new(), "the sweep took a compiler still in use"); + assert!(ca2.stage2.is_dir()); + drop((mine, ca, cb, again, ca2, committed)); + let kept = choose(&a, &rust_dir, &a.join("rust"), fake); + assert_eq!(sweep(&primary, &rust_dir), [orphan], "the sweep took a compiler a worktree names, or left one nobody does"); + assert!(kept.stage2.is_dir()); + + // A placement sweeps too: the compiler an edit replaces goes once nobody + // uses it, and the one another worktree names stays. + let replaced = kept.stage2.parent().unwrap().to_path_buf(); + let named = choose(&b, &rust_dir, &b.join("rust"), fake).stage2; + drop(kept); + write(&a.join("rust/compiler/rustc_target/src/another.rs"), "pub fn u() {}\n"); + let ca3 = choose(&a, &rust_dir, &a.join("rust"), fake); + assert!(!replaced.exists(), "placing a compiler left the one it replaced, which nobody names"); + assert!(ca3.stage2.is_dir() && named.is_dir()); + } + /// **A primary with no record of its compiler is refused by name**, never + /// read as "no compiler": that reading made every worktree build its own. + #[test] + fn a_missing_primary_record_is_refused_and_builds_nothing() { + let scratch = TempDir::new("compiler-record"); + let (_primary, rust_dir, [same, _, _]) = estate(&scratch); + fs::remove_file(primary_record(&rust_dir)).unwrap(); + let builds = Cell::new(0); + let fake = |fork: &Path| { + builds.set(builds.get() + 1); + fork.join("build/toyos-compiler/stage2") + }; + let refused = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + choose(&same, &rust_dir, &same.join("rust"), fake); + })); + let why = refused.expect_err("a worktree resolved a compiler with no primary record"); + let why = why.downcast_ref::().cloned().unwrap_or_default(); + assert!(why.contains("toyos-compiler cannot be read"), "{why}"); + assert_eq!(builds.get(), 0, "a missing record built a compiler"); + } + + /// Every source a compiler is built from moves its key: LLVM by commit, + /// the tools by content. + #[test] + fn llvm_and_the_tools_move_the_key() { + let scratch = TempDir::new("compiler-key"); + let (_primary, _rust_dir, [same, _, _]) = estate(&scratch); + let fork = same.join("rust"); + let before = key(&fork); + write(&fork.join("src/tools/lld-wrapper/src/main.rs"), "fn main() { 1; }\n"); + let tools = key(&fork); + assert_ne!(tools, before, "a tool's source did not move the key"); + git(&fork, &["update-index", "--add", "--cacheinfo", "160000,1111111111111111111111111111111111111111,src/llvm-project"]); + assert_ne!(key(&fork), tools, "another LLVM commit did not move the key"); + } +} diff --git a/src/flags.rs b/src/flags.rs index 182f8eb7ee2..eb8fb50d0c0 100644 --- a/src/flags.rs +++ b/src/flags.rs @@ -75,6 +75,7 @@ declare_flags!(pub CARGO_RUN = { pub DIAG_BOOT = "--diag-boot", None; pub CONSOLE_BOOT = "--console-boot", None; pub BOOT_CONFIG = "--boot-config", Next; + pub ARCH = "--arch", Next; pub REGEN_FONT = "--regen-font", None; pub REGEN_WALLPAPER = "--regen-wallpaper", None; pub REGEN_SOUNDFONT = "--regen-soundfont", Next; diff --git a/src/image.rs b/src/image.rs index efc09239153..0c416a1fa30 100644 --- a/src/image.rs +++ b/src/image.rs @@ -4,6 +4,8 @@ use std::num::NonZeroU64; use std::path::Path; use bcachefs::{BlockBuf, Formatted, FsUuid, Superblock, VecBlockIO}; + +use crate::arch::Arch; use sha2::{Digest, Sha256}; use toyos_fat32::{BlockAccess, Fat32, FatTime, IoError}; @@ -173,6 +175,7 @@ pub fn update_image(kernel: &[u8], root: &[u8], params: &str, signing: Signing<' /// table, the log partition — third, where the metal loop finds it — and then /// slot A's FAT and ROOT, marked, and slot B's where `second` asks for one. pub fn create_boot_image( + arch: Arch, kernel_bytes: &[u8], bl_bytes: &[u8], root_bytes: &[u8], @@ -206,7 +209,7 @@ pub fn create_boot_image( table_volume[..toyos_update::slots::BLOCK].copy_from_slice(&table.encode()); let mut parts = vec![ - Part::full("ESP", "EFI System", gpt::partition_types::EFI, esp_guid, create_esp_volume(bl_bytes, log_guid), Some(Volume::Fat32)), + Part::full("ESP", "EFI System", gpt::partition_types::EFI, esp_guid, create_esp_volume(arch, bl_bytes, log_guid), Some(Volume::Fat32)), Part::full("slot table", "ToyOS slots", TOYOS_SLOTS, table_guid, table_volume, None), // Microsoft Basic Data, and that type is the whole reason this is a // partition of its own: macOS never auto-mounts an EFI-typed partition @@ -657,14 +660,14 @@ fn populate(volume: &mut [u8], label: &str, files: &[(&str, &[u8])]) { /// partition the kernel's log goes on. The kernel and its parameter are a /// slot's (`create_slot_volume`), because the loader is the one part of the /// machine that is not slotted. -fn create_esp_volume(bootloader: &[u8], log_guid: uuid::Uuid) -> Vec { +fn create_esp_volume(arch: Arch, bootloader: &[u8], log_guid: uuid::Uuid) -> Vec { let total_size = round_up_sectors(((bootloader.len() + ESP_FREE_BYTES) * 64 / 63).max(FAT32_MIN_BYTES)); let mut volume = format_fat32(total_size, "TOYOS-BOOT"); populate( &mut volume, "TOYOS-BOOT", &[ - ("EFI/BOOT/BOOTx64.EFI", bootloader), + (arch.removable_loader(), bootloader), // Mirrored in `bootloader/src/main.rs` as `\toyos\log.guid`, which // reads it beside itself and refuses the volume if it is not there. // The sixteen bytes are the GPT entry's own, in the entry's own @@ -1139,7 +1142,7 @@ mod tests { let key = key(); let s = sections(b"kernel", &tiny_root(), "", signing(&key)); for (what, volume) in [ - ("ESP", create_esp_volume(b"bootloader", uuid::Uuid::new_v4())), + ("ESP", create_esp_volume(Arch::X86_64, b"bootloader", uuid::Uuid::new_v4())), ("slot volume", create_slot_volume(Some(&s), FAT32_MIN_BYTES)), ("empty slot volume", create_slot_volume(None, FAT32_MIN_BYTES)), ("log volume", create_log_volume()), @@ -1169,7 +1172,7 @@ mod tests { fn a_damaged_image_is_refused_by_the_reader_that_caught_it() { let root_image = tiny_root(); let key = key(); - let disk = create_boot_image(b"kernel", b"bootloader", &root_image, "", signing(&key), None); + let disk = create_boot_image(Arch::X86_64, b"kernel", b"bootloader", &root_image, "", signing(&key), None); let log = only(&disk, toyos_gpt::Guid::MICROSOFT_BASIC); let root = root_partition_guid_of(&disk); let parts = @@ -1230,7 +1233,7 @@ mod tests { #[test] fn the_esp_and_a_slot_carry_what_the_bootloader_looks_for() { assert_eq!( - files_of(create_esp_volume(b"bootloader", uuid::Uuid::new_v4())), + files_of(create_esp_volume(Arch::X86_64, b"bootloader", uuid::Uuid::new_v4())), ["EFI/BOOT/BOOTx64.EFI", "toyos/log.guid"] ); let key = key(); @@ -1254,7 +1257,7 @@ mod tests { let root = tiny_root(); let dir = toyos_tmpdir::TempDir::new("image-slot"); let path = dir.join("slotted.img"); - let disk = create_boot_image(b"\x7fELF kernel", b"bootloader", &root, "sched-fast-health", signing(&key), Some(SecondSlot { root_bytes: 8 << 20 })); + let disk = create_boot_image(Arch::X86_64, b"\x7fELF kernel", b"bootloader", &root, "sched-fast-health", signing(&key), Some(SecondSlot { root_bytes: 8 << 20 })); std::fs::write(&path, &disk).expect("write the image"); let mut file = std::fs::File::open(&path).expect("open the image"); let table = slot_table_of(&mut file).expect("the slot table"); @@ -1301,7 +1304,7 @@ mod tests { let root_image = tiny_root(); let write = |name: &str, params: &str| { let path = dir.join(name); - std::fs::write(&path, create_boot_image(b"kernel", b"bootloader", &root_image, params, signing(&key()), None)) + std::fs::write(&path, create_boot_image(Arch::X86_64, b"kernel", b"bootloader", &root_image, params, signing(&key()), None)) .expect("write an image"); path }; @@ -1414,7 +1417,7 @@ mod tests { let root_image = create_root_image(&files, &symlinks, true); let key = key(); - let disk = create_boot_image(b"kernel", b"bootloader", &root_image, "", signing(&key), None); + let disk = create_boot_image(Arch::X86_64, b"kernel", b"bootloader", &root_image, "", signing(&key), None); // Located by *type*, through the parser the kernel uses, at the offset // the table gives — never at the one the writer computed. @@ -1513,6 +1516,7 @@ mod tests { files.push((toyos_manifest::PATH.to_string(), manifest.clone())); let disk = create_boot_image( + Arch::X86_64, b"kernel", b"bootloader", &create_root_image(&files, &symlinks, true), diff --git a/src/kernelkeys.rs b/src/kernelkeys.rs index c99d3e8eab9..bc54b91b578 100644 --- a/src/kernelkeys.rs +++ b/src/kernelkeys.rs @@ -59,8 +59,8 @@ pub const DECLARED: &[Declared] = &[ keys: "`IdKey`, which no integer implements: every key is an id this kernel issued", }, Declared { - file: "kernel/src/mm/paging.rs", - ty: "HashMap", + file: "kernel/src/arch/x86_64/paging.rs", + ty: "HashMap", keys: "a physical address the page allocator returned", }, Declared { diff --git a/src/lib.rs b/src/lib.rs index aad25394387..e44e0c0ced2 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -4,12 +4,14 @@ pub mod actuatorstate; /// What the suite's isolated re-run is allowed to call one failure; read by /// `tests/toyos.rs` and by its own tests. pub mod alone; +pub mod arch; pub mod assets; pub mod bootlog; pub mod build; pub mod buildlock; pub mod ci; pub mod clippy; +pub mod compiler; /// What the untouched-disk gate compares a device against, in `tests/`. pub mod fingerprint; pub mod flags; @@ -54,36 +56,6 @@ pub mod worktree; use std::path::{Path, PathBuf}; use std::process::Command; -/// Whether this host will actually let a guest run on KVM. -/// -/// **Presence is not permission**, and `Path::exists` cannot tell the two -/// apart. A GitHub runner ships `/dev/kvm` as `crw-rw---- root:kvm` with the -/// build user outside the group, so a check on existence puts `-accel kvm` on -/// every boot and every boot dies on `failed to initialize kvm: Permission -/// denied` — a whole suite red for a reason no test names. Any Linux box whose -/// user is not in `kvm` is that machine. Opening it is the question QEMU is -/// about to ask. -pub fn kvm_usable() -> bool { - cfg!(target_arch = "x86_64") - && std::fs::OpenOptions::new() - .read(true) - .write(true) - .open("/dev/kvm") - .is_ok() -} - -/// The CPU every guest this repository launches gets, accelerated and not. -/// -/// **One declaration, read by `cargo run` and by the harness both**, because -/// the two drifted: the harness gained `+smep` and the interactive path did -/// not, so the machine an owner looked at differed from the machine the suite -/// judged in exactly the dimension the suite had been changed for. The kernel's -/// own `CR4` comes from one declaration for the same reason. -pub const CPU_KVM: &str = "host,+rdrand,+smap,+fsgsbase,+x2apic,+smep"; -/// [`CPU_KVM`]'s emulated twin — the same features off a base model, because a -/// TCG guest that withholds one is a feature this tree stops exercising. -pub const CPU_TCG: &str = "qemu64,+rdrand,+smap,+fsgsbase,+x2apic,+smep"; - /// The `.git` directory every worktree of this repository shares. /// /// `git rev-parse --git-common-dir` answers relatively from the primary diff --git a/src/libc.rs b/src/libc.rs index 88ce2466ab3..0a3996b1663 100644 --- a/src/libc.rs +++ b/src/libc.rs @@ -2,6 +2,8 @@ use std::fs; use std::path::Path; use std::process::Command; +use crate::arch::Arch; + /// The crate every sysroot links into std, so into every userland binary. pub const CRATE: &str = "userland/libc"; @@ -11,8 +13,8 @@ pub const FEATURES: &str = "std-runtime"; /// Build toyos-libc against the toolchain at `toolchain`, in `target_dir`, and /// install it there as `libtoyos_c.a`. Part of making a sysroot /// (`src/sysroot.rs`), whose key `userland/libc/src` is one of. -pub fn build(root: &Path, toolchain: &Path, target_dir: &Path) { - let dest = toolchain.join("lib/rustlib/x86_64-unknown-toyos/lib/libtoyos_c.a"); +pub fn build(root: &Path, toolchain: &Path, target_dir: &Path, arch: Arch) { + let dest = toolchain.join(format!("lib/rustlib/{}/lib/libtoyos_c.a", arch.userland())); eprintln!("Building toyos-libc for sysroot..."); @@ -30,7 +32,7 @@ pub fn build(root: &Path, toolchain: &Path, target_dir: &Path) { "build", "--release", "--target", - "x86_64-unknown-toyos", + arch.userland(), "--features", FEATURES, "--message-format=json", diff --git a/src/licence.rs b/src/licence.rs index d3c4db3d0fb..020d1951371 100644 --- a/src/licence.rs +++ b/src/licence.rs @@ -412,6 +412,18 @@ pub const COMMITTED_FILES: &[(&str, &str, &str, Terms)] = &[ "NOTICE", Terms::Font("OFL-1.1"), ), + ( + "aavmf/AAVMF_CODE.fd", + "47765fe344818cbc464b1c14ae658fb4b854f5c2ceffa982411731eb4865594d", + "NOTICE", + Terms::Spdx("BSD-2-Clause-Patent AND Apache-2.0"), + ), + ( + "aavmf/AAVMF_VARS.fd", + "b3b855c5a80310168051164986855692d1bdb06e67619856177965cd87c6774f", + "NOTICE", + Terms::Spdx("BSD-2-Clause-Patent AND Apache-2.0"), + ), ( "ovmf/DEBUGX64_OVMF.fd", "800ff5af1220d1232d4da7173ccddbb74a9217600bd8935903d9d534801778b4", diff --git a/src/main.rs b/src/main.rs index 555a53c4dc4..22711b829f2 100644 --- a/src/main.rs +++ b/src/main.rs @@ -2,6 +2,7 @@ mod qemu; use std::env; use std::path::{Path, PathBuf}; +use toyos_build::arch::Arch; use toyos_build::flags::{self, CARGO_RUN}; /// One prerequisite: any of `any` satisfies it, and `why` is what reaches it. @@ -19,7 +20,6 @@ struct Tool { const REQUIRED: &[Tool] = &[ Tool { any: &["git"], why: "every build; the image ships what git says is tracked" }, Tool { any: &["rustup"], why: "the toolchain — install from https://rustup.rs" }, - Tool { any: &["qemu-system-x86_64"], why: "every boot — install QEMU" }, Tool { any: &["cc"], why: "rustc links every host binary through it; no guest binary" }, ]; @@ -51,7 +51,7 @@ fn executable_on_path(name: &str) -> bool { }) } -fn check_prerequisites(root: &Path) { +fn check_prerequisites(root: &Path, arch: Arch) { fn absent(tools: &'static [Tool]) -> Vec<&'static Tool> { tools.iter().filter(|t| !t.any.iter().any(|n| executable_on_path(n))).collect() } @@ -60,11 +60,15 @@ fn check_prerequisites(root: &Path) { eprintln!("Note: no {} — {}", tool.any.join(" or "), tool.why); } - let missing = absent(REQUIRED); + let mut missing: Vec = + absent(REQUIRED).iter().map(|t| format!("{} ({})", t.any.join(" or "), t.why)).collect(); + if !executable_on_path(arch.qemu()) { + missing.push(format!("{} (every {} boot — install QEMU)", arch.qemu(), arch.name())); + } if !missing.is_empty() { eprintln!("Error: missing required tools:"); for tool in &missing { - eprintln!(" - {} ({})", tool.any.join(" or "), tool.why); + eprintln!(" - {tool}"); } std::process::exit(1); } @@ -72,7 +76,7 @@ fn check_prerequisites(root: &Path) { // The one prerequisite whose *version* decides verdicts rather than whether // anything runs at all, so a scan of `PATH` cannot ask it. // `toyos_build::ci` carries why this is a note here and a red in CI. - if let Some(note) = toyos_build::ci::qemu_version_note(root) { + if let Some(note) = toyos_build::ci::qemu_version_note(root, arch) { eprintln!("{note}"); } } @@ -158,7 +162,8 @@ fn main() { } } - check_prerequisites(&root); + let arch = toyos_build::build::arch_for(&args); + check_prerequisites(&root, arch); env::set_current_dir(&root).expect("Failed to cd to project root"); let debug = asked(&flags::DEBUG); @@ -264,7 +269,7 @@ fn main() { println!("Boot image: {}", image.display()); if !build_only { - qemu::launch(&qemu::Options { debug, dump_audio, profile, smp, mute, image }); + qemu::launch(&qemu::Options { arch: plan.arch, debug, dump_audio, profile, smp, mute, image }); } } diff --git a/src/qemu.rs b/src/qemu.rs index d7332c21dee..9f77d7545d0 100644 --- a/src/qemu.rs +++ b/src/qemu.rs @@ -52,6 +52,8 @@ use std::fs::File; use std::path::PathBuf; use std::process::Command; +use toyos_build::arch::Arch; + /// The hardware shape QEMU presents to the guest. /// /// Not a display setting: each variant is a whole machine. `Virtio` and `Gop` @@ -106,6 +108,7 @@ impl Profile { } pub struct Options { + pub arch: Arch, pub debug: bool, pub dump_audio: bool, pub profile: Profile, @@ -123,7 +126,8 @@ pub struct Options { pub fn launch(opts: &Options) { let shape = opts.profile.shape(); - let mut qemu = Command::new("qemu-system-x86_64"); + let arch = opts.arch; + let mut qemu = Command::new(arch.qemu()); // Without this QEMU runs its default-device pass whenever no network // option is given, which is exactly and only the Metal profile: measured @@ -133,29 +137,31 @@ pub fn launch(opts: &Options) { // leaves i8042/ps2-kbd/ps2-mouse alone. qemu.arg("-nodefaults"); - if toyos_build::kvm_usable() { - qemu.arg("-accel").arg("kvm"); - qemu.arg("-cpu").arg(toyos_build::CPU_KVM); - } else { - qemu.arg("-cpu").arg(toyos_build::CPU_TCG); + let accel = arch.accel(); + if accel.is_hardware() { + qemu.arg("-accel").arg(accel.name()); } + qemu.arg("-cpu").arg(arch.cpu(accel)); + let [code, vars] = arch.pflash(std::path::Path::new(".")); + if let Some(boot) = arch.boot() { + qemu.arg("-boot").arg(boot); + } qemu.arg("-machine") - .arg(if shape.iommu { "q35,kernel-irqchip=split" } else { "q35" }) + .arg(machine(arch, shape.iommu)) .arg("-smp") .arg(format!("cores={}", opts.smp)) .arg("-m") .arg("2G") .arg("-drive") - .arg("if=pflash,format=raw,unit=0,file=ovmf/OVMF_CODE-pure-efi.fd,readonly=on") + .arg(code) .arg("-drive") - .arg("if=pflash,format=raw,unit=1,file=ovmf/OVMF_VARS-pure-efi.fd,readonly=on"); + .arg(vars); // Before every other `-device`: a PCI function created ahead of the unit // gets QEMU's bypassing address space and is never decoded by it. - if shape.iommu { - qemu.arg("-device") - .arg("intel-iommu,intremap=on,caching-mode=on,aw-bits=48"); + if let (true, Some(unit)) = (shape.iommu, iommu_device(arch)) { + qemu.arg("-device").arg(unit); } // Without this a virtio function keeps the machine's own address space and // the unit never sees it, whatever the tables say. @@ -192,7 +198,12 @@ pub fn launch(opts: &Options) { .arg("-device") .arg(format!("virtio-gpu-pci,xres=1280,yres=720{platform}")); } else { - qemu.arg("-vga").arg("std"); + match arch { + Arch::X86_64 => qemu.arg("-vga").arg("std"), + // A framebuffer in guest memory that firmware publishes as its + // GOP and that needs no driver after it: `virt` has no VGA. + Arch::Aarch64 => qemu.arg("-device").arg("ramfb"), + }; } if shape.virtio { @@ -270,6 +281,27 @@ pub fn launch(opts: &Options) { qemu.status().expect("failed to execute QEMU"); } +/// The machine a profile runs on, with its IOMMU where the machine carries one +/// as a property rather than a device. +fn machine(arch: Arch, iommu: bool) -> &'static str { + match (arch, iommu) { + (Arch::X86_64, true) => "q35,kernel-irqchip=split", + (Arch::X86_64, false) => "q35", + (Arch::Aarch64, true) => "virt,gic-version=3,iommu=smmuv3", + (Arch::Aarch64, false) => "virt,gic-version=3", + } +} + +/// The IOMMU as a device, in the one configuration this project builds +/// against: interrupt remapping on, caching mode on, 48-bit addresses. `virt`'s +/// SMMUv3 is a machine property ([`machine`]) and has none. +fn iommu_device(arch: Arch) -> Option<&'static str> { + match arch { + Arch::X86_64 => Some("intel-iommu,intremap=on,caching-mode=on,aw-bits=48"), + Arch::Aarch64 => None, + } +} + fn audio_backend() -> &'static str { if cfg!(target_os = "macos") { "coreaudio" diff --git a/src/release.rs b/src/release.rs index d9b08fefc38..d8f40c589b2 100644 --- a/src/release.rs +++ b/src/release.rs @@ -19,6 +19,8 @@ use std::process::{Command, Stdio}; use sha2::{Digest, Sha256}; +use crate::toolchain::HOSTED_ARCH; + /// What the tag hashes, as `git rev-parse HEAD:` names them. The last is /// this file. pub const TREES: [&str; 7] = [ @@ -198,7 +200,7 @@ fn published(root: &Path, tag: &str) -> bool { pub fn ensure_published(root: &Path) -> Result { let tag = tag(root)?; println!("this tree's toolchain: {tag}"); - if !(cfg!(target_os = "linux") && cfg!(target_arch = "x86_64")) { + if !(cfg!(target_os = "linux") && crate::arch::Arch::HOST == Some(crate::arch::Arch::X86_64)) { return Err(format!( "the release is {HOST}'s and this host is not one; a tarball built here would \ install nowhere" @@ -268,10 +270,10 @@ fn build(root: &Path, tag: &str, tmp: &Path) -> Result<(), String> { let mut tar = Command::new("tar") .arg("-C") .arg(&build) - .arg(format!("--exclude=x86_64-unknown-toyos/stage2/lib/rustlib/{HOST}")) + .arg(format!("--exclude={}/stage2/lib/rustlib/{HOST}", HOSTED_ARCH.userland())) .arg(format!("--exclude={sysroot}/bin/cargo")) .arg(format!("--transform=s,^{sysroot},{HOST}/stage2,")) - .args(["-c", &sysroot, "x86_64-unknown-toyos/stage2"]) + .args(["-c", &sysroot, &format!("{}/stage2", HOSTED_ARCH.userland())]) .args(["toyos-sysroot-witness", "toyos-ld-witness", "TOOLCHAIN"]) .stdout(Stdio::piped()) .spawn() diff --git a/src/sourcegate.rs b/src/sourcegate.rs index 9424b0e3bac..29c8ab9606a 100644 --- a/src/sourcegate.rs +++ b/src/sourcegate.rs @@ -3,9 +3,9 @@ //! **Clippy runs now, and these scans are what it cannot say.** `cargo run -- //! --ci host` runs default clippy with warnings denied over three trees //! on every merge — the host workspace -//! (`--workspace --all-targets`), the kernel (`--target x86_64-unknown-none`) -//! and the bootloader (`--target x86_64-unknown-uefi`) — so a `clippy.toml` is -//! no longer a wall with nothing behind it. +//! (`--workspace --all-targets`), and the kernel and the bootloader for every +//! architecture (`src/clippy.rs`) — so a `clippy.toml` is no longer a wall +//! with nothing behind it. //! //! What is behind it is still not these six scans. `disallowed-methods` could //! take the first one, and would lose what makes it useful: the exceptions @@ -61,6 +61,14 @@ //! workflow, script and container recipe into shell commands and refuses a //! package no row declares, the other reads `uses:`. //! +//! The tenth is the architecture rules, [`ARCH_RULES`]: each a set of +//! spellings — assembly and `core::arch` intrinsics, `target_arch`, a path into +//! one architecture's module — stated as the only places they may appear, so an +//! architecture is chosen in one place and reached only through the arch +//! interface. A declared exception is a file row that points at the issue +//! holding what it owes, and one that no longer holds a needle is refused. +//! The rust fork's std is `src/forkcheck.rs`'s to govern and is not read here. +//! //! **What none of them reaches is filed rather than implied**, and each table's //! own doc names its half: the entries under `issues/build/` say so. @@ -193,14 +201,14 @@ const BANS: &[Ban] = &[ with it", allowed: &[ // Cache lines to walk, not bytes. - ("kernel/src/arch/control_regs.rs", 1), + ("kernel/src/arch/x86_64/control_regs.rs", 1), // Two guard-page sizes: the mapping is 4 KiB because a guard is one // hardware page, and `PAGE_SIZE` is 2 MiB territory here. - ("kernel/src/arch/percpu.rs", 2), + ("kernel/src/arch/x86_64/percpu.rs", 2), // A device's TX buffer. ("kernel/src/drivers/virtio_console.rs", 1), // A VT-d table is 4 KiB by the specification, not by this kernel. - ("kernel/src/iommu/vtd/table.rs", 1), + ("kernel/src/arch/x86_64/vtd/table.rs", 1), // Ring entries. ("kernel/src/trace.rs", 1), // A path length, in bytes. @@ -212,7 +220,7 @@ const BANS: &[Ban] = &[ why: "as above, in the other width", allowed: &[ // A VT-d register window, by the specification. - ("kernel/src/iommu/vtd/mod.rs", 1), + ("kernel/src/arch/x86_64/vtd/mod.rs", 1), // The export itself, and the one place the literal lives. ("kernel/src/mm/mod.rs", 1), // One PCIe function's extended config space, by PCI 3.0 §7.2.2 and @@ -409,7 +417,7 @@ const LOG_PRODUCERS: &[&str] = &["log!(", "alert!(", "boot_phase!(", "log::emit( /// `kernel/src/hardlockup/probe.rs`, which is why it is a file of its own — and /// neither is the line the arm writes, which its caller writes for it. const NMI_SILENT: &[&str] = - &["kernel/src/arch/idt/nmi.rs", "kernel/src/hardlockup/mod.rs"]; + &["kernel/src/arch/x86_64/idt/nmi.rs", "kernel/src/hardlockup/mod.rs"]; /// Every `enable_bus_master(` site `kernel/src` holds, by file and count. /// Arming DMA comes after a site's refusals — virtio parses its capability @@ -465,7 +473,7 @@ const AUTO_TRAIT_IMPLS: &[(&str, usize)] = &[ ("kernel/src/drivers/panic_console/mod.rs", 3), ("kernel/src/drivers/virtio_console.rs", 1), ("kernel/src/drivers/virtio_sound.rs", 2), - ("kernel/src/hw.rs", 1), + ("kernel/src/arch/x86_64/hw.rs", 1), ("kernel/src/mm/mmio.rs", 2), ("kernel/src/mm/region.rs", 2), ("kernel/src/pipe.rs", 1), @@ -540,9 +548,10 @@ const HOST_SPAWNS: &[Spawn] = &[ why: "how the toolchain is installed and linked, and `REQUIRED`", }, Spawn { - arg: "\"qemu-system-x86_64\"", - sites: &[], - why: "QEMU, the other half of the bar, and `REQUIRED`", + arg: "arch.qemu()", + sites: &[("src/qemu.rs", 1), ("src/ci.rs", 2), ("tests/common/qemu.rs", 1)], + why: "QEMU, the other half of the bar: `Arch::qemu` names `qemu-system-x86_64` and \ + `qemu-system-aarch64`, and `check_prerequisites` requires the one being booted", }, Spawn { arg: "\"gh\"", @@ -1224,6 +1233,339 @@ fn tracked_rust_files(root: &Path, tree: &str) -> std::collections::BTreeSet Vec { + let code: Vec = text.lines().map(code_only).collect::>().join("\n").chars().collect(); + let word = |c: char| c.is_ascii_alphanumeric() || c == '_'; + let line_of = |at: usize| code[..at].iter().filter(|c| **c == '\n').count(); + let skip_space = |mut at: usize| { + while code.get(at).is_some_and(|c| c.is_whitespace()) { + at += 1; + } + at + }; + let ident_at = |at: usize, name: &str| { + let end = at + name.len(); + code.get(at..end).is_some_and(|s| s.iter().copied().eq(name.chars())) + && !code.get(end).is_some_and(|c| word(*c)) + && !at.checked_sub(1).and_then(|j| code.get(j)).is_some_and(|c| word(*c)) + }; + // The identifier that ends before `at`, over whitespace and `::`. + let ident_before = |at: usize| { + let mut end = at; + while end > 0 && (code[end - 1].is_whitespace() || code[end - 1] == ':') { + end -= 1; + } + let mut start = end; + while start > 0 && word(code[start - 1]) { + start -= 1; + } + (start, code[start..end].iter().collect::()) + }; + // Whether `at` begins an item that imports: `use …`, or `extern crate …`, + // or an element of a `use` group. + let imported = |at: usize| { + let mut at = at; + loop { + let (start, ident) = ident_before(at); + match ident.as_str() { + "use" => return true, + "crate" => return ident_before(start).1 == "extern", + "" => {} + _ => return false, + } + // Not after an identifier: inside a group only if a `{` opens it. + let mut back = start; + while back > 0 && code[back - 1].is_whitespace() { + back -= 1; + } + match back.checked_sub(1).map(|j| code[j]) { + Some('{') => at = back - 1, + Some(',') => { + let mut depth = 0usize; + let mut j = back - 1; + loop { + let Some(k) = j.checked_sub(1) else { return false }; + j = k; + match code[j] { + '}' => depth += 1, + '{' if depth == 0 => break, + '{' => depth -= 1, + ';' => return false, + _ => {} + } + } + at = j; + } + _ => return false, + } + } + }; + let mut lines = Vec::new(); + for at in 0..code.len() { + let Some(root) = ["core", "std"].into_iter().find(|root| ident_at(at, root)) else { continue }; + let after = skip_space(at + root.len()); + if ident_at(after, "as") && imported(at) { + lines.push(line_of(at)); + continue; + } + let colons = after; + if code.get(colons..colons + 2) != Some(&[':', ':'][..]) { + continue; + } + let next = skip_space(colons + 2); + if ident_at(next, "arch") || code.get(next) == Some(&'*') { + lines.push(line_of(next)); + } else if code.get(next) == Some(&'{') { + // Each element of the group begins after its `{` or a `,` at depth one. + let (mut depth, mut i, mut element) = (0usize, next, true); + while let Some(&c) = code.get(i) { + match c { + '{' => { + depth += 1; + element = depth == 1; + } + '}' => { + depth -= 1; + if depth == 0 { + break; + } + } + ',' if depth == 1 => element = true, + c if c.is_whitespace() => {} + _ => { + let renamed_self = + ident_at(i, "self") && ident_at(skip_space(i + "self".len()), "as"); + if element && depth == 1 && (ident_at(i, "arch") || c == '*' || renamed_self) { + lines.push(line_of(i)); + } + element = false; + } + } + i += 1; + } + } + } + lines +} + +/// The spelling of the first needle of `rule` that line `n` of a file holds, +/// `code` being that line's code and `arch_lines` the file's +/// [`arch_module_lines`]. +#[cfg(test)] +fn needle_on(rule: &PlaceRule, code: &str, n: usize, arch_lines: &[usize]) -> Option<&'static str> { + rule.needles + .iter() + .copied() + .find(|needle| code.contains(needle)) + .or_else(|| (rule.arch_module && arch_lines.contains(&n)).then_some(ARCH_MODULE)) +} + +/// The spellings that name one architecture's module from outside it. +#[cfg(test)] +const ARCH_PATHS: &[&str] = &["arch::x86_64", "arch::aarch64"]; + +/// The pure crates: `CLAUDE.md`'s layout marks every one of them pure, and +/// `toyos-sched` is the scheduler core behind `Machine`/`Hw`. +#[cfg(test)] +const PURE_CRATES: &[&str] = &[ + "toyos-blockhold/", + "toyos-desktop/", + "toyos-dma/", + "toyos-elide/", + "toyos-hda/", + "toyos-mixer/", + "toyos-pci/", + "toyos-proclife/", + "toyos-sched/", + "toyos-userbound/", + "toyos-wallclock/", +]; + +/// Where issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md +/// holds what is owed, for the rows that cite it. +#[cfg(test)] +const USERLAND_ASM: &str = + "declared, not placed: issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md"; + +#[cfg(test)] +const ARCH_RULES: &[PlaceRule] = &[ + PlaceRule { + name: "assembly lives in an architecture's own module", + needles: ASSEMBLY, + arch_module: true, + scope: &[], + places: &[ + ("kernel/src/arch/x86_64/", "the kernel's x86-64 module"), + ("kernel/src/arch/aarch64/", "the kernel's AArch64 module"), + ("bootloader/src/arch/", "the loader's architecture module"), + ("toyos-abi/src/syscall.rs", "toyos-abi's per-architecture syscall entry"), + ("userland/libc/src/arch/", "libc's architecture modules"), + ("userland/metalprobe/src/arch/", "metalprobe's architecture modules"), + ("userland/toyos-window/src/arch/", "toyos-window's architecture modules"), + ("tests/toyos-rust-tests/src/bin/abuse_kernel_addr.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/abuse_tls_alloc.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/debug_trap.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/demand_paging_sse.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/fault_gate_child.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/fpu_isolation.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/gsbase_probe.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/inventory_bounds.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/log_hold.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/mmap_prot.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/nmi_window_spin.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/partition_claimant.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/process_lifecycle.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/syscall_cost.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/tlb_shootdown_waits.rs", USERLAND_ASM), + ("tests/toyos-rust-tests/src/bin/wake_storm_cost.rs", USERLAND_ASM), + ], + }, + PlaceRule { + name: "an architecture is selected in one place", + needles: &["target_arch"], + arch_module: false, + scope: &[], + places: &[ + ("kernel/src/arch/mod.rs", "the kernel's one selector"), + ("kernel/src/arch/x86_64/", "the kernel's x86-64 module"), + ("kernel/src/arch/aarch64/", "the kernel's AArch64 module"), + ("bootloader/src/arch/", "the loader's architecture module and its selector"), + ("toyos-abi/src/syscall.rs", "toyos-abi's per-architecture syscall entry"), + ("src/arch.rs", "the build system's one reading of the host it runs on"), + ( + "toyos-cc/src/preprocess/mod.rs", + "the C compiler's default target is its host's, when the command line names none", + ), + ( + "src/licence.rs", + "the licence gate resolves a dependency's `cfg(target_arch = ...)` against the target an image is built for: the architecture is the data it reads, not a choice it makes", + ), + ("userland/libc/src/arch/", "libc's architecture modules and their selector"), + ("userland/toyos-window/src/arch/", "toyos-window's architecture modules and their selector"), + ("userland/metalprobe/src/arch/", "metalprobe's architecture modules and their selector"), + ], + }, + PlaceRule { + name: "generic kernel code reaches the machine through the arch interface", + needles: ARCH_PATHS, + arch_module: false, + scope: &["kernel/src/"], + places: &[("kernel/src/arch/", "the architecture's own modules")], + }, + PlaceRule { + name: "a pure crate names no architecture", + needles: &[ + "asm!", "global_asm!", "naked_asm!", "#[naked]", "unsafe(naked)", "target_arch", "arch::x86_64", + "arch::aarch64", + ], + arch_module: true, + scope: PURE_CRATES, + places: &[], + }, +]; + +/// Every `file:line` in `files` that breaks `rule`, as the red names it. +#[cfg(test)] +fn place_violations(rule: &PlaceRule, files: &[(String, String)]) -> Vec { + let in_scope = + |at: &str| rule.scope.is_empty() || rule.scope.iter().any(|tree| at.starts_with(tree)); + let placed = |at: &str| rule.places.iter().any(|(place, _)| at.starts_with(place)); + let mut found = Vec::new(); + for (at, text) in files { + if !in_scope(at) || placed(at) { + continue; + } + let arch_lines = if rule.arch_module { arch_module_lines(text) } else { Vec::new() }; + for (n, line) in text.lines().enumerate() { + let code = code_only(line); + if let Some(needle) = needle_on(rule, &code, n, &arch_lines) { + let places: Vec = + rule.places.iter().map(|(place, why)| format!("{place} ({why})")).collect(); + found.push(format!( + "{at}:{}: `{needle}` breaks \"{}\"; it may appear only in {}", + n + 1, + rule.name, + if places.is_empty() { "no file at all".to_string() } else { places.join(", ") }, + )); + } + } + } + found +} + +/// Every `.rs` file this repository holds, as `(repository-relative path, +/// contents)`: everything outside [`NOT_OURS`], build output and dotted +/// directories. +#[cfg(test)] +fn every_rust_file() -> Vec<(String, String)> { + let root = repo_root(); + let mut files = Vec::new(); + let Ok(entries) = std::fs::read_dir(&root) else { return Vec::new() }; + let mut entries: Vec<_> = entries.filter_map(Result::ok).map(|e| e.path()).collect(); + entries.sort(); + for path in entries { + let name = path.file_name().unwrap_or_default().to_string_lossy().to_string(); + if name.starts_with('.') || name == "target" || name == NOT_OURS { + continue; + } + if path.is_dir() { + rust_files(&path, &mut files); + } else if path.extension().is_some_and(|e| e == "rs") { + files.push(path); + } + } + files + .into_iter() + .filter_map(|path| Some((rel(&root, &path), std::fs::read_to_string(&path).ok()?))) + .collect() +} + /// The third-party C corpus, whose attribution is per *population* rather than /// per file: `tests/testcases/LICENSE` states how many files each of these /// directories holds, and `NOTICE` points at that file for the terms. @@ -1268,6 +1610,133 @@ fn digest(bytes: &[u8]) -> String { mod tests { use super::*; + /// **The architecture rules hold over the whole repository**, and no place + /// a rule names is a place that no longer holds a needle — a row nobody + /// needs any more is a permission nobody re-argued. + #[test] + fn every_architecture_rule_holds() { + let files = every_rust_file(); + assert!( + files.iter().any(|(at, _)| at == "kernel/src/arch/mod.rs"), + "the walk did not reach the kernel's selector, so it is reading no tree" + ); + let mut complaints = Vec::new(); + for rule in ARCH_RULES { + complaints.extend(place_violations(rule, &files)); + // A module is a place a needle may stand; a file is an exception + // somebody declared, and one that holds nothing any more is stale. + for (place, _) in rule.places.iter().filter(|(place, _)| !place.ends_with('/')) { + let used = files.iter().filter(|(at, _)| at.starts_with(place)).any(|(_, text)| { + let arch_lines = if rule.arch_module { arch_module_lines(text) } else { Vec::new() }; + text.lines().enumerate().any(|(n, l)| needle_on(rule, &code_only(l), n, &arch_lines).is_some()) + }); + if !used { + complaints.push(format!( + "\"{}\" names {place} as a place, and nothing there spells any of {:?}", + rule.name, rule.needles + )); + } + } + } + assert!(complaints.is_empty(), "{}", complaints.join("\n")); + } + + fn rule(name: &str) -> &'static PlaceRule { + ARCH_RULES.iter().find(|r| r.name == name).unwrap_or_else(|| panic!("no rule {name}")) + } + + fn file(at: &str, text: &str) -> (String, String) { + (at.to_string(), text.to_string()) + } + + /// Each rule's own fixture: a violation planted where the rule forbids it + /// reds, naming the rule and the places, and the same line where the rule + /// allows it does not. + #[test] + fn assembly_outside_an_arch_module_is_red() { + let asm = rule("assembly lives in an architecture's own module"); + let planted = [ + file("kernel/src/sched/driver.rs", " unsafe { core::arch::asm!(\"nop\") };\n"), + file("bootloader/src/main.rs", "#[unsafe(naked)]\n"), + file("toyos/src/window.rs", " let t = core::arch::x86_64::_rdtsc();\n"), + // The module named by any path: renamed, grouped, over lines. + file("toyos/src/net.rs", "use core::arch as isa;\n"), + file("toyos/src/shm.rs", "use core::{arch::x86_64::_rdtsc};\n"), + file("toyos/src/ipc.rs", "use ::std::{\n fmt,\n arch::asm,\n};\n"), + ]; + let said = place_violations(asm, &planted); + assert_eq!(said.len(), 6, "{said:?}"); + assert!(said.iter().any(|s| s.starts_with("toyos/src/ipc.rs:3:")), "{said:?}"); + assert!(said.iter().all(|s| s.contains(asm.name) && s.contains("kernel/src/arch/x86_64/")), "{said:?}"); + let placed = [ + file("kernel/src/arch/aarch64/cpu.rs", " unsafe { core::arch::asm!(\"wfi\") };\n"), + file("kernel/src/sched/driver.rs", " // core::arch::asm! is the architecture's.\n"), + file("toyos/src/lib.rs", "use core::{fmt::{self, arch}, ptr};\nlet s = \"core::arch\";\nuse mycore::arch;\n"), + ]; + assert_eq!(place_violations(asm, &placed), Vec::::new()); + } + + #[test] + fn an_architecture_chosen_outside_the_selector_is_red() { + let select = rule("an architecture is selected in one place"); + let said = place_violations(select, &[file("kernel/src/main.rs", "#[cfg(target_arch = \"x86_64\")]\n")]); + assert_eq!(said.len(), 1, "{said:?}"); + assert!(said[0].contains(select.name) && said[0].contains("kernel/src/arch/mod.rs"), "{said:?}"); + assert!(place_violations(select, &[file("kernel/src/arch/mod.rs", "#[cfg(target_arch = \"aarch64\")]\n")]).is_empty()); + } + + #[test] + fn a_path_into_one_architecture_from_generic_code_is_red() { + let reach = rule("generic kernel code reaches the machine through the arch interface"); + let said = place_violations(reach, &[file("kernel/src/process.rs", "use crate::arch::x86_64::percpu;\n")]); + assert_eq!(said.len(), 1, "{said:?}"); + assert!(said[0].contains(reach.name) && said[0].contains("kernel/src/arch/"), "{said:?}"); + // Outside the kernel the rule does not read; inside `arch/` it allows. + assert!(place_violations(reach, &[file("kernel/src/arch/x86_64/boot.rs", "use crate::arch::x86_64::percpu;\n")]).is_empty()); + } + + #[test] + fn a_pure_crate_that_names_an_architecture_is_red() { + let pure = rule("a pure crate names no architecture"); + let planted = [ + file("toyos-sched/src/cpu.rs", "#[cfg(target_arch = \"aarch64\")]\n"), + file("toyos-desktop/src/lib.rs", " core::arch::asm!(\"nop\");\n"), + file("toyos-dma/src/lib.rs", "use crate::arch::aarch64::x;\n"), + file("toyos-sched/src/lib.rs", "use core::arch as isa;\nfn f() { unsafe { isa::aarch64::vdupq_n_u8(0) }; }\n"), + file("toyos-mixer/src/lib.rs", "use core::{arch::x86_64::_rdtsc};\n"), + // A glob or a rename of the root brings `arch` in under a name no + // line scan resolves; each is red at the import, and the use after + // it is spelled so that no other needle matches. + file("toyos-elide/src/lib.rs", "use core::*;\nuse arch::{aarch64::vdupq_n_u8};\n"), + file("toyos-hda/src/lib.rs", "use core as k;\nuse k::arch::{aarch64::vdupq_n_u8};\n"), + file("toyos-pci/src/lib.rs", "extern crate core as k;\nuse k::arch::{aarch64::vdupq_n_u8};\n"), + file("toyos-wallclock/src/lib.rs", "use core::{self as k};\nuse k::arch::{aarch64::vdupq_n_u8};\n"), + file("toyos-userbound/src/lib.rs", "use std::{\n *,\n};\n"), + file("toyos-proclife/src/lib.rs", "use {alloc::vec, std as k};\n"), + ]; + let said = place_violations(pure, &planted); + assert_eq!(said.len(), 11, "{said:?}"); + assert!(said.iter().any(|s| s.starts_with("toyos-sched/src/lib.rs:1:")), "{said:?}"); + for at in [ + "toyos-elide/src/lib.rs:1:", + "toyos-hda/src/lib.rs:1:", + "toyos-pci/src/lib.rs:1:", + "toyos-wallclock/src/lib.rs:1:", + "toyos-userbound/src/lib.rs:2:", + "toyos-proclife/src/lib.rs:1:", + ] { + assert!(said.iter().any(|s| s.starts_with(at) && s.contains(ARCH_MODULE)), "{at} is not red: {said:?}"); + } + assert!(said.iter().all(|s| s.contains(pure.name) && s.contains("no file at all")), "{said:?}"); + + // What is not the root, or not an import, is not a rename of it. + let unrelated = [ + file("toyos-sched/src/cpu.rs", "fn f(core: u8) -> u32 { g(1, core as u32) + { core as u32 } }\n"), + file("toyos-dma/src/lib.rs", "use crate::core as c;\nuse a::{b, core as k};\n"), + ]; + assert_eq!(place_violations(pure, &unrelated), Vec::::new()); + } + /// **Two clauses, one rule each, and neither is checkable any other way.** /// /// The first: the NMI handler must not log. It would reenter its own CPU's @@ -1653,7 +2122,7 @@ mod tests { } } assert!( - found.iter().any(|(arg, _, _)| arg == "\"qemu-system-x86_64\""), + found.iter().any(|(arg, _, _)| arg == "arch.qemu()"), "the walk did not find the QEMU launch, so it is reading no host tree" ); assert!( diff --git a/src/sysroot.rs b/src/sysroot.rs index 9d1bf204483..0c5905e20e1 100644 --- a/src/sysroot.rs +++ b/src/sysroot.rs @@ -6,12 +6,12 @@ //! built from: the three trees std and `libtoyos_c.a` compile //! ([`SYSROOT_SOURCES`]), the std fork's `library/` and `src/bootstrap/` in the //! checkout that builds it, and the compiler that builds it. `rust/build/ -//! sysroots//` is a whole toolchain — the compiler's files cloned from the -//! primary's `stage2`, the guest targets' libraries built from this key's -//! sources — and nothing writes it after its [`SOURCES`] file exists. A build -//! compiles against the directory its own key names, so two worktrees with -//! different ABIs never refuse or wait for each other, and main and every branch -//! matching it share one copy. +//! sysroots//` is a whole toolchain — the compiler's files cloned from its +//! `stage2`, the guest targets' libraries built from this key's sources — and +//! nothing writes it after its [`SOURCES`] file exists. A build compiles against +//! the directory its own key names, so two worktrees with different ABIs or +//! different compilers never refuse or wait for each other, and main and every +//! branch matching it share one copy. //! //! **Each worktree builds std in its own fork checkout, and nothing but the //! primary's own sync moves the primary's.** The primary builds in its `rust/`; @@ -19,16 +19,16 @@ //! the primary's fork repository at the commit this tree pins ([`fork_checkout`]). //! `library/std` names `toyos-abi` and `toyos` as `../../../`, so each //! checkout's std compiles against its own worktree's ABI with nothing -//! rewritten. The build is bootstrap's stage-0 local rebuild: the primary's -//! `stage2` compiler compiles the checkout's `library/` for the guest targets -//! into `/build/toyos-std/`. The compiler is built once, by the -//! primary, and a fork commit whose `compiler/` is not the one it was built from -//! is refused by name ([`check_compiler`]). +//! rewritten. The build is bootstrap's stage-0 local rebuild: the compiler the +//! checkout names (`src/compiler.rs` — the primary's `stage2`, or one of the +//! worktree's own where its `compiler/` differs) compiles the checkout's +//! `library/` for the guest targets into `/build/toyos-std/`. //! -//! Locks, in the one order every acquirer takes them: the key's -//! (`buildlock::sysroot_*`), with this worktree's build lock put down; then, to -//! build, this worktree's exclusively (its fork build directory is written); -//! then the global one shared, because the primary's compiler is read. +//! Locks, in the one order every acquirer takes them: the compiler key's, if the +//! compiler is a worktree's own; the sysroot key's (`buildlock::keyed_*`), with +//! this worktree's build lock put down; then, to build, this worktree's +//! exclusively (its fork build directory is written); then, if the compiler is +//! the primary's, the global one shared, because it is read. //! //! A sysroot no worktree names any more is removed by [`sweep`], which //! `--worktree remove` runs: each build records the key it used in its @@ -42,7 +42,9 @@ use std::process::Command; use sha2::{Digest, Sha256}; -use crate::buildlock::{self, Guard, Held}; +use crate::arch::Arch; +use crate::buildlock::{self, Guard, Held, Keyed}; +use crate::compiler::{self, Compiler}; use crate::identity; use crate::toolchain::{self, host_triple, Owner, GUEST_TARGETS}; @@ -60,18 +62,12 @@ const SOURCES: &str = "SOURCES"; /// What changes how a key's sources become a sysroot and is none of them: the /// std build's recipe below. Moving it moves every key. -const RECIPE: &str = "bootstrap stage-0 local rebuild, profile compiler, targets toyos none uefi, \ +const RECIPE: &str = "bootstrap stage-0 local rebuild, profile compiler, \ libtoyos_c merged, libraries from the stamp; 2"; /// Where each build records the key it compiled against, for [`sweep`]. const RECORD: &str = "target/toyos-sysroot-key"; -/// The compiler `stage2` was built from, written by the primary: see -/// [`record_compiler`]. -fn compiler_record(rust_dir: &Path) -> PathBuf { - rust_dir.join("build/toyos-compiler") -} - /// Every sysroot on this host. pub fn sysroots_dir(rust_dir: &Path) -> PathBuf { rust_dir.join("build/sysroots") @@ -81,6 +77,9 @@ pub fn sysroots_dir(rust_dir: &Path) -> PathBuf { pub struct Sysroot { /// A toolchain directory: `RUSTUP_TOOLCHAIN` names it. pub dir: PathBuf, + /// Whether its compiler is the primary's, which the ToyOS-hosted rustc is + /// built from. + pub primary_compiler: bool, _using: Option, } @@ -89,7 +88,7 @@ impl Sysroot { /// artifact's, which `toolchain::check_installed_toolchain` has matched to /// these sources. pub(crate) fn installed(stage2: PathBuf) -> Self { - Self { dir: stage2, _using: None } + Self { dir: stage2, primary_compiler: true, _using: None } } } @@ -98,7 +97,7 @@ fn hex(digest: &[u8]) -> String { } /// The first 16 hex digits of the SHA-256 of `data`. -fn short(data: &[u8]) -> String { +pub(crate) fn short(data: &[u8]) -> String { hex(&Sha256::digest(data))[..16].to_string() } @@ -155,7 +154,7 @@ pub fn witness(root: &Path) -> String { /// covers, into every submodule checked out there — never what a build or the /// desktop leaves beside them (bootstrap's `__pycache__`, Finder's /// `.DS_Store`), which would make a key that moves while it is being built. -fn tree_identity(base: &Path, paths: &[&str]) -> String { +pub(crate) fn tree_identity(base: &Path, paths: &[&str]) -> String { let mut files = Vec::new(); source_files(base, paths, &mut files); files.sort(); @@ -188,80 +187,14 @@ fn source_files(checkout: &Path, paths: &[&str], out: &mut Vec) { } } -/// The compiler as the key sees it: the source it was built from, as -/// [`record_compiler`] wrote it, and the driver that build left, so a rebuild -/// of the same source is a new compiler too. -fn compiler_identity(rust_dir: &Path) -> String { - let record = compiler_record(rust_dir); - let source = fs::read_to_string(&record).unwrap_or_else(|_| { - panic!( - "{} is missing, so no sysroot can say which compiler it was built with.\n\ - The primary checkout writes it: run `cargo run -- --build-only` in {} once.", - record.display(), - rust_dir.parent().unwrap_or(rust_dir).display(), - ) - }); - let lib = toolchain::stage2(rust_dir).join("lib"); - let driver = fs::read_dir(&lib) - .unwrap_or_else(|e| panic!("read {}: {e}", lib.display())) - .flatten() - .find(|e| e.file_name().to_string_lossy().starts_with("librustc_driver")) - .unwrap_or_else(|| panic!("{} holds no librustc_driver", lib.display())); - let meta = driver.metadata().unwrap_or_else(|e| panic!("stat the driver: {e}")); - let mtime = meta - .modified() - .ok() - .and_then(|t| t.duration_since(std::time::UNIX_EPOCH).ok()) - .map_or(0, |d| d.as_nanos()); - format!("{} {} {} {mtime}", source.trim(), driver.file_name().to_string_lossy(), meta.len()) -} - -/// What `checkout`'s `compiler/` is: its commit's tree, and whatever the working -/// tree changes in it. -fn compiler_source(checkout: &Path) -> String { - let tree = git_out(checkout, &["rev-parse", "HEAD:compiler"]); - let local = git_bytes(checkout, &["diff", "HEAD", "--", "compiler"]); - if local.is_empty() { - tree.trim().to_string() - } else { - format!("{} with local changes {}", tree.trim(), short(&local)) - } -} - -/// Record which compiler `stage2` is. The primary calls this after a toolchain -/// build, and when the record is missing — its compiler stamp has just said -/// `stage2` is built from what its `rust/` holds. -pub fn record_compiler(rust_dir: &Path) { - let record = compiler_record(rust_dir); - let want = compiler_source(rust_dir); - if fs::read_to_string(&record).ok().as_deref() != Some(want.as_str()) { - fs::write(&record, &want).unwrap_or_else(|e| panic!("write {}: {e}", record.display())); - } -} - -/// Refuse a fork checkout whose `compiler/` is not the one `stage2` was built -/// from: its std would be compiled by a compiler it was not written for. -fn check_compiler(rust_dir: &Path, fork: &Path) { - let record = compiler_record(rust_dir); - let built = fs::read_to_string(&record).unwrap_or_default(); - let here = compiler_source(fork); - assert!( - built.trim() == here, - "{} holds the fork at a `compiler/` ({here}) the shared compiler was not built from \ - ({}).\nA compiler change is built once, by the primary: land it, and the primary's \ - sync and next build rebuild the compiler every worktree uses.", - fork.display(), - if built.is_empty() { "nothing recorded" } else { built.trim() }, - ); -} - -/// The key of the sysroot `root` builds against with its std fork at `fork`. -pub fn key(root: &Path, rust_dir: &Path, fork: &Path) -> String { +/// The key of the sysroot `root` builds against with its std fork at `fork`, +/// compiled by `compiler`. +pub fn key(root: &Path, compiler: &Compiler, fork: &Path) -> String { let parts = [ - format!("{RECIPE}; cargo {STAGE0_CARGO}"), + format!("{RECIPE}; cargo {STAGE0_CARGO}; targets {}", GUEST_TARGETS.join(" ")), witness(root), tree_identity(fork, &["library", "src/bootstrap"]), - compiler_identity(rust_dir), + compiler.identity(), ]; short(parts.join("\n\0\n").as_bytes()) } @@ -374,53 +307,55 @@ fn finished(dir: &Path) -> bool { /// held in use for as long as the returned value lives. pub fn ensure(root: &Path, rust_dir: &Path, lock: &mut Held) -> Sysroot { let fork = fork_checkout(root); - if fork != rust_dir { - check_compiler(rust_dir, &fork); - } - let key = key(root, rust_dir, &fork); + let compiler = compiler::resolve(root, rust_dir, &fork, lock); + let key = key(root, &compiler, &fork); let dir = sysroots_dir(rust_dir).join(&key); let record = root.join(RECORD); fs::create_dir_all(record.parent().expect("a file under target/")).ok(); fs::write(&record, &key).unwrap_or_else(|e| panic!("write {}: {e}", record.display())); let using = lock.without_shared(|| loop { - let using = buildlock::sysroot_using(root, &key); + let using = buildlock::keyed_using(root, Keyed::Sysroot, &key); if finished(&dir) { break using; } drop(using); - let _building = buildlock::sysroot_building(root, &key); + let _building = buildlock::keyed_building(root, Keyed::Sysroot, &key); if !finished(&dir) { - build(root, rust_dir, &fork, &key, &dir); + build(root, &compiler, &fork, &key, &dir); } }); toolchain::assert_toolchain_is_honest(&dir); - Sysroot { dir, _using: Some(using) } + Sysroot { dir, primary_compiler: compiler.primary, _using: Some(using) } } /// Make the sysroot `key` names at `dir`, from `root`'s sources and the std fork -/// at `fork`. The caller holds the key's lock. -fn build(root: &Path, rust_dir: &Path, fork: &Path, key: &str, dir: &Path) { +/// at `fork`, with `compiler`. The caller holds the key's lock. +fn build(root: &Path, compiler: &Compiler, fork: &Path, key: &str, dir: &Path) { let what = format!("building sysroot {key}"); let _worktree = buildlock::worktree_exclusive(root, &what); - let _compiler = buildlock::compiler_shared(root, &what); - eprintln!("Building sysroot {key}: std from {}, the compiler from {}", fork.display(), rust_dir.display()); + // Only the primary's compiler is rebuilt in place; one of a worktree's own + // is written once and held in use by `compiler`. + let _compiler = compiler.primary.then(|| buildlock::compiler_shared(root, &what)); + eprintln!("Building sysroot {key}: std from {}, the compiler {}", fork.display(), compiler.stage2.display()); - let built = build_std(root, rust_dir, fork); + let built = build_std(root, compiler, fork); let partial = dir.with_extension("partial"); if partial.exists() { fs::remove_dir_all(&partial).unwrap_or_else(|e| panic!("remove {}: {e}", partial.display())); } - clone_tree(&toolchain::stage2(rust_dir), &partial); + clone_tree(&compiler.stage2, &partial); for target in GUEST_TARGETS { place_std(&stamp(&built, target), &partial.join("lib/rustlib").join(target).join("lib")); } let libc_target = dir.with_extension("libc-target"); - crate::libc::build(root, &partial, &libc_target); + for arch in Arch::ALL { + crate::libc::build(root, &partial, &libc_target, arch); + } let _ = fs::remove_dir_all(&libc_target); // The sources the key named are the ones built, or this is not that key's. - let again = self::key(root, rust_dir, fork); + let again = self::key(root, compiler, fork); assert!( again == key, "the sources moved while sysroot {key} was being built (they are now {again}); \ @@ -435,12 +370,10 @@ fn build(root: &Path, rust_dir: &Path, fork: &Path, key: &str, dir: &Path) { .unwrap_or_else(|e| panic!("rename {} -> {}: {e}", partial.display(), dir.display())); } -/// Compile the guest targets' libraries from `fork`'s `library/` with the -/// primary's compiler, and return the directory each target's is under. -fn build_std(root: &Path, rust_dir: &Path, fork: &Path) -> PathBuf { - if fork == rust_dir { - crate::ensure_submodule(fork, "library/backtrace"); - } +/// Compile the guest targets' libraries from `fork`'s `library/` with +/// `compiler`, and return the directory each target's is under. +fn build_std(root: &Path, compiler: &Compiler, fork: &Path) -> PathBuf { + crate::ensure_submodule(fork, "library/backtrace"); let host = host_triple(); let build_dir = fork.join("build/toyos-std"); fs::create_dir_all(&build_dir).unwrap_or_else(|e| panic!("create {}: {e}", build_dir.display())); @@ -450,7 +383,7 @@ fn build_std(root: &Path, rust_dir: &Path, fork: &Path) -> PathBuf { let _ = fs::remove_dir_all(build_dir.join(&host).join("stage0-std").join(target)); } let config = build_dir.join("bootstrap.toml"); - fs::write(&config, std_config(rust_dir, &build_dir, &host, &toolchain::toyos_ld_binary(root))) + fs::write(&config, std_config(&compiler.stage2, &build_dir, &host, &toolchain::toyos_ld_binary(root))) .unwrap_or_else(|e| panic!("write {}: {e}", config.display())); // Bootstrap re-locks `library/Cargo.lock` to this worktree's `toyos-abi` and @@ -464,7 +397,9 @@ fn build_std(root: &Path, rust_dir: &Path, fork: &Path) -> PathBuf { let (ok, log) = toolchain::x_build(fork, &args, "std"); toolchain::refuse_on_compile_error(&log, "std"); assert!(ok, "the std build failed, and nothing in its output was a compile error"); - toolchain::assert_std_built_from(root, &build_dir.join(&host).join("stage0-std/x86_64-unknown-toyos")); + for arch in Arch::ALL { + toolchain::assert_std_built_from(root, &build_dir.join(&host).join("stage0-std").join(arch.userland())); + } build_dir.join(&host).join("stage0-std") } @@ -517,8 +452,20 @@ fn place_std(stamp: &Path, lib: &Path) { /// `local-rebuild` is what lets stage 0 compile the library for a target the /// stage-0 compiler has none for, and `profile = "compiler"` is the primary's, /// so these libraries are built with the options `stage2`'s own were. -fn std_config(rust_dir: &Path, build_dir: &Path, host: &str, toyos_ld: &Path) -> String { +fn std_config(compiler: &Path, build_dir: &Path, host: &str, toyos_ld: &Path) -> String { let targets = GUEST_TARGETS.iter().map(|t| format!("\"{t}\"")).collect::>().join(", "); + let linker = toyos_ld.display(); + let userland: String = Arch::ALL + .iter() + .map(|arch| { + let linker = if arch.links_through_toyos_ld() { + format!("linker = \"{linker}\"") + } else { + format!("linker = \"{}\"\nrpath = false", toolchain::rust_lld(compiler).display()) + }; + format!("\n[target.{}]\n{linker}\n", arch.userland()) + }) + .collect(); format!( r#"change-id = "ignore" profile = "compiler" @@ -533,14 +480,10 @@ target = [{targets}] [rust] lld = false - -[target.x86_64-unknown-toyos] -linker = "{linker}" -"#, - rustc = toolchain::stage2(rust_dir).join("bin/rustc").display(), +{userland}"#, + rustc = compiler.join("bin/rustc").display(), cargo = bootstrap_cargo().display(), build_dir = build_dir.display(), - linker = toyos_ld.display(), ) } @@ -570,13 +513,13 @@ fn bootstrap_cargo() -> PathBuf { } /// A file put back to its bytes when this drops, however the scope ends. -struct Restore { +pub(crate) struct Restore { path: PathBuf, bytes: Vec, } impl Restore { - fn holding(path: &Path) -> Self { + pub(crate) fn holding(path: &Path) -> Self { let bytes = fs::read(path).unwrap_or_else(|e| panic!("read {}: {e}", path.display())); Self { path: path.to_path_buf(), bytes } } @@ -592,7 +535,7 @@ impl Drop for Restore { /// Copy `from` to `to`, a symbolic link as a link: `stage2`'s own point at /// things that outlive it. `fs::copy` clones on APFS and reflinks where Linux /// can, so a sysroot costs the bytes its own libraries differ by. -fn clone_tree(from: &Path, to: &Path) { +pub(crate) fn clone_tree(from: &Path, to: &Path) { fs::create_dir_all(to).unwrap_or_else(|e| panic!("create {}: {e}", to.display())); for entry in fs::read_dir(from).unwrap_or_else(|e| panic!("read {}: {e}", from.display())).flatten() { let src = entry.path(); @@ -633,7 +576,7 @@ pub fn sweep(root: &Path) -> Vec { if whole && named.contains(&key) { continue; } - let Some(_idle) = buildlock::sysroot_idle(root, &key) else { continue }; + let Some(_idle) = buildlock::keyed_idle(root, Keyed::Sysroot, &key) else { continue }; let path = entry.path(); fs::remove_dir_all(&path).unwrap_or_else(|e| panic!("remove {}: {e}", path.display())); removed.push(path); @@ -645,7 +588,7 @@ fn path_str(path: &Path) -> &str { path.to_str().unwrap_or_else(|| panic!("{} is not UTF-8", path.display())) } -fn git_bytes(dir: &Path, args: &[&str]) -> Vec { +pub(crate) fn git_bytes(dir: &Path, args: &[&str]) -> Vec { let out = Command::new("git") .args(args) .current_dir(dir) @@ -660,7 +603,7 @@ fn git_bytes(dir: &Path, args: &[&str]) -> Vec { out.stdout } -fn git_out(dir: &Path, args: &[&str]) -> String { +pub(crate) fn git_out(dir: &Path, args: &[&str]) -> String { String::from_utf8_lossy(&git_bytes(dir, args)).into_owned() } @@ -712,7 +655,7 @@ mod tests { write(&fork.join(".gitignore"), "__pycache__\n.DS_Store\n"); git(&fork, &["init", "-q"]); let rust_dir = base.join("rust"); - write(&compiler_record(&rust_dir), "tree-1"); + write(&rust_dir.join("build/toyos-compiler"), "tree-1"); write(&toolchain::stage2(&rust_dir).join("lib/librustc_driver-1.dylib"), "a driver"); (root, rust_dir, fork) } @@ -724,7 +667,7 @@ mod tests { fn a_comment_is_the_same_sysroot_and_a_signature_is_another() { let base = TempDir::new("key"); let (root, rust_dir, fork) = keyed(&base); - let k = || key(&root, &rust_dir, &fork); + let k = || key(&root, &Compiler::primary(&rust_dir), &fork); let base = k(); assert_eq!(base.len(), 16, "{base}"); @@ -758,7 +701,7 @@ mod tests { write(&root.join("toyos-abi/Cargo.toml"), "[package]\nversion = \"0.1.0\"\n"); assert_eq!(k(), base); - write(&compiler_record(&rust_dir), "tree-2"); + write(&rust_dir.join("build/toyos-compiler"), "tree-2"); assert_ne!(k(), base, "another compiler kept the old sysroot"); } @@ -865,7 +808,7 @@ mod tests { } write(&root.join(RECORD), "named"); write(&linked.join(RECORD), "linked-named"); - let using = buildlock::sysroot_using(&root, "in-use"); + let using = buildlock::keyed_using(&root, Keyed::Sysroot, "in-use"); let mut removed = sweep(&root); removed.sort(); diff --git a/src/tiers.rs b/src/tiers.rs index 6713cb8999c..f0c57216b9f 100644 --- a/src/tiers.rs +++ b/src/tiers.rs @@ -11,6 +11,9 @@ //! enforces it: a name that is Nightly because its verdict is anchored to real //! time, or because it shares a boot with one that is slow, stays where it is //! whatever it measures. +//! +//! The local tier is the third, and the only one CI never runs: its guests are +//! of an architecture no hosted runner has been measured to boot. /// Which run a registered test belongs to. #[derive(Clone, Copy, PartialEq, Eq, Debug)] @@ -19,4 +22,21 @@ pub enum Tier { Fast, /// `cargo test --test toyos-build -- --nightly`. Nightly, + /// Every `cargo test` on a developer's machine and no sharded run: a guest + /// of an architecture no CI runner boots yet. The AArch64 port's stage 8 + /// (`issues/kernel/toyos-runs-on-arm64.md`) measures the runners and + /// moves these rows to `Fast` or `Nightly`. + Local, +} + +impl Tier { + /// Whether a run selects this tier: `nightly` is `--nightly`, `sharded` + /// a `--shard`, which only CI's jobs pass. + pub fn selected(self, nightly: bool, sharded: bool) -> bool { + match self { + Self::Fast => true, + Self::Nightly => nightly, + Self::Local => !sharded, + } + } } diff --git a/src/toolchain.rs b/src/toolchain.rs index fc3b80e8ef3..413519e64f8 100644 --- a/src/toolchain.rs +++ b/src/toolchain.rs @@ -4,6 +4,7 @@ use std::path::{Path, PathBuf}; use std::process::Command; use std::sync::OnceLock; +use crate::arch::Arch; use crate::buildlock; use crate::buildlock::Scope; use crate::stamps; @@ -85,8 +86,18 @@ const STD_SOURCES: [&str; 2] = ["toyos-abi/src", "toyos/src"]; /// `src/build.rs`'s external fingerprint. A fifth spelling would silently leave /// one of them building or fingerprinting a different set of targets than the /// others. -pub const GUEST_TARGETS: [&str; 3] = - ["x86_64-unknown-toyos", "x86_64-unknown-none", "x86_64-unknown-uefi"]; +pub const GUEST_TARGETS: [&str; 6] = [ + Arch::X86_64.userland(), + Arch::X86_64.kernel(), + Arch::X86_64.loader(), + Arch::Aarch64.userland(), + Arch::Aarch64.kernel(), + Arch::Aarch64.loader(), +]; + +/// The one ToyOS the hosted rustc (`system.toml`'s `hosted-rustc`) is built to +/// run on. +pub const HOSTED_ARCH: Arch = Arch::X86_64; /// The primary's compiler, which every sysroot is cloned from and compiled by. pub(crate) fn stage2(rust_dir: &Path) -> PathBuf { @@ -217,7 +228,7 @@ pub(crate) fn rustup_home() -> Option { } /// Where the machine-global `toyos` rustup toolchain currently points. -fn rustup_link() -> Option { +pub(crate) fn rustup_link() -> Option { fs::read_link(rustup_home()?.join("toolchains/toyos")).ok() } @@ -306,7 +317,7 @@ fn cargo_link_stale(stage2: &Path) -> bool { /// has, and a copy would put a 32 MB host binary into a 401 MiB artifact to /// stand in for a file the consumer can make in a microsecond. `Owner::Installed` /// makes it, exactly as it makes the host target. -fn provision_toolchain_cargo(stage2: &Path) { +pub(crate) fn provision_toolchain_cargo(stage2: &Path) { let at = stage2.join("bin/cargo"); let _ = fs::remove_file(&at); std::os::unix::fs::symlink(host_cargo(), &at).unwrap_or_else(|e| { @@ -314,7 +325,8 @@ fn provision_toolchain_cargo(stage2: &Path) { }); } -/// Refuse a toolchain layout that would make rustup narrate. +/// Refuse a toolchain layout that would make rustup narrate, or that has no +/// linker for the guest targets that name `rust-lld`. /// /// Unconditional and after the step that provisions, because the defect being /// gated is a provisioning step that silently stopped running: a check that only @@ -331,6 +343,21 @@ pub(crate) fn assert_toolchain_is_honest(stage2: &Path) { narrated.join(" and "), if narrated.len() == 1 { "it" } else { "them" }, ); + let lld = rust_lld(stage2); + assert!( + lld.is_file(), + "the toyos toolchain at {} carries no {}, the linker every guest target that does not \ + link through toyos-ld names: bootstrap puts it there when `write_config` says \ + `lld = true`, and it did not", + stage2.display(), + lld.display(), + ); +} + +/// The linker the guest targets name, as the toolchain at `toolchain` carries +/// it: `lib/rustlib//bin/rust-lld`, where rustc itself looks for it. +pub(crate) fn rust_lld(toolchain: &Path) -> PathBuf { + toolchain.join("lib/rustlib").join(host_triple()).join("bin/rust-lld") } /// Ensure the toolchain is up to date, and return the sysroot this checkout's @@ -464,7 +491,7 @@ pub fn ensure(root: &Path, force_rebuild: bool, lock: &mut buildlock::Held) -> S eprintln!("Building full toolchain (this takes a while on first run)..."); full_bootstrap(root, &rust_dir); stamps::write_dir_stamp(&rust_dir.join("compiler"), &compiler_stamp); - sysroot::record_compiler(&rust_dir); + crate::compiler::record(&rust_dir); if kind.invalidate_hosted { let _ = fs::remove_file(&hosted_stamp); } @@ -476,10 +503,10 @@ pub fn ensure(root: &Path, force_rebuild: bool, lock: &mut buildlock::Held) -> S Scope::Global, "record which compiler the toolchain is", || (!rust_dir.join("build/toyos-compiler").exists()).then_some(()), - |()| sysroot::record_compiler(&rust_dir), + |()| crate::compiler::record(&rust_dir), ); - let hosted_rustc = rust_dir.join("build/x86_64-unknown-toyos/stage2/bin/rustc"); + let hosted_rustc = rust_dir.join(format!("build/{}/stage2/bin/rustc", HOSTED_ARCH.userland())); lock.act_if( Scope::Global, "build the ToyOS-hosted rustc", @@ -733,10 +760,12 @@ fn full_bootstrap(root: &Path, rust_dir: &Path) { ); tolerated_failure(&log, "the toolchain build"); } - assert_std_built_from( - root, - &rust_dir.join(format!("build/{host}/stage1-std/x86_64-unknown-toyos")), - ); + for arch in Arch::ALL { + assert_std_built_from( + root, + &rust_dir.join(format!("build/{host}/stage1-std/{}", arch.userland())), + ); + } } fn build_hosted_rustc(rust_dir: &Path, toyos_ld: &Path) { @@ -748,7 +777,7 @@ fn build_hosted_rustc(rust_dir: &Path, toyos_ld: &Path) { refuse_on_compile_error(&log, "the hosted rustc"); // rustdoc for ToyOS may fail to link; rustc and librustc_driver may not. - let toyos_stage2 = rust_dir.join("build/x86_64-unknown-toyos/stage2"); + let toyos_stage2 = rust_dir.join(format!("build/{}/stage2", HOSTED_ARCH.userland())); assert!( toyos_stage2.join("bin/rustc").exists(), "the hosted rustc build failed and {} is not there.\n\ @@ -766,15 +795,44 @@ fn build_hosted_rustc(rust_dir: &Path, toyos_ld: &Path) { if !ok { tolerated_failure(&log, "the hosted rustc build"); } - // No config restore needed — full_bootstrap writes the - // cross-only config before they run, so the next non-hosted build - // always starts with the correct config regardless of what's on disk. + + // That build reassembled the host's `stage2` without `rust-lld` + // (`write_config` says why), so the host-only build runs once more to put + // it back: everything it would compile is already built. + write_config(rust_dir, &host_triple(), toyos_ld, false); + let (ok, log) = x_build( + rust_dir, + &["build", "--stage", "2", "--warnings", "warn"], + "the toolchain, reassembled", + ); + refuse_on_compile_error(&log, "the toolchain, reassembled"); + assert!( + rust_lld(&stage2(rust_dir)).is_file(), + "the toolchain's reassembly after the hosted rustc left no {}", + rust_lld(&stage2(rust_dir)).display() + ); + if !ok { + tolerated_failure(&log, "the toolchain's reassembly"); + } } +/// `bootstrap.toml` for the host-only toolchain, or with the ToyOS-hosted rustc. +/// +/// `lld = true` is what puts `rust-lld` in every stage's sysroot, where rustc +/// finds the linker the targets that do not link through toyos-ld name. The +/// hosted rustc's build cannot have it: bootstrap would then build LLD for the +/// ToyOS host from C++, which nothing here can compile. Every assemble removes +/// the host's `stage2` first, so [`build_hosted_rustc`] reassembles it under +/// the host-only config after. +/// +/// The host's `default-linker-linux-override` is pinned off because bootstrap +/// otherwise ties it to `lld` for `x86_64-unknown-linux-gnu`, and a host rustc +/// whose build environment flips with the config is rebuilt by each of those +/// two builds. fn write_config(rust_dir: &Path, host: &str, toyos_ld: &Path, with_hosted_rustc: bool) { let linker = toyos_ld.display(); let host_line = if with_hosted_rustc { - format!("host = [\"{host}\", \"x86_64-unknown-toyos\"]") + format!("host = [\"{host}\", \"{}\"]", HOSTED_ARCH.userland()) } else { format!("host = [\"{host}\"]") }; @@ -788,6 +846,21 @@ fn write_config(rust_dir: &Path, host: &str, toyos_ld: &Path, with_hosted_rustc: .map(|t| format!("\"{t}\"")) .collect::>() .join(", "); + let userland: String = Arch::ALL + .iter() + .map(|arch| { + let backends = if *arch == HOSTED_ARCH { codegen_backends } else { "" }; + // rust-lld by name: rustc finds it in the sysroot of the stage that + // links, which `lld = true` puts it in. It takes no `-Wl,` rpath, + // and a guest std has no host library path to record. + let linker = if arch.links_through_toyos_ld() { + format!("linker = \"{linker}\"") + } else { + "linker = \"rust-lld\"\nrpath = false".to_string() + }; + format!("[target.{}]\n{linker}{backends}\n\n", arch.userland()) + }) + .collect(); let config = format!( r#"change-id = "ignore" profile = "compiler" @@ -798,16 +871,21 @@ target = [{targets}] [rust] incremental = true -lld = false +lld = {lld} -[target.x86_64-unknown-toyos] -linker = "{linker}"{codegen_backends} +[target.{host}] +{HOST_LINKER_PIN} -"# +{userland}"#, + lld = !with_hosted_rustc, ); fs::write(rust_dir.join("bootstrap.toml"), config).unwrap(); } +/// What the host rustc links its own binaries with, held to one answer in every +/// `bootstrap.toml` that builds a host compiler: [`write_config`] says why. +pub(crate) const HOST_LINKER_PIN: &str = "default-linker-linux-override = \"off\""; + /// Path to the host toyos-ld binary (stable location, never wiped by sysroot rebuilds). /// /// The workspace root's `target/`, not `toyos-ld/target/`: `toyos-ld` is a @@ -955,14 +1033,14 @@ fn host_sysroot() -> PathBuf { /// Whether the ToyOS sysroot is missing the host target proc-macros compile against. fn host_target_missing(rust_dir: &Path) -> bool { - let toyos_sysroot = rust_dir.join("build/x86_64-unknown-toyos/stage2/lib/rustlib"); + let toyos_sysroot = rust_dir.join(format!("build/{}/stage2/lib/rustlib", HOSTED_ARCH.userland())); toyos_sysroot.exists() && !toyos_sysroot.join(host_triple()).exists() } fn link_host_target(rust_dir: &Path) { let host = host_triple(); let host_target_dir = rust_dir - .join("build/x86_64-unknown-toyos/stage2/lib/rustlib") + .join(format!("build/{}/stage2/lib/rustlib", HOSTED_ARCH.userland())) .join(&host); let source = host_sysroot().join("lib/rustlib").join(&host); @@ -1046,6 +1124,16 @@ mod tests { provision_toolchain_cargo(&stage2); assert!(narrated_binaries(&bin).is_empty()); assert!(!cargo_link_stale(&stage2)); + + // Nothing narrates, and the toolchain is still refused: it has no linker. + let refused = std::panic::catch_unwind(|| assert_toolchain_is_honest(&stage2)) + .expect_err("a toolchain with no rust-lld is refused"); + let said = refused.downcast_ref::().expect("a formatted refusal"); + assert!(said.contains("rust-lld"), "the refusal names the linker: {said}"); + + let lld = rust_lld(&stage2); + fs::create_dir_all(lld.parent().unwrap()).unwrap(); + fs::write(&lld, b"").unwrap(); assert_toolchain_is_honest(&stage2); } diff --git a/src/worktree.rs b/src/worktree.rs index 75fc40e1182..cb2e968cee5 100644 --- a/src/worktree.rs +++ b/src/worktree.rs @@ -308,7 +308,8 @@ fn gib(bytes: u64) -> String { /// a worktree of anything. Once git has let go of it nothing in it is work, /// so the whole directory goes. /// -/// Then every sysroot no remaining worktree names goes too (`src/sysroot.rs`). +/// Then every sysroot and every compiler no remaining worktree names goes too +/// (`src/sysroot.rs`, `src/compiler.rs`). pub(crate) fn remove(root: &Path, path: &str) { let at = root.join(path); remove_fork_checkout(root, &at); @@ -326,6 +327,10 @@ pub(crate) fn remove(root: &Path, path: &str) { if !swept.is_empty() { eprintln!("removed {} sysroot(s) no worktree names any more", swept.len()); } + let swept = crate::compiler::sweep(root, &crate::toolchain::rust_dir(root)); + if !swept.is_empty() { + eprintln!("removed {} compiler(s) no worktree names any more", swept.len()); + } } /// A linked worktree's own fork checkout (`src/sysroot.rs`'s `fork_checkout`) diff --git a/tests/common/audio.rs b/tests/common/audio.rs index 583fefa38ba..fa9a199a4d4 100644 --- a/tests/common/audio.rs +++ b/tests/common/audio.rs @@ -448,7 +448,7 @@ pub struct SounddCounters { /// The two numbers this boot drew for its clocks, off the kernel's own boot /// lines: the TSC period against the HPET (`kernel/src/clock.rs`) and the LAPIC -/// timer's tick rate against that (`kernel/src/arch/apic.rs`). +/// timer's tick rate against that (`kernel/src/arch/x86_64/apic.rs`). /// /// **They are here because they are the only per-boot draws that scale every /// armed timer for the boot's whole life**, which is the shape diff --git a/tests/common/compile.rs b/tests/common/compile.rs index 7a003507a67..58f2b35eeaa 100644 --- a/tests/common/compile.rs +++ b/tests/common/compile.rs @@ -28,7 +28,7 @@ fn libc_archive_toyos() -> PathBuf { ARCHIVE .get_or_init(|| { let libc_dir = libc_dir(); - let target = "x86_64-unknown-toyos"; + let target = super::qemu::SUITE_ARCH.userland(); let _slot = toyos_build::buildlock::build_slot(&repo_root(), "the libc archive"); let mut lock = toyos_build::buildlock::shared(&repo_root(), "toyos-libc archive"); @@ -85,7 +85,7 @@ pub fn compile_c(name: &str) -> (Vec, Vec>) { let opts = toyos_cc::CompileOptions { include_paths, defines: Vec::new(), - target: Some("x86_64-unknown-toyos".to_string()), + target: Some(super::qemu::SUITE_ARCH.userland().to_string()), opt_level: 0, force_includes: Vec::new(), }; diff --git a/tests/common/faults.rs b/tests/common/faults.rs index b30ea23488f..a00c2e86b56 100644 --- a/tests/common/faults.rs +++ b/tests/common/faults.rs @@ -899,7 +899,7 @@ pub fn syscall_window_nmi( /// from. /// /// **The decision itself rather than a second reading of it**: `qemu_command` -/// puts `-accel kvm` there when `toyos_build::kvm_usable()` says so, and +/// puts `-accel kvm` there when `SUITE_ARCH.accel()` says so, and /// `profile_argv` is that same builder. A CPUID probe in the guest would be a /// second place that can be told the wrong answer, and `virtio_net_no_msix` and /// `diskless_boot` already assert about a boot by reading its argv. diff --git a/tests/common/logread.rs b/tests/common/logread.rs index 0fa51bb8f80..731f349b003 100644 --- a/tests/common/logread.rs +++ b/tests/common/logread.rs @@ -233,7 +233,7 @@ pub fn log_reserve_window( /// `log::nested`'s handler emits: exactly one shard generation. const BURST: u64 = 512; -/// The negative control on [`log_reserve_window`], and on `LogCommitGuard` +/// The negative control on [`log_reserve_window`], and on `arch::IrqGuard` /// itself: the same boot with the reserve bracket removed. /// /// **The one thing that can make the log's correctness claim fail on purpose.** diff --git a/tests/common/metal.rs b/tests/common/metal.rs index 11f5ca965af..3e3b86c0c9a 100644 --- a/tests/common/metal.rs +++ b/tests/common/metal.rs @@ -800,14 +800,14 @@ fn build( let identity = super::ssh::Identity::mint_in(&talk_home(&home))?; extra.push((super::ssh::KEYS_ON_ROOT.to_string(), identity.authorized_line().into_bytes())); } - let plan = toyos_build::build::Plan::new(&config, features, ¶ms); + let plan = toyos_build::build::Plan::new(toyos_build::arch::Arch::X86_64, &config, features, ¶ms); let bytes = toyos_build::build::build_test_image(root, &plan, quiet, &extra); let image = home.join("image.img"); std::fs::write(&image, &bytes).map_err(|e| format!("{}: {e}", image.display()))?; // The binary the second invocation sends, copied now so it is this // build's and not whatever the tree holds when the machine is reached. if let Some(service) = batch.swap { - toyos_build::build::copy_guest_program(root, service, &home.join(service))?; + toyos_build::build::copy_guest_program(root, toyos_build::arch::Arch::X86_64, service, &home.join(service))?; } Ok(image) } diff --git a/tests/common/passcost.rs b/tests/common/passcost.rs index 2dd72008c46..ff741280c06 100644 --- a/tests/common/passcost.rs +++ b/tests/common/passcost.rs @@ -232,7 +232,7 @@ pub const TCG: Baseline = Baseline { /// The recorded sample for the accelerator this run is actually using. pub fn baseline() -> &'static Baseline { - if toyos_build::kvm_usable() { + if super::qemu::SUITE_ARCH.accel().is_hardware() { &KVM } else { &TCG diff --git a/tests/common/power.rs b/tests/common/power.rs index 28dee8d8106..970ad8b78cb 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -685,7 +685,7 @@ pub fn watchdog_fed( } /// The kernel's read-back above its own arm, in -/// `kernel/src/drivers/watchdog.rs`: whole clauses, one per branch. +/// `kernel/src/arch/x86_64/watchdog.rs`: whole clauses, one per branch. const ARMED_ON_ARRIVAL: &str = "so the bootloader had already armed the timer"; /// Unreachable from this suite: every guest that reaches the kernel's arm /// passed the parameter, and the loader read the same one first. @@ -968,7 +968,7 @@ pub fn panic_before_peripherals_reboots( // The one thing this branch removed. A guest reaching the other held branch // for the other reason must not be read as this one passing. boot.must_not_say("decoded no reset register")?; - if !held.contains("states no TSC frequency") { + if !held.contains("states no counter frequency") { return Err(format!( "the panel held for a reason this guest was not expected to reach\n{held}" )); @@ -1728,7 +1728,7 @@ const PANIC_OUTLIVES_DEADLINE: &str = "boot-deadline=4000"; /// page that crosses the reset says which of the two ended the machine. /// /// **What this cannot judge, stated rather than implied.** Reverting -/// `deadline::stand_down` alone leaves this green: after `apic::halt_all_cpus` +/// `deadline::stand_down` alone leaves this green: after `panic::halt_all_cpus` /// every CPU is halted or spinning with `IF` clear, so nothing reaches the poll /// and an armed deadline cannot expire whether or not it was disarmed. The /// window the stand-down closes is the one *before* that — the panicking CPU has @@ -2495,7 +2495,7 @@ pub fn blackbox_early_panic_sealed_muted( Ok(()) } -/// `kernel/src/arch/apic.rs`'s `LOG_DRAIN_EXPIRED`, which a boot that never had +/// `kernel/src/panic.rs`'s `LOG_DRAIN_EXPIRED`, which a boot that never had /// a drainer may not print. const LOG_DRAIN_EXPIRED: &str = "the report did not reach /log"; diff --git a/tests/common/qemu.rs b/tests/common/qemu.rs index cfa4e181a02..ba20f12e25a 100644 --- a/tests/common/qemu.rs +++ b/tests/common/qemu.rs @@ -9,6 +9,12 @@ use std::time::{Duration, Instant}; use std::{fs, thread}; use super::compile; +use toyos_build::arch::{Accel, Arch}; + +/// The architecture every machine this suite builds and boots is: the suite's +/// q35 shapes, i8042 and VT-d are x86-64's, and the aarch64 bring-up boots +/// through its own launcher ([`boot_bringup`]). +pub const SUITE_ARCH: Arch = Arch::X86_64; /// When true, serial output is printed to stderr as it arrives. pub static VERBOSE: AtomicBool = AtomicBool::new(false); @@ -1368,6 +1374,69 @@ pub enum Profile { /// that is the driver's work. The negative control on the whole bind path /// — a first-match kernel would go green on every other HDA test. HdaTwoLive, + /// QEMU `virt` on AArch64 (GICv3, AAVMF): a GOP from `ramfb`, the boot + /// stick on an xHCI, the PL011, and nothing else — no virtio, NIC, NVMe or + /// IOMMU. The machine the AArch64 port reaches its console on, and the only + /// profile that is not a q35. + Virt, + /// [`Profile::Virt`] with EL2 (`virtualization=on`), emulated on `-cpu max`: + /// firmware then hands the loader the CPU at EL2, and the kernel's entry + /// has to drop from it. HVF gives a guest EL1 only. + VirtEl2, +} + +impl Profile { + /// The architecture this machine is. + pub fn arch(self) -> Arch { + match self { + Self::Virt | Self::VirtEl2 => Arch::Aarch64, + Self::Headless + | Self::HeadlessNoIommu + | Self::VirtioNetNoMsix + | Self::E1000e + | Self::E1000eNoServer + | Self::E1000eBesideIgb + | Self::Gop + | Self::VirtioGpu + | Self::Metal + | Self::MetalNoUsb + | Self::InternalDisk + | Self::MetalUsb + | Self::MetalDisk + | Self::Diskless + | Self::NvmeWideSector + | Self::UsbDisk + | Self::UsbDisk4k + | Self::UsbDiskHuge + | Self::UsbDiskReadOnly + | Self::NvmeBootUsbDisk + | Self::UsbDiskRefusedFirst + | Self::UsbDiskCrowd + | Self::MetalFullSpeed + | Self::MetalXhciSecond + | Self::MetalXhciBoth + | Self::MetalXhciMsi + | Self::MetalXhciNoIrq + | Self::MetalXhciDeaf + | Self::MetalHotplug + | Self::NoIommu + | Self::IommuNarrow + | Self::IommuNoIntremap + | Self::IommuEim + | Self::Hda + | Self::HdaTwoLive => Arch::X86_64, + } + } + + /// How this host provides the machine: [`Profile::VirtEl2`] emulated, + /// since only emulation gives a guest EL2; every other as its + /// architecture's own. + pub fn accel(self) -> Accel { + match self { + Self::VirtEl2 => Accel::Tcg, + _ => self.arch().accel(), + } + } } /// The vIOMMU a profile puts on the machine. @@ -1691,6 +1760,22 @@ pub const NVME_T14_BLOCKS: u64 = NVME_T14_BYTES / 4096; impl Profile { fn shape(self) -> Shape { match self { + Self::VirtEl2 => Self::Virt.shape(), + Self::Virt => Shape { + vga: "std", + panel: None, + gpu: None, + virtio: Virtio::Absent, + nic: Nic::Absent, + xhci: &[XHCI_DEFAULT], + storage_bus: "xhci.0", + usb: &[], + nvme_bytes: 0, + nvme_lba_bytes: NVME_LBA_DEFAULT, + usb_disks: &[], + hda: &[], + iommu: None, + }, Self::Headless => Shape { vga: "none", panel: None, @@ -2735,7 +2820,7 @@ pub fn build_boot_image_carrying( } else { toyos_build::build::TEST_KERNEL }; - build_boot_image_with(test_crate, c_tests, rust_tests, staged, kernel, kernel_params, false) + build_boot_image_with(SUITE_ARCH, test_crate, c_tests, rust_tests, staged, kernel, kernel_params, false) } /// Refuse a staged [`BootOptions::boot_image`] that is not the image this @@ -2817,7 +2902,10 @@ fn kernel_of(options: &BootOptions) -> Vec<&'static str> { toyos_build::build::TEST_KERNEL.to_vec() } +// Eight, because an image is its architecture as much as its files and its kernel. +#[allow(clippy::too_many_arguments)] fn build_boot_image_with( + arch: Arch, test_crate: &Path, c_tests: &[(String, Vec)], rust_tests: &[(String, Vec)], @@ -2866,6 +2954,14 @@ fn build_boot_image_with( toyos_build::build::DEBUG_KERNEL_BUILD, ); KERNELS.lock().expect("the kernel census").insert(joined); + // The suite's programs are built for one architecture, and a ROOT of + // another carries none of them. + assert!( + arch == SUITE_ARCH || (c_tests.is_empty() && rust_tests.is_empty()), + "a {} image was handed programs built for {}", + arch.name(), + SUITE_ARCH.name() + ); let mut extra_files: Vec<(String, Vec)> = Vec::new(); for (name, data) in c_tests { extra_files.push((format!("bin/test_c_{name}"), data.clone())); @@ -2887,7 +2983,7 @@ fn build_boot_image_with( ); let quiet = !VERBOSE.load(Ordering::Relaxed); - let plan = toyos_build::build::Plan::new(&config_path, kernel_features, kernel_params); + let plan = toyos_build::build::Plan::new(arch, &config_path, kernel_features, kernel_params); toyos_build::build::build_test_image(&compile::repo_root(), &plan, quiet, &extra_files) } @@ -2895,7 +2991,7 @@ fn build_boot_image_with( pub fn build_toyos_bins(crate_path: &Path) -> Vec<(String, Vec)> { let repo = compile::repo_root(); let quiet = !VERBOSE.load(Ordering::Relaxed); - toyos_build::build::build_toyos_bins(&repo, crate_path, quiet) + toyos_build::build::build_toyos_bins(&repo, SUITE_ARCH, crate_path, quiet) } /// All kernel serial output goes through log!() which prepends "[kernel ...]". @@ -2982,6 +3078,7 @@ impl QemuInstance { let params = options.params(); let params: Vec<&str> = params.iter().map(String::as_str).collect(); build_boot_image_with( + options.profile.arch(), test_crate, c_tests, rust_tests, @@ -4311,14 +4408,18 @@ fn qemu_command( "mute removes the only console a virtio profile has" ); + let arch = options.profile.arch(); let repo = compile::repo_root(); - let ovmf_dir = repo.join("ovmf"); + let [firmware_code, firmware_vars] = arch.pflash(&repo); - let mut qemu = Command::new("qemu-system-x86_64"); + let mut qemu = Command::new(arch.qemu()); + if let Some(boot) = arch.boot() { + qemu.arg("-boot").arg(boot); + } - let kvm = toyos_build::kvm_usable(); - if kvm { - qemu.arg("-accel").arg("kvm"); + let accel = options.profile.accel(); + if accel.is_hardware() { + qemu.arg("-accel").arg(accel.name()); } // Without this QEMU runs its default-device pass whenever no network @@ -4334,7 +4435,18 @@ fn qemu_command( // `kernel-irqchip=split` only when there is a unit: interrupt remapping // needs the userspace half of the irqchip, and a machine with no unit has // no reason to be built differently from the one it has always been. - let mut machine = String::from("q35"); + let mut machine = match arch { + Arch::X86_64 => String::from("q35"), + Arch::Aarch64 => { + // `virt` has no i8042 to take away, and the unit a profile declares + // is VT-d, which it has none of either. + assert!(options.i8042 && shape.iommu.is_none(), "`virt` has neither an i8042 nor VT-d"); + String::from(match options.profile { + Profile::VirtEl2 => "virt,gic-version=3,virtualization=on", + _ => "virt,gic-version=3", + }) + } + }; if !options.i8042 { machine.push_str(",i8042=off"); } @@ -4346,26 +4458,27 @@ fn qemu_command( qemu.arg("-rtc").arg(format!("base={base}")); } + // `virt` puts RAM at 1 GiB and AAVMF allocates from its top, so with 4 GiB + // the loader's allocations land past the 4 GiB its boot map reaches and it + // refuses the boot: issues/boot-media/the-boot-map-reaches-4-gib-and-firmware-decides-what-lands-in-it.md. + let memory = match arch { + Arch::X86_64 => "4G", + Arch::Aarch64 => "2G", + }; qemu.arg("-machine") .arg(&machine) .arg("-cpu") - .arg(if kvm { toyos_build::CPU_KVM } else { toyos_build::CPU_TCG }) + .arg(arch.cpu(accel)) .arg("-smp") .arg(options.smp.to_string()) .arg("-m") - .arg("4G") + .arg(memory) .arg("-drive") - .arg(format!( - "if=pflash,format=raw,unit=0,file={},readonly=on", - ovmf_dir.join("OVMF_CODE-pure-efi.fd").display() - )) + .arg(firmware_code) .arg("-drive") .arg(match &options.firmware_vars { Some(vars) => format!("if=pflash,format=raw,unit=1,file={},readonly=off", vars.display()), - None => format!( - "if=pflash,format=raw,unit=1,file={},readonly=on", - ovmf_dir.join("OVMF_VARS-pure-efi.fd").display() - ), + None => firmware_vars, }) .arg("-drive") .arg(format!( @@ -4480,11 +4593,24 @@ fn qemu_command( ); qemu.arg("-device").arg(format!("{gpu}{platform}")); } - qemu.arg("-vga").arg(shape.vga).arg("-display").arg("none"); + match (arch, shape.vga) { + (Arch::X86_64, vga) => { + qemu.arg("-vga").arg(vga); + } + // `virt` has no VGA: a GOP there is firmware's over `ramfb`, a + // framebuffer in guest memory that needs no driver after it. + (Arch::Aarch64, "std") => { + qemu.arg("-device").arg("ramfb"); + } + (Arch::Aarch64, "none") => {} + (Arch::Aarch64, other) => panic!("`virt` has no `-vga {other}`"), + } + qemu.arg("-display").arg("none"); if !options.takes_the_reset { qemu.arg("-no-reboot"); } if let Some((w, h)) = shape.panel { + assert_eq!(arch, Arch::X86_64, "a panel is declared through VGA's EDID, and `virt` has no VGA"); // A panel on a machine with no VGA adapter is a declaration nothing // emits, which is the silently-inert field this suite refuses by name. assert_eq!( diff --git a/tests/common/serial.rs b/tests/common/serial.rs index e497d02d723..84bec400f00 100644 --- a/tests/common/serial.rs +++ b/tests/common/serial.rs @@ -36,7 +36,7 @@ use super::qemu::{is_kernel_line, QemuInstance}; #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub enum Died { /// The kernel itself. Every path that writes one of these words ends at - /// `apic::halt_all_cpus` — **unless** the panic handler finds the panic + /// `panic::halt_all_cpus` — **unless** the panic handler finds the panic /// recoverable, which it does for a `panic!` taken in syscall context /// (`kernel/src/main.rs`: the caller is killed and the machine carries on, /// which is what `panic_recovery`, `heap_ceiling` and @@ -45,7 +45,7 @@ pub enum Died { /// dying", and the guest going quiet afterwards is what says it meant it. Kernel, /// A process the kernel killed: a Ring 3 fault, reported by name in - /// `kernel/src/arch/idt/exceptions.rs`. The machine is fine — a test whose + /// `kernel/src/arch/x86_64/idt/exceptions.rs`. The machine is fine — a test whose /// whole subject is a process dying (`handle_kill_policy` and every /// `faults.rs` probe) produces these deliberately. Before a boot's ready /// marker it still ends the boot: whatever died was `init` or one of its @@ -93,7 +93,7 @@ pub enum Died { /// recursive kernel fault is still found by the guard, one silent ceiling later. const DEATHS: &[(&str, Died, Died)] = &[ // spelling the kernel wrote it anybody else wrote it - // kernel/src/arch/idt/exceptions.rs — a Ring 0 exception. Always fatal. + // kernel/src/arch/x86_64/idt/exceptions.rs — a Ring 0 exception. Always fatal. ("KERNEL PANIC", Died::Kernel, Died::Kernel), // `double_fault_handler`, which is `-> !` and ends at `halt_all_cpus`. It // writes none of the words above it, which is how a staged `#DF` inside a @@ -103,7 +103,7 @@ const DEATHS: &[(&str, Died, Died)] = &[ // `machine_check_handler`, the one exception a Ring 3 frame does not make // the process's fault. Also `-> !`. ("MACHINE CHECK", Died::Kernel, Died::Kernel), - // kernel/src/iommu/vtd/fault.rs — a fault on a stream this kernel drives + // kernel/src/arch/x86_64/vtd/fault.rs — a fault on a stream this kernel drives // has nobody to hand it to, so the handler halts. One a *process* drives // says `owner=slot` and the machine goes on, which is why the needle is // the owner rather than the fault. @@ -119,13 +119,13 @@ const DEATHS: &[(&str, Died, Died)] = &[ // rather than the console, and is here so that a capture carrying it is // never read as anything else. ("PANIC REENTRY", Died::Kernel, Died::Kernel), - // kernel/src/arch/idt/exceptions.rs `crash_report_panic` — a Rust `panic!`. + // kernel/src/arch/x86_64/idt/exceptions.rs `crash_report_panic` — a Rust `panic!`. ("PANIC:", Died::Kernel, Died::Panicked), // `PanicInfo`'s `Display` newlines this out of the record above, so the // kernel writes it too — and so does every program's panic handler. ("panicked at", Died::Kernel, Died::Panicked), ("libc panic:", Died::Panicked, Died::Panicked), - // kernel/src/arch/idt/exceptions.rs — a Ring 3 fault, by name. + // kernel/src/arch/x86_64/idt/exceptions.rs — a Ring 3 fault, by name. ("SEGFAULT", Died::Faulted, Died::Faulted), ("SIGILL tid=", Died::Faulted, Died::Faulted), ("SIGFPE tid=", Died::Faulted, Died::Faulted), @@ -405,7 +405,7 @@ impl Serial { /// all — so the only capture allowed to hold one is the capture of the test /// that staged it, and every other boot in the estate reds. const NEVER_CLEAN: &[&str] = &[ - // kernel/src/iommu/vtd/fault.rs — a function a *process* drives reached an + // kernel/src/arch/x86_64/vtd/fault.rs — a function a *process* drives reached an // address its own domain does not map. The machine goes on and the claim // refuses every later call, so this is not a death; it is a driver whose // descriptors are wrong, and a netd that did it on every boot would @@ -613,7 +613,7 @@ pub fn self_check() -> Result<(), String> { // **The report, which is the artefact a verdict used to drop.** Staged as // the lines `double_fault_handler` really writes - // (`kernel/src/arch/idt/exceptions.rs`), with the ordinary run in front of + // (`kernel/src/arch/x86_64/idt/exceptions.rs`), with the ordinary run in front of // it and a daemon still talking after the header — a capture that begins at // the death would be a capture nobody has. const DF_HEADER: &str = diff --git a/tests/common/swap.rs b/tests/common/swap.rs index f57defc2d0b..8ca0a32fab2 100644 --- a/tests/common/swap.rs +++ b/tests/common/swap.rs @@ -172,7 +172,7 @@ impl Rig { /// rehearsal sends as that service's rebuild. fn rebuilt(name: &str, dir: &Path) -> Result { let to = dir.join(format!("{name}.rebuilt")); - toyos_build::build::copy_guest_program(&super::compile::repo_root(), name, &to)?; + toyos_build::build::copy_guest_program(&super::compile::repo_root(), super::qemu::SUITE_ARCH, name, &to)?; Ok(to) } diff --git a/tests/common/update.rs b/tests/common/update.rs index 15b8bc03f56..5ce91a9d5e5 100644 --- a/tests/common/update.rs +++ b/tests/common/update.rs @@ -72,7 +72,7 @@ struct Rig { /// The plan for this config's image: `features` is the kernel build, `params` /// the actuators its slot arms, `version` its signed header's. fn plan(features: &[&str], params: &[&str], version: u64, second: Option) -> Plan { - let mut plan = Plan::new(&super::compile::repo_root().join(CONFIG).join("system.toml"), features, params); + let mut plan = Plan::new(toyos_build::arch::Arch::X86_64, &super::compile::repo_root().join(CONFIG).join("system.toml"), features, params); plan.version = version; plan.second = second; plan @@ -98,6 +98,7 @@ impl Rig { let parts = build::build_test_parts(&root, &base, true, &files); let room = SecondSlot { root_bytes: 2 * parts.root.len() as u64 }; let disk = image::create_boot_image( + toyos_build::arch::Arch::X86_64, &parts.kernel, &parts.bootloader, &parts.root, diff --git a/tests/iced-counter/Cargo.lock b/tests/iced-counter/Cargo.lock index 0c505a36315..a9832d4828f 100644 --- a/tests/iced-counter/Cargo.lock +++ b/tests/iced-counter/Cargo.lock @@ -2976,7 +2976,7 @@ version = "0.1.0" [[package]] name = "toyos-window" -version = "0.18.0" +version = "0.19.0" dependencies = [ "toyos", "toyos-abi", diff --git a/tests/toyos-rust-tests/Cargo.lock b/tests/toyos-rust-tests/Cargo.lock index facdf122562..2d84a982c64 100644 --- a/tests/toyos-rust-tests/Cargo.lock +++ b/tests/toyos-rust-tests/Cargo.lock @@ -2180,7 +2180,7 @@ version = "0.1.0" [[package]] name = "toyos-window" -version = "0.18.0" +version = "0.19.0" dependencies = [ "toyos", "toyos-abi", diff --git a/tests/toyos-rust-tests/src/bin/abuse_thread_name.rs b/tests/toyos-rust-tests/src/bin/abuse_thread_name.rs index 313b82811db..4bdc43a3b37 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_thread_name.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_thread_name.rs @@ -2,7 +2,7 @@ //! usize).min(THREAD_NAME_LEN)` and set the truncated prefix — a silent //! clamp, and the shape //! `issues/isolation/untrusted-sites-not-yet-adopted.md` named for the -//! whole of `kernel/src/arch/syscall/`. `Untrusted::at_most` replaced it +//! whole of `kernel/src/syscall/`. `Untrusted::at_most` replaced it //! with a refusal, which is a behaviour change worth its own gate: this //! proves the refusal actually fires, rather than the clamp it replaced. //! diff --git a/tests/toyos-rust-tests/src/bin/fault_gates.rs b/tests/toyos-rust-tests/src/bin/fault_gates.rs index 71c9140086c..5fd6d5db3f8 100644 --- a/tests/toyos-rust-tests/src/bin/fault_gates.rs +++ b/tests/toyos-rust-tests/src/bin/fault_gates.rs @@ -32,7 +32,7 @@ const ARMS: &[(&str, Expect)] = &[ ("ss", Expect::Killed), ("ss_rsp", Expect::Killed), // Trappable at all only because `CR0.NE` is in the declaration every CPU - // is held to (`arch/control_regs.rs`): with it clear the exception is + // is held to (`arch/x86_64/control_regs.rs`): with it clear the exception is // signalled on FERR#, which nothing in a modern machine listens to. ("mf", Expect::Killed), // TCG raises no #XM whatever MXCSR says. `CR4.OSXMMEXCPT` is declared set, diff --git a/tests/toyos-rust-tests/src/bin/fpu_isolation.rs b/tests/toyos-rust-tests/src/bin/fpu_isolation.rs index 2e09a701640..ff81834e671 100644 --- a/tests/toyos-rust-tests/src/bin/fpu_isolation.rs +++ b/tests/toyos-rust-tests/src/bin/fpu_isolation.rs @@ -1,7 +1,7 @@ //! What a transition out of Ring 3 preserves. //! //! Three arms, all positive assertions, and each one fails on the tree that -//! came before the bracket in `kernel/src/arch/entry.rs`: +//! came before the bracket in `kernel/src/arch/x86_64/entry.rs`: //! //! 1. **Leak.** One process pins a distinctive FP state and exits without //! restoring it; the next asserts the *declared* state at its own entry. diff --git a/tests/toyos-rust-tests/src/bin/log_hold.rs b/tests/toyos-rust-tests/src/bin/log_hold.rs index 510f73382ca..e4263dc3021 100644 --- a/tests/toyos-rust-tests/src/bin/log_hold.rs +++ b/tests/toyos-rust-tests/src/bin/log_hold.rs @@ -8,7 +8,7 @@ const RECORDS: usize = 192; /// A retired syscall's number: each call is refused and is one kernel record -/// naming it (`kernel/src/arch/syscall/dispatch.rs`'s `retired_syscall`). +/// naming it (`kernel/src/syscall/dispatch.rs`'s `retired_syscall`). const RETIRED: u64 = 26; fn main() { diff --git a/tests/toyos.rs b/tests/toyos.rs index 2da449c7b7d..839da4b2331 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -542,6 +542,10 @@ const AUDIO_TESTS: &[(&str, Tier)] = // first-class single-CPU case, smp=8 the full-SMP case. const AUDIO_SMP: &[u32] = &[1, 8]; +/// What `test-early-panic` panics with (`kernel/src/main.rs`): the last line its +/// report puts on serial. +const EARLY_PANIC_MESSAGE: &str = "test-early-panic: on-screen console check"; + // Tests that read a decoded screendump, which is exactly the set for which // the screen is the device under test: the panic console. On a machine with // no serial port the rendered report is the only diagnostic that exists, so @@ -604,6 +608,10 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ // than through `halt_all_cpus`. ("screen_fatal_halt_composited", Sched::Parallel, Tier::Nightly), ("screen_pager_keys", Sched::Serial, Tier::Nightly), + // AArch64 guests on QEMU `virt`: local, because no CI runner boots one yet. + ("virt_early_panic", Sched::Parallel, Tier::Local), + ("virt_early_fault", Sched::Parallel, Tier::Local), + ("virt_el2_drop", Sched::Parallel, Tier::Local), ]; /// What `screen_console_shell` types, and what it then looks for on its own. @@ -1583,7 +1591,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("quarantine_exit_status", Sched::Parallel, Tier::Fast), ("quarantine_entries", Sched::Parallel, Tier::Fast), // Same: the control-register verdict, against the machine this tree - // actually booted before `arch/control_regs.rs`. + // actually booted before `arch/x86_64/control_regs.rs`. ("control_regs_verdict", Sched::Parallel, Tier::Fast), // Same: which of the two shared boots each binary belongs on, asked of the // binaries rather than of the list that claims to name them. @@ -2541,7 +2549,7 @@ const T14_COLS: usize = 1920 / 8; /// The line `SYS_DEBUG` action 3 logs immediately before halting every CPU. /// It exists only on a `test-actuators` kernel — every other action costs the /// caller its own process, this one costs the machine. Kept in sync with -/// `kernel/src/arch/syscall/debug.rs` by this comment and by screen_fatal_halt +/// `kernel/src/syscall/debug.rs` by this comment and by screen_fatal_halt /// failing loudly if it drifts. const FATAL_HALT_NONCE: &str = "SYS_DEBUG: fatal halt 4b1d9e2c"; @@ -2550,7 +2558,7 @@ const FATAL_HALT_NONCE: &str = "SYS_DEBUG: fatal halt 4b1d9e2c"; /// /// `screen_fatal_halt_composited` reads it off the *panel*, because the machine /// that wait exists for has no serial port and `/log` is the thing that did not -/// answer. Kept in sync with `kernel/src/arch/apic.rs::LOG_DRAIN_EXPIRED` by +/// answer. Kept in sync with `kernel/src/panic.rs::LOG_DRAIN_EXPIRED` by /// this comment and by that test turning every spent budget into a red if it /// drifts. const LOG_DRAIN_EXPIRED: &str = "the report did not reach /log"; @@ -3173,7 +3181,7 @@ fn check_symbols_were_read(test: &str, serial: &str) -> bool { /// that held the lock rather than the scheduler that caught it — which is the /// only thing `#[track_caller]` on `assert_baseline` buys. /// -/// A whole-buffer `contains("arch/syscall/dispatch.rs")` certifies none of that: the +/// A whole-buffer `contains("syscall/dispatch.rs")` certifies none of that: the /// same boot's `test_syscall_panic` panics in that file too, so the needle is /// already present before the tripwire runs. Scope it instead to the window /// between this panic's header and its message — `panicked at ` is @@ -3189,7 +3197,7 @@ fn check_tripwire_attribution(serial: &str) -> Result<(), String> { .rfind(HEADER) .ok_or("tripwire message with no panic header before it")?; let location = &serial[header_at..msg_at]; - if !location.contains("arch/syscall/dispatch.rs") { + if !location.contains("syscall/dispatch.rs") { return Err(format!( "expected the tripwire to name the guilty call site, not scheduler.rs; got: {}", location.trim() @@ -5919,6 +5927,135 @@ fn run_screen_test( )?; Ok(()) } + "virt_early_panic" => { + // The AArch64 port's stage 3, whole: the loader on AAVMF, the entry's + // drop and declaration, the PL011 SPCR names, the boot's survey of + // the machine, and a panic on both channels, before the kernel + // reaches the AArch64 userland its ROOT carries. + let started = std::time::Instant::now(); + let mut qemu = QemuInstance::boot_with_options( + test_config, + &[], + &[], + BootOptions { + profile: qemu::Profile::Virt, + qmp: true, + kernel_params: &["test-early-panic"], + ready_marker: "EARLY PANIC:", + ..Default::default() + }, + ); + let dump = qemu.screendump_until("EARLY PANIC:", Duration::from_secs(30)); + let rest = qemu.drain_until(Duration::from_secs(10), |l| l.contains(EARLY_PANIC_MESSAGE)); + let serial = format!("{}\n{rest}", qemu.boot_log()); + eprintln!(" [virt] the panel is up {} ms after the boot began", started.elapsed().as_millis()); + // What stage 3 prints before it panics: every item is a record + // only the AArch64 side of the loader or the kernel writes. + for want in [ + "CPU: entered at EL1", + "serial: PL011 at", + "control registers: SCTLR_EL1=", + "as declared; entered at EL", + "memory: 0x0000400", + "ACPI: MADT GICD at 0x8000000, GIC version 3", + "ACPI: MADT GICC uid=0 mpidr=0x0 enabled=true", + "ACPI: GTDT timers:", + "EARLY PANIC: panicked at", + EARLY_PANIC_MESSAGE, + ] { + if !serial.contains(want) { + return Err(format!("{want:?} not on the PL011\nserial:\n{serial}")); + } + } + let text = dump.text(); + print_screen(name, &text); + for want in ["EARLY PANIC:", "test-early-panic: on-screen console check"] { + if !text.contains(want) { + return Err(format!("{want:?} not on the ramfb panel\ndecoded screen:\n{text}")); + } + } + check_colors( + &dump, + FILL_FATAL, + &["EARLY PANIC:", "test-early-panic: on-screen console check"], + "ACPI: GTDT timers:", + )?; + Ok(()) + } + "virt_el2_drop" => { + // The entry's drop from EL2, which HVF never exercises: `virt` with + // EL2 under TCG, where firmware hands the loader the CPU at EL2. A + // loader that refuses the CPU says so and stops; a drop that leaves + // `HCR_EL2` other than declared halts in a named refusal and says + // nothing; one that lands anywhere but EL1 on `SP_EL1` panics in the + // declaration's read-back. Each way the line this waits for never + // comes. + let mut qemu = QemuInstance::boot_with_options( + test_config, + &[], + &[], + BootOptions { + profile: qemu::Profile::VirtEl2, + kernel_params: &["test-early-panic"], + ready_marker: "EARLY PANIC:", + ..Default::default() + }, + ); + let rest = qemu.drain_until(Duration::from_secs(10), |l| l.contains(EARLY_PANIC_MESSAGE)); + let serial = format!("{}\n{rest}", qemu.boot_log()); + for want in [ + "CPU: entered at EL2, HCR_EL2.E2H ", + "ID_AA64MMFR4_EL1.E2H0 0x0: the kernel's entry writes E2H clear", + "as declared; entered at EL2, HCR_EL2 read back as declared", + "EARLY PANIC: panicked at", + EARLY_PANIC_MESSAGE, + ] { + if !serial.contains(want) { + return Err(format!("{want:?} not on the PL011\nserial:\n{serial}")); + } + } + Ok(()) + } + "virt_early_fault" => { + // The vectors, judged by the one thing a broken table cannot do: + // report. An undefined instruction right after the console step + // reaches `trap::exception`, which says what was taken and panics, + // and the panic reaches both channels. A table that is misaligned, + // never installed, or whose entry does not reach the handler + // leaves the guest silent, and this waits for a line that never + // comes. + let mut qemu = QemuInstance::boot_with_options( + test_config, + &[], + &[], + BootOptions { + profile: qemu::Profile::Virt, + qmp: true, + kernel_params: &["test-early-fault"], + ready_marker: "EARLY PANIC:", + ..Default::default() + }, + ); + let dump = qemu.screendump_until("EARLY PANIC:", Duration::from_secs(30)); + const FAULT_MESSAGE: &str = "synchronous from EL1 on SP_EL1: unknown reason (an undefined instruction) at 0x"; + let rest = qemu.drain_until(Duration::from_secs(10), |l| l.contains(FAULT_MESSAGE)); + let serial = format!("{}\n{rest}", qemu.boot_log()); + for want in [ + "KERNEL PANIC: synchronous from EL1 on SP_EL1: unknown reason (an undefined instruction)", + "EARLY PANIC: panicked at", + FAULT_MESSAGE, + ] { + if !serial.contains(want) { + return Err(format!("{want:?} not on the PL011\nserial:\n{serial}")); + } + } + let text = dump.text(); + print_screen(name, &text); + if !text.contains("EARLY PANIC:") || !text.contains("undefined instruction") { + return Err(format!("the fault's report is not on the ramfb panel\ndecoded screen:\n{text}")); + } + Ok(()) + } "screen_early_panic" => { // The window the console exists for: percpu is not up, mm::init // has not run, and on a machine with no UART nothing else can @@ -13372,9 +13509,9 @@ fn run_machine_test( "hash_seed_precedes_every_map" => { // `kernel/src/hasher.rs`'s `UNSEEDED`, as a prefix: the wrong seed // the compiler cannot reach, because the container works. Its other - // two are unrepresented here — both CPU models carry `+rdrand` - // (`src/lib.rs:73-76`) and QEMU's DRNG always answers — so - // `NO_RDRAND` and `NO_ENTROPY` are mutation-measured. + // two are unrepresented here — both x86-64 CPU models carry `+rdrand` + // (`Arch::cpu`) and QEMU's DRNG always answers — so + // the no-source refusal and `NO_ENTROPY` are mutation-measured. const UNSEEDED: &str = "kernel hasher: a hash container was built before hasher::seed()"; let qemu = QemuInstance::boot_with_options( test_config, @@ -13692,7 +13829,7 @@ fn run_machine_test( // touching it. With one CPU there is nowhere else. // // The actuator is SYS_DEBUG 5, 6 and 7, and the reason it is not - // an ordinary workload is beside them in `arch/syscall/dispatch.rs`: routes + // an ordinary workload is beside them in `syscall/dispatch.rs`: routes // past the ceiling do still exist, // and each of them holds the VFS lock when it dies, so the // machine wedges either way and the allocator's recovery cannot @@ -17111,8 +17248,8 @@ fn control_regs(log: &str, cpus: u32) -> Result<(), String> { // Vol. 3A §2.5, Vol. 2 `WRGSBASE`), so no Ring 3 thread aims `GS.base`. (16, "FSGSBASE", false), (18, "OSXSAVE", false), - // Not a bit the machine may withhold: `toyos_build::qemu::CPU_KVM` and - // `CPU_TCG` are the only two CPUs this repository launches and both name + // Not a bit the machine may withhold: `Arch::cpu`'s two x86-64 CPUs + // are the only x86-64 CPUs this repository launches and both name // `+smep`, so a boot without supervisor-mode execution prevention is a // kernel that stopped enabling it or a launcher that stopped asking. (20, "SMEP", true), @@ -17743,7 +17880,7 @@ fn control_regs_negative( } // Where the host does leave `CD` set, it is demanded, so the arm that *can* // see the caching defect does not quietly become the weaker of the two. - if !toyos_build::kvm_usable() && !refusal.contains("CD") { + if !common::qemu::SUITE_ARCH.accel().is_hardware() && !refusal.contains("CD") { return Err(format!( "TCG leaves an AP's `CD` set and the refusal does not name it: {refusal}" )); @@ -18332,7 +18469,8 @@ fn run_debug_mode(c_tests: &[(String, Vec)], rust_bins: &[(String, Vec)] let repo = compile::repo_root(); let kernel_elf = repo.join(format!( - "kernel/target/x86_64-unknown-none/{}/kernel", + "kernel/target/{}/{}/kernel", + common::qemu::SUITE_ARCH.kernel(), toyos_build::build::PROFILE )); @@ -20214,7 +20352,8 @@ fn build_tasks<'a>( fn check_shard_partition(all_tests: &[TestDef]) { let pricing = shard_pricing(); for &nightly in &[false, true] { - let in_tier = |tier: Tier| nightly || tier == Tier::Fast; + // Sharded, because every run this partition is for is one. + let in_tier = |tier: Tier| tier.selected(nightly, true); let tests_to_run: Vec<&TestDef> = all_tests.iter().filter(|_| in_tier(SHARED_TIER)).collect(); let machine_to_run: Vec<(&str, Sched)> = MACHINE_TESTS @@ -20861,7 +21000,7 @@ fn main() { // nobody remembers. `cargo test -- desktop_window_child` refuses below and // says what to type instead, which is the same information a silent skip // would have withheld. - let in_tier = |tier: Tier| nightly || tier == Tier::Fast; + let in_tier = |tier: Tier| tier.selected(nightly, shard.is_some()); let tests_to_run: Vec<&TestDef> = all_tests .iter() .filter(|t| keep(t.name.as_str()) && in_tier(SHARED_TIER)) @@ -20887,18 +21026,30 @@ fn main() { // introduces, so the names are printed rather than counted, and the line // carries both the command that runs them and the record that says what each // one guarded. - let held_back: Vec<&str> = MACHINE_TESTS - .iter() - .chain(SCREEN_TESTS) - .filter(|(n, _, tier)| keep(n) && !in_tier(*tier)) - .map(|(n, _, _)| *n) - .chain( - AUDIO_TESTS - .iter() - .filter(|(name, tier)| keep(name) && !in_tier(*tier)) - .map(|(name, _)| *name), - ) - .collect(); + let held = |which: Tier| -> Vec<&str> { + MACHINE_TESTS + .iter() + .chain(SCREEN_TESTS) + .filter(|(n, _, tier)| keep(n) && *tier == which && !in_tier(*tier)) + .map(|(n, _, _)| *n) + .chain( + AUDIO_TESTS + .iter() + .filter(|(name, tier)| keep(name) && *tier == which && !in_tier(*tier)) + .map(|(name, _)| *name), + ) + .collect() + }; + let held_back = held(Tier::Nightly); + let held_local = held(Tier::Local); + if !held_local.is_empty() { + eprintln!( + "[toyos] local tier: {} test(s) NOT run, because a sharded run is CI's and no CI \ + runner boots their architecture yet. An unsharded `cargo test` runs them.", + held_local.len(), + ); + eprintln!("[toyos] {}", held_local.join(", ")); + } if !held_back.is_empty() { eprintln!( "[toyos] nightly tier: {} test(s) NOT run. \ diff --git a/toyos-acpi/src/gtdt.rs b/toyos-acpi/src/gtdt.rs new file mode 100644 index 00000000000..edac464cd5a --- /dev/null +++ b/toyos-acpi/src/gtdt.rs @@ -0,0 +1,57 @@ +//! The GTDT (signature `GTDT`): ACPI 6.5 §5.2.25, Table 5.121 — the Arm +//! generic timer's interrupts. + +use crate::{find_table, Phys, TableError}; + +/// Secure EL1 timer GSIV at 48 and its flags at 52; non-secure EL1 at 56/60; +/// virtual EL1 at 64/68; EL2 at 72/76. +const SECURE_EL1: usize = 48; +const NON_SECURE_EL1: usize = 56; +const VIRTUAL_EL1: usize = 64; +const EL2: usize = 72; +/// Every field this decoder reads lies below the EL2 flags' end. +pub const GTDT_NEEDED: usize = EL2 + 8; + +/// One timer's interrupt: its GSIV (a PPI) and ACPI 6.5 Table 5.122's flags — +/// bit 0 edge-triggered, bit 1 active-low, bit 2 always-on capable. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub struct TimerInterrupt { + pub gsiv: u32, + pub flags: u32, +} + +impl TimerInterrupt { + pub fn edge(self) -> bool { + self.flags & 1 != 0 + } + + pub fn active_low(self) -> bool { + self.flags & 2 != 0 + } +} + +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub struct Gtdt { + pub secure_el1: TimerInterrupt, + pub non_secure_el1: TimerInterrupt, + pub virtual_el1: TimerInterrupt, + pub el2: TimerInterrupt, +} + +/// The GTDT at `rsdp_addr`, decoded. +pub fn gtdt(phys: P, rsdp_addr: u64) -> Result { + let table = find_table(phys, rsdp_addr, b"GTDT", GTDT_NEEDED)?; + let short = TableError::Length { declared: table.len() as u32, needed: GTDT_NEEDED }; + let timer = |at: usize| -> Result { + Ok(TimerInterrupt { + gsiv: table.u32_at(at).ok_or(short)?, + flags: table.u32_at(at + 4).ok_or(short)?, + }) + }; + Ok(Gtdt { + secure_el1: timer(SECURE_EL1)?, + non_secure_el1: timer(NON_SECURE_EL1)?, + virtual_el1: timer(VIRTUAL_EL1)?, + el2: timer(EL2)?, + }) +} diff --git a/toyos-acpi/src/lib.rs b/toyos-acpi/src/lib.rs index ed75b64af3b..90ad4a8e3ba 100644 --- a/toyos-acpi/src/lib.rs +++ b/toyos-acpi/src/lib.rs @@ -7,7 +7,7 @@ //! //! Multi-byte fields are composed from bytes, little-endian, so no firmware //! byte is transmuted into a type. Field offsets cite ACPI 6.5, except MCFG's -//! (PCI Firmware Specification) and HPET's (IA-PC HPET Specification). +//! (PCI Firmware Specification), HPET's (IA-PC HPET Specification) and SPCR's (Microsoft's SPCR specification). //! //! `no_std`, no allocation, no `unsafe`. @@ -15,17 +15,21 @@ #![forbid(unsafe_code)] mod fadt; +mod gtdt; mod madt; mod resource; +mod spcr; pub use fadt::{ century_of, dsdt_address, iapc_boot_arch, reset_register, rtc_century, Century, Reset, CMOS_RAM, FADT_FOR_RESET, FADT_PM1A_CNT_BLK, FADT_X_DSDT, }; pub use madt::{ - madt_entries, IoApicEntry, MadtEntries, MadtEntry, MadtHalt, SourceOverride, MADT_ENTRIES, + madt_entries, Gicc, IoApicEntry, MadtEntries, MadtEntry, MadtHalt, SourceOverride, MADT_ENTRIES, }; +pub use gtdt::{gtdt, Gtdt, TimerInterrupt, GTDT_NEEDED}; pub use resource::{memory_windows, ResourceError, Walk, MAX_LIST_BYTES}; +pub use spcr::{spcr, Gas, SerialInterface, Spcr, GAS_SYSTEM_MEMORY, SPCR_NEEDED}; /// Physical memory, as this decoder reads it. /// diff --git a/toyos-acpi/src/madt.rs b/toyos-acpi/src/madt.rs index fc6f9ca9b3c..921dae67fd9 100644 --- a/toyos-acpi/src/madt.rs +++ b/toyos-acpi/src/madt.rs @@ -4,7 +4,7 @@ //! one running past the list ends the walk with a [`MadtHalt`]: neither can be //! resynchronised, and a walk that tried would not terminate. -use crate::{u16le, u32le, Phys, Table, SDT_HEADER_LEN}; +use crate::{u16le, u32le, u64le, Phys, Table, SDT_HEADER_LEN}; /// ACPI 6.5 Table 5.19: `Local Interrupt Controller Address` (4) and `Flags` /// (4) follow the header; the interrupt controller structures start here. @@ -38,10 +38,29 @@ pub enum MadtEntry { LocalApic { apic_id: u32, enabled: bool }, IoApic(IoApicEntry), SourceOverride(SourceOverride), + /// Type 0xB (Table 5.37): one CPU's GIC CPU interface; `enabled` is flags bit 0. + Gicc(Gicc), + /// Type 0xC (Table 5.39): the GIC distributor. + Gicd { base: u64, version: u8 }, + /// Type 0xE (Table 5.43): a range holding GICv3 redistributors. + Gicr { base: u64, length: u32 }, + /// Type 0xF (Table 5.44): a GICv3 Interrupt Translation Service. + Its { id: u32, base: u64 }, /// A type this kernel does not act on, or one too short to hold its own fields. Other(u8), } +/// MADT type 0xB, the fields a GICv3 kernel reads: which CPU it is +/// (`MPIDR`), whether it may be started, and where its redistributor is when +/// firmware names one per CPU rather than as a range. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub struct Gicc { + pub uid: u32, + pub enabled: bool, + pub gicr_base: u64, + pub mpidr: u64, +} + /// The entry whose declared length the list cannot hold; the last item a walk yields. #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub struct MadtHalt { @@ -116,6 +135,20 @@ impl Iterator for MadtEntries

{ apic_id: u32le(phys, base + 4), enabled: u32le(phys, base + 8) & 1 != 0, }, + // Table 5.37: ACPI Processor UID (8..12), Flags (12..16), GICR Base + // Address (60..68), MPIDR (68..76). 76 bytes is ACPI 5.1's length. + (0xB, 76..) => MadtEntry::Gicc(Gicc { + uid: u32le(phys, base + 8), + enabled: u32le(phys, base + 12) & 1 != 0, + gicr_base: u64le(phys, base + 60), + mpidr: u64le(phys, base + 68), + }), + // Table 5.39: Physical Base Address (8..16), GIC Version (20). + (0xC, 24..) => MadtEntry::Gicd { base: u64le(phys, base + 8), version: phys.byte(base + 20) }, + // Table 5.43: Discovery Range Base Address (4..12), Length (12..16). + (0xE, 16..) => MadtEntry::Gicr { base: u64le(phys, base + 4), length: u32le(phys, base + 12) }, + // Table 5.44: GIC ITS ID (4..8), Physical Base Address (8..16). + (0xF, 20..) => MadtEntry::Its { id: u32le(phys, base + 4), base: u64le(phys, base + 8) }, _ => MadtEntry::Other(entry_type), })) } diff --git a/toyos-acpi/src/spcr.rs b/toyos-acpi/src/spcr.rs new file mode 100644 index 00000000000..727a0a23da5 --- /dev/null +++ b/toyos-acpi/src/spcr.rs @@ -0,0 +1,83 @@ +//! The SPCR (Microsoft "Serial Port Console Redirection Table", revision 2 +//! and later): which UART firmware's console is on, and where. + +use crate::{find_table, Phys, TableError, SDT_HEADER_LEN}; + +/// Interface Type (1) at 36, reserved to 40, Base Address (a 12-byte Generic +/// Address Structure) at 40, Interrupt Type at 52, IRQ at 53, Global System +/// Interrupt at 54..58. +const INTERFACE_TYPE: usize = SDT_HEADER_LEN; +const BASE_ADDRESS: usize = 40; +const GSIV: usize = 54; +/// Every field this decoder reads lies below the GSIV's end. +pub const SPCR_NEEDED: usize = GSIV + 4; + +/// ACPI 6.5 §5.2.3.2, Table 5.1: a register block's place and access shape. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub struct Gas { + /// 0 is system memory, 1 system I/O. + pub space: u8, + pub bit_width: u8, + pub bit_offset: u8, + /// 1..=4 are byte, word, dword and qword access; 0 is undefined. + pub access_size: u8, + pub address: u64, +} + +/// The Generic Address Structure's system-memory space id. +pub const GAS_SYSTEM_MEMORY: u8 = 0; + +/// The UART kinds the DBG2 table's serial subtypes name, as far as a kernel +/// that drives them needs to tell them apart. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum SerialInterface { + /// 0x00: a full 16550. + Ns16550, + /// 0x03: an Arm PL011. + Pl011, + /// 0x0E: the Arm SBSA generic UART, a PL011 subset firmware configured. + SbsaGeneric, + /// 0x0D: the same, restricted to 32-bit accesses. + SbsaGeneric32, + /// Any other subtype, by number. + Other(u8), +} + +impl SerialInterface { + fn from_raw(raw: u8) -> Self { + match raw { + 0x00 => Self::Ns16550, + 0x03 => Self::Pl011, + 0x0D => Self::SbsaGeneric32, + 0x0E => Self::SbsaGeneric, + other => Self::Other(other), + } + } +} + +/// What the SPCR says about firmware's console. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub struct Spcr { + pub interface: SerialInterface, + pub base: Gas, + /// The UART's interrupt as a global system interrupt. + pub gsiv: u32, +} + +/// The SPCR at `rsdp_addr`, decoded. +pub fn spcr(phys: P, rsdp_addr: u64) -> Result { + let table = find_table(phys, rsdp_addr, b"SPCR", SPCR_NEEDED)?; + let short = TableError::Length { declared: table.len() as u32, needed: SPCR_NEEDED }; + let byte = |at| table.byte(at).ok_or(short); + Ok(Spcr { + interface: SerialInterface::from_raw(byte(INTERFACE_TYPE)?), + base: Gas { + space: byte(BASE_ADDRESS)?, + bit_width: byte(BASE_ADDRESS + 1)?, + bit_offset: byte(BASE_ADDRESS + 2)?, + access_size: byte(BASE_ADDRESS + 3)?, + address: table.u64_at(BASE_ADDRESS + 4).ok_or(short)?, + }, + gsiv: table.u32_at(GSIV).ok_or(short)?, + }) +} diff --git a/toyos-acpi/tests/fixtures.rs b/toyos-acpi/tests/fixtures.rs index c34ed71fb7f..c7f4c2c171d 100644 --- a/toyos-acpi/tests/fixtures.rs +++ b/toyos-acpi/tests/fixtures.rs @@ -48,6 +48,10 @@ fn the_madt_names_the_two_cpus_that_boot_and_the_chip_that_interrupts_them() { MadtEntry::IoApic(e) => io_apics.push(e), MadtEntry::SourceOverride(o) => overrides.push(o), MadtEntry::Other(_) => {} + gic @ (MadtEntry::Gicc(_) + | MadtEntry::Gicd { .. } + | MadtEntry::Gicr { .. } + | MadtEntry::Its { .. }) => panic!("q35 publishes no GIC structure, and the walk found {gic:?}"), } } diff --git a/toyos-bootmap/src/aarch64.rs b/toyos-bootmap/src/aarch64.rs new file mode 100644 index 00000000000..9d833667a01 --- /dev/null +++ b/toyos-bootmap/src/aarch64.rs @@ -0,0 +1,65 @@ +//! AArch64's encoding of a [`Plan`](crate::Plan): the VMSAv8-64 stage 1 +//! descriptors of a 4 KiB granule (Arm ARM K.a, D8.3), and the one `MAIR_EL1` +//! whose indices they name — the loader writes the one, the kernel's entry +//! loads the other, and both read them here. + +use crate::Cache; + +/// `MAIR_EL1` index 0: Device-nGnRE, registers. +pub const ATTR_DEVICE: u64 = 0; +/// `MAIR_EL1` index 1: Normal, inner and outer write-back, read- and +/// write-allocate — RAM. +pub const ATTR_NORMAL: u64 = 1; +/// `MAIR_EL1` index 2: Normal, inner and outer non-cacheable — the scanout, +/// whose stores gather as write-combining ones do. +pub const ATTR_NORMAL_NC: u64 = 2; +/// Each index's attribute byte (D24.2.110), in index order: the value +/// `PAR_EL1.ATTR` also reports a translation's type in. +pub const ATTRS: [u8; 3] = [0x04, 0xFF, 0x44]; +/// `MAIR_EL1` whole. +pub const MAIR: u64 = ATTRS[0] as u64 | (ATTRS[1] as u64) << 8 | (ATTRS[2] as u64) << 16; + +/// A table descriptor (bits 1:0 = 0b11) naming the next table down. +const TABLE: u64 = 0b11; +/// A block descriptor (bits 1:0 = 0b01) at level 1 or 2. +const BLOCK: u64 = 0b01; +/// `SH` = inner shareable, bits 9:8. +const INNER_SHAREABLE: u64 = 0b11 << 8; +/// `SH` = outer shareable. +const OUTER_SHAREABLE: u64 = 0b10 << 8; +/// The access flag, bit 10: set, so the first access does not fault. +const AF: u64 = 1 << 10; +/// Privileged execute-never, bit 53. +const PXN: u64 = 1 << 53; +/// Unprivileged execute-never, bit 54. +const UXN: u64 = 1 << 54; + +/// A page descriptor (bits 1:0 = 0b11) at level 3. +const PAGE: u64 = 0b11; + +/// A descriptor naming the next table down, at `phys`. +pub const fn table(phys: u64) -> u64 { + phys | TABLE +} + +/// A 2 MiB block at `phys`: EL1 read-write and EL0 nothing (`AP` = 0b00), and +/// executable at EL1 only where it is memory — the kernel's image is. A +/// device is never executable, since a speculative fetch from one is a read +/// of its registers. +pub const fn block(phys: u64, cache: Cache) -> u64 { + phys | BLOCK | AF | attributes(cache) +} + +const fn attributes(cache: Cache) -> u64 { + match cache { + Cache::Memory => ATTR_NORMAL << 2 | INNER_SHAREABLE | UXN, + Cache::Device => ATTR_DEVICE << 2 | PXN | UXN, + Cache::Scanout => ATTR_NORMAL_NC << 2 | OUTER_SHAREABLE | PXN | UXN, + Cache::Firmware => panic!("AArch64 has no range registers to type memory: its plan types by the map"), + } +} + +/// A 4 KiB page at `phys`, with [`block`]'s attributes. +pub const fn page(phys: u64, cache: Cache) -> u64 { + phys | PAGE | AF | attributes(cache) +} diff --git a/toyos-bootmap/src/lib.rs b/toyos-bootmap/src/lib.rs index 30be68a996c..a585814053b 100644 --- a/toyos-bootmap/src/lib.rs +++ b/toyos-bootmap/src/lib.rs @@ -1,34 +1,44 @@ //! The bootloader's transient page tables, decided rather than built. //! -//! Between the loader's `mov cr3` and the kernel's `mm::init` there is one -//! mapping in the machine, and everything the kernel touches in that window has -//! to be in it — its own image, the boot parameter, and the panel it reports a -//! wedge on. +//! Between the loader's switch to these tables and the kernel's `mm::init` +//! there is one mapping in the machine, and everything the kernel touches in +//! that window has to be in it — its own image, the boot parameter, and the +//! panel it reports a wedge on. //! -//! Pure: three numbers in, a [`Plan`] out. The loader allocates the pages and -//! writes the entries. +//! Both architectures walk the same shape — a root (x86-64's PML4, AArch64's +//! L0), a table per 512 GiB (PDPT, L1), a table per GiB (page directory, L2) +//! of 2 MiB leaves — so one [`Plan`] serves both. What differs is how a leaf +//! is typed ([`Typing`]) and how an entry is encoded ([`x86_64`], +//! [`aarch64`]). +//! +//! Pure: the scanout and, where the architecture types memory by firmware's +//! map, the write-back ranges in; a [`Plan`] out. The loader allocates the +//! pages and writes the entries. #![no_std] #![forbid(unsafe_code)] use core::fmt; +pub mod aarch64; +pub mod x86_64; + /// The page every entry in this map describes. pub const PAGE_2M: u64 = 2 * 1024 * 1024; const GIB: u64 = 1 << 30; -/// One PDPT reaches 512 GiB, and the map has two: the identity view at PML4[0] -/// and the high-half view at PML4[256]. +/// One second-level table reaches 512 GiB, and the map has two: the identity +/// view at root slot 0 and the high-half view at root slot 256. const GIB_PER_PDPT: u64 = 512; -/// PML4[0]: physical memory at its own address, which the `mov cr3` itself -/// runs from. -pub const PML4_IDENTITY: usize = 0; +/// Root slot 0: physical memory at its own address, which the switch to these +/// tables itself runs from. +pub const ROOT_IDENTITY: usize = 0; -/// PML4[256]: the same memory at `PHYS_OFFSET`, whose bits 39..48 are 256 and -/// which is where the kernel's entry point is called. -pub const PML4_HIGH_HALF: usize = 256; +/// Root slot 256: the same memory at `PHYS_OFFSET`, whose bits 39..48 are 256 +/// and which is where the kernel runs from. +pub const ROOT_HIGH_HALF: usize = 256; /// How much physical memory the map covers, at identity and at `PHYS_OFFSET` /// alike. Everything the entry jump needs, not everything `KernelArgs` names. @@ -44,8 +54,18 @@ const SCANOUT_DIRECTORIES: usize = 2; /// Every page directory a [`Plan`] can name. pub const MAX_DIRECTORIES: usize = LOW_DIRECTORIES + SCANOUT_DIRECTORIES; -/// The pool a builder needs: a PML4, a PDPT per view, and every directory. -pub const MAX_PAGES: usize = 3 + MAX_DIRECTORIES; +/// The small page a split 2 MiB page is mapped in. +pub const PAGE_4K: u64 = 4096; + +/// Leaves in one table. +const PAGES_PER_TABLE: u64 = PAGE_2M / PAGE_4K; + +/// The 2 MiB pages a scanout covers only in part: at most its first and its last. +const FINE_TABLES: usize = 2; + +/// The pool a builder needs: a root, a second-level table per view, every +/// directory, and a table of 4 KiB leaves per split page. +pub const MAX_PAGES: usize = 3 + MAX_DIRECTORIES + FINE_TABLES; /// Why a machine's memory does not fit these tables. #[derive(Clone, Copy, PartialEq, Eq, Debug)] @@ -55,10 +75,13 @@ pub enum Refusal { Unaligned(u64), /// The range's own end does not fit an address. Extent { base: u64, len: u64 }, - /// Past the reach of the two PDPTs this map has. + /// Past the reach of the two second-level tables this map has. PastPdpt(u64), /// More directories than [`MAX_DIRECTORIES`]. Directories(usize), + /// A 2 MiB page of the low map that is write-back memory in part and not + /// in the rest: either type is wrong for some of it. + Mixed(u64), } impl fmt::Display for Refusal { @@ -72,98 +95,180 @@ impl fmt::Display for Refusal { } Self::PastPdpt(gib) => write!( f, - "GiB {gib} is past the {GIB_PER_PDPT} this map's two PDPTs reach" + "GiB {gib} is past the {GIB_PER_PDPT} this map's two second-level tables reach" ), Self::Directories(needed) => { write!(f, "{needed} page directories are needed and {MAX_DIRECTORIES} may be named") } + Self::Mixed(phys) => write!( + f, + "the 2 MiB page at {phys:#x} is part write-back memory and part not, so no one \ + memory type is right for all of it" + ), } } } -/// What a 2 MiB entry selects. +/// What a 2 MiB entry is, which is what its memory type follows from. #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub enum Cache { - /// PAT entry 0, whose type is the MTRR's — plain memory. - DeferToMtrr, - /// PCD and PWT set with the PAT bit clear: PAT entry 3, uncacheable under - /// every MTRR type and under an unprogrammed PAT, which is what the machine - /// has until the kernel's `pat::init`. - Uncacheable, + /// Typed by the architecture's own range registers rather than by the + /// entry: x86-64's MTRRs, beneath an entry that selects plain memory. + Firmware, + /// Write-back memory, every byte of it, by firmware's map. + Memory, + /// Not memory by firmware's map: registers, or nothing at all. + Device, + /// The scanout. + Scanout, } -impl Cache { - /// The bits a 2 MiB entry carries for this type, beside its address and the - /// present, writable and page-size bits every entry has. - pub const fn bits(self) -> u64 { - /// Page-level cache disable, bit 4. - const PCD: u64 = 1 << 4; - /// Page-level write-through, bit 3. - const PWT: u64 = 1 << 3; - match self { - Self::DeferToMtrr => 0, - Self::Uncacheable => PCD | PWT, +/// How the pages that are not the scanout are typed. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Typing<'a> { + /// By the architecture's range registers: every such page is [`Cache::Firmware`]. + Firmware, + /// By firmware's memory map: these ranges, `(base, length)`, are + /// write-back memory; a page wholly outside them is a device's, and a page + /// partly inside is [`Refusal::Mixed`]. + ByMap(&'a [(u64, u64)]), +} + +impl Typing<'_> { + /// The type of the `size`-byte page at `phys`. + fn of(self, phys: u64, size: u64) -> Result { + let Typing::ByMap(memory) = self else { return Ok(Cache::Firmware) }; + let end = phys + size; + // Bytes of the page the ranges cover; firmware's ranges do not overlap. + let covered: u64 = memory + .iter() + .map(|&(base, len)| base.saturating_add(len).min(end).saturating_sub(base.max(phys))) + .sum(); + match covered { + 0 => Ok(Cache::Device), + c if c == size => Ok(Cache::Memory), + _ => Err(Refusal::Mixed(phys)), } } } -/// One 2 MiB entry the map holds. +/// Where a leaf sits. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Slot { + /// A 2 MiB leaf: in which of [`Plan::directories`], by position, and at + /// which index. + Directory { directory: usize, index: usize }, + /// A 4 KiB leaf: in which of [`Plan::fine_tables`], by position, and at + /// which index. + Fine { table: usize, index: usize }, +} + +/// One leaf the map holds. #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub struct Entry { pub phys: u64, - /// Which of [`Plan::directories`] holds it, by position. - pub directory: usize, - /// Its index in that directory. - pub index: usize, + pub slot: Slot, pub cache: Cache, } /// Where a machine's memory and its scanout go in the loader's two views. #[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub struct Plan { +pub struct Plan<'a> { gibs: [u64; MAX_DIRECTORIES], directories: usize, + /// The 2 MiB pages split into 4 KiB leaves, by base. + fine: [u64; FINE_TABLES], + fines: usize, scanout: Option<(u64, u64)>, + typing: Typing<'a>, } -impl Plan { - /// Lay out [`BOOT_MAP_BYTES`] and `scanout`, or say why neither can be. +impl<'a> Plan<'a> { + /// Lay out [`BOOT_MAP_BYTES`] and `scanout`, typed by `typing`, or say why + /// they cannot be. /// - /// `scanout` is firmware's framebuffer as firmware reports it. Its base - /// must be 2 MiB aligned; its length is rounded up, because a page cannot - /// end anywhere else and what the last one covers past the framebuffer is - /// the same aperture firmware put it in. - pub fn new(scanout: Option<(u64, u64)>) -> Result { - let mut plan = Self { gibs: [0; MAX_DIRECTORIES], directories: 0, scanout: None }; + /// `scanout` is firmware's framebuffer as firmware reports it. Under + /// [`Typing::Firmware`] its base must be 2 MiB aligned and its length is + /// rounded up to whole 2 MiB pages, because what the last one covers past + /// the framebuffer is the same aperture firmware put it in. Under + /// [`Typing::ByMap`] it need only be 4 KiB aligned: a 2 MiB page it covers + /// in part is split into 4 KiB leaves, so the scanout's type reaches no + /// byte beside it — a framebuffer firmware carved out of RAM sits between + /// memory other owners hold. + pub fn new(scanout: Option<(u64, u64)>, typing: Typing<'a>) -> Result { + let mut plan = Self { + gibs: [0; MAX_DIRECTORIES], + directories: 0, + fine: [0; FINE_TABLES], + fines: 0, + scanout: None, + typing, + }; for gib in 0..BOOT_MAP_BYTES / GIB { plan.claim(gib); } - let Some((base, len)) = scanout else { return Ok(plan) }; - if base % PAGE_2M != 0 { + if let Some((base, len)) = scanout { + plan.place_scanout(base, len)?; + } + // Every leaf that is not the scanout typed once here, so `entries` + // cannot meet a refusal. + for page in 0..BOOT_MAP_BYTES / PAGE_2M { + let phys = page * PAGE_2M; + if !plan.is_split(phys) && !plan.in_scanout(phys, PAGE_2M) { + typing.of(phys, PAGE_2M)?; + } + } + for &base in plan.fine_tables() { + for page in 0..PAGES_PER_TABLE { + let phys = base + page * PAGE_4K; + if !plan.in_scanout(phys, PAGE_4K) { + typing.of(phys, PAGE_4K)?; + } + } + } + Ok(plan) + } + + /// Claim the directories and the split pages `base..base + len` needs, + /// and record the extent the map gives it. + fn place_scanout(&mut self, base: u64, len: u64) -> Result<(), Refusal> { + let granule = match self.typing { + Typing::Firmware => PAGE_2M, + Typing::ByMap(_) => PAGE_4K, + }; + if !base.is_multiple_of(granule) { return Err(Refusal::Unaligned(base)); } let end = base.checked_add(len).ok_or(Refusal::Extent { base, len })?; - let covered = - end.checked_next_multiple_of(PAGE_2M).ok_or(Refusal::Extent { base, len })? - base; - let last = (base + covered - PAGE_2M) / GIB; - if last >= GIB_PER_PDPT { - return Err(Refusal::PastPdpt(last)); + let end = end.checked_next_multiple_of(granule).ok_or(Refusal::Extent { base, len })?; + let first = base / PAGE_2M * PAGE_2M; + let last = (end - 1) / PAGE_2M * PAGE_2M; + if last / GIB >= GIB_PER_PDPT { + return Err(Refusal::PastPdpt(last / GIB)); } // Counted whole before one is claimed, so a machine needing fourteen is // told fourteen rather than that one more than the budget was wanted. let low = LOW_DIRECTORIES as u64; - let fresh = if last < low { 0 } else { last - (base / GIB).max(low) + 1 }; + let fresh = if last / GIB < low { 0 } else { last / GIB - (first / GIB).max(low) + 1 }; let required = LOW_DIRECTORIES + fresh as usize; if required > MAX_DIRECTORIES { return Err(Refusal::Directories(required)); } - let mut phys = base; - while phys < base + covered { - plan.claim(phys / GIB); - phys += PAGE_2M; + let mut page = first; + while page <= last { + self.claim(page / GIB); + page += PAGE_2M; } - plan.scanout = Some((base, covered)); - Ok(plan) + // At most the first and the last page are covered in part. + for page in [first, last] { + let whole = base <= page && page + PAGE_2M <= end; + if !whole && !self.is_split(page) { + self.fine[self.fines] = page; + self.fines += 1; + } + } + self.scanout = Some((base, end - base)); + Ok(()) } /// The GiB each directory covers, in the order a builder allocates them. @@ -171,48 +276,81 @@ impl Plan { &self.gibs[..self.directories] } - /// The scanout as this map covers it: firmware's base, rounded up to whole - /// pages. `None` is a machine with no framebuffer. + /// The 2 MiB pages mapped by a table of 4 KiB leaves rather than one + /// leaf, by base, in the order a builder allocates those tables. + pub fn fine_tables(&self) -> &[u64] { + &self.fine[..self.fines] + } + + /// Where each of [`Plan::fine_tables`] is named from: its directory, by + /// position, and its index there. + pub fn fine_slots(&self) -> impl Iterator + '_ { + self.fine_tables().iter().map(move |&base| self.directory_slot(base)) + } + + /// The scanout as this map covers it: firmware's base, and its length + /// rounded up to whole leaves. `None` is a machine with no framebuffer. pub fn scanout(&self) -> Option<(u64, u64)> { self.scanout } - /// Every entry the map holds, low memory first and each place once: a - /// scanout inside the low map retypes the pages already there rather than - /// adding a second entry for them. + /// Every leaf the map holds, low memory first and each place once: a + /// scanout inside the low map retypes the leaves already there rather than + /// adding a second one for them. pub fn entries(&self) -> impl Iterator + '_ { - let plan = *self; let low = (0..BOOT_MAP_BYTES / PAGE_2M) - .map(move |page| plan.entry(page * PAGE_2M, plan.cache_of(page * PAGE_2M))); - let scanout = self.scanout.into_iter().flat_map(move |(base, covered)| { - (0..covered / PAGE_2M) - .map(move |page| base + page * PAGE_2M) - .filter(|phys| *phys >= BOOT_MAP_BYTES) - .map(move |phys| plan.entry(phys, Cache::Uncacheable)) + .map(|page| page * PAGE_2M) + .filter(move |phys| !self.is_split(*phys)) + .map(move |phys| self.leaf(phys, self.cache_of(phys, PAGE_2M))); + let scanout = self.scanout.into_iter().flat_map(move |(base, len)| { + let first = base / PAGE_2M * PAGE_2M; + (0..(base + len - first).div_ceil(PAGE_2M)) + .map(move |page| first + page * PAGE_2M) + .filter(move |phys| *phys >= BOOT_MAP_BYTES && !self.is_split(*phys)) + .map(move |phys| self.leaf(phys, Cache::Scanout)) + }); + let fine = self.fine_tables().iter().enumerate().flat_map(move |(table, &base)| { + (0..PAGES_PER_TABLE).map(move |index| { + let phys = base + index * PAGE_4K; + Entry { + phys, + slot: Slot::Fine { table, index: index as usize }, + cache: self.cache_of(phys, PAGE_4K), + } + }) }); - low.chain(scanout) + low.chain(scanout).chain(fine) } - /// What a page of the low map selects: the scanout's type where the two - /// overlap, and plain memory everywhere else. - fn cache_of(&self, phys: u64) -> Cache { - match self.scanout { - Some((base, covered)) if phys >= base && phys < base + covered => Cache::Uncacheable, - _ => Cache::DeferToMtrr, - } + fn in_scanout(&self, phys: u64, size: u64) -> bool { + self.scanout.is_some_and(|(base, len)| phys >= base && phys + size <= base + len) + } + + fn is_split(&self, page: u64) -> bool { + self.fine_tables().contains(&page) } - fn entry(&self, phys: u64, cache: Cache) -> Entry { - Entry { - phys, - directory: self - .directories() - .iter() - .position(|gib| *gib == phys / GIB) - .expect("every entry's GiB was claimed"), - index: ((phys / PAGE_2M) % 512) as usize, - cache, + /// What a leaf is: the scanout where the scanout covers it, and what the + /// typing says everywhere else. + fn cache_of(&self, phys: u64, size: u64) -> Cache { + if self.in_scanout(phys, size) { + return Cache::Scanout; } + self.typing.of(phys, size).expect("`new` typed every leaf") + } + + fn directory_slot(&self, phys: u64) -> (usize, usize) { + let directory = self + .directories() + .iter() + .position(|gib| *gib == phys / GIB) + .expect("every entry's GiB was claimed"); + (directory, ((phys / PAGE_2M) % 512) as usize) + } + + fn leaf(&self, phys: u64, cache: Cache) -> Entry { + let (directory, index) = self.directory_slot(phys); + Entry { phys, slot: Slot::Directory { directory, index }, cache } } /// Name the directory for `gib`, unless it is already named. Infallible: diff --git a/toyos-bootmap/src/x86_64.rs b/toyos-bootmap/src/x86_64.rs new file mode 100644 index 00000000000..6c138526df2 --- /dev/null +++ b/toyos-bootmap/src/x86_64.rs @@ -0,0 +1,38 @@ +//! x86-64's encoding of a [`Plan`](crate::Plan): Intel SDM Vol. 3A §4.5, +//! Tables 4-15 (a PML4 or PDPT entry naming a table) and 4-17 (a page +//! directory entry mapping a 2 MiB page). + +use crate::Cache; + +const PRESENT: u64 = 1 << 0; +const WRITABLE: u64 = 1 << 1; +/// Page-level write-through, bit 3. +const PWT: u64 = 1 << 3; +/// Page-level cache disable, bit 4. +const PCD: u64 = 1 << 4; +/// A page directory entry that maps 2 MiB rather than naming a table, bit 7. +const PAGE_SIZE: u64 = 1 << 7; + +/// An entry naming the next table down, at `phys`. +pub const fn table(phys: u64) -> u64 { + phys | PRESENT | WRITABLE +} + +/// A 2 MiB leaf at `phys`. Memory and the MTRR-typed page select PAT entry 0, +/// whose type is the MTRR's; a device and the scanout select entry 3 (PCD and +/// PWT with the PAT bit clear), uncacheable under every MTRR type and under an +/// unprogrammed PAT, which is what the machine has until the kernel's +/// `pat::init`. +pub const fn block(phys: u64, cache: Cache) -> u64 { + let typed = match cache { + Cache::Firmware | Cache::Memory => 0, + Cache::Device | Cache::Scanout => PCD | PWT, + }; + phys | PRESENT | WRITABLE | PAGE_SIZE | typed +} + +/// A 4 KiB leaf at `phys` (Table 4-19), typed as [`block`] types a 2 MiB one; +/// a 4 KiB entry's PAT bit is bit 7, and it stays clear. +pub const fn page(phys: u64, cache: Cache) -> u64 { + block(phys, cache) & !PAGE_SIZE +} diff --git a/toyos-bootmap/tests/plan.rs b/toyos-bootmap/tests/plan.rs index a1e68e9bda7..465326c60cf 100644 --- a/toyos-bootmap/tests/plan.rs +++ b/toyos-bootmap/tests/plan.rs @@ -1,33 +1,55 @@ //! The machines this decision is made for: a framebuffer inside the low map, //! one above it, one on the boundary between them, and one that fits no map. -use toyos_bootmap::{Cache, Plan, Refusal, BOOT_MAP_BYTES, MAX_PAGES, PAGE_2M}; +use toyos_bootmap::{ + aarch64, x86_64, Cache, Entry, Plan, Refusal, Slot, Typing, BOOT_MAP_BYTES, MAX_PAGES, PAGE_2M, + PAGE_4K, +}; const GIB: u64 = 1 << 30; /// 1920x1080x4, which is neither a whole page nor a whole GiB. const PANEL: u64 = 0x7e9000; -/// Every entry lands in a directory the plan named, at an index inside it, and -/// no two entries land in the same place — the property the loader's writes -/// rest on, since it stores each one where the plan says without looking. +/// Every leaf lands in a table the plan named, at an index inside it, no two +/// land in the same place, and no leaf lands where a split page's table is +/// named — the property the loader's writes rest on, since it stores each one +/// where the plan says without looking. fn is_consistent(plan: &Plan) { - let mut seen: Vec<(usize, usize)> = Vec::new(); + let mut seen: Vec = Vec::new(); + let tables: Vec<(usize, usize)> = plan.fine_slots().collect(); for entry in plan.entries() { - assert!(entry.directory < plan.directories().len(), "{entry:?}"); - assert!(entry.index < 512, "{entry:?}"); - assert!(seen.iter().all(|at| *at != (entry.directory, entry.index)), "{entry:?} twice"); - seen.push((entry.directory, entry.index)); + match entry.slot { + Slot::Directory { directory, index } => { + assert!(directory < plan.directories().len(), "{entry:?}"); + assert!(index < 512, "{entry:?}"); + assert!(!tables.contains(&(directory, index)), "{entry:?} over a split page's table"); + } + Slot::Fine { table, index } => { + assert!(table < plan.fine_tables().len(), "{entry:?}"); + assert!(index < 512, "{entry:?}"); + } + } + assert!(!seen.contains(&entry.slot), "{entry:?} twice"); + seen.push(entry.slot); + } + assert!(3 + plan.directories().len() + plan.fine_tables().len() <= MAX_PAGES); +} + +/// A 2 MiB leaf's directory, by position. +fn directory(e: &Entry) -> usize { + match e.slot { + Slot::Directory { directory, .. } => directory, + Slot::Fine { .. } => panic!("{e:?} is a 4 KiB leaf"), } - assert!(3 + plan.directories().len() <= MAX_PAGES); } #[test] fn a_machine_with_no_framebuffer_is_the_low_map_and_nothing_else() { - let plan = Plan::new(None).expect("the low map alone"); + let plan = Plan::new(None, Typing::Firmware).expect("the low map alone"); assert_eq!(plan.directories(), [0, 1, 2, 3]); assert_eq!(plan.scanout(), None); assert_eq!(plan.entries().count() as u64, BOOT_MAP_BYTES / PAGE_2M); - assert!(plan.entries().all(|e| e.cache == Cache::DeferToMtrr)); + assert!(plan.entries().all(|e| e.cache == Cache::Firmware)); is_consistent(&plan); } @@ -35,12 +57,12 @@ fn a_machine_with_no_framebuffer_is_the_low_map_and_nothing_else() { /// retypes pages the low map already holds. #[test] fn a_framebuffer_inside_the_low_map_adds_no_directory() { - let plan = Plan::new(Some((0xc000_0000, PANEL))).expect("inside the low map"); + let plan = Plan::new(Some((0xc000_0000, PANEL)), Typing::Firmware).expect("inside the low map"); assert_eq!(plan.directories(), [0, 1, 2, 3]); // Rounded up to whole pages, and up only. assert_eq!(plan.scanout(), Some((0xc000_0000, 0x80_0000))); let uncacheable: Vec = - plan.entries().filter(|e| e.cache == Cache::Uncacheable).map(|e| e.phys).collect(); + plan.entries().filter(|e| e.cache == Cache::Scanout).map(|e| e.phys).collect(); assert_eq!(uncacheable, [0xc000_0000, 0xc020_0000, 0xc040_0000, 0xc060_0000]); is_consistent(&plan); } @@ -49,15 +71,16 @@ fn a_framebuffer_inside_the_low_map_adds_no_directory() { /// the low map untouched. #[test] fn a_framebuffer_above_the_low_map_adds_its_own_directory() { - let plan = Plan::new(Some((256 * GIB, PANEL))).expect("above the low map"); + let plan = Plan::new(Some((256 * GIB, PANEL)), Typing::Firmware).expect("above the low map"); assert_eq!(plan.directories(), [0, 1, 2, 3, 256]); assert_eq!(plan.scanout(), Some((256 * GIB, 0x80_0000))); let mine: Vec = plan .entries() - .filter(|e| e.cache == Cache::Uncacheable) + .filter(|e| e.cache == Cache::Scanout) .map(|e| { - assert_eq!(e.directory, 4, "the scanout is not in the low map's directories"); - e.index + let Slot::Directory { directory, index } = e.slot else { panic!("{e:?}") }; + assert_eq!(directory, 4, "the scanout is not in the low map's directories"); + index }) .collect(); assert_eq!(mine, [0, 1, 2, 3]); @@ -68,13 +91,13 @@ fn a_framebuffer_above_the_low_map_adds_its_own_directory() { /// past it, not the last GiB of it. #[test] fn a_framebuffer_at_the_boundary_is_outside_the_low_map() { - let plan = Plan::new(Some((BOOT_MAP_BYTES, PANEL))).expect("at the boundary"); + let plan = Plan::new(Some((BOOT_MAP_BYTES, PANEL)), Typing::Firmware).expect("at the boundary"); assert_eq!(plan.directories(), [0, 1, 2, 3, 4]); - assert!(plan.entries().filter(|e| e.cache == Cache::Uncacheable).all(|e| e.directory == 4)); + assert!(plan.entries().filter(|e| e.cache == Cache::Scanout).all(|e| directory(&e) == 4)); is_consistent(&plan); // One page below it is the low map's last page, and adds nothing. - let inside = Plan::new(Some((BOOT_MAP_BYTES - PAGE_2M, PAGE_2M))).expect("the last page"); + let inside = Plan::new(Some((BOOT_MAP_BYTES - PAGE_2M, PAGE_2M)), Typing::Firmware).expect("the last page"); assert_eq!(inside.directories(), [0, 1, 2, 3]); is_consistent(&inside); } @@ -85,12 +108,12 @@ fn a_framebuffer_at_the_boundary_is_outside_the_low_map() { #[test] fn a_framebuffer_that_straddles_the_low_maps_end_is_mapped_from_both_arms() { let base = BOOT_MAP_BYTES - PAGE_2M; - let plan = Plan::new(Some((base, 4 * PAGE_2M))).expect("across the end"); + let plan = Plan::new(Some((base, 4 * PAGE_2M)), Typing::Firmware).expect("across the end"); assert_eq!(plan.directories(), [0, 1, 2, 3, 4]); let mine: Vec<(u64, usize)> = plan .entries() - .filter(|e| e.cache == Cache::Uncacheable) - .map(|e| (e.phys, e.directory)) + .filter(|e| e.cache == Cache::Scanout) + .map(|e| (e.phys, directory(&e))) .collect(); assert_eq!( mine, @@ -111,37 +134,27 @@ fn a_framebuffer_that_straddles_the_low_maps_end_is_mapped_from_both_arms() { #[test] fn a_framebuffer_that_straddles_a_gib_claims_both() { let base = 8 * GIB - PAGE_2M; - let plan = Plan::new(Some((base, 4 * PAGE_2M))).expect("straddling"); + let plan = Plan::new(Some((base, 4 * PAGE_2M)), Typing::Firmware).expect("straddling"); assert_eq!(plan.directories(), [0, 1, 2, 3, 7, 8]); - assert_eq!(3 + plan.directories().len(), MAX_PAGES); + // Every directory the budget has; the two split-page tables are a + // map-typed plan's, and this one splits nothing. + assert_eq!(3 + plan.directories().len() + 2, MAX_PAGES); + assert!(plan.fine_tables().is_empty()); is_consistent(&plan); } -/// The bits the loader stores, decided here: PAT entry 3 is PCD and PWT with -/// the PAT bit clear, and entry 0 is none of the three. -#[test] -fn uncacheable_is_pat_entry_three_and_plain_memory_is_entry_zero() { - const PWT: u64 = 1 << 3; - const PCD: u64 = 1 << 4; - const PAT_2M: u64 = 1 << 12; - assert_eq!(Cache::DeferToMtrr.bits(), 0); - assert_eq!(Cache::Uncacheable.bits(), PCD | PWT); - // The PAT bit is what would select entry 4, and nothing here sets it. - assert_eq!(Cache::Uncacheable.bits() & PAT_2M, 0); - assert_eq!(Cache::DeferToMtrr.bits() & PAT_2M, 0); -} /// The two views, so the loader reads the slots rather than knowing them. #[test] fn the_high_half_slot_is_the_top_nine_bits_of_phys_offset() { const PHYS_OFFSET: u64 = 0xFFFF_8000_0000_0000; - assert_eq!(toyos_bootmap::PML4_IDENTITY, 0); - assert_eq!(toyos_bootmap::PML4_HIGH_HALF, ((PHYS_OFFSET >> 39) & 0x1ff) as usize); + assert_eq!(toyos_bootmap::ROOT_IDENTITY, 0); + assert_eq!(toyos_bootmap::ROOT_HIGH_HALF, ((PHYS_OFFSET >> 39) & 0x1ff) as usize); } #[test] fn a_base_off_the_page_is_refused_rather_than_rounded_down() { - assert_eq!(Plan::new(Some((0xc000_1000, PANEL))), Err(Refusal::Unaligned(0xc000_1000))); + assert_eq!(Plan::new(Some((0xc000_1000, PANEL)), Typing::Firmware), Err(Refusal::Unaligned(0xc000_1000))); // The refusal names the address, because a machine owner reads it. assert!(Refusal::Unaligned(0xc000_1000).to_string().contains("0xc0001000")); } @@ -149,13 +162,141 @@ fn a_base_off_the_page_is_refused_rather_than_rounded_down() { #[test] fn a_range_no_map_can_hold_is_refused_by_name() { // Past the two PDPTs' 512 GiB. - assert_eq!(Plan::new(Some((512 * GIB, PANEL))), Err(Refusal::PastPdpt(512))); + assert_eq!(Plan::new(Some((512 * GIB, PANEL)), Typing::Firmware), Err(Refusal::PastPdpt(512))); // Its own end does not fit an address, either as it is given... let base = u64::MAX - PAGE_2M + 1; - assert_eq!(Plan::new(Some((base, u64::MAX))), Err(Refusal::Extent { base, len: u64::MAX })); + assert_eq!(Plan::new(Some((base, u64::MAX)), Typing::Firmware), Err(Refusal::Extent { base, len: u64::MAX })); // ...or once rounded up to the page it must end on. - assert_eq!(Plan::new(Some((base, 1))), Err(Refusal::Extent { base, len: 1 })); + assert_eq!(Plan::new(Some((base, 1)), Typing::Firmware), Err(Refusal::Extent { base, len: 1 })); // Wider than the directories a plan may name, and told what it would need. - assert_eq!(Plan::new(Some((16 * GIB, 10 * GIB))), Err(Refusal::Directories(14))); + assert_eq!(Plan::new(Some((16 * GIB, 10 * GIB)), Typing::Firmware), Err(Refusal::Directories(14))); assert!(Refusal::Directories(14).to_string().contains("14 page directories")); } + +/// The bits the x86-64 loader stores: PAT entry 3 is PCD and PWT with the PAT +/// bit clear, and entry 0 is none of the three. +#[test] +fn uncacheable_is_pat_entry_three_and_plain_memory_is_entry_zero() { + const PRESENT_WRITABLE_2M: u64 = 1 | 1 << 1 | 1 << 7; + const PWT: u64 = 1 << 3; + const PCD: u64 = 1 << 4; + const PAT_2M: u64 = 1 << 12; + let at = 0x4020_0000; + assert_eq!(x86_64::block(at, Cache::Firmware), at | PRESENT_WRITABLE_2M); + assert_eq!(x86_64::block(at, Cache::Scanout), at | PRESENT_WRITABLE_2M | PCD | PWT); + // The PAT bit is what would select entry 4, and nothing here sets it. + for cache in [Cache::Firmware, Cache::Memory, Cache::Device, Cache::Scanout] { + assert_eq!(x86_64::block(at, cache) & PAT_2M, 0, "{cache:?}"); + } + assert_eq!(x86_64::table(0x1000), 0x1003); +} + +/// QEMU `virt` with 1 GiB: flash and devices below 1 GiB, RAM from 1 GiB to 2 +/// GiB, nothing above. What the map calls memory is memory, the rest a device's. +#[test] +fn a_map_typed_plan_is_memory_where_firmware_says_write_back_and_a_device_elsewhere() { + let ram = [(0x4000_0000, 0x4000_0000)]; + let plan = Plan::new(None, Typing::ByMap(&ram)).expect("virt's map"); + for entry in plan.entries() { + let want = if (0x4000_0000..0x8000_0000).contains(&entry.phys) { Cache::Memory } else { Cache::Device }; + assert_eq!(entry.cache, want, "{entry:?}"); + } + is_consistent(&plan); +} + +/// A page firmware says is write-back in part: no one type is right for it. +#[test] +fn a_page_that_is_part_memory_is_refused_by_address() { + let ram = [(0x4000_0000, 0x4000_0000 + 0x1000)]; + assert_eq!(Plan::new(None, Typing::ByMap(&ram)), Err(Refusal::Mixed(0x8000_0000))); + assert!(Refusal::Mixed(0x8000_0000).to_string().contains("0x80000000")); + // Two adjacent ranges that meet inside a page make it whole. + let split = [(0x4000_0000, 0x4010_0000 - 0x4000_0000), (0x4010_0000, 0x3ff0_0000)]; + assert!(Plan::new(None, Typing::ByMap(&split)).is_ok()); +} + +/// A scanout inside memory is the scanout's leaves, not a mixed page. +#[test] +fn a_scanout_inside_memory_is_typed_as_the_scanout() { + let ram = [(0x4000_0000, 0x4000_0000)]; + let plan = Plan::new(Some((0x5000_0000, 0x10_0000)), Typing::ByMap(&ram)).expect("inside"); + let scanout: Vec = + plan.entries().filter(|e| e.cache == Cache::Scanout).map(|e| e.phys).collect(); + assert_eq!(scanout, (0..256).map(|page| 0x5000_0000 + page * PAGE_4K).collect::>()); + is_consistent(&plan); + // And a scanout over a page that is part memory takes that page whole. + let part = [(0x4000_0000, 0x1000)]; + assert!(Plan::new(Some((0x4000_0000, PAGE_2M)), Typing::ByMap(&part)).is_ok()); + assert_eq!(Plan::new(None, Typing::ByMap(&part)), Err(Refusal::Mixed(0x4000_0000))); +} + +/// The AArch64 descriptors, against the Arm ARM's bit positions (K.a, D8.3.1 +/// and D8.3.2): a table is 0b11; a block 0b01 with `AttrIndx` at 4:2, `SH` at +/// 9:8, `AF` at 10, `PXN` at 53 and `UXN` at 54; and `MAIR_EL1`'s bytes are the +/// Device-nGnRE, Normal write-back and Normal non-cacheable encodings of +/// D24.2.110. +#[test] +fn the_aarch64_descriptors_carry_the_attribute_index_the_mair_names() { + let at = 0x4020_0000; + let attr_index = |d: u64| (d >> 2) & 0b111; + let memory = aarch64::block(at, Cache::Memory); + let device = aarch64::block(at, Cache::Device); + let scanout = aarch64::block(at, Cache::Scanout); + for d in [memory, device, scanout] { + assert_eq!(d & 0b11, 0b01, "a block"); + assert_ne!(d & 1 << 10, 0, "the access flag"); + assert_eq!(d & 0b11 << 6, 0, "EL1 read-write, EL0 nothing"); + assert_ne!(d & 1 << 54, 0, "never executable at EL0"); + assert_eq!(d & 0x0000_FFFF_FFE0_0000, at, "the output address"); + } + assert_eq!(aarch64::ATTRS[attr_index(memory) as usize], 0xFF); + assert_eq!(aarch64::ATTRS[attr_index(device) as usize], 0x04); + assert_eq!(aarch64::ATTRS[attr_index(scanout) as usize], 0x44); + assert_eq!(memory & 1 << 53, 0, "the kernel executes from memory"); + assert_ne!(device & 1 << 53, 0, "and never from a device"); + assert_eq!((memory >> 8) & 0b11, 0b11, "memory is inner shareable"); + assert_eq!(aarch64::MAIR, 0x44_FF_04); + // A level-3 page is 0b11 with the same attributes. + assert_eq!(aarch64::page(0x5000_1000, Cache::Scanout), (scanout & !0x0000_FFFF_FFFF_F000 & !0b11) | 0x5000_1000 | 0b11); + assert_eq!(aarch64::table(0x1000), 0x1003); +} + +/// QEMU `virt`'s ramfb as AAVMF placed it on a real boot: 3 MiB carved out of +/// RAM at 0xbc7a0000, on no 2 MiB boundary. Its first and last 2 MiB pages +/// are split, so the scanout's type reaches exactly its own bytes and the RAM +/// beside it stays memory; the page wholly inside it stays one leaf. +#[test] +fn a_scanout_carved_out_of_ram_splits_the_pages_it_shares() { + let ram = [(0x4000_0000, 0x8000_0000)]; + let (base, len) = (0xbc7a_0000, 0x30_0000); + let plan = Plan::new(Some((base, len)), Typing::ByMap(&ram)).expect("ramfb"); + assert_eq!(plan.fine_tables(), [0xbc60_0000, 0xbca0_0000]); + let scanout: Vec = plan.entries().filter(|e| e.cache == Cache::Scanout).collect(); + let bytes: u64 = scanout + .iter() + .map(|e| match e.slot { + Slot::Directory { .. } => PAGE_2M, + Slot::Fine { .. } => PAGE_4K, + }) + .sum(); + assert_eq!(bytes, len, "the scanout's type covers its own bytes and none beside them"); + assert!(scanout.iter().all(|e| e.phys >= base && e.phys < base + len)); + // The 4 KiB leaf just below it and just past it are memory. + let at = |phys: u64| plan.entries().find(|e| e.phys == phys).expect("mapped").cache; + assert_eq!(at(base - PAGE_4K), Cache::Memory); + assert_eq!(at(base + len), Cache::Memory); + assert_eq!(at(0xbc80_0000), Cache::Scanout); + is_consistent(&plan); + // The same framebuffer under range-register typing is refused, as before. + assert_eq!(Plan::new(Some((base, len)), Typing::Firmware), Err(Refusal::Unaligned(base))); +} + +/// A map-typed scanout must still start on a 4 KiB page. +#[test] +fn a_map_typed_scanout_off_a_small_page_is_refused() { + let ram = [(0x4000_0000, 0x8000_0000)]; + assert_eq!( + Plan::new(Some((0xbc7a_0800, 0x1000)), Typing::ByMap(&ram)), + Err(Refusal::Unaligned(0xbc7a_0800)) + ); +} diff --git a/toyos-elf/src/header.rs b/toyos-elf/src/header.rs index e5e1ed872c2..73714819706 100644 --- a/toyos-elf/src/header.rs +++ b/toyos-elf/src/header.rs @@ -22,15 +22,36 @@ const ELFDATA2LSB: u8 = 1; const EV_CURRENT: u8 = 1; const ET_DYN: u16 = 3; const EM_X86_64: u16 = 62; +const EM_AARCH64: u16 = 183; /// The instruction set an image is built for. /// -/// One variant, because one architecture boots. ARM64 adds a variant here and -/// a relocation set to [`crate::rela`]; nothing else in this crate is +/// Each machine names its own relocation numbers ([`crate::rela`]) and its own +/// TLS variant ([`crate::tls`]); nothing else in this crate is /// per-architecture. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum Machine { X86_64, + Aarch64, +} + +impl Machine { + /// The machine an `e_machine` names, or `None` for one ToyOS does not run. + pub const fn from_raw(e_machine: u16) -> Option { + match e_machine { + EM_X86_64 => Some(Machine::X86_64), + EM_AARCH64 => Some(Machine::Aarch64), + _ => None, + } + } + + /// The `e_machine` a file built for this machine carries. + pub const fn raw(self) -> u16 { + match self { + Machine::X86_64 => EM_X86_64, + Machine::Aarch64 => EM_AARCH64, + } + } } #[derive(Clone, Copy, Debug)] @@ -72,9 +93,8 @@ impl FileHeader { if e_type != ET_DYN { return Err(Error::NotPie); } - if read::u16_at(data, 18).ok_or(Error::TooSmall)? != EM_X86_64 { - return Err(Error::WrongMachine); - } + let machine = Machine::from_raw(read::u16_at(data, 18).ok_or(Error::TooSmall)?) + .ok_or(Error::UnknownMachine)?; let phnum = read::u16_at(data, 56).ok_or(Error::TooSmall)?; if phnum == 0 { return Err(Error::NoProgramHeaders); @@ -97,7 +117,7 @@ impl FileHeader { } Ok(FileHeader { - machine: Machine::X86_64, + machine, entry: read::u64_at(data, 24).ok_or(Error::TooSmall)?, phoff, phnum, diff --git a/toyos-elf/src/layout.rs b/toyos-elf/src/layout.rs index e754b334ae1..2d364ad36f9 100644 --- a/toyos-elf/src/layout.rs +++ b/toyos-elf/src/layout.rs @@ -6,7 +6,8 @@ //! site. use crate::header::{ - FileHeader, ProgramHeader, PT_DYNAMIC, PT_GNU_EH_FRAME, PT_LOAD, PT_TLS, SECTION_HEADER_SIZE, + FileHeader, Machine, ProgramHeader, PT_DYNAMIC, PT_GNU_EH_FRAME, PT_LOAD, PT_TLS, + SECTION_HEADER_SIZE, }; use crate::{Error, MAX_LOAD_SEGMENTS, MAX_TLS_ALIGN}; @@ -151,8 +152,12 @@ impl Layout { /// /// `data` need only reach the end of the program header table; the loader /// hands it 4 KiB and never reads a segment's contents to get here. - pub fn parse(data: &[u8]) -> Result { + /// `machine` is the one the caller runs: an image for any other is refused. + pub fn parse(data: &[u8], machine: Machine) -> Result { let ehdr = FileHeader::parse(data)?; + if ehdr.machine != machine { + return Err(Error::WrongMachine); + } let phdrs = ehdr.program_headers(data)?; let mut segments = [Segment { diff --git a/toyos-elf/src/lib.rs b/toyos-elf/src/lib.rs index ef0f167a393..29e68681c8f 100644 --- a/toyos-elf/src/lib.rs +++ b/toyos-elf/src/lib.rs @@ -24,11 +24,11 @@ //! //! # Scope //! -//! ELF64, little-endian, `ET_DYN`, `EM_X86_64`. Everything else is refused by -//! name rather than tolerated: ToyOS emits PIE binaries only, has no 32-bit -//! mode and no big-endian target. A second architecture adds a machine to -//! [`header::Machine`] and a relocation set to [`rela`], not a class or an -//! endianness. +//! ELF64, little-endian, `ET_DYN`, `EM_X86_64` or `EM_AARCH64`. Everything else +//! is refused by name rather than tolerated: ToyOS emits PIE binaries only, has +//! no 32-bit mode and no big-endian target. A loader names the machine it runs +//! ([`Layout::parse`]), so an image for the other one is refused as +//! [`Error::WrongMachine`] before anything of it is mapped. #![no_std] #![forbid(unsafe_code)] @@ -44,7 +44,7 @@ pub mod tls; pub use dynamic::{Dynamic, Table}; pub use gnu_hash::GnuHash; -pub use header::FileHeader; +pub use header::{FileHeader, Machine}; pub use layout::{Layout, Segment, SegmentFlags, SectionTableRef, TlsSegment}; pub use rela::{Rela, RelaCounts, RelaTable, RelocError, RelocKind}; pub use section::{SectionHeader, SectionTable}; @@ -92,7 +92,9 @@ pub enum Error { BadVersion, /// Not `ET_DYN`. ToyOS loads position-independent executables only. NotPie, - /// Not `EM_X86_64`. + /// `e_machine` names no machine ToyOS runs. + UnknownMachine, + /// Built for a machine other than the one loading it. WrongMachine, /// `e_phnum` is zero: nothing to map. NoProgramHeaders, @@ -140,7 +142,8 @@ impl Error { Error::NotLittleEndian => "ELF: not ELFDATA2LSB", Error::BadVersion => "ELF: e_ident version is not EV_CURRENT", Error::NotPie => "ELF: not PIE (expected ET_DYN)", - Error::WrongMachine => "ELF: not x86_64", + Error::UnknownMachine => "ELF: e_machine is neither x86_64 nor aarch64", + Error::WrongMachine => "ELF: built for another machine", Error::NoProgramHeaders => "ELF: no program headers", Error::BadProgramHeaderSize => "ELF: e_phentsize is not 56", Error::ProgramHeadersInsideFileHeader => "ELF: e_phoff points inside the file header", diff --git a/toyos-elf/src/rela.rs b/toyos-elf/src/rela.rs index 936df370baf..b5759eb1215 100644 --- a/toyos-elf/src/rela.rs +++ b/toyos-elf/src/rela.rs @@ -7,45 +7,52 @@ //! a type in it that no writer handles would be validated for a write that //! never happens. Neither can drift, because there is one table. +use crate::header::Machine; use crate::read; /// Bytes in one `Elf64_Rela`. pub const ENTRY_SIZE: usize = 24; -/// The x86-64 relocations this loader knows about. +/// The dynamic relocations this loader knows about, by what they ask for +/// rather than by any one machine's number: [`RelocKind::from_raw`] is the +/// only place a number is read. /// /// `Other` carries the raw type rather than dropping it, so a log line can name /// what it skipped. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum RelocKind { - /// `R_X86_64_GLOB_DAT` + /// `R_X86_64_GLOB_DAT`, `R_AARCH64_GLOB_DAT` GlobDat, - /// `R_X86_64_JUMP_SLOT` + /// `R_X86_64_JUMP_SLOT`, `R_AARCH64_JUMP_SLOT` JumpSlot, - /// `R_X86_64_RELATIVE` + /// `R_X86_64_RELATIVE`, `R_AARCH64_RELATIVE` Relative, - /// `R_X86_64_DTPMOD64` + /// `R_X86_64_DTPMOD64`, `R_AARCH64_TLS_DTPMOD` DtpMod64, - /// `R_X86_64_DTPOFF64` + /// `R_X86_64_DTPOFF64`, `R_AARCH64_TLS_DTPREL` DtpOff64, - /// `R_X86_64_TPOFF64` + /// `R_X86_64_TPOFF64`, `R_AARCH64_TLS_TPREL` Tpoff64, - /// `R_X86_64_TPOFF32` + /// `R_X86_64_TPOFF32`; AArch64 has no 32-bit thread-pointer offset. Tpoff32, + /// `R_AARCH64_TLSDESC`: a TLS descriptor, whose resolver this loader does + /// not have, so [`validate`] refuses it. + TlsDesc, Other(u32), } impl RelocKind { - pub const fn from_raw(r_type: u32) -> RelocKind { - match r_type { - 6 => RelocKind::GlobDat, - 7 => RelocKind::JumpSlot, - 8 => RelocKind::Relative, - 16 => RelocKind::DtpMod64, - 17 => RelocKind::DtpOff64, - 18 => RelocKind::Tpoff64, - 23 => RelocKind::Tpoff32, - other => RelocKind::Other(other), + pub const fn from_raw(machine: Machine, r_type: u32) -> RelocKind { + match (machine, r_type) { + (Machine::X86_64, 6) | (Machine::Aarch64, 1025) => RelocKind::GlobDat, + (Machine::X86_64, 7) | (Machine::Aarch64, 1026) => RelocKind::JumpSlot, + (Machine::X86_64, 8) | (Machine::Aarch64, 1027) => RelocKind::Relative, + (Machine::X86_64, 16) | (Machine::Aarch64, 1028) => RelocKind::DtpMod64, + (Machine::X86_64, 17) | (Machine::Aarch64, 1029) => RelocKind::DtpOff64, + (Machine::X86_64, 18) | (Machine::Aarch64, 1030) => RelocKind::Tpoff64, + (Machine::X86_64, 23) => RelocKind::Tpoff32, + (Machine::Aarch64, 1031) => RelocKind::TlsDesc, + (_, other) => RelocKind::Other(other), } } @@ -60,7 +67,7 @@ impl RelocKind { | RelocKind::DtpOff64 | RelocKind::Tpoff64 => Some(8), RelocKind::Tpoff32 => Some(4), - RelocKind::Other(_) => None, + RelocKind::TlsDesc | RelocKind::Other(_) => None, } } @@ -93,11 +100,13 @@ pub struct Rela { #[derive(Clone, Copy, Debug)] pub struct RelaTable<'a> { data: &'a [u8], + machine: Machine, } impl<'a> RelaTable<'a> { - pub const fn new(data: &'a [u8]) -> RelaTable<'a> { - RelaTable { data } + /// `machine` decides what each entry's type number means. + pub const fn new(data: &'a [u8], machine: Machine) -> RelaTable<'a> { + RelaTable { data, machine } } /// Whole entries the bytes hold. A trailing partial entry is not an entry. @@ -118,7 +127,7 @@ impl<'a> RelaTable<'a> { Some(Rela { offset: read::u64_at(self.data, off)?, sym: (info >> 32) as u32, - kind: RelocKind::from_raw(info as u32), + kind: RelocKind::from_raw(self.machine, info as u32), addend: read::i64_at(self.data, off + 16)?, }) } @@ -144,6 +153,7 @@ pub struct RelaCounts { pub tpoff32: usize, pub dtpmod64: usize, pub dtpoff64: usize, + pub tlsdesc: usize, } impl RelaCounts { @@ -157,6 +167,7 @@ impl RelaCounts { RelocKind::Tpoff32 => &mut counts.tpoff32, RelocKind::DtpMod64 => &mut counts.dtpmod64, RelocKind::DtpOff64 => &mut counts.dtpoff64, + RelocKind::TlsDesc => &mut counts.tlsdesc, RelocKind::Other(_) => continue, }; *slot += 1; @@ -175,6 +186,26 @@ impl RelaCounts { kinds.iter().map(|&k| self.count_of(k)).max().unwrap_or(0) } + /// What an executable's loader reserves for each group it keeps, at `width` + /// bytes an entry, or why it keeps none: a TLS descriptor, which only a + /// resolver this loader does not have can fill, or a group that would + /// not fit `max_bytes`. The reservation is had only through this refusal. + pub fn for_executable(&self, width: usize, max_bytes: usize) -> Result { + if self.tlsdesc != 0 { + return Err(ExeRefusal::TlsDescriptor); + } + let kept = [RelocKind::Relative, RelocKind::GlobDat, RelocKind::Tpoff64, RelocKind::Tpoff32]; + if self.max_of(&kept).checked_mul(width).is_none_or(|b| b > max_bytes) { + return Err(ExeRefusal::TooLarge); + } + Ok(ExeReservation { + relative: self.relative, + bind: self.bind, + tpoff64: self.tpoff64, + tpoff32: self.tpoff32, + }) + } + pub fn count_of(&self, kind: RelocKind) -> usize { match kind { RelocKind::Relative => self.relative, @@ -183,6 +214,7 @@ impl RelaCounts { RelocKind::Tpoff32 => self.tpoff32, RelocKind::DtpMod64 => self.dtpmod64, RelocKind::DtpOff64 => self.dtpoff64, + RelocKind::TlsDesc => self.tlsdesc, RelocKind::Other(_) => 0, } } @@ -201,6 +233,36 @@ pub struct FillLattice { /// The demand-fault page an executable's relocations are filled in. pub const FILL_GRANULE: u64 = 4096; +/// How many entries of each group an executable's loader keeps: made only by +/// [`RelaCounts::for_executable`]. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[non_exhaustive] +pub struct ExeReservation { + pub relative: usize, + /// `GLOB_DAT` and `JUMP_SLOT`. + pub bind: usize, + pub tpoff64: usize, + pub tpoff32: usize, +} + +/// Why an executable's relocations are refused before any is kept. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ExeRefusal { + /// [`RelocError::TlsDescriptor`]'s reason. + TlsDescriptor, + /// A group would not fit one allocation. + TooLarge, +} + +impl ExeRefusal { + pub const fn as_str(self) -> &'static str { + match self { + ExeRefusal::TlsDescriptor => RelocError::TlsDescriptor.as_str(), + ExeRefusal::TooLarge => "ELF: a relocation group does not fit one allocation", + } + } +} + /// Why a relocation cannot be applied. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum RelocError { @@ -212,6 +274,9 @@ pub enum RelocError { SymbolPastTable, /// The write would cross a fill page, so a chunked writer would drop it. StraddlesFillPage, + /// A TLS descriptor, which only a resolver this loader does not have can + /// fill. + TlsDescriptor, } impl RelocError { @@ -221,6 +286,7 @@ impl RelocError { RelocError::OutsideWindow => "ELF: relocation r_offset outside the writable image", RelocError::SymbolPastTable => "ELF: relocation r_sym past .dynsym", RelocError::StraddlesFillPage => "ELF: relocation crosses a fill-page boundary", + RelocError::TlsDescriptor => "ELF: R_AARCH64_TLSDESC has no resolver in this loader", } } } @@ -291,6 +357,9 @@ pub fn validate( ) -> Result<(), RelocError> { let (lo, hi) = window; for rela in entries { + if rela.kind == RelocKind::TlsDesc { + return Err(RelocError::TlsDescriptor); + } let Some(width) = rela.kind.write_width() else { continue; }; diff --git a/toyos-elf/src/tls.rs b/toyos-elf/src/tls.rs index b85b597f57f..91806cb2fab 100644 --- a/toyos-elf/src/tls.rs +++ b/toyos-elf/src/tls.rs @@ -1,18 +1,60 @@ -//! Where everything goes inside a thread's TLS allocation. -//! -//! x86-64 variant II, plus the DTV the kernel writes in front of it: +//! Where everything goes inside a thread's TLS allocation, in the layout each +//! machine's psABI names ([`Variant`]), with the DTV the kernel writes at the +//! front of the allocation: //! //! ```text -//! [DTV] [alignment padding] [TLS data (.tdata + .tbss)] [TCB] -//! ^ data_start ^ thread pointer +//! variant II (x86-64): [DTV] [pad] [TLS data (.tdata + .tbss)] [TCB] +//! ^ tls_start ^ thread pointer +//! variant I (AArch64): [DTV] [pad] [TCB] [TLS data (.tdata + .tbss)] +//! ^ thread pointer, tls_start - gap //! ``` //! -//! The linker computes `TPOFF = sym_offset - memsz` raw, so the thread pointer -//! sits at `data_start + memsz` and `data_start` must carry the largest -//! alignment any module asked for. Both of those are sums of numbers a file -//! declared, which is why every step here is checked and the whole thing is a -//! pure function: `dtv_bytes <= tls_start` is the property, and it used to be -//! an assertion in the kernel reached from a crafted `PT_TLS`. +//! Variant II: the linker computes `TPOFF = sym_offset - memsz` raw, so the +//! thread pointer sits at `tls_start + memsz`. Variant I: the linker computes +//! `TPOFF = align_up(16, p_align) + sym_offset` from the executable's own +//! `PT_TLS`, so the executable's block is the first one, `gap` above the +//! thread pointer. Either way `tls_start` carries the largest alignment any +//! module asked for. Every input is a sum of numbers a file declared, which is +//! why every step here is checked and the whole thing is a pure function: +//! `dtv_bytes <= tls_start` is the property, and it used to be an assertion in +//! the kernel reached from a crafted `PT_TLS`. + +use crate::header::Machine; + +/// Which of the psABIs' two TLS layouts a machine uses. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Variant { + /// The thread pointer addresses a two-word TCB, `[DTV, reserved]`, and the + /// first module's data follows it. + I, + /// The thread pointer addresses a TCB, `[self, DTV]`, after the last + /// module's data. + II, +} + +impl Variant { + pub const fn of(machine: Machine) -> Variant { + match machine { + Machine::X86_64 => Variant::II, + Machine::Aarch64 => Variant::I, + } + } +} + +/// A thread's static TLS, as much of it as its layout depends on. Its +/// alignments are powers of two, or it does not exist ([`Static::new`]). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Static { + variant: Variant, + /// Every static module's bytes, placed by [`place_module`]. + total_memsz: usize, + /// The largest `p_align` any static module declared, "no constraint" as 8. + max_align: usize, + /// The `p_align` of the module at offset 0, "no constraint" as 8: variant + /// I's executable, whose linker fixed its distance from the thread pointer + /// from it. + first_align: usize, +} /// A planned TLS allocation, in offsets from the base of one block. #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -22,56 +64,88 @@ pub struct TlsBlock { /// Where the first module's TLS data begins. Aligned to the requested /// alignment, and never below `dtv_bytes`. pub tls_start: usize, - /// Where the thread pointer goes: `tls_start + total_memsz`. + /// Where the thread pointer goes. pub tp_offset: usize, } -/// Lay out one thread's TLS block, or `None` for a layout no allocation can -/// hold. -/// -/// `align` is `PT_TLS`'s `p_align` — zero or one mean "no constraint" and -/// become 8. It has already been established a power of two no larger than -/// [`crate::MAX_TLS_ALIGN`] by [`crate::Layout::parse`], which is what makes -/// `!(align - 1)` a mask here. -/// -/// `granule` is the allocation unit the block is rounded up to, and -/// `dtv_bytes` the fixed header the kernel writes at offset zero. Both are -/// kernel constants; everything else came out of a file. -pub fn plan( - total_memsz: usize, - align: usize, - tcb_size: usize, - dtv_bytes: usize, - granule: usize, -) -> Option { - debug_assert!(granule.is_power_of_two()); - if align != 0 && !align.is_power_of_two() { - return None; +impl Static { + /// `None` for an alignment that is not a power of two: a mask that is not + /// a mask can place the data anywhere. Zero and one mean "no constraint". + pub fn new(variant: Variant, total_memsz: usize, max_align: usize, first_align: usize) -> Option { + Some(Static { variant, total_memsz, max_align: effective(max_align)?, first_align: effective(first_align)? }) + } + + pub fn total_memsz(self) -> usize { + self.total_memsz } - let align = if align > 1 { align } else { 8 }; - let block_size = total_memsz.checked_add(tcb_size)?; - // The DTV goes at the start of this same allocation and the TLS data is - // placed `align`-aligned above it, so both belong in the size. Sizing from - // the block and the alignment alone left `tls_start` free to land inside - // the DTV. - let alloc_size = align_up(block_size.checked_add(dtv_bytes)?.checked_add(align)?, granule)?; + pub fn max_align(self) -> usize { + self.max_align + } - // Rounding *down* by `align` loses less than `align`, and `align` was one - // of the addends — so this is at least `dtv_bytes + 1` and the DTV can - // never be overwritten by TLS data. - let tls_start = (alloc_size - block_size) & !(align - 1); + /// No module at all: what a thread of a program without TLS is given. + pub const fn empty(variant: Variant) -> Static { + Static { variant, total_memsz: 0, max_align: 8, first_align: 8 } + } + + /// Lay out one thread's TLS block, or `None` for a layout no allocation + /// can hold. + /// + /// `granule` is the allocation unit the block is rounded up to, + /// `dtv_bytes` the fixed header the kernel writes at offset zero, and + /// `tcb_size` variant II's TCB; variant I's is the `gap` below its data. + /// All three are kernel constants; everything else came out of a file. + pub fn plan(self, tcb_size: usize, dtv_bytes: usize, granule: usize) -> Option { + debug_assert!(granule.is_power_of_two()); + let align = self.max_align; + match self.variant { + Variant::II => { + let block_size = self.total_memsz.checked_add(tcb_size)?; + // The DTV goes at the start of this same allocation and the TLS + // data is placed `align`-aligned above it, so both belong in + // the size. Sizing from the block and the alignment alone left + // `tls_start` free to land inside the DTV. + let alloc_size = + align_up(block_size.checked_add(dtv_bytes)?.checked_add(align)?, granule)?; + // Rounding *down* by `align` loses less than `align`, and + // `align` was one of the addends — so this is at least + // `dtv_bytes + 1` and the DTV can never be overwritten by TLS + // data. + let tls_start = (alloc_size - block_size) & !(align - 1); + Some(TlsBlock { alloc_size, tls_start, tp_offset: tls_start + self.total_memsz }) + } + Variant::I => { + let gap = self.gap(); + // At least 16, so the thread pointer `gap` below is 16-aligned. + let tls_start = align_up(dtv_bytes.checked_add(gap)?, align.max(16))?; + let alloc_size = align_up(tls_start.checked_add(self.total_memsz)?, granule)?; + Some(TlsBlock { alloc_size, tls_start, tp_offset: tls_start - gap }) + } + } + } - Some(TlsBlock { - alloc_size, - tls_start, - tp_offset: tls_start + total_memsz, - }) + /// A static-TLS datum's initial-exec offset from the thread pointer: psABI + /// `S + A - tp`, where `module_addr` is `S` (its module's `base_offset` + /// plus the datum's offset) from `tls_start`. Every `TPOFF` branch passes + /// the addend here, so none can drop `A`. + pub fn tpoff(self, module_addr: u64, addend: i64) -> i64 { + let from_start = module_addr as i64 + addend; + match self.variant { + Variant::II => from_start - self.total_memsz as i64, + Variant::I => from_start + self.gap() as i64, + } + } + + /// Variant I's distance from the thread pointer to the first module's + /// data: `align_up(16, p_align)` of that module, the linker's own. + fn gap(self) -> usize { + self.first_align.max(16) + } } /// One module's placement in a combined block: `cursor` rounded up to the -/// module's own `p_align` (psABI variant II, not a shared constant), floored at -/// the 16 `cmpxchg16b` needs. `align` is a power of two ≤ [`crate::MAX_TLS_ALIGN`] +/// module's own `p_align` (psABI, not a shared constant), floored at the 16 +/// `cmpxchg16b` needs. `align` is a power of two ≤ [`crate::MAX_TLS_ALIGN`] /// by [`crate::Layout::parse`]; `tls_start` carries the max, so a base on the /// module's own align lands the module on it. pub fn place_module(cursor: usize, memsz: usize, align: usize) -> Option<(usize, usize)> { @@ -80,12 +154,12 @@ pub fn place_module(cursor: usize, memsz: usize, align: usize) -> Option<(usize, Some((base, base.checked_add(memsz)?)) } -/// A static-TLS datum's initial-exec offset from the thread pointer: psABI -/// `S + A - tp`, where `module_addr` is `S` (its module's `base_offset` plus the -/// datum's offset) and `tp` sits `total_memsz` past the same origin. Every -/// `TPOFF` branch passes the addend here, so none can drop `A`. -pub fn tpoff(module_addr: u64, addend: i64, total_memsz: usize) -> i64 { - module_addr as i64 + addend - total_memsz as i64 +/// `align`, with "no constraint" as 8; `None` for one that is not a power of two. +fn effective(align: usize) -> Option { + if align != 0 && !align.is_power_of_two() { + return None; + } + Some(if align > 1 { align } else { 8 }) } fn align_up(value: usize, granule: usize) -> Option { diff --git a/toyos-elf/tests/common/mod.rs b/toyos-elf/tests/common/mod.rs index cc9e3b6fa7a..6f3817659c5 100644 --- a/toyos-elf/tests/common/mod.rs +++ b/toyos-elf/tests/common/mod.rs @@ -11,6 +11,7 @@ pub const ET_DYN: u16 = 3; pub const ET_EXEC: u16 = 2; pub const EM_X86_64: u16 = 62; pub const EM_AARCH64: u16 = 183; +pub const EM_386: u16 = 3; pub const PT_LOAD: u32 = 1; pub const PT_DYNAMIC: u32 = 2; diff --git a/toyos-elf/tests/crafted.rs b/toyos-elf/tests/crafted.rs index 44801e24ff4..9ae008d8537 100644 --- a/toyos-elf/tests/crafted.rs +++ b/toyos-elf/tests/crafted.rs @@ -15,37 +15,37 @@ mod common; use common::*; use toyos_elf::header::PROGRAM_HEADER_SIZE; -use toyos_elf::{Error, Layout, MAX_LOAD_SEGMENTS}; +use toyos_elf::{Error, Layout, Machine, MAX_LOAD_SEGMENTS}; fn refused(bytes: Vec) -> Error { - match Layout::parse(&bytes) { + match Layout::parse(&bytes, Machine::X86_64) { Ok(_) => panic!("accepted a file that must be refused"), Err(e) => e, } } fn accepted(bytes: Vec) -> Layout { - Layout::parse(&bytes).expect("refused a file that must be accepted") + Layout::parse(&bytes, Machine::X86_64).expect("refused a file that must be accepted") } // ── The bytes before the program headers ──────────────────────────────── #[test] fn an_empty_file_is_too_small() { - assert_eq!(Layout::parse(&[]).unwrap_err(), Error::TooSmall); + assert_eq!(Layout::parse(&[], Machine::X86_64).unwrap_err(), Error::TooSmall); } #[test] fn a_header_one_byte_short_is_too_small() { let bytes = Elf::honest(0x1000).build(); - assert_eq!(Layout::parse(&bytes[..63]).unwrap_err(), Error::TooSmall); + assert_eq!(Layout::parse(&bytes[..63], Machine::X86_64).unwrap_err(), Error::TooSmall); } #[test] fn the_magic_is_checked() { let mut bytes = Elf::honest(0x1000).build(); bytes[1] = b'e'; - assert_eq!(Layout::parse(&bytes).unwrap_err(), Error::BadMagic); + assert_eq!(Layout::parse(&bytes, Machine::X86_64).unwrap_err(), Error::BadMagic); } #[test] @@ -54,7 +54,16 @@ fn class_endianness_version_type_and_machine_are_each_refused_by_name() { assert_eq!(refused(Elf::honest(0x1000).endian(2).build()), Error::NotLittleEndian); assert_eq!(refused(Elf::honest(0x1000).version(0).build()), Error::BadVersion); assert_eq!(refused(Elf::honest(0x1000).kind(ET_EXEC).build()), Error::NotPie); - assert_eq!(refused(Elf::honest(0x1000).machine(EM_AARCH64).build()), Error::WrongMachine); + assert_eq!(refused(Elf::honest(0x1000).machine(EM_386).build()), Error::UnknownMachine); +} + +#[test] +fn an_image_for_the_other_machine_is_refused_and_one_for_this_machine_is_not() { + let arm = Elf::honest(0x1000).machine(EM_AARCH64).build(); + assert_eq!(refused(arm.clone()), Error::WrongMachine); + assert!(Layout::parse(&arm, Machine::Aarch64).is_ok()); + let x86 = Elf::honest(0x1000).build(); + assert_eq!(Layout::parse(&x86, Machine::Aarch64).unwrap_err(), Error::WrongMachine); } #[test] diff --git a/toyos-elf/tests/real.rs b/toyos-elf/tests/real.rs index aa54dee1f8c..4e2530797b1 100644 --- a/toyos-elf/tests/real.rs +++ b/toyos-elf/tests/real.rs @@ -10,13 +10,13 @@ //! of=toyos-elf/tests/fixtures/toyos-ld-headers.bin bs=1 count=4096`, and //! expect the entry point below to move. -use toyos_elf::Layout; +use toyos_elf::{Layout, Machine}; const HEADERS: &[u8] = include_bytes!("fixtures/toyos-ld-headers.bin"); #[test] fn a_toyos_ld_binary_parses_to_what_readelf_says() { - let layout = Layout::parse(HEADERS).expect("toyos-ld's own output"); + let layout = Layout::parse(HEADERS, Machine::X86_64).expect("toyos-ld's own output"); assert_eq!(layout.entry, 0x3261c); assert_eq!(layout.vaddr_min, 0); diff --git a/toyos-elf/tests/tables.rs b/toyos-elf/tests/tables.rs index 7a2395d6f23..a93dbd3a33a 100644 --- a/toyos-elf/tests/tables.rs +++ b/toyos-elf/tests/tables.rs @@ -18,6 +18,7 @@ use toyos_elf::gnu_hash::{self, GnuHash}; use toyos_elf::rela::{self, RelaCounts, RelaTable, RelocError, RelocKind}; use toyos_elf::section::{SectionTable, SHT_DYNSYM, SHT_RELA, SHT_SYMTAB}; use toyos_elf::sym::SymTab; +use toyos_elf::Machine; /// `st_info` for a global `STT_FUNC`, and for the data object it is told apart /// from. @@ -80,7 +81,7 @@ fn everything_after_dt_null_is_ignored() { fn a_trailing_partial_relocation_is_not_a_relocation() { let mut bytes = rela(0x10, 0, 8, 4).to_vec(); bytes.extend_from_slice(&[0u8; 23]); - let table = RelaTable::new(&bytes); + let table = RelaTable::new(&bytes, Machine::X86_64); assert_eq!(table.len(), 1); assert_eq!(table.get(1), None); assert_eq!(table.iter().count(), 1); @@ -88,15 +89,15 @@ fn a_trailing_partial_relocation_is_not_a_relocation() { #[test] fn relocation_types_map_to_the_width_the_writers_use() { - assert_eq!(RelocKind::from_raw(8).write_width(), Some(8)); - assert_eq!(RelocKind::from_raw(23).write_width(), Some(4)); - assert_eq!(RelocKind::from_raw(0).write_width(), None); - assert_eq!(RelocKind::from_raw(42).write_width(), None); + assert_eq!(RelocKind::from_raw(Machine::X86_64, 8).write_width(), Some(8)); + assert_eq!(RelocKind::from_raw(Machine::X86_64, 23).write_width(), Some(4)); + assert_eq!(RelocKind::from_raw(Machine::X86_64, 0).write_width(), None); + assert_eq!(RelocKind::from_raw(Machine::X86_64, 42).write_width(), None); // RELATIVE is the one written type that resolves no symbol, so it is the // one whose `r_sym` needs no bound. assert!(!RelocKind::Relative.needs_symbol()); for raw in [6u32, 7, 16, 17, 18, 23] { - assert!(RelocKind::from_raw(raw).needs_symbol(), "type {raw}"); + assert!(RelocKind::from_raw(Machine::X86_64, raw).needs_symbol(), "type {raw}"); } } @@ -106,25 +107,25 @@ fn validation_refuses_a_write_outside_the_window_by_name() { let overflowing = [rela(u64::MAX - 3, 0, 8, 0)].concat(); assert_eq!( - rela::validate(RelaTable::new(&overflowing).iter(), window, 4, None), + rela::validate(RelaTable::new(&overflowing, Machine::X86_64).iter(), window, 4, None), Err(RelocError::OffsetOverflows), ); let below = [rela(0xFF8, 0, 8, 0)].concat(); assert_eq!( - rela::validate(RelaTable::new(&below).iter(), window, 4, None), + rela::validate(RelaTable::new(&below, Machine::X86_64).iter(), window, 4, None), Err(RelocError::OutsideWindow), ); // One byte of an eight-byte write past the end. let straddling = [rela(0x1FF9, 0, 8, 0)].concat(); assert_eq!( - rela::validate(RelaTable::new(&straddling).iter(), window, 4, None), + rela::validate(RelaTable::new(&straddling, Machine::X86_64).iter(), window, 4, None), Err(RelocError::OutsideWindow), ); let fits = [rela(0x1FF8, 0, 8, 0)].concat(); - assert_eq!(rela::validate(RelaTable::new(&fits).iter(), window, 4, None), Ok(())); + assert_eq!(rela::validate(RelaTable::new(&fits, Machine::X86_64).iter(), window, 4, None), Ok(())); } /// A table the loader reads while it writes must not lie inside the range it @@ -187,12 +188,12 @@ fn a_relocation_crossing_a_fill_page_is_refused_only_for_a_chunked_writer() { for off in 0xFF9u64..=0xFFF { let straddles = [rela(off, 0, 8, 0)].concat(); assert_eq!( - rela::validate(RelaTable::new(&straddles).iter(), window, 4, Some(lattice)), + rela::validate(RelaTable::new(&straddles, Machine::X86_64).iter(), window, 4, Some(lattice)), Err(RelocError::StraddlesFillPage), "offset {off:#x} straddles the page but was accepted", ); assert_eq!( - rela::validate(RelaTable::new(&straddles).iter(), window, 4, None), + rela::validate(RelaTable::new(&straddles, Machine::X86_64).iter(), window, 4, None), Ok(()), "offset {off:#x} refused for a contiguous writer", ); @@ -200,17 +201,17 @@ fn a_relocation_crossing_a_fill_page_is_refused_only_for_a_chunked_writer() { // A write ending at the boundary fits; a 4-byte TPOFF32 fits in the last 4. assert_eq!( - rela::validate(RelaTable::new(&[rela(0xFF8, 0, 8, 0)].concat()).iter(), window, 4, Some(lattice)), + rela::validate(RelaTable::new(&[rela(0xFF8, 0, 8, 0)].concat(), Machine::X86_64).iter(), window, 4, Some(lattice)), Ok(()), ); assert_eq!( - rela::validate(RelaTable::new(&[rela(0xFFC, 0, 23, 0)].concat()).iter(), window, 4, Some(lattice)), + rela::validate(RelaTable::new(&[rela(0xFFC, 0, 23, 0)].concat(), Machine::X86_64).iter(), window, 4, Some(lattice)), Ok(()), ); let shifted = rela::FillLattice { base: 3, granule: 4096 }; assert_eq!( - rela::validate(RelaTable::new(&[rela(0x1000, 0, 8, 0)].concat()).iter(), window, 4, Some(shifted)), + rela::validate(RelaTable::new(&[rela(0x1000, 0, 8, 0)].concat(), Machine::X86_64).iter(), window, 4, Some(shifted)), Err(RelocError::StraddlesFillPage), ); } @@ -224,14 +225,54 @@ fn every_written_type_is_validated_and_no_other_is() { for raw in [6u32, 7, 8, 16, 17, 18, 23] { let bytes = [rela(0x1000, 0, raw, 0)].concat(); assert_eq!( - rela::validate(RelaTable::new(&bytes).iter(), window, 4, None), + rela::validate(RelaTable::new(&bytes, Machine::X86_64).iter(), window, 4, None), Err(RelocError::OutsideWindow), "type {raw} was not validated", ); } // A type nobody patches may name any offset at all. let ignored = [rela(u64::MAX, 0, 42, 0)].concat(); - assert_eq!(rela::validate(RelaTable::new(&ignored).iter(), window, 0, None), Ok(())); + assert_eq!(rela::validate(RelaTable::new(&ignored, Machine::X86_64).iter(), window, 0, None), Ok(())); +} + +/// A TLS descriptor is filled by a resolver this loader does not have, so an +/// AArch64 image that needs one is refused by name rather than left with a +/// descriptor nobody wrote. The same number on x86-64 is a type nobody patches. +#[test] +fn an_aarch64_tls_descriptor_is_refused_by_name() { + let window = (0u64, 0x100u64); + let desc = [rela(0x10, 1, 1031, 0)].concat(); + assert_eq!(RelocKind::from_raw(Machine::Aarch64, 1031), RelocKind::TlsDesc); + assert_eq!( + rela::validate(RelaTable::new(&desc, Machine::Aarch64).iter(), window, 4, None), + Err(RelocError::TlsDescriptor), + ); + assert!(RelocError::TlsDescriptor.as_str().contains("R_AARCH64_TLSDESC")); + assert_eq!(rela::validate(RelaTable::new(&desc, Machine::X86_64).iter(), window, 4, None), Ok(())); + let counts = RelaCounts::of(RelaTable::new(&desc, Machine::Aarch64).iter()); + assert_eq!(counts.count_of(RelocKind::TlsDesc), 1); +} + +/// An executable's loader keeps no group until the counts allow it: one TLS +/// descriptor refuses the image outright, before anything is reserved, and a +/// group too large for one allocation refuses it too. +#[test] +fn an_executable_s_reservation_is_had_only_through_its_refusals() { + let mixed = [rela(0, 0, 8, 0), rela(8, 1, 6, 0), rela(16, 2, 18, 0), rela(24, 3, 7, 0)].concat(); + let desc = [rela(0, 0, 1027, 0), rela(0x10, 1, 1031, 0)].concat(); + + let aarch64 = RelaCounts::of(RelaTable::new(&desc, Machine::Aarch64).iter()); + assert_eq!(aarch64.for_executable(24, usize::MAX), Err(rela::ExeRefusal::TlsDescriptor)); + assert!(rela::ExeRefusal::TlsDescriptor.as_str().contains("R_AARCH64_TLSDESC")); + + let kept = RelaCounts::of(RelaTable::new(&mixed, Machine::X86_64).iter()) + .for_executable(24, 48) + .expect("two entries of 24 bytes fit 48"); + assert_eq!((kept.relative, kept.bind, kept.tpoff64, kept.tpoff32), (1, 2, 1, 0)); + + let x86 = RelaCounts::of(RelaTable::new(&mixed, Machine::X86_64).iter()); + assert_eq!(x86.for_executable(24, 47), Err(rela::ExeRefusal::TooLarge)); + assert_eq!(x86.for_executable(usize::MAX, usize::MAX), Err(rela::ExeRefusal::TooLarge)); } #[test] @@ -240,23 +281,23 @@ fn a_symbol_index_past_the_table_is_refused_except_for_relative() { let bind = [rela(0x10, 4, 6, 0)].concat(); assert_eq!( - rela::validate(RelaTable::new(&bind).iter(), window, 4, None), + rela::validate(RelaTable::new(&bind, Machine::X86_64).iter(), window, 4, None), Err(RelocError::SymbolPastTable), ); - assert_eq!(rela::validate(RelaTable::new(&bind).iter(), window, 5, None), Ok(())); + assert_eq!(rela::validate(RelaTable::new(&bind, Machine::X86_64).iter(), window, 5, None), Ok(())); let relative = [rela(0x10, u32::MAX, 8, 0)].concat(); - assert_eq!(rela::validate(RelaTable::new(&relative).iter(), window, 0, None), Ok(())); + assert_eq!(rela::validate(RelaTable::new(&relative, Machine::X86_64).iter(), window, 0, None), Ok(())); } #[test] fn counts_are_per_kind_over_every_table() { let a = [rela(0, 0, 8, 0), rela(8, 1, 6, 0), rela(16, 2, 18, 0)].concat(); let b = [rela(24, 3, 7, 0), rela(32, 0, 23, 0), rela(40, 0, 99, 0)].concat(); - let counts = RelaCounts::of(RelaTable::new(&a).iter().chain(RelaTable::new(&b).iter())); + let counts = RelaCounts::of(RelaTable::new(&a, Machine::X86_64).iter().chain(RelaTable::new(&b, Machine::X86_64).iter())); assert_eq!( counts, - RelaCounts { relative: 1, bind: 2, tpoff64: 1, tpoff32: 1, dtpmod64: 0, dtpoff64: 0 }, + RelaCounts { relative: 1, bind: 2, tpoff64: 1, tpoff32: 1, dtpmod64: 0, dtpoff64: 0, tlsdesc: 0 }, ); // The ceiling is over the kinds a caller reserves for, never over every // kind: a bound on one nothing stores refuses a file for a collection that @@ -372,7 +413,7 @@ fn one_byte_past_a_sized_symbol_is_not_that_symbol() { /// A symbol with no size bounds nothing, so every address above it is inside /// it until a later symbol takes over. That is the assembly case — hand-written /// entry points carry `st_size` 0 — and losing it would leave every frame in -/// `arch/entry.rs` unnamed. +/// `arch/x86_64/entry.rs` unnamed. #[test] fn a_symbol_with_no_size_owns_everything_above_it() { let syms = [sym(0, 0, 0, 0), sym_sized(1, FUNC, 1, 0x1000, 0)].concat(); @@ -481,8 +522,8 @@ fn rela_dyn_is_found_by_shape_and_only_by_shape() { .concat(); let table = SectionTable::new(&bytes); let mut reader = |off: u64| match off { - 0x300 => RelaTable::new(&rela(0, 1, 6, 0)).get(0), - 0x400 => RelaTable::new(&rela(0, 0, 8, 0)).get(0), + 0x300 => RelaTable::new(&rela(0, 1, 6, 0), Machine::X86_64).get(0), + 0x400 => RelaTable::new(&rela(0, 0, 8, 0), Machine::X86_64).get(0), _ => None, }; assert_eq!(table.rela_dyn(&mut reader), Some((0x400, 72))); @@ -609,3 +650,26 @@ fn hash_table( fn bloom_word_for(h: u32, shift: u32) -> u64 { (1u64 << (h % 64)) | (1u64 << ((h as u64 >> shift) % 64)) } + +#[test] +fn each_machine_reads_its_own_relocation_numbers() { + // The AArch64 psABI's dynamic types, and one x86-64 number each machine + // must not read as the other's. + for (raw, kind) in [ + (1025u32, RelocKind::GlobDat), + (1026, RelocKind::JumpSlot), + (1027, RelocKind::Relative), + (1028, RelocKind::DtpMod64), + (1029, RelocKind::DtpOff64), + (1030, RelocKind::Tpoff64), + ] { + assert_eq!(RelocKind::from_raw(Machine::Aarch64, raw), kind, "type {raw}"); + assert_eq!(RelocKind::from_raw(Machine::X86_64, raw), RelocKind::Other(raw), "type {raw}"); + } + assert_eq!(RelocKind::from_raw(Machine::Aarch64, 8), RelocKind::Other(8)); + assert_eq!(RelocKind::from_raw(Machine::Aarch64, 23), RelocKind::Other(23)); + + let bytes = rela(0x10, 0, 1027, 4); + let entry = RelaTable::new(&bytes, Machine::Aarch64).get(0).unwrap(); + assert_eq!((entry.kind, entry.addend), (RelocKind::Relative, 4)); +} diff --git a/toyos-elf/tests/tls.rs b/toyos-elf/tests/tls.rs index b95e582a468..78979e052e3 100644 --- a/toyos-elf/tests/tls.rs +++ b/toyos-elf/tests/tls.rs @@ -5,7 +5,7 @@ //! kernel-bug assert reached from a crafted `PT_TLS`. The property is proved //! here instead, which is what lets the assert go. -use toyos_elf::tls; +use toyos_elf::tls::{self, Static, Variant}; const TCB: usize = 64; const DTV: usize = 16 + 64 * 8; @@ -30,7 +30,8 @@ fn the_dtv_is_never_overwritten_by_tls_data() { let aligns = [0usize, 1, 2, 8, 16, 64, 4096, 65536, GRANULE]; for &memsz in &sizes { for &align in &aligns { - let plan = tls::plan(memsz, align, TCB, DTV, GRANULE) + let plan = Static::new(Variant::II, memsz, align, align) + .and_then(|s| s.plan(TCB, DTV, GRANULE)) .unwrap_or_else(|| panic!("no plan for memsz {memsz} align {align}")); let effective = if align > 1 { align } else { 8 }; assert!( @@ -51,9 +52,12 @@ fn the_dtv_is_never_overwritten_by_tls_data() { #[test] fn a_size_no_allocation_can_hold_has_no_plan() { - assert_eq!(tls::plan(usize::MAX, 8, TCB, DTV, GRANULE), None); - assert_eq!(tls::plan(usize::MAX - TCB, 8, TCB, DTV, GRANULE), None); - assert_eq!(tls::plan(usize::MAX - GRANULE, GRANULE, TCB, DTV, GRANULE), None); + for variant in [Variant::I, Variant::II] { + let plan = |memsz, align| Static::new(variant, memsz, align, align).unwrap().plan(TCB, DTV, GRANULE); + assert_eq!(plan(usize::MAX, 8), None, "{variant:?}"); + assert_eq!(plan(usize::MAX - TCB, 8), None, "{variant:?}"); + assert_eq!(plan(usize::MAX - GRANULE, GRANULE), None, "{variant:?}"); + } } /// `!(align - 1)` is only a mask for a power of two, and a mask that is not a @@ -61,7 +65,10 @@ fn a_size_no_allocation_can_hold_has_no_plan() { #[test] fn an_alignment_that_is_not_a_power_of_two_has_no_plan() { for align in [3usize, 5, 6, 100, usize::MAX] { - assert_eq!(tls::plan(64, align, TCB, DTV, GRANULE), None, "align {align}"); + for variant in [Variant::I, Variant::II] { + assert_eq!(Static::new(variant, 64, align, 8), None, "align {align}"); + assert_eq!(Static::new(variant, 64, 8, align), None, "first align {align}"); + } } } @@ -95,17 +102,61 @@ fn a_module_lands_on_its_own_declared_alignment() { #[test] fn tpoff_carries_the_addend() { let total = 0x200usize; + let two = Static::new(Variant::II, total, 64, 64).unwrap(); + let one = Static::new(Variant::I, total, 64, 64).unwrap(); for &module_addr in &[0u64, 8, 0x40, 0x1F0] { - assert_eq!(tls::tpoff(module_addr, 0, total), module_addr as i64 - total as i64); + assert_eq!(two.tpoff(module_addr, 0), module_addr as i64 - total as i64); + assert_eq!(one.tpoff(module_addr, 0), module_addr as i64 + 64); for &addend in &[0i64, 8, -8, 0x100, -0x100] { - assert_eq!( - tls::tpoff(module_addr, addend, total), - module_addr as i64 + addend - total as i64, - ); - assert_eq!( - tls::tpoff(module_addr, addend, total) - tls::tpoff(module_addr, 0, total), - addend, - ); + for s in [one, two] { + assert_eq!(s.tpoff(module_addr, addend) - s.tpoff(module_addr, 0), addend, "{s:?}"); + } + } + } +} + +/// Variant I, as lld resolves an AArch64 executable's own local-exec access +/// at link time: `TPOFF = align_up(16, p_align) + offset in its PT_TLS` — +/// lld's `getTlsTpOffset`, and the AArch64 ELF ABI's 16-byte TCB. The +/// executable is the module at offset 0, so its data has to start exactly +/// that far above the thread pointer, whatever a later module asks for. +#[test] +fn variant_i_puts_the_first_module_where_its_linker_put_it() { + let sizes = [0usize, 1, 8, 0x48, 4096, DTV + 1, GRANULE - 1, GRANULE + 1]; + let aligns = [0usize, 1, 2, 8, 16, 64, 4096, GRANULE]; + for &memsz in &sizes { + for &first in &aligns { + for &max in aligns.iter().filter(|&&a| a.max(8) >= first.max(8)) { + let s = Static::new(Variant::I, memsz, max, first).unwrap(); + let plan = s.plan(TCB, DTV, GRANULE).unwrap(); + let gap = 16usize.max(first); + let at = format!("memsz {memsz} first {first} max {max}: {plan:?}"); + assert_eq!(plan.tls_start - plan.tp_offset, gap, "{at}"); + assert_eq!(s.tpoff(0, 0), gap as i64, "{at}"); + assert!(plan.tp_offset >= DTV, "{at}: the TCB overlaps the DTV"); + assert_eq!(plan.tp_offset % 16, 0, "{at}"); + assert_eq!(plan.tls_start % max.max(16), 0, "{at}"); + assert!(plan.tls_start + memsz <= plan.alloc_size, "{at}"); + assert_eq!(plan.alloc_size % GRANULE, 0, "{at}"); + } } } } + +/// rust-lld's own answer, read off an `aarch64-unknown-toyos` executable whose +/// `PT_TLS` is 64-aligned: `add x0, x8, #0x40` for the datum at offset 0 of +/// `.tdata`, `#0x80` for the one at 0x40. +#[test] +fn variant_i_agrees_with_what_lld_linked() { + let s = Static::new(Variant::I, 0xb0, 64, 64).unwrap(); + assert_eq!(s.tpoff(0, 0), 0x40); + assert_eq!(s.tpoff(0x40, 0), 0x80); +} + +/// The machine names the variant: x86-64's psABI is variant II, AArch64's +/// variant I. +#[test] +fn each_machine_has_its_own_variant() { + assert_eq!(Variant::of(toyos_elf::Machine::X86_64), Variant::II); + assert_eq!(Variant::of(toyos_elf::Machine::Aarch64), Variant::I); +} diff --git a/toyos-libc-copies/Cargo.toml b/toyos-libc-copies/Cargo.toml new file mode 100644 index 00000000000..97fe87c4b00 --- /dev/null +++ b/toyos-libc-copies/Cargo.toml @@ -0,0 +1,16 @@ +[package] +name = "toyos-libc-copies" +version = "0.1.0" +edition = "2021" +license = "MIT OR Apache-2.0" +publish = false + +# `userland/libc/src/arch/`'s copies, fills and square roots, compiled for the +# host this runs on and held against Rust's own. A package of its own because +# libc cross-compiles and exports `memcpy`: this compiles only the architecture +# module, the way `kernel-loom` compiles kernel files. `std-runtime` is libc's +# own switch, on here so the module's `_start` stays out of a host binary that +# has an entry of its own. +[features] +default = ["std-runtime"] +std-runtime = [] diff --git a/toyos-libc-copies/src/lib.rs b/toyos-libc-copies/src/lib.rs new file mode 100644 index 00000000000..371c8d880e3 --- /dev/null +++ b/toyos-libc-copies/src/lib.rs @@ -0,0 +1,102 @@ +//! libc's architecture module on the host, differentially: every copy and fill +//! it has, over every length to 300 and every source and destination offset to +//! 20, against `copy_within` and `fill`, with the two buffers overlapping both +//! ways; and its square roots against `f64::sqrt` and `f32::sqrt`. Each host +//! architecture checks its own module. + +#[cfg(test)] +#[path = "../../userland/libc/src/arch/mod.rs"] +mod arch; + +#[cfg(test)] +mod tests { + use super::arch; + + const LENGTHS: usize = 300; + const OFFSETS: usize = 20; + + fn pattern(len: usize) -> Vec { + (0..len).map(|i| (i.wrapping_mul(31) ^ (i >> 3)) as u8).collect() + } + + /// One buffer, so the copy's two ends can overlap by any amount either way. + #[test] + fn every_copy_agrees_with_copy_within() { + let mut cases = 0u32; + for n in 0..LENGTHS { + for from in 0..OFFSETS { + for to in 0..OFFSETS { + let base = pattern(n + 2 * OFFSETS); + let mut want = base.clone(); + want.copy_within(from..from + n, to); + let mut got = base.clone(); + let p = got.as_mut_ptr(); + // SAFETY: both ranges are inside `got`, and the direction is + // the one each overlap needs: forward when the destination + // is below the source, backward when above. + unsafe { + if to <= from { + arch::copy_forward(p.add(to), p.add(from), n); + } else { + arch::copy_backward(p.add(to), p.add(from), n); + } + } + assert_eq!(got, want, "n {n} from {from} to {to}"); + cases += 1; + } + } + } + assert_eq!(cases, (LENGTHS * OFFSETS * OFFSETS) as u32); + } + + /// Two buffers, so neither direction leans on the other's order. + #[test] + fn a_disjoint_copy_agrees_either_way() { + for n in 0..LENGTHS { + for at in 0..OFFSETS { + let src = pattern(n + OFFSETS); + for backward in [false, true] { + let mut got = vec![0xAAu8; n + OFFSETS]; + // SAFETY: `n` bytes at `at` in each of two buffers of `n + OFFSETS`. + unsafe { + let (d, s) = (got.as_mut_ptr().add(at), src.as_ptr().add(at)); + if backward { + arch::copy_backward(d, s, n); + } else { + arch::copy_forward(d, s, n); + } + } + let mut want = vec![0xAAu8; n + OFFSETS]; + want[at..at + n].copy_from_slice(&src[at..at + n]); + assert_eq!(got, want, "n {n} at {at} backward {backward}"); + } + } + } + } + + #[test] + fn every_fill_agrees_with_fill() { + for n in 0..LENGTHS { + for at in 0..OFFSETS { + for byte in [0u8, 0x5A, 0xFF] { + let mut got = pattern(n + 2 * OFFSETS); + let mut want = got.clone(); + want[at..at + n].fill(byte); + // SAFETY: `n` bytes at `at` inside `got`. + unsafe { arch::fill(got.as_mut_ptr().add(at), byte, n) }; + assert_eq!(got, want, "n {n} at {at} byte {byte:#x}"); + } + } + } + } + + #[test] + fn the_square_roots_are_the_correctly_rounded_ones() { + for x in [0.0f64, 1.0, 2.0, 0.5, 1e-300, 1e300, f64::MAX, f64::MIN_POSITIVE, 123456.789] { + assert_eq!(arch::sqrt_f64(x).to_bits(), x.sqrt().to_bits(), "{x}"); + let y = x as f32; + assert_eq!(arch::sqrt_f32(y).to_bits(), y.sqrt().to_bits(), "{y}"); + } + assert!(arch::sqrt_f64(-1.0).is_nan() && arch::sqrt_f32(-1.0).is_nan()); + } +} diff --git a/toyos-ps2/src/key.rs b/toyos-ps2/src/key.rs index 48160f25ab3..0930f87c6c1 100644 --- a/toyos-ps2/src/key.rs +++ b/toyos-ps2/src/key.rs @@ -180,7 +180,7 @@ impl KeyDecoder { // Shift *release*, which is accidentally the right direction // for the one state that could stick. Untested on metal. If it // does bite, the answer is a controller-side reconnect probe — - // `0xF2` identify on a timer, from `kernel/src/drivers/i8042/` + // `0xF2` identify on a timer, from `kernel/src/arch/x86_64/i8042/` // — and never a wire heuristic, because no wire heuristic // exists. 0x00 | 0xFF => KeyOutcome::Lost, diff --git a/toyos-ps2/src/lib.rs b/toyos-ps2/src/lib.rs index 916c44a057f..f95b8175088 100644 --- a/toyos-ps2/src/lib.rs +++ b/toyos-ps2/src/lib.rs @@ -8,7 +8,7 @@ //! a bug that is invisible in QEMU and un-single-steppable on the laptop //! this exists for. //! -//! The kernel side of the driver (`kernel/src/drivers/i8042/`) owns the +//! The kernel side of the driver (`kernel/src/arch/x86_64/i8042/`) owns the //! controller, the interrupt and the queues. Nothing in here touches //! hardware, allocates, or knows what a lock is. diff --git a/toyos-userbound/src/lib.rs b/toyos-userbound/src/lib.rs index de10023d8a2..c39f963899f 100644 --- a/toyos-userbound/src/lib.rs +++ b/toyos-userbound/src/lib.rs @@ -14,8 +14,8 @@ //! //! Pure. No I/O, no allocation, no `unsafe`, nothing read from a device and //! nothing named outside this crate. The kernel is the only caller — -//! `user_ptr.rs`, `mm/`, `arch/syscall/`, `loader/` and -//! `arch/idt/exceptions.rs` — and this is a crate rather than files inside it so +//! `user_ptr.rs`, `mm/`, `syscall/`, `loader/` and +//! `arch/x86_64/idt/exceptions.rs` — and this is a crate rather than files inside it so //! that the boundary table below runs on the host in milliseconds instead of in //! a boot. //! diff --git a/userland/Cargo.lock b/userland/Cargo.lock index f6a1502c3b6..7372352279d 100644 --- a/userland/Cargo.lock +++ b/userland/Cargo.lock @@ -4157,7 +4157,7 @@ version = "0.1.0" [[package]] name = "toyos-window" -version = "0.18.0" +version = "0.19.0" dependencies = [ "toyos", "toyos-abi", diff --git a/userland/libc/src/arch/aarch64.rs b/userland/libc/src/arch/aarch64.rs new file mode 100644 index 00000000000..011358f11ae --- /dev/null +++ b/userland/libc/src/arch/aarch64.rs @@ -0,0 +1,119 @@ +//! AArch64: the entry, the copies and fills as pair loads and stores with a +//! byte tail, and the FP square roots. An unaligned `ldp`/`stp` is legal on the +//! Normal memory every user buffer is. + +/// Entry point for C programs. Stack layout at entry (set up by kernel), with +/// the stack pointer 16-byte aligned: +/// [sp] = argc +/// [sp+8] = argv[0], argv[1], ..., NULL +#[cfg(not(feature = "std-runtime"))] +#[unsafe(no_mangle)] +#[unsafe(naked)] +unsafe extern "C" fn _start() -> ! { + core::arch::naked_asm!( + "ldr x0, [sp]", // argc + "add x1, sp, #8", // argv + // The outermost frame record: a backtrace ends here. + "mov x29, xzr", + "mov x30, xzr", + "bl {start_c}", + "brk #0x1", + start_c = sym crate::runtime::start_c, + ); +} + +/// Copy `n` bytes from `src` to `dest`, lowest address first. +pub(crate) unsafe fn copy_forward(dest: *mut u8, src: *const u8, n: usize) { + unsafe { + core::arch::asm!( + "2:", + "cmp {n}, #16", + "b.lo 3f", + "ldp {a}, {b}, [{src}], #16", + "stp {a}, {b}, [{dest}], #16", + "sub {n}, {n}, #16", + "b 2b", + "3:", + "cbz {n}, 4f", + "ldrb {a:w}, [{src}], #1", + "strb {a:w}, [{dest}], #1", + "sub {n}, {n}, #1", + "b 3b", + "4:", + dest = inout(reg) dest => _, + src = inout(reg) src => _, + n = inout(reg) n => _, + a = out(reg) _, + b = out(reg) _, + options(nostack), + ); + } +} + +/// Copy `n` bytes from `src` to `dest`, highest address first: the order a +/// `dest` overlapping `src` from above needs. +pub(crate) unsafe fn copy_backward(dest: *mut u8, src: *const u8, n: usize) { + unsafe { + core::arch::asm!( + "add {src}, {src}, {n}", + "add {dest}, {dest}, {n}", + "2:", + "cmp {n}, #16", + "b.lo 3f", + "ldp {a}, {b}, [{src}, #-16]!", + "stp {a}, {b}, [{dest}, #-16]!", + "sub {n}, {n}, #16", + "b 2b", + "3:", + "cbz {n}, 4f", + "ldrb {a:w}, [{src}, #-1]!", + "strb {a:w}, [{dest}, #-1]!", + "sub {n}, {n}, #1", + "b 3b", + "4:", + dest = inout(reg) dest => _, + src = inout(reg) src => _, + n = inout(reg) n => _, + a = out(reg) _, + b = out(reg) _, + options(nostack), + ); + } +} + +/// Set `n` bytes at `dest` to `byte`. +pub(crate) unsafe fn fill(dest: *mut u8, byte: u8, n: usize) { + let word = u64::from(byte) * 0x0101_0101_0101_0101; + unsafe { + core::arch::asm!( + "2:", + "cmp {n}, #16", + "b.lo 3f", + "stp {w}, {w}, [{dest}], #16", + "sub {n}, {n}, #16", + "b 2b", + "3:", + "cbz {n}, 4f", + "strb {w:w}, [{dest}], #1", + "sub {n}, {n}, #1", + "b 3b", + "4:", + dest = inout(reg) dest => _, + n = inout(reg) n => _, + w = in(reg) word, + options(nostack), + ); + } +} + +pub(crate) fn sqrt_f64(x: f64) -> f64 { + let result: f64; + unsafe { core::arch::asm!("fsqrt {0:d}, {0:d}", inout(vreg) x => result, options(pure, nomem, nostack)) }; + result +} + +pub(crate) fn sqrt_f32(x: f32) -> f32 { + let result: f32; + unsafe { core::arch::asm!("fsqrt {0:s}, {0:s}", inout(vreg) x => result, options(pure, nomem, nostack)) }; + result +} diff --git a/userland/libc/src/arch/mod.rs b/userland/libc/src/arch/mod.rs new file mode 100644 index 00000000000..03e946d2a59 --- /dev/null +++ b/userland/libc/src/arch/mod.rs @@ -0,0 +1,14 @@ +//! The C runtime's architecture-specific pieces, one module per architecture, +//! each answering the same names: the process entry, the three copies and +//! fills that must not be written in Rust (the compiler lowers a Rust loop that +//! copies back into a call to `memcpy`), and the square roots. + +#[cfg(target_arch = "aarch64")] +mod aarch64; +#[cfg(target_arch = "aarch64")] +pub(crate) use aarch64::*; + +#[cfg(target_arch = "x86_64")] +mod x86_64; +#[cfg(target_arch = "x86_64")] +pub(crate) use x86_64::*; diff --git a/userland/libc/src/arch/x86_64.rs b/userland/libc/src/arch/x86_64.rs new file mode 100644 index 00000000000..f76430eca0e --- /dev/null +++ b/userland/libc/src/arch/x86_64.rs @@ -0,0 +1,76 @@ +//! x86-64: the entry, the string instructions, and SSE2's square roots. + +/// Entry point for C programs. Stack layout at entry (set up by kernel): +/// [RSP] = argc +/// [RSP+8] = argv[0], argv[1], ..., NULL +#[cfg(not(feature = "std-runtime"))] +#[unsafe(no_mangle)] +#[unsafe(naked)] +unsafe extern "C" fn _start() -> ! { + core::arch::naked_asm!( + "mov rdi, [rsp]", // argc + "lea rsi, [rsp + 8]", // argv + "call {start_c}", + "ud2", + start_c = sym crate::runtime::start_c, + ); +} + +/// Copy `n` bytes from `src` to `dest`, lowest address first. +pub(crate) unsafe fn copy_forward(dest: *mut u8, src: *const u8, n: usize) { + unsafe { + core::arch::asm!( + "rep movsb", + inout("rdi") dest => _, + inout("rsi") src => _, + inout("rcx") n => _, + options(nostack), + ); + } +} + +/// Copy `n` bytes from `src` to `dest`, highest address first: the order a +/// `dest` overlapping `src` from above needs. +pub(crate) unsafe fn copy_backward(dest: *mut u8, src: *const u8, n: usize) { + if n == 0 { + return; + } + // The direction flag is set only across this one `rep`; every Ring 0 entry + // clears it for the kernel's own sake. + unsafe { + core::arch::asm!( + "std", + "rep movsb", + "cld", + inout("rdi") dest.add(n - 1) => _, + inout("rsi") src.add(n - 1) => _, + inout("rcx") n => _, + options(nostack), + ); + } +} + +/// Set `n` bytes at `dest` to `byte`. +pub(crate) unsafe fn fill(dest: *mut u8, byte: u8, n: usize) { + unsafe { + core::arch::asm!( + "rep stosb", + inout("rdi") dest => _, + in("al") byte, + inout("rcx") n => _, + options(nostack), + ); + } +} + +pub(crate) fn sqrt_f64(x: f64) -> f64 { + let result: f64; + unsafe { core::arch::asm!("sqrtsd {0}, {0}", inout(xmm_reg) x => result, options(pure, nomem, nostack)) }; + result +} + +pub(crate) fn sqrt_f32(x: f32) -> f32 { + let result: f32; + unsafe { core::arch::asm!("sqrtss {0}, {0}", inout(xmm_reg) x => result, options(pure, nomem, nostack)) }; + result +} diff --git a/userland/libc/src/lib.rs b/userland/libc/src/lib.rs index 862821d165f..219d9bb256d 100644 --- a/userland/libc/src/lib.rs +++ b/userland/libc/src/lib.rs @@ -2,6 +2,7 @@ extern crate alloc; +mod arch; mod ctype; mod math; mod memory; @@ -14,34 +15,18 @@ mod stdio; mod string; mod time; -// C runtime: _start entry point, panic handler, and global allocator. +// C runtime: the entry `arch::_start` calls, panic handler, and global allocator. // Only for pure C programs (no Rust std). When linked into a Rust program // with std, std provides these. #[cfg(not(feature = "std-runtime"))] mod runtime { use core::panic::PanicInfo; - // Entry point for C programs. The kernel pushes argc and argv onto the stack. - #[unsafe(no_mangle)] - #[unsafe(naked)] - unsafe extern "C" fn _start() -> ! { - // Stack layout at entry (set up by kernel): - // [RSP] = argc - // [RSP+8] = argv[0], argv[1], ..., NULL - core::arch::naked_asm!( - "mov rdi, [rsp]", // argc - "lea rsi, [rsp + 8]", // argv - "call {start_c}", - "ud2", - start_c = sym start_c, - ); - } - // Returning from `main` is defined as calling `exit` with its value, so this // goes through libc's `exit` rather than the syscall: the atexit table and // `fflush(NULL)` are what stand between a program's last unterminated line // and the fd. - extern "C" fn start_c(argc: i32, argv: *const *const u8) -> ! { + pub(crate) extern "C" fn start_c(argc: i32, argv: *const *const u8) -> ! { unsafe extern "C" { fn main(argc: i32, argv: *const *const u8) -> i32; } diff --git a/userland/libc/src/math.rs b/userland/libc/src/math.rs index 7abb5c0f25a..2d97c989566 100644 --- a/userland/libc/src/math.rs +++ b/userland/libc/src/math.rs @@ -230,36 +230,10 @@ pub extern "C" fn floorf(x: f32) -> f32 { pub extern "C" fn ceilf(x: f32) -> f32 { -floorf(-x) } #[no_mangle] -pub extern "C" fn sqrt(x: f64) -> f64 { - #[cfg(target_arch = "x86_64")] - { - let result: f64; - unsafe { core::arch::asm!("sqrtsd {0}, {0}", inout(xmm_reg) x => result); } - return result; - } - #[cfg(target_arch = "aarch64")] - { - let result: f64; - unsafe { core::arch::asm!("fsqrt {0:d}, {0:d}", inout(vreg) x => result); } - return result; - } -} +pub extern "C" fn sqrt(x: f64) -> f64 { crate::arch::sqrt_f64(x) } #[no_mangle] -pub extern "C" fn sqrtf(x: f32) -> f32 { - #[cfg(target_arch = "x86_64")] - { - let result: f32; - unsafe { core::arch::asm!("sqrtss {0}, {0}", inout(xmm_reg) x => result); } - return result; - } - #[cfg(target_arch = "aarch64")] - { - let result: f32; - unsafe { core::arch::asm!("fsqrt {0:s}, {0:s}", inout(vreg) x => result); } - return result; - } -} +pub extern "C" fn sqrtf(x: f32) -> f32 { crate::arch::sqrt_f32(x) } #[no_mangle] pub extern "C" fn fabs(x: f64) -> f64 { f64::from_bits(x.to_bits() & !(1u64 << 63)) } diff --git a/userland/libc/src/memory.rs b/userland/libc/src/memory.rs index 2555b2be805..3432aaf6a35 100644 --- a/userland/libc/src/memory.rs +++ b/userland/libc/src/memory.rs @@ -86,21 +86,15 @@ pub unsafe extern "C" fn realloc(p: *mut u8, new_size: usize) -> *mut u8 { unsafe { backend::realloc(p, new_size) } } -// memcpy, memmove, memset, memcmp — implemented in inline asm to avoid -// infinite recursion (Rust's ptr::copy_nonoverlapping emits calls to memcpy). +// memcpy, memmove and memset are the architecture's (`arch`): Rust's +// ptr::copy_nonoverlapping, and a copying loop, are lowered to calls to memcpy. // This libc spells C strings and buffers as `*const u8`, not `c_char`/`c_void`: // same ABI, different element type from the declaration std links against. #[allow(suspicious_runtime_symbol_definitions)] #[no_mangle] pub unsafe extern "C" fn memcpy(dest: *mut u8, src: *const u8, n: usize) -> *mut u8 { - core::arch::asm!( - "rep movsb", - inout("rdi") dest => _, - inout("rsi") src => _, - inout("rcx") n => _, - options(nostack), - ); + crate::arch::copy_forward(dest, src, n); dest } @@ -108,18 +102,10 @@ pub unsafe extern "C" fn memcpy(dest: *mut u8, src: *const u8, n: usize) -> *mut #[no_mangle] pub unsafe extern "C" fn memmove(dest: *mut u8, src: *const u8, n: usize) -> *mut u8 { if (dest as usize) <= (src as usize) || (dest as usize) >= (src as usize) + n { - memcpy(dest, src, n); + crate::arch::copy_forward(dest, src, n); } else { - // Overlap with dest after src — copy backwards - core::arch::asm!( - "std", - "rep movsb", - "cld", - inout("rdi") dest.add(n - 1) => _, - inout("rsi") src.add(n - 1) => _, - inout("rcx") n => _, - options(nostack), - ); + // Overlap with dest after src: copy backwards. + crate::arch::copy_backward(dest, src, n); } dest } @@ -127,13 +113,7 @@ pub unsafe extern "C" fn memmove(dest: *mut u8, src: *const u8, n: usize) -> *mu #[allow(suspicious_runtime_symbol_definitions)] #[no_mangle] pub unsafe extern "C" fn memset(dest: *mut u8, c: i32, n: usize) -> *mut u8 { - core::arch::asm!( - "rep stosb", - inout("rdi") dest => _, - in("al") c as u8, - inout("rcx") n => _, - options(nostack), - ); + crate::arch::fill(dest, c as u8, n); dest } diff --git a/userland/metalprobe/src/arch/aarch64.rs b/userland/metalprobe/src/arch/aarch64.rs new file mode 100644 index 00000000000..b8b12383d60 --- /dev/null +++ b/userland/metalprobe/src/arch/aarch64.rs @@ -0,0 +1,8 @@ +//! AArch64. + +/// Complete this CPU's stores: DSB ST waits until every store before it has +/// completed. +pub fn drain_stores() { + // SAFETY: a barrier with no operands and no memory it can misuse. + unsafe { std::arch::asm!("dsb st", options(nostack, preserves_flags)) }; +} diff --git a/userland/metalprobe/src/arch/mod.rs b/userland/metalprobe/src/arch/mod.rs new file mode 100644 index 00000000000..099b435e4b5 --- /dev/null +++ b/userland/metalprobe/src/arch/mod.rs @@ -0,0 +1,11 @@ +//! What the probes ask of the CPU, one module per architecture. + +#[cfg(target_arch = "aarch64")] +mod aarch64; +#[cfg(target_arch = "aarch64")] +pub use aarch64::*; + +#[cfg(target_arch = "x86_64")] +mod x86_64; +#[cfg(target_arch = "x86_64")] +pub use x86_64::*; diff --git a/userland/metalprobe/src/arch/x86_64.rs b/userland/metalprobe/src/arch/x86_64.rs new file mode 100644 index 00000000000..888d2affe57 --- /dev/null +++ b/userland/metalprobe/src/arch/x86_64.rs @@ -0,0 +1,8 @@ +//! x86-64. + +/// Drain this CPU's stores out of the write-combining fill buffers. +pub fn drain_stores() { + // SAFETY: `sfence` has no operands and no memory it can misuse; the + // target is x86-64, where the instruction always exists. + unsafe { std::arch::x86_64::_mm_sfence() }; +} diff --git a/userland/metalprobe/src/fb.rs b/userland/metalprobe/src/fb.rs index fd1cb476571..e70632a72e1 100644 --- a/userland/metalprobe/src/fb.rs +++ b/userland/metalprobe/src/fb.rs @@ -103,9 +103,7 @@ impl Scanout { /// Every write is in the fill buffers until this runs: a weakly-ordered /// mapping owes a fence before anything reads what was put there. fn fence(&self) { - // SAFETY: `sfence` has no operands and no memory it can misuse; the - // target is x86-64, where the instruction always exists. - unsafe { std::arch::x86_64::_mm_sfence() }; + crate::arch::drain_stores(); } /// Paint the whole scanout black, so the kernel's panel is legible again diff --git a/userland/metalprobe/src/main.rs b/userland/metalprobe/src/main.rs index 617df2b4110..fd6dbf5553e 100644 --- a/userland/metalprobe/src/main.rs +++ b/userland/metalprobe/src/main.rs @@ -21,6 +21,7 @@ //! A command that could not measure exits with one of [`Refusal`]'s negative //! codes instead, so a missing number is never a small one. +mod arch; mod fb; mod usb; diff --git a/userland/toybox/src/mv.rs b/userland/toybox/src/mv.rs index cabf753f291..70ce7c9b3bd 100644 --- a/userland/toybox/src/mv.rs +++ b/userland/toybox/src/mv.rs @@ -6,7 +6,7 @@ use std::process; /// /// There is no copy-and-delete fallback for a move between mounts, and the /// reason is not that the pieces are missing — `cp` is right there. It is that -/// `sys_rename` in `kernel/src/arch/syscall.rs` collapses all five of +/// `sys_rename` in `kernel/src/syscall/fs.rs` collapses all five of /// `Vfs::rename`'s errors into `NotFound`, and `Stat` carries no mount /// identity, so this process cannot tell "different mounts" from "the rename /// is broken". A fallback keyed on *any* rename failure would quietly copy its diff --git a/userland/toyos-window/Cargo.toml b/userland/toyos-window/Cargo.toml index fe216549bfe..2992ce9bbb0 100644 --- a/userland/toyos-window/Cargo.toml +++ b/userland/toyos-window/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "toyos-window" -version = "0.18.0" +version = "0.19.0" edition = "2021" license = "MIT OR Apache-2.0" description = "Client library for ToyOS windows: create one, draw into it, read its events." diff --git a/userland/toyos-window/src/arch/aarch64.rs b/userland/toyos-window/src/arch/aarch64.rs new file mode 100644 index 00000000000..e22f2010985 --- /dev/null +++ b/userland/toyos-window/src/arch/aarch64.rs @@ -0,0 +1,7 @@ +//! AArch64. + +/// Complete this CPU's stores to the scanout: DSB ST waits until every store +/// before it has completed. +pub(crate) fn drain_stores() { + unsafe { core::arch::asm!("dsb st", options(nostack, preserves_flags)) }; +} diff --git a/userland/toyos-window/src/arch/mod.rs b/userland/toyos-window/src/arch/mod.rs new file mode 100644 index 00000000000..8e7aaf91f6d --- /dev/null +++ b/userland/toyos-window/src/arch/mod.rs @@ -0,0 +1,11 @@ +//! What a surface's writer asks of the CPU, one module per architecture. + +#[cfg(target_arch = "aarch64")] +mod aarch64; +#[cfg(target_arch = "aarch64")] +pub(crate) use aarch64::*; + +#[cfg(target_arch = "x86_64")] +mod x86_64; +#[cfg(target_arch = "x86_64")] +pub(crate) use x86_64::*; diff --git a/userland/toyos-window/src/arch/x86_64.rs b/userland/toyos-window/src/arch/x86_64.rs new file mode 100644 index 00000000000..cb42608a074 --- /dev/null +++ b/userland/toyos-window/src/arch/x86_64.rs @@ -0,0 +1,8 @@ +//! x86-64. + +/// Drain this CPU's stores to a write-combining scanout: SFENCE empties the +/// WC buffers (SDM Vol. 3A §11.3.1), where the tail of a blit otherwise waits +/// for something unrelated to evict it. +pub(crate) fn drain_stores() { + unsafe { core::arch::x86_64::_mm_sfence() }; +} diff --git a/userland/toyos-window/src/framebuffer.rs b/userland/toyos-window/src/framebuffer.rs index 95d4ca7e1f6..9b20bba153f 100644 --- a/userland/toyos-window/src/framebuffer.rs +++ b/userland/toyos-window/src/framebuffer.rs @@ -134,12 +134,11 @@ impl Screen { } } // A scanout is write-combining, so the tail of the last row can sit in - // a partly filled WC buffer until something unrelated evicts it — a - // sliver of the previous frame left on the panel for as long as its - // owner has nothing else to draw. SFENCE is what drains one (SDM - // Vol. 3A §11.3.1), and this is the call that says the pixels are the + // a store buffer until something unrelated evicts it — a sliver of the + // previous frame left on the panel for as long as its owner has nothing + // else to draw. This is the call that says the pixels are the // surface's now. - unsafe { core::arch::x86_64::_mm_sfence() }; + crate::arch::drain_stores(); self.written.set(self.written.get() + (row_bytes * h) as u64); self.blits.set(self.blits.get() + 1); let painted = Rect { x: x as u32, y: y as u32, w: w as u32, h: h as u32 }; diff --git a/userland/toyos-window/src/lib.rs b/userland/toyos-window/src/lib.rs index 9fbb1439b7d..6350e5853c6 100644 --- a/userland/toyos-window/src/lib.rs +++ b/userland/toyos-window/src/lib.rs @@ -1,3 +1,4 @@ +mod arch; pub mod framebuffer; pub mod wait;