diff --git a/kernel/src/loader/mod.rs b/kernel/src/loader/mod.rs index deb308799aa..cfa90ffcbd2 100644 --- a/kernel/src/loader/mod.rs +++ b/kernel/src/loader/mod.rs @@ -15,8 +15,7 @@ mod symbols; mod tls; pub use start::{build_child_handles, PendingHandles, SLOT_PAIR_LEN}; -pub(crate) use start::alloc_kernel_stack; -pub(crate) use crate::arch::entry::{kernel_start, process_start, thread_start}; +pub(crate) use start::{alloc_kernel_stack, Start}; pub use tls::{TlsBlock, DTV_INITIAL_CAPACITY, VARIANT as TLS_VARIANT}; use alloc::string::String; @@ -568,7 +567,12 @@ pub fn spawn( return Err(SyscallError::ResourceExhausted.into()); }; - let entry = (image_start + layout.entry().get()).raw(); + // Inside the image, which `rebase_base` placed inside the user half. + let Some(entry) = toyos_userbound::Entry::new((image_start + layout.entry().get()).raw()) + else { + log!("spawn: {}: the entry is outside the user half", path); + return Err(SyscallError::InvalidArgument.into()); + }; let image_end = (image_start + layout.span()).raw(); let sp = user_stack.write_argv(argv); let t_tls = crate::clock::nanos_since_boot(); @@ -585,7 +589,7 @@ pub fn spawn( bias: base, }); - let (ks_alloc, ks_sp) = match alloc_kernel_stack(process_start, entry, sp, 0) { + let (ks_alloc, ks_sp) = match alloc_kernel_stack(Start::Process { entry, sp }) { Some(ks) => ks, None => { log!("spawn: {}: failed to allocate kernel stack", path); @@ -683,7 +687,7 @@ pub fn spawn( let t3 = crate::clock::nanos_since_boot(); log!("spawn: {} pid={} tid={} dst={} base={:#x} entry={:#x} root={:#x} (layout={}ms relocs={}ms deps={}ms tls={}ms total={}ms)", - path, pid, tid, dst.0, base, entry, child_pt.lock().root().phys(), + path, pid, tid, dst.0, base, entry.addr(), child_pt.lock().root().phys(), (t1 - t0) / 1_000_000, (t2 - t1) / 1_000_000, (t_deps - t2) / 1_000_000, (t_tls - t_deps) / 1_000_000, (t3 - t0) / 1_000_000); diff --git a/kernel/src/loader/start.rs b/kernel/src/loader/start.rs index 5d42b4c7ae8..4d67e159b35 100644 --- a/kernel/src/loader/start.rs +++ b/kernel/src/loader/start.rs @@ -1,7 +1,9 @@ //! Loads a built process onto a CPU, builds the handle table it starts with //! and puts its spawner's handle to it in the spawner's table. The frame a new //! stack starts from and the trampolines it returns into are the -//! architecture's (`arch::entry`). +//! architecture's (`arch::entry`), and [`Start`] is the only way to one: a +//! trampoline that returns to userland is reached with an +//! [`Entry`](toyos_userbound::Entry) and with nothing else. use alloc::sync::Arc; use alloc::vec::Vec; @@ -21,13 +23,24 @@ use toyos_abi::syscall::{ /// One `[child_slot, parent_handle]` pair of `SpawnArgs::slot_map_ptr`, in bytes. pub const SLOT_PAIR_LEN: usize = 8; +/// Where a new context first runs. +pub(crate) enum Start { + /// A program's first instruction, on the stack its loader wrote. + Process { entry: toyos_userbound::Entry, sp: u64 }, + /// A thread's, on a stack its process chose, with its argument. + Thread { entry: toyos_userbound::Entry, sp: u64, arg: u64 }, + /// A kernel thread's body, which never returns to userland. + Kernel { body: extern "C" fn(u64) -> !, arg: u64 }, +} + /// Allocate a kernel stack and lay out the frame `context_switch` will restore. -pub(crate) fn alloc_kernel_stack( - trampoline: unsafe extern "C" fn(), - user_entry: u64, - user_sp: u64, - arg: u64, -) -> Option<(OwnedAlloc, u64)> { +pub(crate) fn alloc_kernel_stack(start: Start) -> Option<(OwnedAlloc, u64)> { + use crate::arch::entry::{kernel_start, process_start, thread_start}; + let (trampoline, user_entry, user_sp, arg): (unsafe extern "C" fn(), _, _, _) = match start { + Start::Process { entry, sp } => (process_start, entry.addr(), sp, 0), + Start::Thread { entry, sp, arg } => (thread_start, entry.addr(), sp, arg), + Start::Kernel { body, arg } => (kernel_start, body as usize as u64, 0, arg), + }; let alloc = OwnedAlloc::new(KERNEL_STACK_SIZE, 4096)?; scheduler::write_stack_canary(&alloc); let top = alloc.ptr() as u64 + KERNEL_STACK_SIZE as u64; diff --git a/kernel/src/process.rs b/kernel/src/process.rs index 78cfa1ded34..79b7a85a7a4 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -21,7 +21,7 @@ use crate::sched::payload::ThreadSched; use crate::time::{Deadline, Duration}; use crate::{elf, pipe, scheduler}; use crate::UserAddr; -use crate::loader::{alloc_kernel_stack, thread_start, TlsBlock}; +use crate::loader::{alloc_kernel_stack, Start, TlsBlock}; pub use toyos_abi::{Pid, Tid}; pub use crate::scheduler::TaskId; @@ -923,7 +923,12 @@ impl Drop for Admission { } /// Spawn a thread within the current process. -pub fn spawn_thread(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> Option { +pub fn spawn_thread( + entry: toyos_userbound::Entry, + stack_ptr: u64, + arg: u64, + stack_base: u64, +) -> Option { // Phase 1: parent's data + address space (table lock dropped after). let parent_process = current_process(); let (parent_addr_space, process_data_arc) = { @@ -962,7 +967,7 @@ pub fn spawn_thread(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> Op }; let tls_alloc_tcb = tls_alloc.ptr().wrapping_add(tp_offset); - let (ks_alloc, ks_sp) = match alloc_kernel_stack(thread_start, entry, stack_ptr, arg) { + let (ks_alloc, ks_sp) = match alloc_kernel_stack(Start::Thread { entry, sp: stack_ptr, arg }) { Some(ks) => ks, None => { tls_alloc.release(&parent_addr_space); diff --git a/kernel/src/sched/kthread.rs b/kernel/src/sched/kthread.rs index e63843293c5..55a7119020a 100644 --- a/kernel/src/sched/kthread.rs +++ b/kernel/src/sched/kthread.rs @@ -1,5 +1,5 @@ //! Kernel threads: ordinary tasks that name `mm::paging::kernel` as their -//! address space, enter through `loader::kernel_start`, and hold a process-table +//! address space, enter through `arch::entry::kernel_start`, and hold a process-table //! entry. One is preempted or stolen only at a preemption point its body reaches, //! and a Ring 0 loop reaches none. [`ROWS`] holds every one. @@ -74,13 +74,8 @@ pub fn is_kernel_task(id: TaskId) -> bool { /// Start a kernel thread running `body(arg)` on its own kernel stack and return its scheduler faces. pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64) -> ThreadSched { - let (stack, entry_sp) = crate::loader::alloc_kernel_stack( - crate::loader::kernel_start, - body as usize as u64, - 0, - arg, - ) - .unwrap_or_else(|| panic!("kthread: no kernel stack for {name}")); + let (stack, entry_sp) = crate::loader::alloc_kernel_stack(crate::loader::Start::Kernel { body, arg }) + .unwrap_or_else(|| panic!("kthread: no kernel stack for {name}")); // Before the table lock: a panic holding the process table hangs the machine. let claim = Claim::take(name); diff --git a/kernel/src/syscall/proc.rs b/kernel/src/syscall/proc.rs index e856d42a1c1..3251510eac1 100644 --- a/kernel/src/syscall/proc.rs +++ b/kernel/src/syscall/proc.rs @@ -128,11 +128,16 @@ pub(super) fn sys_endowments(out: &mut crate::user_ptr::UserBytesMut) -> u64 { needed as u64 } -/// Spawn a thread; refuses a `stack_base` above `stack_ptr` (no stack to clamp to). +/// Spawn a thread; refuses a `stack_base` above `stack_ptr` (no stack to clamp to) +/// and an `entry` outside the user half, which the return to it would fault on +/// in the kernel's ring. The stack pointer is the thread's own to fault on. pub(super) fn sys_thread_spawn(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> u64 { if stack_base > stack_ptr { return SyscallError::InvalidArgument.to_u64(); } + let Some(entry) = toyos_userbound::Entry::new(entry) else { + return SyscallError::InvalidArgument.to_u64(); + }; // A None here is a resource failure or teardown race, never a bad argument. process::spawn_thread(entry, stack_ptr, arg, stack_base) .map_or(SyscallError::ResourceExhausted.to_u64(), |t| t.raw() as u64) diff --git a/tests/toyos-rust-tests/src/bin/abuse_thread_table.rs b/tests/toyos-rust-tests/src/bin/abuse_thread_table.rs index 679420e8313..912ce62d2c1 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_thread_table.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_thread_table.rs @@ -1,6 +1,7 @@ //! A process's threads are bounded at `MAX_THREADS`, exited ones not yet //! joined among them: past it a spawn is refused by name, and a join is room -//! for one. +//! for one. An entry outside the user half is refused by name whatever room +//! there is. use std::sync::atomic::{AtomicU32, Ordering}; use std::time::{Duration, Instant}; @@ -81,6 +82,17 @@ fn main() { let ran = AtomicU32::new(0); let mut stacks = Stacks { chunk: 0, used: 0 }; + // Canonical under neither 48-bit nor 57-bit addressing; the first address + // past the user half; the first of the kernel half. + let (base, top) = stacks.next(); + for entry in [0x0100_0000_0000_0000, 0x0000_8000_0000_0000, 0xFFFF_8000_0000_0000] { + assert_eq!( + SyscallError::from_u64(unsafe { syscall::thread_spawn(entry, top, 0, base) }), + Some(SyscallError::InvalidArgument), + "a thread entry of {entry:#x} was not refused", + ); + } + let mut first = None; let mut spawned = 0usize; let mut refusal = None; diff --git a/toyos-abi/src/syscall.rs b/toyos-abi/src/syscall.rs index 3e3260d780e..259fe56cfce 100644 --- a/toyos-abi/src/syscall.rs +++ b/toyos-abi/src/syscall.rs @@ -990,7 +990,8 @@ pub fn mark_tty(handle: RawHandle) { pub const MAX_THREADS: usize = 4096; /// Spawn a new thread with the given entry point, stack pointer, argument, and stack base. -/// `stack_base` is the bottom of the user stack (for stack info queries). +/// `stack_base` is the bottom of the user stack (for stack info queries). An +/// `entry` outside the user half is `InvalidArgument`. /// /// # Safety /// `entry` must be a valid function pointer and `stack`/`stack_base` must diff --git a/toyos-userbound/src/lib.rs b/toyos-userbound/src/lib.rs index bf63df0e219..fe80d070c22 100644 --- a/toyos-userbound/src/lib.rs +++ b/toyos-userbound/src/lib.rs @@ -38,6 +38,6 @@ pub use place::{PageSpan, Window}; pub use port::{port_access, IoBitmap, PortAccess, Ports, Reserved, Undeclared, IO_PORTS}; pub use segment::{pieces, segments, Pinned, Pins, Segment}; pub use span::{ - align_2m_checked, in_user_half, is_user_addr, is_user_object, rebase_base, Access, PAGE_2M, - PAGE_4K, USER_TOP, + align_2m_checked, in_user_half, is_user_addr, is_user_object, rebase_base, Access, Entry, + PAGE_2M, PAGE_4K, USER_TOP, }; diff --git a/toyos-userbound/src/place.rs b/toyos-userbound/src/place.rs index 992d7c7fd72..274ddb78939 100644 --- a/toyos-userbound/src/place.rs +++ b/toyos-userbound/src/place.rs @@ -17,7 +17,7 @@ //! What a region already registered says about itself is the kernel's own //! ledger and is not re-checked: a wrap there is a kernel bug and traps. -use crate::span::{align_2m_checked, PAGE_2M, USER_TOP}; +use crate::span::{align_2m_checked, PAGE_2M, PLACED_TOP}; /// The part of the user half anonymous ranges are placed in, and the guard /// left above each. Built in a `const`, so a window that breaks @@ -44,15 +44,15 @@ impl PageSpan { } impl Window { - /// Every bound 2 MiB-aligned, inside the user half, with room for at least - /// the guard: the kernel's own layout, so a window that breaks any of them + /// Every bound 2 MiB-aligned, at or below [`PLACED_TOP`], with room for at + /// least the guard: the kernel's own layout, so a window that breaks any of them /// is a kernel bug. pub const fn new(floor: u64, ceiling: u64, guard: u64) -> Self { assert!( floor.is_multiple_of(PAGE_2M) && ceiling.is_multiple_of(PAGE_2M) && guard.is_multiple_of(PAGE_2M), "a placement window's bounds and guard are whole 2 MiB pages" ); - assert!(ceiling <= USER_TOP, "a placement window is inside the user half"); + assert!(ceiling <= PLACED_TOP, "a placement window stops below the last page of the user half"); assert!(floor < ceiling && ceiling - floor >= guard, "a placement window holds its guard"); Self { floor, ceiling, guard } } @@ -108,6 +108,7 @@ impl Window { #[cfg(test)] mod tests { use super::*; + use crate::span::USER_TOP; /// The kernel's own window, by value: `vma::WINDOW`. const FLOOR: u64 = 0x0002_0000_0000; @@ -132,6 +133,9 @@ mod tests { } } + /// The highest window there is compiles; the test below is one page past it. + const _: Window = Window::new(FLOOR, PLACED_TOP, PAGE_2M); + #[test] fn a_zero_length_is_refused() { assert_eq!(WINDOW.span(0), None); @@ -210,9 +214,9 @@ mod tests { } #[test] - #[should_panic(expected = "inside the user half")] - fn a_window_past_the_user_half_is_a_kernel_bug() { - let _ = Window::new(FLOOR, USER_TOP + PAGE_2M, PAGE_2M); + #[should_panic(expected = "below the last page of the user half")] + fn a_window_reaching_the_last_page_of_the_user_half_is_a_kernel_bug() { + let _ = Window::new(FLOOR, USER_TOP, PAGE_2M); } /// `find_gap` no longer pre-filters by the ceiling; `gap` is the one diff --git a/toyos-userbound/src/span.rs b/toyos-userbound/src/span.rs index e7ef60af548..e4a458950a3 100644 --- a/toyos-userbound/src/span.rs +++ b/toyos-userbound/src/span.rs @@ -40,6 +40,34 @@ pub fn is_user_addr(addr: u64) -> bool { addr < USER_TOP } +/// An instruction pointer the kernel may return to userland at. +/// +/// A return to an address that is not canonical faults in the returning +/// instruction itself, in the kernel's ring and not the thread's, so one that +/// crossed the trust boundary is refused before any frame carries it. No +/// other constructor: a first return to userland takes this and nothing else. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub struct Entry(u64); + +impl Entry { + /// `None` for an address outside the user half; what is mapped at one + /// inside it is the thread's own fault to take. + pub fn new(addr: u64) -> Option { + is_user_addr(addr).then_some(Self(addr)) + } + + pub const fn addr(self) -> u64 { + self.0 + } +} + +/// One past the highest address the kernel places anything at, an image or a +/// [`Window`](crate::Window)'s range: the user half less its last page. An +/// instruction that ends at [`USER_TOP`] leaves that address, no canonical +/// one, as where a syscall or an interrupt taken after it returns to, so +/// nothing executable is placed where one could end there. +pub(crate) const PLACED_TOP: u64 = USER_TOP - PAGE_2M; + /// Whether `[ptr, ptr + len)` is entirely in the user half. /// /// Also the bound `sys_mmap` applies to a range it will install rather than @@ -54,12 +82,13 @@ pub fn in_user_half(ptr: u64, len: u64) -> bool { /// The load base an `ET_DYN` image rebases to `vm_base` (`vm_base - vaddr_min`), /// or `None` when it cannot: a `vaddr_min` above `vm_base` underflows the -/// subtraction, and a `span` reaching from `vm_base` past the user half does not +/// subtraction, and a `span` reaching from `vm_base` past [`PLACED_TOP`] does not /// fit. The ELF spec leaves an `ET_DYN` `p_vaddr` unconstrained, so the kernel /// that picks `vm_base` is the only place that can refuse one. pub fn rebase_base(vm_base: u64, vaddr_min: u64, span: u64) -> Option { let base = vm_base.checked_sub(vaddr_min)?; - in_user_half(vm_base, span).then_some(base) + let end = vm_base.checked_add(span)?; + (end <= PLACED_TOP).then_some(base) } /// Whether the kernel may read or write a `size`-byte value of alignment @@ -111,6 +140,19 @@ mod tests { assert!(is_user_addr(0)); } + /// The last address of the user half is one, and the first that is not + /// canonical, the first of the kernel half and the last of all are not. + #[test] + fn an_entry_is_an_address_of_the_user_half_and_nothing_else() { + assert_eq!(Entry::new(USER_TOP - 1).map(Entry::addr), Some(USER_TOP - 1)); + assert_eq!(Entry::new(0).map(Entry::addr), Some(0)); + assert_eq!(Entry::new(USER_TOP), None); + assert_eq!(Entry::new(0x0100_0000_0000_0000), None); + assert_eq!(Entry::new(0xFFFF_7FFF_FFFF_FFFF), None); + assert_eq!(Entry::new(0xFFFF_8000_0000_0000), None); + assert_eq!(Entry::new(u64::MAX), None); + } + #[test] fn a_range_ending_past_the_bound_is_refused_and_a_wrapping_one_too() { assert!(in_user_half(USER_TOP - 8, 8)); @@ -185,9 +227,11 @@ mod tests { } #[test] - fn an_image_that_rebases_past_the_user_half_is_refused() { - assert_eq!(rebase_base(USER_VM_BASE, 0, USER_TOP - USER_VM_BASE), Some(USER_VM_BASE)); - assert_eq!(rebase_base(USER_VM_BASE, 0, USER_TOP - USER_VM_BASE + 1), None); + fn an_image_that_reaches_the_last_page_of_the_user_half_is_refused() { + assert_eq!(rebase_base(USER_VM_BASE, 0, PLACED_TOP - USER_VM_BASE), Some(USER_VM_BASE)); + assert_eq!(rebase_base(USER_VM_BASE, 0, PLACED_TOP - USER_VM_BASE + 1), None); + assert_eq!(rebase_base(USER_VM_BASE, 0, USER_TOP - USER_VM_BASE), None); assert_eq!(rebase_base(USER_VM_BASE, 0, u64::MAX), None); + assert_eq!(rebase_base(u64::MAX, 0, 1), None); } }