Skip to main content

strat9_kernel/process/
thread_ops.rs

1//! Thread lifecycle operations shared by the syscall layer and `/thread` VFS scheme.
2//!
3//! Extracted from `syscall/process.rs` so that both `SYS_THREAD_CREATE/JOIN/EXIT`
4//! and the Plan 9 style `/thread/*` scheme forward to the exact same internal
5//! implementation (no duplicated WaitQueue, no duplicated validation).
6//!
7//! Also owns the **kernel-owned user stack** bookkeeping: threads created via
8//! `/thread/create` get their user stack allocated and mapped by the kernel,
9//! which reclaims it deterministically when the thread dies. Threads created
10//! through raw `SYS_THREAD_CREATE` keep providing their own stack and leave
11//! the field untouched.
12
13use super::{
14    block_current_task, current_task_clone, current_task_id, current_tid, get_child_task_id_by_tid,
15    get_task_id_by_tid, kill_task,
16    scheduler::add_task_with_parent,
17    task::{CpuContext, ExtendedState, KernelStack, SyncUnsafeCell, Task},
18    try_wait_child, WaitChildResult,
19};
20use crate::{
21    memory::address_space::{VmaFlags, VmaPageSize, VmaType},
22    syscall::{error::SyscallError, SyscallFrame},
23};
24use alloc::{boxed::Box, sync::Arc};
25use core::{
26    mem::offset_of,
27    sync::atomic::{AtomicU64, Ordering},
28};
29
30// ============================================================================
31// Kernel-owned user stack accounting (leak-proofing / verification)
32// ============================================================================
33
34/// Total number of kernel-owned user stacks ever allocated.
35static KERNEL_USER_STACKS_ALLOCATED: AtomicU64 = AtomicU64::new(0);
36/// Number of kernel-owned user stacks currently alive (allocated - reclaimed).
37static KERNEL_USER_STACKS_ACTIVE: AtomicU64 = AtomicU64::new(0);
38
39/// Snapshot of kernel-owned user stack counters `(allocated_total, active)`.
40///
41/// Exposed through `/thread/stats` for the zero-leak regression test.
42pub fn kernel_user_stack_stats() -> (u64, u64) {
43    (
44        KERNEL_USER_STACKS_ALLOCATED.load(Ordering::Relaxed),
45        KERNEL_USER_STACKS_ACTIVE.load(Ordering::Relaxed),
46    )
47}
48
49/// Bounds of a kernel-owned user stack request.
50pub const THREAD_MIN_STACK_BYTES: u64 = 4 * 1024;
51pub const THREAD_MAX_STACK_BYTES: u64 = 16 * 1024 * 1024;
52/// Default size when the caller passes `stack_size == 0`.
53pub const THREAD_DEFAULT_STACK_BYTES: u64 = 64 * 1024;
54
55#[inline]
56const fn page_align_up(addr: u64) -> u64 {
57    addr.wrapping_add(4095) & !4095u64
58}
59
60/// Allocate and lazily map a user stack in the parent process address space.
61///
62/// Returns `(vaddr_base, aligned_size)`; the initial RSP is the top of the
63/// region. Pages are demand-paged (`reserve_region`) exactly like `sys_mmap`.
64fn alloc_kernel_user_stack(parent: &Arc<Task>, requested: u64) -> Result<(u64, u64), SyscallError> {
65    let size = if requested == 0 {
66        THREAD_DEFAULT_STACK_BYTES
67    } else {
68        page_align_up(requested.clamp(THREAD_MIN_STACK_BYTES, THREAD_MAX_STACK_BYTES))
69    };
70    let n_pages = (size / 4096) as usize;
71
72    let addr_space = parent.process.address_space_arc();
73    let base = addr_space
74        .find_free_vma_range(crate::kaslr::mmap_base(), n_pages, VmaPageSize::Small)
75        .ok_or(SyscallError::OutOfMemory)?;
76    addr_space
77        .reserve_region(
78            base,
79            n_pages,
80            VmaFlags {
81                readable: true,
82                writable: true,
83                executable: false,
84                user_accessible: true,
85            },
86            VmaType::Anonymous,
87            VmaPageSize::Small,
88        )
89        .map_err(|_| SyscallError::OutOfMemory)?;
90
91    // Advance the mmap hint past the new mapping (atomically forward-only).
92    let _ = parent
93        .process
94        .mmap_hint
95        .fetch_max(base + n_pages as u64 * 4096, Ordering::Relaxed);
96
97    KERNEL_USER_STACKS_ALLOCATED.fetch_add(1, Ordering::Relaxed);
98    KERNEL_USER_STACKS_ACTIVE.fetch_add(1, Ordering::Relaxed);
99
100    Ok((base, size))
101}
102
103/// Reclaim the kernel-owned user stack of `task`, if any.
104///
105/// Idempotent: the mapping is `take()`n from the task so repeated calls from
106/// both the exit path and the reap path are safe. Called from
107/// `exit_current_task` (normal exit) and `cleanup_task_resources`
108/// (forced kills / reaping).
109pub fn reclaim_kernel_user_stack(task: &Arc<Task>) {
110    let Some((base, size)) = task.take_kernel_stack_user() else {
111        return;
112    };
113    let addr_space = task.process.address_space_arc();
114    if !addr_space.is_kernel() {
115        // The address space may already be torn down (whole-process exit);
116        // unmapping a stale range is harmless, so errors are ignored.
117        let _ = addr_space.unmap_range(base, size);
118    }
119    KERNEL_USER_STACKS_ACTIVE.fetch_sub(1, Ordering::Relaxed);
120}
121
122// ============================================================================
123// Userspace entry bootstrap (moved verbatim from syscall/process.rs)
124// ============================================================================
125
126#[repr(C)]
127#[derive(Clone, Copy)]
128struct ThreadUserContext {
129    entry: u64,
130    stack_top: u64,
131    arg0: u64,
132    user_cs: u64,
133    user_rflags: u64,
134    user_ss: u64,
135}
136
137const THREAD_OFF_ENTRY: usize = offset_of!(ThreadUserContext, entry);
138const THREAD_OFF_STACK_TOP: usize = offset_of!(ThreadUserContext, stack_top);
139const THREAD_OFF_ARG0: usize = offset_of!(ThreadUserContext, arg0);
140const THREAD_OFF_USER_CS: usize = offset_of!(ThreadUserContext, user_cs);
141const THREAD_OFF_USER_RFLAGS: usize = offset_of!(ThreadUserContext, user_rflags);
142const THREAD_OFF_USER_SS: usize = offset_of!(ThreadUserContext, user_ss);
143
144/// Performs the thread child start operation.
145extern "C" fn thread_child_start(ctx_ptr: u64) -> ! {
146    // SAFETY: `ctx_ptr` is allocated with Box::into_raw in `build_user_thread_task`
147    // and passed as immutable bootstrap data for this task only.
148    let boxed = unsafe { Box::from_raw(ctx_ptr as *mut ThreadUserContext) };
149    let ctx = *boxed;
150    // SAFETY: Assembly routine performs an iretq into userspace with validated context.
151    unsafe { thread_iret_from_ctx(&ctx as *const ThreadUserContext) }
152}
153
154/// Performs the thread iret from ctx operation.
155#[unsafe(naked)]
156unsafe extern "C" fn thread_iret_from_ctx(_ctx: *const ThreadUserContext) -> ! {
157    core::arch::naked_asm!(
158        // Mask IRQs before touching GS. The user RFLAGS frame re-enables IF.
159        "cli",
160        "mov rsi, rdi",
161        // Build iret frame: SS, RSP, RFLAGS, CS, RIP
162        "mov r8, [rsi + {off_user_ss}]",
163        "push r8",
164        "mov r8, [rsi + {off_stack_top}]",
165        "push r8",
166        "mov r8, [rsi + {off_user_rflags}]",
167        "push r8",
168        "mov r8, [rsi + {off_user_cs}]",
169        "push r8",
170        "mov r8, [rsi + {off_entry}]",
171        "push r8",
172        // Argument convention for userspace entry: rdi = arg0
173        "mov rdi, [rsi + {off_arg0}]",
174        // Child thread returns 0 if entry routine ever reads rax.
175        "xor rax, rax",
176        "swapgs",
177        "iretq",
178        off_entry = const THREAD_OFF_ENTRY,
179        off_stack_top = const THREAD_OFF_STACK_TOP,
180        off_arg0 = const THREAD_OFF_ARG0,
181        off_user_cs = const THREAD_OFF_USER_CS,
182        off_user_rflags = const THREAD_OFF_USER_RFLAGS,
183        off_user_ss = const THREAD_OFF_USER_SS,
184    );
185}
186
187/// Caller's userspace segment/flag state used to seed the child iret frame.
188#[derive(Debug, Clone, Copy)]
189pub struct UserEntryContext {
190    pub cs: u64,
191    pub rflags: u64,
192    pub ss: u64,
193}
194
195impl UserEntryContext {
196    /// Build the context from a syscall frame (dispatcher path).
197    pub fn from_frame(frame: &SyscallFrame) -> Self {
198        UserEntryContext {
199            cs: frame.iret_cs,
200            rflags: frame.iret_rflags,
201            ss: frame.iret_ss,
202        }
203    }
204
205    /// Build a synthetic ring-3 context from the GDT (scheme path, where the
206    /// caller's frame is not passed down through the VFS layer).
207    pub fn ring3() -> Self {
208        UserEntryContext {
209            cs: crate::arch::x86_64::gdt::user_code_selector().0 as u64,
210            rflags: 0x202, // IF set + reserved bit 1
211            ss: crate::arch::x86_64::gdt::user_data_selector().0 as u64,
212        }
213    }
214}
215
216/// Performs the build user thread task operation.
217fn build_user_thread_task(
218    parent: &Arc<Task>,
219    bootstrap_ctx: Box<ThreadUserContext>,
220    tls_base: u64,
221) -> Result<Arc<Task>, SyscallError> {
222    let kernel_stack =
223        KernelStack::allocate(Task::DEFAULT_STACK_SIZE).map_err(|_| SyscallError::OutOfMemory)?;
224    let context = CpuContext::new(thread_child_start as *const () as u64, &kernel_stack);
225    let (pid, tid, _) = Task::allocate_process_ids();
226
227    let parent_fpu = unsafe { &*parent.fpu_state.get() };
228    let mut child_fpu = ExtendedState::new();
229    child_fpu.copy_from(parent_fpu);
230    let interrupt_frame = SyscallFrame {
231        r15: 0,
232        r14: 0,
233        r13: 0,
234        r12: 0,
235        rbp: 0,
236        rbx: 0,
237        r11: bootstrap_ctx.user_rflags,
238        r10: 0,
239        r9: 0,
240        r8: 0,
241        rsi: 0,
242        rdi: bootstrap_ctx.arg0,
243        rdx: 0,
244        rcx: bootstrap_ctx.entry,
245        rax: 0,
246        iret_rip: bootstrap_ctx.entry,
247        iret_cs: bootstrap_ctx.user_cs,
248        iret_rflags: bootstrap_ctx.user_rflags,
249        iret_rsp: bootstrap_ctx.stack_top,
250        iret_ss: bootstrap_ctx.user_ss,
251    };
252
253    let task = Arc::new(Task {
254        id: crate::process::TaskId::new(),
255        pid,
256        tid,
257        tgid: parent.tgid,
258        pgid: core::sync::atomic::AtomicU32::new(parent.pgid.load(Ordering::Relaxed)),
259        sid: core::sync::atomic::AtomicU32::new(parent.sid.load(Ordering::Relaxed)),
260        uid: core::sync::atomic::AtomicU32::new(parent.uid.load(Ordering::Relaxed)),
261        euid: core::sync::atomic::AtomicU32::new(parent.euid.load(Ordering::Relaxed)),
262        gid: core::sync::atomic::AtomicU32::new(parent.gid.load(Ordering::Relaxed)),
263        egid: core::sync::atomic::AtomicU32::new(parent.egid.load(Ordering::Relaxed)),
264        state: core::sync::atomic::AtomicU8::new(crate::process::TaskState::Ready as u8),
265        priority: parent.priority,
266        context: SyncUnsafeCell::new(context),
267        resume_kind: SyncUnsafeCell::new(crate::process::task::ResumeKind::RetFrame),
268        interrupt_rsp: core::sync::atomic::AtomicU64::new(0),
269        kernel_stack,
270        user_stack: None,
271        stack_canary: core::sync::atomic::AtomicU64::new(0),
272        stack_canary_addr: core::sync::atomic::AtomicU64::new(0),
273        kernel_stack_user: SyncUnsafeCell::new(None),
274        name: "user-thread",
275        process: parent.process.clone(),
276        pending_signals: crate::process::signal::SignalSet::new(),
277        blocked_signals: parent.blocked_signals.clone(),
278        irq_signal_delivery_blocked: core::sync::atomic::AtomicBool::new(false),
279        signal_stack: SyncUnsafeCell::new(None),
280        itimers: crate::process::timer::ITimers::new(),
281        wake_pending: core::sync::atomic::AtomicBool::new(false),
282        wake_deadline_ns: core::sync::atomic::AtomicU64::new(0),
283        trampoline_entry: core::sync::atomic::AtomicU64::new(0),
284        trampoline_stack_top: core::sync::atomic::AtomicU64::new(0),
285        trampoline_arg0: core::sync::atomic::AtomicU64::new(0),
286        ticks: core::sync::atomic::AtomicU64::new(0),
287        sched_policy: SyncUnsafeCell::new(parent.sched_policy()),
288        home_cpu: core::sync::atomic::AtomicUsize::new(usize::MAX),
289        last_cpu: core::sync::atomic::AtomicUsize::new(usize::MAX),
290        affinity_mask: core::sync::atomic::AtomicU64::new(0),
291        vruntime: core::sync::atomic::AtomicU64::new(parent.vruntime()),
292        fair_rq_generation: core::sync::atomic::AtomicU64::new(0),
293        fair_on_rq: core::sync::atomic::AtomicBool::new(false),
294        clear_child_tid: core::sync::atomic::AtomicU64::new(0),
295        robust_list_head: core::sync::atomic::AtomicU64::new(0),
296        robust_list_len: core::sync::atomic::AtomicUsize::new(0),
297        user_fs_base: core::sync::atomic::AtomicU64::new(tls_base),
298        fpu_state: SyncUnsafeCell::new(child_fpu),
299        xcr0_mask: core::sync::atomic::AtomicU64::new(parent.xcr0_mask.load(Ordering::Relaxed)),
300        rt_link: intrusive_collections::LinkedListLink::new(),
301        rt_budget_remaining: core::sync::atomic::AtomicU64::new(
302            parent.rt_budget_remaining.load(Ordering::Relaxed),
303        ),
304        rt_budget_period_start: core::sync::atomic::AtomicU64::new(
305            parent.rt_budget_period_start.load(Ordering::Relaxed),
306        ),
307        rt_degraded: core::sync::atomic::AtomicBool::new(
308            parent.rt_degraded.load(Ordering::Relaxed),
309        ),
310        fair_wait_ticks: core::sync::atomic::AtomicU64::new(0),
311    });
312
313    // CpuContext initial stack layout: r15, r14, r13(arg), r12(entry), rbp, rbx, ret
314    // Seed r13 with bootstrap context pointer for `thread_child_start`.
315    unsafe {
316        let ctx = &mut *task.context.get();
317        let frame = ctx.saved_rsp as *mut u64;
318        *frame.add(2) = Box::into_raw(bootstrap_ctx) as u64;
319    }
320
321    task.seed_interrupt_frame(interrupt_frame);
322
323    Ok(task)
324}
325
326// ============================================================================
327// Public helpers (shared by dispatcher syscalls and the /thread scheme)
328// ============================================================================
329
330const USER_TOP_EXCLUSIVE: u64 = crate::memory::userslice::USER_SPACE_END;
331
332/// Core of `SYS_THREAD_CREATE` and `/thread/create`.
333///
334/// Validates arguments with the exact same rules as the raw syscall, builds
335/// the child task, registers it as a child of the current task, and returns
336/// it (the caller decides what to expose: TID via syscall return value, or
337/// kernel-owned stack + TID via the scheme).
338///
339/// `stack_top` is supplied by the caller (raw path). For the scheme path use
340/// [`create_user_thread_with_kernel_stack`] instead, which allocates the
341/// user stack in the kernel.
342pub fn create_user_thread(
343    entry_ctx: UserEntryContext,
344    entry: u64,
345    stack_top: u64,
346    arg0: u64,
347    flags: u64,
348    tls_base: u64,
349) -> Result<Arc<Task>, SyscallError> {
350    if flags != 0 {
351        return Err(SyscallError::InvalidArgument);
352    }
353
354    if entry == 0
355        || stack_top == 0
356        || entry >= USER_TOP_EXCLUSIVE
357        || stack_top >= USER_TOP_EXCLUSIVE
358        || (stack_top & 0x7) != 0
359    // 8-byte alignment (x86_64 entry: 8 mod 16 is valid)
360    {
361        return Err(SyscallError::InvalidArgument);
362    }
363
364    let parent = current_task_clone().ok_or(SyscallError::Fault)?;
365    if parent.is_kernel() {
366        return Err(SyscallError::PermissionDenied);
367    }
368
369    let user_ctx = Box::new(ThreadUserContext {
370        entry,
371        stack_top,
372        arg0,
373        user_cs: entry_ctx.cs,
374        user_rflags: entry_ctx.rflags | (1 << 9),
375        user_ss: entry_ctx.ss,
376    });
377
378    let child = build_user_thread_task(&parent, user_ctx, tls_base)?;
379    add_task_with_parent(child.clone(), parent.id);
380    Ok(child)
381}
382
383/// `/thread/create` variant: the kernel allocates and owns the user stack.
384///
385/// Same validation rules as [`create_user_thread`] except `stack_size` replaces
386/// `stack_top`. On success the child carries a `kernel_stack_user` mapping that
387/// is reclaimed automatically when the thread dies (making `detach()` free of
388/// any userspace cleanup).
389pub fn create_user_thread_with_kernel_stack(
390    entry_ctx: UserEntryContext,
391    entry: u64,
392    stack_size: u64,
393    arg0: u64,
394    tls_base: u64,
395) -> Result<Arc<Task>, SyscallError> {
396    if entry == 0 || entry >= USER_TOP_EXCLUSIVE {
397        return Err(SyscallError::InvalidArgument);
398    }
399    if arg0 != 0 && arg0 >= USER_TOP_EXCLUSIVE {
400        return Err(SyscallError::InvalidArgument);
401    }
402    if tls_base != 0 && tls_base >= USER_TOP_EXCLUSIVE {
403        return Err(SyscallError::InvalidArgument);
404    }
405
406    let parent = current_task_clone().ok_or(SyscallError::Fault)?;
407    if parent.is_kernel() {
408        return Err(SyscallError::PermissionDenied);
409    }
410
411    let (stack_base, stack_len) = alloc_kernel_user_stack(&parent, stack_size)?;
412    let stack_top = stack_base + stack_len;
413
414    let child = match create_user_thread_inner(&parent, entry_ctx, entry, stack_top, arg0, tls_base)
415    {
416        Ok(child) => child,
417        Err(e) => {
418            // Roll back the mapping on failure.
419            let _ = parent
420                .process
421                .address_space_arc()
422                .unmap_range(stack_base, stack_len);
423            KERNEL_USER_STACKS_ALLOCATED.fetch_sub(1, Ordering::Relaxed);
424            KERNEL_USER_STACKS_ACTIVE.fetch_sub(1, Ordering::Relaxed);
425            return Err(e);
426        }
427    };
428    child.set_kernel_stack_user(stack_base, stack_len);
429
430    add_task_with_parent(child.clone(), parent.id);
431    Ok(child)
432}
433
434fn create_user_thread_inner(
435    parent: &Arc<Task>,
436    entry_ctx: UserEntryContext,
437    entry: u64,
438    stack_top: u64,
439    arg0: u64,
440    tls_base: u64,
441) -> Result<Arc<Task>, SyscallError> {
442    let user_ctx = Box::new(ThreadUserContext {
443        entry,
444        stack_top,
445        arg0,
446        user_cs: entry_ctx.cs,
447        user_rflags: entry_ctx.rflags | (1 << 9),
448        user_ss: entry_ctx.ss,
449    });
450    build_user_thread_task(parent, user_ctx, tls_base)
451}
452
453/// Core of `SYS_THREAD_JOIN` and `/thread/join/<tid>`.
454///
455/// Blocking reap loop: only children of the calling task can be joined,
456/// self-join is `EINVAL`, absent or already-reaped targets yield `ENOENT`.
457/// Returns `(tid, exit_code)`.
458pub fn join_task(wait_tid: u32) -> Result<(u32, i32), SyscallError> {
459    let current = current_task_clone().ok_or(SyscallError::Fault)?;
460    if wait_tid == current.tid {
461        return Err(SyscallError::InvalidArgument);
462    }
463
464    let parent_id = current_task_id().ok_or(SyscallError::Fault)?;
465    let child_id = get_child_task_id_by_tid(parent_id, wait_tid).ok_or(SyscallError::NotFound)?;
466
467    loop {
468        match try_wait_child(parent_id, Some(child_id)) {
469            WaitChildResult::Reaped { status, .. } => return Ok((wait_tid, status)),
470            WaitChildResult::NoChildren => return Err(SyscallError::NotFound),
471            WaitChildResult::StillRunning => block_current_task(),
472        }
473    }
474}
475
476/// Core of `SYS_THREAD_EXIT` and `/thread/exit`.
477///
478/// Terminates only the calling thread; never returns.
479pub fn exit_current_thread(code: i32) -> ! {
480    crate::process::scheduler::exit_current_task(code)
481}
482
483/// Kill a thread by TID, restricted to threads of the caller's own process.
484///
485/// Errors: `ESRCH` when the TID does not exist, `EPERM` when the target lives
486/// in another thread group (mirrors the parent/child restriction of join).
487pub fn kill_thread(caller_tid: u32, target_tid: u32) -> Result<(), SyscallError> {
488    let caller = current_task_clone().ok_or(SyscallError::Fault)?;
489    if caller.tid != caller_tid {
490        return Err(SyscallError::InvalidArgument);
491    }
492    if target_tid == caller.tid {
493        // Killing yourself is exit(), not kill(): refuse explicitly.
494        return Err(SyscallError::InvalidArgument);
495    }
496    let target_id = get_task_id_by_tid(target_tid).ok_or(SyscallError::NoSuchProcess)?;
497    let target = super::get_task_by_id(target_id).ok_or(SyscallError::NoSuchProcess)?;
498    if target.tgid != caller.tgid {
499        return Err(SyscallError::PermissionDenied);
500    }
501    if !kill_task(target_id) {
502        return Err(SyscallError::NoSuchProcess);
503    }
504    Ok(())
505}
506
507/// Current TID of the calling task (scheme `/thread/current` fast path).
508pub fn current_thread_tid() -> Result<u32, SyscallError> {
509    current_tid().ok_or(SyscallError::Fault)
510}