Skip to main content

strat9_kernel/process/
task.rs

1//! Task Management
2//!
3//! Defines the Task structure and related types for the Strat9-OS scheduler.
4
5use crate::{
6    arch::xshim::{PhysAddr, VirtAddr},
7    memory::AddressSpace,
8};
9use alloc::sync::Arc;
10use core::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, AtomicU8, AtomicUsize, Ordering};
11use intrusive_collections::LinkedListLink;
12
13/// POSIX process ID.
14pub type Pid = u32;
15/// POSIX thread ID.
16pub type Tid = u32;
17
18/// Performs the next pid operation.
19#[inline]
20fn next_pid() -> Pid {
21    static NEXT_PID: AtomicU32 = AtomicU32::new(1);
22    NEXT_PID.fetch_add(1, Ordering::SeqCst)
23}
24
25/// Performs the next tid operation.
26#[inline]
27fn next_tid() -> Tid {
28    static NEXT_TID: AtomicU32 = AtomicU32::new(1);
29    NEXT_TID.fetch_add(1, Ordering::SeqCst)
30}
31
32/// Unique identifier for a task
33#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
34pub struct TaskId(pub u64);
35
36impl TaskId {
37    /// Generate a new unique task ID
38    pub fn new() -> Self {
39        static NEXT_ID: AtomicU64 = AtomicU64::new(0);
40        TaskId(NEXT_ID.fetch_add(1, Ordering::SeqCst))
41    }
42
43    /// Get the raw u64 value
44    pub fn as_u64(self) -> u64 {
45        self.0
46    }
47
48    /// Create a TaskId from a raw u64 (for IPC reply routing).
49    pub fn from_u64(raw: u64) -> Self {
50        TaskId(raw)
51    }
52}
53
54impl core::fmt::Display for TaskId {
55    /// Performs the fmt operation.
56    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
57        write!(f, "{}", self.0)
58    }
59}
60
61/// Priority levels for tasks
62#[derive(Debug, Clone, Copy, PartialEq, Eq)]
63pub enum TaskPriority {
64    Idle = 0,
65    Low = 1,
66    Normal = 2,
67    High = 3,
68    Realtime = 4,
69}
70
71/// State of a task in the scheduler
72#[repr(u8)]
73#[derive(Debug, Clone, Copy, PartialEq, Eq)]
74pub enum TaskState {
75    /// Task is ready to be scheduled
76    Ready = 0,
77    /// Task is currently running
78    Running = 1,
79    /// Task is blocked waiting for an event
80    Blocked = 2,
81    /// Task has exited
82    Dead = 3,
83}
84
85/// How this task must be resumed the next time the scheduler selects it.
86///
87/// - `RetFrame`: legacy kernel-only context switch using `ret`
88/// - `IretFrame`: interrupt/syscall-like frame restored with `iretq`
89#[derive(Debug, Clone, Copy, PartialEq, Eq)]
90pub enum ResumeKind {
91    RetFrame,
92    IretFrame,
93}
94
95use core::cell::UnsafeCell;
96
97/// A wrapper around UnsafeCell that implements Sync for TaskState
98pub struct SyncUnsafeCell<T> {
99    inner: UnsafeCell<T>,
100}
101
102unsafe impl<T> Sync for SyncUnsafeCell<T> {}
103
104impl<T> SyncUnsafeCell<T> {
105    /// Creates a new instance.
106    pub const fn new(value: T) -> Self {
107        Self {
108            inner: UnsafeCell::new(value),
109        }
110    }
111
112    /// Performs the get operation.
113    pub fn get(&self) -> *mut T {
114        self.inner.get()
115    }
116}
117
118/// FPU/SSE/AVX extended state, saved and restored on context switch.
119///
120/// When XSAVE is available, uses `xsave`/`xrstor` with a variable-size area.
121/// Falls back to `fxsave`/`fxrstor` (512 bytes) on older CPUs.
122#[repr(C, align(64))]
123pub struct ExtendedState {
124    pub data: [u8; Self::MAX_XSAVE_SIZE],
125    pub size: usize,
126    pub uses_xsave: bool,
127    pub xcr0_mask: u64,
128}
129
130// XSAVE/XRSTOR #GP on non-64-byte-aligned operands (SDM Vol. 2A, "XSAVE"):
131// the save area MUST stay 64-byte aligned for its whole lifetime. These
132// compile-time assertions pin the layout contract:
133//  - the struct itself is 64B-aligned (`repr(align)`) and `data` is at
134//    offset 0, so any &self.data / self-as-*mut u8 inherits the alignment;
135//  - MAX_XSAVE_SIZE keeps the total struct size a multiple of 64 so arrays
136//    of ExtendedState never break per-element alignment.
137const _: () = assert!(core::mem::align_of::<ExtendedState>() == 64);
138const _: () = assert!(core::mem::offset_of!(ExtendedState, data) == 0);
139const _: () = assert!(core::mem::size_of::<ExtendedState>() % 64 == 0);
140
141impl core::fmt::Debug for ExtendedState {
142    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
143        f.debug_struct("ExtendedState")
144            .field("size", &self.size)
145            .field("uses_xsave", &self.uses_xsave)
146            .field("xcr0_mask", &self.xcr0_mask)
147            .finish()
148    }
149}
150
151impl ExtendedState {
152    pub const FXSAVE_SIZE: usize = 512;
153    pub const MAX_XSAVE_SIZE: usize = 2688;
154
155    /// Create a new default state using the host's maximum capabilities.
156    ///
157    /// Uses the frozen boot-time XSAVE profile (see `cpuid::boot_xsave_profile`)
158    /// so task creation never walks CPUID leaf 0x0D sub-leaves.
159    pub fn new() -> Self {
160        let profile = crate::arch::x86_64::cpuid::boot_xsave_profile();
161        let uses_xsave = profile.xcr0_mask != 0 && crate::arch::x86_64::cpuid::host_uses_xsave();
162        let size = if uses_xsave {
163            // Profile area size is already 64B-rounded and capped by MAX.
164            profile.area_size.min(Self::MAX_XSAVE_SIZE)
165        } else {
166            Self::FXSAVE_SIZE
167        };
168        let default_xcr0 = if uses_xsave { profile.xcr0_mask } else { 0x3 };
169
170        let mut state = Self {
171            data: [0u8; Self::MAX_XSAVE_SIZE],
172            size,
173            uses_xsave,
174            xcr0_mask: default_xcr0,
175        };
176        state.set_defaults();
177        state
178    }
179
180    /// Create a state for a specific XCR0 mask (per-silo feature restriction).
181    pub fn for_xcr0(xcr0: u64) -> Self {
182        let uses_xsave = crate::arch::cpuid::host_uses_xsave();
183        let size = if uses_xsave {
184            crate::arch::cpuid::xsave_size_for_xcr0(xcr0).min(Self::MAX_XSAVE_SIZE)
185        } else {
186            Self::FXSAVE_SIZE
187        };
188
189        let mut state = Self {
190            data: [0u8; Self::MAX_XSAVE_SIZE],
191            size,
192            uses_xsave,
193            xcr0_mask: xcr0,
194        };
195        state.set_defaults();
196        state
197    }
198
199    fn set_defaults(&mut self) {
200        // Set x87 FCW = 0x037F and MXCSR = 0x1F80 directly in the buffer.
201        // These match the x87/SSE INIT values, so even though the XSAVE
202        // header's XSTATE_BV field (bytes 512-519) remains 0, the first
203        // XRSTOR will restore the same values from INIT defaults.
204        // This is correct and safe by design.
205        self.data[0] = 0x7F;
206        self.data[1] = 0x03;
207        self.data[24] = 0x80;
208        self.data[25] = 0x1F;
209    }
210
211    /// Copy the state from another `ExtendedState`.
212    pub fn copy_from(&mut self, other: &ExtendedState) {
213        let len = other.size.min(self.size);
214        self.data[..len].copy_from_slice(&other.data[..len]);
215    }
216}
217
218#[inline]
219fn normalized_xcr0(xcr0: u64) -> u64 {
220    if !crate::arch::cpuid::host_uses_xsave() {
221        return 0x3;
222    }
223
224    // Use the lock-free cache to avoid acquiring the HOST_CPU spinlock
225    // with interrupts disabled in the context-switch hot path.
226    let host_xcr0 = crate::arch::cpuid::host_default_xcr0_fast();
227    (xcr0 & host_xcr0).max(0x3)
228}
229
230/// Represents a single task/thread in the system
231pub struct Task {
232    /// Unique identifier for this task
233    pub id: TaskId,
234    /// Process identifier visible to userspace.
235    pub pid: Pid,
236    /// Thread identifier visible to userspace.
237    pub tid: Tid,
238    /// Thread-group identifier (equals process leader PID).
239    pub tgid: Pid,
240    /// Process group id (job-control group).
241    pub pgid: AtomicU32,
242    /// Session id.
243    pub sid: AtomicU32,
244    /// real user id.
245    pub uid: AtomicU32,
246    /// effective user id.
247    pub euid: AtomicU32,
248    /// real group id.
249    pub gid: AtomicU32,
250    /// effective group id.
251    pub egid: AtomicU32,
252    /// Current state of the task. Stored as AtomicU8 for lock-free cross-CPU visibility.
253    /// Use `get_state()` / `set_state()` for typed access.
254    pub state: AtomicU8,
255    /// Priority level of the task
256    pub priority: TaskPriority,
257    /// Saved CPU context for this task (just the stack pointer)
258    pub context: SyncUnsafeCell<CpuContext>,
259    /// Resume convention for this task's saved kernel stack frame.
260    pub resume_kind: SyncUnsafeCell<ResumeKind>,
261    /// Saved interrupt/syscall-compatible frame pointer for `iretq`-based resume.
262    pub interrupt_rsp: AtomicU64,
263    /// Kernel stack for this task
264    pub kernel_stack: KernelStack,
265    /// User stack for this task (if applicable)
266    pub user_stack: Option<UserStack>,
267    /// Random per-process canary written at the top word of the user stack
268    /// (issue #63). 0 = no canary (kernel tasks, threads without user stack).
269    pub stack_canary: AtomicU64,
270    /// User address of the canary slot (stack_top - 8); 0 = none.
271    pub stack_canary_addr: AtomicU64,
272    /// Kernel-owned user thread stack: `(vaddr_base, size_bytes)` in the
273    /// process address space. Set by `/thread/create`; the kernel reclaims
274    /// (munmaps) it when the task dies. `None` for tasks created by other
275    /// paths (fork, exec, raw SYS_THREAD_CREATE).
276    pub kernel_stack_user: SyncUnsafeCell<Option<(u64, u64)>>,
277    /// Task name for debugging purposes
278    pub name: &'static str,
279    /// Capabilities granted to this task
280    /// Address space for this task (kernel tasks share the kernel AS)
281    pub process: Arc<crate::process::process::Process>,
282    /// File descriptor table for this task
283    /// Pending signals for this task
284    pub pending_signals: super::signal::SignalSet,
285    /// Blocked signals mask for this task
286    pub blocked_signals: super::signal::SignalSet,
287    /// Suppress repeated IRQ-return delivery attempts until a normal delivery
288    /// path runs again.
289    pub irq_signal_delivery_blocked: AtomicBool,
290    /// Signal actions (handlers) for this task
291    /// Signal alternate stack for this task
292    pub signal_stack: SyncUnsafeCell<Option<super::signal::SigStack>>,
293    /// Interval timers (ITIMER_REAL, ITIMER_VIRTUAL, ITIMER_PROF)
294    pub itimers: super::timer::ITimers,
295    /// Pending wakeup flag: set by `wake_task()` when the task is not yet
296    /// in `blocked_tasks` (it is still transitioning to Blocked state).
297    /// Checked by `block_current_task()` : if set, the task skips blocking
298    /// and continues execution, preventing a lost-wakeup race.
299    pub wake_pending: AtomicBool,
300    /// Sleep deadline in nanoseconds (monotonic). If non-zero, the task
301    /// is sleeping until this time. Checked by the scheduler to auto-wake.
302    pub wake_deadline_ns: AtomicU64,
303    /// Program break (end of heap), in bytes. 0 = not yet initialised.
304    /// Lazily set to `BRK_BASE` on the first `sys_brk` call.
305    /// mmap_hint: next candidate virtual address for anonymous mmap allocations
306    /// User-space entry point for ring3 trampoline (ELF tasks only, 0 otherwise).
307    pub trampoline_entry: AtomicU64,
308    /// User-space stack top for ring3 trampoline (ELF tasks only, 0 otherwise).
309    pub trampoline_stack_top: AtomicU64,
310    /// First argument (RDI) passed to the user process on entry (e.g. bootstrap cap handle).
311    pub trampoline_arg0: AtomicU64,
312    /// Total CPU ticks consumed by this task
313    pub ticks: AtomicU64,
314    /// Scheduling policy (Fair, RealTime, Idle)
315    pub sched_policy: SyncUnsafeCell<crate::process::sched::SchedPolicy>,
316    /// Home CPU index for this task. Set when the task is first scheduled
317    /// or explicitly assigned. Used by `wake_task()` to route to the correct
318    /// per-CPU runqueue without acquiring `GLOBAL_SCHED_STATE`.
319    pub home_cpu: AtomicUsize,
320    /// Last CPU this task ran on. Updated on every pick-next. Used by the
321    /// scheduler to prefer cache-warm placement when waking tasks.
322    pub last_cpu: AtomicUsize,
323    /// Soft CPU affinity bitmask. Bit N set = CPU N is allowed.
324    /// 0 means "no restriction" (all CPUs allowed).
325    /// Derived from the task's silo `cpu_affinity_mask` at creation time,
326    /// or set via `sched_setaffinity` syscall. The scheduler *prefers*
327    /// CPUs in this mask but does not hard-restrict (falls back to any
328    /// CPU if no affinity-eligible CPU has capacity).
329    pub affinity_mask: AtomicU64,
330    /// Virtual runtime for CFS
331    pub vruntime: AtomicU64,
332    /// Monotonic token identifying the currently valid FAIR runqueue entry.
333    pub fair_rq_generation: AtomicU64,
334    /// Whether this task is logically present in the FAIR runqueue.
335    pub fair_on_rq: AtomicBool,
336    /// TID address for futex-based thread join (set_tid_address).
337    /// The kernel writes 0 here when the thread exits, then futex_wake.
338    pub clear_child_tid: AtomicU64,
339    /// Robust list head pointer (set by set_robust_list).
340    /// Points to a userspace robust_list_head used for dead-thread mutex cleanup.
341    pub robust_list_head: AtomicU64,
342    /// Length of the robust list head structure (as passed to set_robust_list).
343    pub robust_list_len: AtomicUsize,
344    /// Current working directory (POSIX, inherited by children).
345    /// File creation mask (inherited by children, NOT reset by exec).
346    /// User-space FS.base (TLS on x86_64, set via arch_prctl ARCH_SET_FS).
347    /// Saved/restored across context switches.
348    pub user_fs_base: AtomicU64,
349    /// FPU/SSE/AVX extended state saved during context switch.
350    pub fpu_state: SyncUnsafeCell<ExtendedState>,
351    /// XCR0 mask for this task (inherited from its silo).
352    pub xcr0_mask: AtomicU64,
353    /// Intrusive linked-list link for the RT run queue.
354    ///
355    /// Only touched while holding the per-CPU scheduler spinlock.
356    pub rt_link: LinkedListLink,
357
358    // ── RT budget per period ──────────────────────────────────────────────
359    /// Remaining ticks in the current RT budget period.
360    /// Decremented by `update_current`; when 0 the task is preempted and
361    /// flagged degraded until the period expires.
362    pub rt_budget_remaining: AtomicU64,
363    /// Tick at which the current RT budget period started.
364    /// Used to determine when to reset `rt_budget_remaining`.
365    pub rt_budget_period_start: AtomicU64,
366    /// Whether this RT task has exhausted its budget and is temporarily
367    /// degraded (treated as Fair) until the period expires.
368    pub rt_degraded: AtomicBool,
369
370    // ── Fair starvation protection ────────────────────────────────────────
371    /// Ticks this task has spent waiting in the Fair run queue without being
372    /// selected.  Reset to 0 when the task is picked.  If this exceeds
373    /// `FAIR_STARVATION_THRESHOLD_TICKS`, the task is boosted to the front
374    /// of its priority class.
375    pub fair_wait_ticks: AtomicU64,
376}
377
378// SAFETY: `LinkedListLink` uses `UnsafeCell` internally and is therefore
379// `!Sync` by default, but all mutations to `rt_link` are performed under the
380// per-CPU scheduler spinlock.  Every other non-atomic field in `Task` is
381// similarly protected by the appropriate lock or by the task's own atomics.
382unsafe impl Sync for Task {}
383
384impl Task {
385    /// Re-checks the user-stack canary written by the ELF loader / execve
386    /// (issue #63).
387    ///
388    /// Called on task exit while the address space is still alive. A mismatch
389    /// means something wrote past the top of the boot stack area (classic
390    /// stack-smashing direction for the initial frames) : we log a security
391    /// warning rather than panicking, to keep false positives from turning
392    /// into a denial of service.
393    pub fn verify_user_stack_canary(&self) {
394        let canary = self.stack_canary.load(Ordering::Relaxed);
395        let addr = self.stack_canary_addr.load(Ordering::Relaxed);
396        if canary == 0 || addr == 0 {
397            return; // no canary (kernel tasks, threads, ...)
398        }
399        let as_ref = self.process.address_space_arc();
400        let Some(phys) = as_ref.translate(x86_64::VirtAddr::new(addr)) else {
401            log::warn!(
402                "[security] tid={} stack canary slot {:#x} unmapped at exit",
403                self.id.as_u64(),
404                addr
405            );
406            return;
407        };
408        // SAFETY: `translate()` proved the page is mapped in this address
409        // space; HHDM makes it readable from the kernel. Unaligned u64 read:
410        // the slot was placed 8 bytes below a page-aligned top.
411        let observed = unsafe {
412            core::ptr::read_volatile(crate::memory::phys_to_virt(phys.as_u64()) as *const u64)
413        };
414        if observed != canary {
415            log::warn!(
416                "[security] tid={} USER STACK CANARY SMASHED: expected={:#x} found={:#x} at {:#x}",
417                self.id.as_u64(),
418                canary,
419                observed,
420                addr
421            );
422            crate::serial_println!(
423                "[security] tid={} USER STACK CANARY SMASHED at {:#x}",
424                self.id.as_u64(),
425                addr
426            );
427        }
428    }
429
430    /// Leave this much headroom above the synthetic `SyscallFrame`.
431    ///
432    /// The raw IRQ switch path does `mov rsp, next_rsp` and then `call
433    /// finish_interrupt_switch`, so `next_rsp` must be close to the top of the
434    /// kernel stack to preserve downward growth room for the call chain.
435    const BOOTSTRAP_INTERRUPT_FRAME_TOP_HEADROOM: usize = 0x1000;
436
437    /// Canary placed below the interrupt frame to detect stack underflow
438    /// (interrupt handler overflowing downward past the frame)
439    const STACK_UNDERFLOW_CANARY_OFFSET: usize = 0x100; // 256 bytes from base
440
441    /// Performs the default sched policy operation.
442    pub fn default_sched_policy(priority: TaskPriority) -> crate::process::sched::SchedPolicy {
443        use crate::process::sched::{nice::Nice, real_time::RealTimePriority, SchedPolicy};
444        match priority {
445            TaskPriority::Idle => SchedPolicy::Idle,
446            TaskPriority::Realtime => SchedPolicy::RealTimeRR {
447                prio: RealTimePriority::new(50),
448            },
449            TaskPriority::High => SchedPolicy::Fair(Nice::new(-10)),
450            TaskPriority::Low => SchedPolicy::Fair(Nice::new(10)),
451            TaskPriority::Normal => SchedPolicy::Fair(Nice::default()),
452        }
453    }
454
455    /// Get the current scheduling policy of the task
456    pub fn sched_policy(&self) -> crate::process::sched::SchedPolicy {
457        unsafe { *self.sched_policy.get() }
458    }
459
460    /// Set the scheduling policy of the task
461    pub fn set_sched_policy(&self, policy: crate::process::sched::SchedPolicy) {
462        unsafe {
463            *self.sched_policy.get() = policy;
464        }
465    }
466
467    /// Returns the current resume convention for this task.
468    pub fn resume_kind(&self) -> ResumeKind {
469        unsafe { *self.resume_kind.get() }
470    }
471
472    /// Sets the resume convention for this task.
473    pub fn set_resume_kind(&self, kind: ResumeKind) {
474        unsafe {
475            *self.resume_kind.get() = kind;
476        }
477    }
478
479    /// Returns the saved `iretq`-compatible frame pointer for this task.
480    pub fn interrupt_rsp(&self) -> u64 {
481        self.interrupt_rsp.load(Ordering::Acquire)
482    }
483
484    /// Updates the saved `iretq`-compatible frame pointer for this task.
485    pub fn set_interrupt_rsp(&self, rsp: u64) {
486        self.interrupt_rsp.store(rsp, Ordering::Release);
487    }
488
489    /// Seed a synthetic interrupt frame for tasks that have not yet been
490    /// preempted from an IRQ path but must still be resumable via `iretq`.
491    pub fn seed_interrupt_frame(&self, frame: crate::syscall::SyscallFrame) {
492        let stack_base = self.kernel_stack.virt_base.as_u64();
493        let stack_top = stack_base + self.kernel_stack.size as u64;
494        let frame_size = core::mem::size_of::<crate::syscall::SyscallFrame>() as u64;
495        let raw_frame_addr = stack_top
496            .saturating_sub(Self::BOOTSTRAP_INTERRUPT_FRAME_TOP_HEADROOM as u64)
497            .saturating_sub(frame_size);
498        let frame_addr = raw_frame_addr & !0xF;
499        let frame_end = frame_addr + core::mem::size_of::<crate::syscall::SyscallFrame>() as u64;
500        assert!(
501            frame_addr >= stack_base && frame_end <= stack_top,
502            "kernel stack too small for bootstrap interrupt frame"
503        );
504        unsafe {
505            (frame_addr as *mut crate::syscall::SyscallFrame).write(frame);
506
507            // Place underflow canary below the frame (at lower address)
508            // This detects if interrupt handler overflows downward past expected range
509            let canary_addr = stack_base + Self::STACK_UNDERFLOW_CANARY_OFFSET as u64;
510            *(canary_addr as *mut u64) = 0xBAD57ACBAD57AC;
511        }
512        self.set_interrupt_rsp(frame_addr);
513    }
514
515    /// Seed an `iretq`-compatible frame from the legacy `CpuContext` bootstrap
516    /// layout used by kernel tasks (`ret` into `task_entry_trampoline`).
517    ///
518    /// The synthesised frame always sets IF=1 so that IRQ-driven resumes keep
519    /// receiving timer interrupts. First-launch tasks still enter through the
520    /// legacy `ret` trampoline and must explicitly re-enable interrupts in
521    /// `task_post_switch_enter`.
522    pub fn seed_kernel_interrupt_frame_from_context(&self) {
523        let stack_base = self.kernel_stack.virt_base.as_u64();
524        let stack_top = stack_base + self.kernel_stack.size as u64;
525        let saved_rsp = unsafe { (*self.context.get()).saved_rsp as *const u64 };
526        let saved_rsp_val = saved_rsp as u64;
527        debug_assert!(
528            saved_rsp_val >= stack_base && saved_rsp_val.saturating_add(7 * 8) <= stack_top,
529            "saved_rsp outside kernel stack while seeding interrupt frame"
530        );
531        let ret_target = unsafe { *saved_rsp.add(6) };
532        // Always set IF=1 (bit 9) so IRQ-driven resumes keep interrupts enabled.
533        // First-launch tasks still need an explicit sti() in
534        // task_post_switch_enter because the legacy bootstrap path reaches the
535        // entry point through a plain ret, not an iretq restoring RFLAGS.
536        let rflags = 0x202u64; // bit 9 = IF, bit 1 = reserved (always 1)
537        let frame = unsafe {
538            crate::syscall::SyscallFrame {
539                r15: *saved_rsp.add(0),
540                r14: *saved_rsp.add(1),
541                r13: *saved_rsp.add(2),
542                r12: *saved_rsp.add(3),
543                rbp: *saved_rsp.add(4),
544                rbx: *saved_rsp.add(5),
545                r11: 0,
546                r10: 0,
547                r9: 0,
548                r8: 0,
549                rsi: 0,
550                rdi: 0,
551                rdx: 0,
552                rcx: 0,
553                rax: 0,
554                iret_rip: ret_target,
555                iret_cs: crate::arch::gdt::kernel_code_selector().0 as u64,
556                iret_rflags: rflags,
557                // Resume the task at the stack pointer it would have had after
558                // the legacy `ret`-based switch consumed its 7-word frame
559                // (6 callee-saved + return target). For a never-launched task
560                // this equals stack_top (the fake frame sits exactly there);
561                // for an already-launched task it equals the RSP captured by
562                // switch_context_fxsave at its last switch, which is where the
563                // interrupted code expects to resume. Using stack_top here
564                // instead desynchronizes the resumed task and derails it into
565                // unrelated kernel code (observed as the boot_alloc V-storm).
566                iret_rsp: saved_rsp_val + 7 * 8,
567                iret_ss: crate::arch::gdt::kernel_data_selector().0 as u64,
568            }
569        };
570        // TEMP DEBUG: pulse the seeded frame's r12 (entry) for tracing.
571        unsafe {
572            let hex = b"0123456789abcdef";
573            core::arch::asm!("out 0xe9, al", in("al") b'@', options(nomem, nostack));
574            core::arch::asm!("out 0xe9, al", in("al") b'E', options(nomem, nostack));
575            let v = frame.r12;
576            for sh in [28usize, 24, 20, 16, 12, 8, 4, 0] {
577                let nib = hex[((v >> sh) & 0xF) as usize];
578                core::arch::asm!("out 0xe9, al", in("al") nib, options(nomem, nostack));
579            }
580            core::arch::asm!("out 0xe9, al", in("al") b'\n', options(nomem, nostack));
581        }
582        self.seed_interrupt_frame(frame);
583    }
584
585    /// Get virtual runtime
586    pub fn vruntime(&self) -> u64 {
587        self.vruntime.load(Ordering::Relaxed)
588    }
589
590    /// Set virtual runtime
591    pub fn set_vruntime(&self, vruntime: u64) {
592        self.vruntime.store(vruntime, Ordering::Relaxed);
593    }
594
595    /// Prepare a new FAIR runqueue entry and return its generation token.
596    pub fn fair_prepare_enqueue(&self) -> (u64, bool) {
597        let was_queued = self.fair_on_rq.swap(true, Ordering::AcqRel);
598        let generation = self.fair_rq_generation.fetch_add(1, Ordering::Relaxed) + 1;
599        (generation, was_queued)
600    }
601
602    /// Returns the generation of the currently valid FAIR entry.
603    pub fn fair_generation(&self) -> u64 {
604        self.fair_rq_generation.load(Ordering::Relaxed)
605    }
606
607    /// Returns whether the task is logically queued in FAIR.
608    pub fn fair_is_on_rq(&self) -> bool {
609        self.fair_on_rq.load(Ordering::Acquire)
610    }
611
612    /// Marks the task as dequeued from FAIR.
613    pub fn fair_mark_dequeued(&self) -> bool {
614        self.fair_on_rq.swap(false, Ordering::AcqRel)
615    }
616
617    /// Invalidates the current FAIR entry so stale heap nodes can be skipped lazily.
618    pub fn fair_invalidate_rq_entry(&self) -> bool {
619        let was_queued = self.fair_on_rq.swap(false, Ordering::AcqRel);
620        self.fair_rq_generation.fetch_add(1, Ordering::Relaxed);
621        was_queued
622    }
623
624    /// Read the current task state atomically.
625    #[inline]
626    pub fn get_state(&self) -> TaskState {
627        let raw = self.state.load(Ordering::Acquire);
628        debug_assert!(
629            raw <= TaskState::Dead as u8,
630            "get_state: invalid TaskState discriminant {:#x}",
631            raw
632        );
633        // SAFETY: `raw` is always one of the four valid `#[repr(u8)]`
634        // discriminants (0..=3); the only writer is `set_state` which stores
635        // a cast from the same enum.
636        unsafe { core::mem::transmute(raw) }
637    }
638
639    /// Write the task state atomically. Uses Release ordering so the new state
640    /// is visible to any CPU that subsequently does an Acquire load.
641    #[inline]
642    pub fn set_state(&self, new_state: TaskState) {
643        self.state.store(new_state as u8, Ordering::Release);
644    }
645}
646
647/// CPU context saved/restored during context switches.
648///
649/// Only stores the saved RSP. All callee-saved registers (rbx, rbp, r12-r15)
650/// are pushed onto the task's kernel stack by `switch_context()`.
651#[repr(C)]
652
653pub struct CpuContext {
654    /// Saved stack pointer (points into the task's kernel stack)
655    pub saved_rsp: u64,
656}
657
658impl CpuContext {
659    /// Create a new CPU context for a task starting at the given entry point.
660    ///
661    /// Sets up a fake stack frame on the kernel stack that looks like
662    /// `switch_context()` just pushed callee-saved registers. When
663    /// `switch_context()` or `restore_first_task()` pops them and does `ret`,
664    /// it will jump to `task_entry_trampoline`, which enables interrupts
665    /// and jumps to the real entry point (stored in r12).
666    ///
667    /// Stack layout (growing downward):
668    /// ```text
669    /// [stack_top]
670    ///   0xDEADBEEFCAFEBABE      <- stack canary
671    ///   task_entry_trampoline   <- ret target
672    ///   0  (r15)
673    ///   0  (r14)
674    ///   0  (r13)
675    ///   entry_point (r12)      <- trampoline reads this
676    ///   0  (rbp)
677    ///   0  (rbx)
678    ///   <- saved_rsp points here
679    /// ```
680    pub fn new(entry_point: u64, kernel_stack: &KernelStack) -> Self {
681        let stack_top = kernel_stack.virt_base.as_u64() + kernel_stack.size as u64;
682
683        // Reserve space for the stack canary before building the fake frame.
684        const STACK_CANARY: u64 = 0xDEADBEEFCAFEBABE;
685        let canary_addr = stack_top - 8;
686        let initial_rsp = canary_addr - 7 * 8;
687
688        // SAFETY: We own this stack memory and it's properly allocated and zeroed.
689        // The stack region [virt_base, virt_base + size) is valid.
690        unsafe {
691            let stack = initial_rsp as *mut u64;
692            // Push order must match switch_context pops (LIFO, but we write linearly from RSP up):
693            // [RSP+0]  = r15
694            // [RSP+8]  = r14
695            // [RSP+16] = r13
696            // [RSP+24] = r12 (entry point)
697            // [RSP+32] = rbp
698            // [RSP+40] = rbx
699            // [RSP+48] = ret (trampoline)
700            *stack.add(0) = 0; // r15
701            *stack.add(1) = 0; // r14
702            *stack.add(2) = 0; // r13
703            *stack.add(3) = entry_point; // r12 (trampoline target)
704            *stack.add(4) = 0; // rbp
705            *stack.add(5) = 0; // rbx
706            *stack.add(6) = task_entry_trampoline as *const () as u64; // ret address
707        }
708
709        // Add stack canary at the very top (leave the frame below it so `ret` still points
710        // to `task_entry_trampoline`). The canary slot must be reserved before writing the
711        // frame to avoid overwriting the trampoline address.
712        unsafe {
713            let canary_ptr = canary_addr as *mut u64;
714            *canary_ptr = STACK_CANARY;
715        }
716
717        // Verify canary is still intact
718        unsafe {
719            let canary_ptr = canary_addr as *const u64;
720            let canary = *canary_ptr;
721            if canary != STACK_CANARY {
722                crate::serial_force_println!(
723                    "[PANIC] Stack canary corrupted at setup! entry_point={:#x} canary={:#x}",
724                    entry_point,
725                    canary
726                );
727            }
728        }
729
730        // Debug: verify entire stack frame
731        unsafe {
732            let stack = initial_rsp as *const u64;
733            crate::serial_println!(
734                "[CpuContext] frame verify: r15={:#x} r14={:#x} r13={:#x} r12={:#x} rbp={:#x} rbx={:#x} ret={:#x}",
735                *stack.add(0),
736                *stack.add(1),
737                *stack.add(2),
738                *stack.add(3),
739                *stack.add(4),
740                *stack.add(5),
741                *stack.add(6)
742            );
743            // Verify canary one more time
744            let canary_ptr = canary_addr as *const u64;
745            let canary = *canary_ptr;
746            if canary != STACK_CANARY {
747                crate::serial_force_println!(
748                    "[CpuContext] CANARY CORRUPTED AFTER FRAME SETUP! canary={:#x}",
749                    canary
750                );
751            }
752
753            // Debug: check if stack memory overlaps with another task
754            crate::serial_println!(
755                "[CpuContext] stack range: base={:#x} top={:#x} initial_rsp={:#x}",
756                kernel_stack.virt_base.as_u64(),
757                stack_top,
758                initial_rsp
759            );
760        }
761
762        CpuContext {
763            saved_rsp: initial_rsp,
764        }
765    }
766}
767
768/// Trampoline for newly created tasks.
769///
770/// When a new task is first scheduled, `switch_context()` pops the fake
771/// callee-saved registers and `ret`s here, then tail-jumps into the actual
772/// post-switch entry helper.
773#[unsafe(naked)]
774pub unsafe extern "C" fn task_entry_trampoline() -> ! {
775    core::arch::naked_asm!(
776        "mov al, 'T'",
777        "out 0xe9, al",
778        "call {finish_switch}",
779        "mov al, '1'",
780        "out 0xe9, al",
781        "mov rdi, r12", // entry_point
782        "mov rsi, r13", // arg0
783        "and rsp, -16",
784        "sub rsp, 8",
785        "jmp {post_switch_enter}",
786        finish_switch = sym crate::process::scheduler::finish_switch,
787        post_switch_enter = sym task_post_switch_enter,
788    );
789}
790
791fn task_post_switch_enter(entry: u64, arg0: u64) -> ! {
792    // breadcrumb: 'P' = reached post_switch_enter (no serial lock needed).
793    crate::arch::serial::putc(b'P');
794
795    crate::arch::percpu::mark_tlb_ready_current();
796
797    let cpu = crate::arch::percpu::current_cpu_index();
798
799    let is_user_entry = crate::process::scheduler::current_task_clone_try()
800        .map(|task| task.trampoline_entry.load(Ordering::Relaxed) != 0)
801        .unwrap_or(false);
802
803    // Single diagnostic print (IF may be 0 or 1 depending on RFLAGS seed; either
804    // way E9 is IRQ-safe and this is the LAST trace call before entry_fn).
805    if let Some(task) = crate::process::scheduler::current_task_clone_try() {
806        crate::e9_println!(
807            "[pse] cpu={} tid={} user={} entry={:#x}",
808            cpu,
809            task.id.as_u64(),
810            is_user_entry,
811            entry
812        );
813        if is_user_entry {
814            crate::serial_println!(
815                "[trace][task] post_switch_enter cpu={} tid={} entry={:#x}",
816                cpu,
817                task.id.as_u64(),
818                entry
819            );
820        }
821    }
822
823    // First-launch tasks arrive here via the legacy `ret` bootstrap path, which
824    // does not restore RFLAGS. Re-enable interrupts now that `finish_switch()`
825    // has completed and the task is running on its own stack.
826    crate::arch::sti();
827
828    // User tasks still transition to Ring 3 via iretq later and will restore
829    // their own RFLAGS there.
830
831    let entry_fn: extern "C" fn(u64) -> ! = unsafe { core::mem::transmute(entry as usize) };
832    entry_fn(arg0)
833}
834
835/// Kernel stack for a task
836pub struct KernelStack {
837    /// Physical address of the stack
838    pub base: PhysAddr,
839    /// Virtual address of the stack
840    pub virt_base: VirtAddr,
841    /// Size of the stack
842    pub size: usize,
843}
844
845impl KernelStack {
846    /// Allocate a new kernel stack using the buddy allocator
847    pub fn allocate(size: usize) -> Result<Self, &'static str> {
848        // Calculate number of pages needed (round up)
849        let pages = (size + 4095) / 4096;
850        let order = pages.next_power_of_two().trailing_zeros() as u8;
851
852        crate::serial_println!("[trace][task] kstack allocate begin size={}", size);
853        crate::serial_println!(
854            "[trace][task] kstack allocate pages={} order={}",
855            pages,
856            order
857        );
858
859        crate::serial_println!(
860            "[trace][task] kstack allocate calling allocate_frames order={}",
861            order
862        );
863        let frame = crate::sync::with_irqs_disabled(|token| {
864            crate::memory::allocate_kernel_stack_frames(token, order)
865        })
866        .map_err(|_| "Failed to allocate kernel stack")?;
867        crate::serial_println!(
868            "[trace][task] kstack allocate frame phys={:#x}",
869            frame.start_address.as_u64()
870        );
871
872        let phys_base = frame.start_address;
873        let virt_base = VirtAddr::new(crate::memory::phys_to_virt(phys_base.as_u64()));
874        crate::serial_println!(
875            "[trace][task] kstack allocate virt_base={:#x}",
876            virt_base.as_u64()
877        );
878
879        // Zero out the stack for safety
880        unsafe {
881            core::ptr::write_bytes(virt_base.as_mut_ptr::<u8>(), 0, size);
882        }
883        crate::serial_println!("[trace][task] kstack allocate memset done");
884
885        // Debug: verify zeroing worked
886        unsafe {
887            let first_word = *(virt_base.as_ptr::<u64>());
888            let mid_offset = size / 2;
889            let mid_word = *((virt_base.as_u64() + mid_offset as u64) as *const u64);
890            let last_offset = size - 8;
891            let last_word = *((virt_base.as_u64() + last_offset as u64) as *const u64);
892            if first_word != 0 || mid_word != 0 || last_word != 0 {
893                crate::serial_force_println!(
894                    "[WARN] kstack zeroing failed! first={:#x} mid={:#x} last={:#x}",
895                    first_word,
896                    mid_word,
897                    last_word
898                );
899            }
900        }
901
902        Ok(KernelStack {
903            base: phys_base,
904            virt_base,
905            size,
906        })
907    }
908
909    /// Debug: check if this stack overlaps with another range
910    pub fn overlaps(&self, other_base: u64, other_size: usize) -> bool {
911        let self_end = self.virt_base.as_u64() + self.size as u64;
912        let other_end = other_base + other_size as u64;
913        !(self_end <= other_base || other_end <= self.virt_base.as_u64())
914    }
915}
916
917impl Drop for KernelStack {
918    /// Performs the drop operation.
919    fn drop(&mut self) {
920        use crate::memory::frame::PhysFrame;
921
922        let pages = (self.size + 4095) / 4096;
923        let order = pages.next_power_of_two().trailing_zeros() as u8;
924        let frame = PhysFrame {
925            start_address: self.base,
926        };
927
928        crate::sync::with_irqs_disabled(|token| {
929            crate::memory::free_kernel_stack_frames(token, frame, order);
930        });
931    }
932}
933
934/// User stack for a task (when running in userspace)
935pub struct UserStack {
936    /// Virtual address of the user stack
937    pub virt_base: VirtAddr,
938    /// Size of the stack
939    pub size: usize,
940}
941
942impl Task {
943    /// Default kernel stack size (64 KB - increased from 16KB due to overflow)
944    pub const DEFAULT_STACK_SIZE: usize = 65536;
945
946    /// Create a new kernel task with a real allocated stack
947    pub fn new_kernel_task(
948        entry_point: extern "C" fn() -> !,
949        name: &'static str,
950        priority: TaskPriority,
951    ) -> Result<Arc<Self>, &'static str> {
952        Self::new_kernel_task_with_stack(entry_point, name, priority, Self::DEFAULT_STACK_SIZE)
953    }
954
955    /// Create a new kernel task with a custom kernel stack size.
956    pub fn new_kernel_task_with_stack(
957        entry_point: extern "C" fn() -> !,
958        name: &'static str,
959        priority: TaskPriority,
960        stack_size: usize,
961    ) -> Result<Arc<Self>, &'static str> {
962        crate::serial_println!(
963            "[trace][task] new_kernel_task_with_stack begin name={} stack_size={}",
964            name,
965            stack_size
966        );
967        // Allocate a real kernel stack
968        let kernel_stack = KernelStack::allocate(stack_size)?;
969        crate::serial_println!("[trace][task] new_kernel_task_with_stack kstack done");
970
971        // Create CPU context with the allocated stack
972        let context = CpuContext::new(entry_point as *const () as u64, &kernel_stack);
973        crate::serial_println!("[trace][task] new_kernel_task_with_stack context done");
974        let id = TaskId::new();
975        let (pid, tid, tgid) = Self::allocate_process_ids();
976        crate::serial_println!(
977            "[trace][task] new_kernel_task_with_stack ids done id={} pid={} tid={} tgid={}",
978            id.as_u64(),
979            pid,
980            tid,
981            tgid
982        );
983        let fpu_state = ExtendedState::new();
984        let xcr0_mask = fpu_state.xcr0_mask;
985
986        let process = Arc::new(crate::process::process::Process::new(
987            pid,
988            crate::memory::kernel_address_space().clone(),
989        ));
990        crate::serial_println!("[trace][task] new_kernel_task_with_stack process done");
991
992        log::debug!(
993            "[task][create] name={} id={} pid={} tid={} kstack={:?} kstack_kib={}",
994            name,
995            id.as_u64(),
996            pid,
997            tid,
998            kernel_stack.virt_base,
999            kernel_stack.size / 1024
1000        );
1001
1002        let task = Arc::new(Task {
1003            id,
1004            pid,
1005            tid,
1006            tgid,
1007            pgid: AtomicU32::new(pid),
1008            sid: AtomicU32::new(pid),
1009            uid: AtomicU32::new(0),
1010            euid: AtomicU32::new(0),
1011            gid: AtomicU32::new(0),
1012            egid: AtomicU32::new(0),
1013            state: AtomicU8::new(TaskState::Ready as u8),
1014            priority,
1015            context: SyncUnsafeCell::new(context),
1016            resume_kind: SyncUnsafeCell::new(ResumeKind::RetFrame),
1017            interrupt_rsp: AtomicU64::new(0),
1018            kernel_stack,
1019            user_stack: None,
1020            stack_canary: AtomicU64::new(0),
1021            stack_canary_addr: AtomicU64::new(0),
1022            kernel_stack_user: SyncUnsafeCell::new(None),
1023            name,
1024            process,
1025            pending_signals: super::signal::SignalSet::new(),
1026            blocked_signals: super::signal::SignalSet::new(),
1027            irq_signal_delivery_blocked: AtomicBool::new(false),
1028            signal_stack: SyncUnsafeCell::new(None),
1029            itimers: super::timer::ITimers::new(),
1030            wake_pending: AtomicBool::new(false),
1031            wake_deadline_ns: AtomicU64::new(0),
1032            trampoline_entry: AtomicU64::new(0),
1033            trampoline_stack_top: AtomicU64::new(0),
1034            trampoline_arg0: AtomicU64::new(0),
1035            ticks: AtomicU64::new(0),
1036            sched_policy: SyncUnsafeCell::new(Self::default_sched_policy(priority)),
1037            home_cpu: AtomicUsize::new(usize::MAX),
1038            last_cpu: AtomicUsize::new(usize::MAX),
1039            affinity_mask: AtomicU64::new(0),
1040            vruntime: AtomicU64::new(0),
1041            fair_rq_generation: AtomicU64::new(0),
1042            fair_on_rq: AtomicBool::new(false),
1043            clear_child_tid: AtomicU64::new(0),
1044            robust_list_head: AtomicU64::new(0),
1045            robust_list_len: AtomicUsize::new(0),
1046            user_fs_base: AtomicU64::new(0),
1047            fpu_state: SyncUnsafeCell::new(fpu_state),
1048            xcr0_mask: AtomicU64::new(xcr0_mask),
1049            rt_link: LinkedListLink::new(),
1050            rt_budget_remaining: AtomicU64::new(0),
1051            rt_budget_period_start: AtomicU64::new(0),
1052            rt_degraded: AtomicBool::new(false),
1053            fair_wait_ticks: AtomicU64::new(0),
1054        });
1055        task.seed_kernel_interrupt_frame_from_context();
1056        Ok(task)
1057    }
1058
1059    /// Create a new user task with its own address space (stub for future use).
1060    ///
1061    /// The entry point and user stack must already be mapped in the given address space.
1062    pub fn new_user_task(
1063        entry_point: u64,
1064        address_space: Arc<AddressSpace>,
1065        name: &'static str,
1066        priority: TaskPriority,
1067    ) -> Result<Arc<Self>, &'static str> {
1068        let kernel_stack = KernelStack::allocate(Self::DEFAULT_STACK_SIZE)?;
1069        let context = CpuContext::new(entry_point, &kernel_stack);
1070        let id = TaskId::new();
1071        let (pid, tid, tgid) = Self::allocate_process_ids();
1072        let fpu_state = ExtendedState::new();
1073        let xcr0_mask = fpu_state.xcr0_mask;
1074
1075        log::debug!(
1076            "[task][create] name={} id={} pid={} tid={} user_as_cr3={:#x}",
1077            name,
1078            id.as_u64(),
1079            pid,
1080            tid,
1081            address_space.cr3().as_u64()
1082        );
1083
1084        Ok(Arc::new(Task {
1085            id,
1086            pid,
1087            tid,
1088            tgid,
1089            pgid: AtomicU32::new(pid),
1090            sid: AtomicU32::new(pid),
1091            uid: AtomicU32::new(0),
1092            euid: AtomicU32::new(0),
1093            gid: AtomicU32::new(0),
1094            egid: AtomicU32::new(0),
1095            state: AtomicU8::new(TaskState::Ready as u8),
1096            priority,
1097            context: SyncUnsafeCell::new(context),
1098            resume_kind: SyncUnsafeCell::new(ResumeKind::RetFrame),
1099            interrupt_rsp: AtomicU64::new(0),
1100            kernel_stack,
1101            user_stack: None,
1102            stack_canary: AtomicU64::new(0),
1103            stack_canary_addr: AtomicU64::new(0),
1104            kernel_stack_user: SyncUnsafeCell::new(None),
1105            name,
1106            process: Arc::new(crate::process::process::Process::new(pid, address_space)),
1107            pending_signals: super::signal::SignalSet::new(),
1108            blocked_signals: super::signal::SignalSet::new(),
1109            irq_signal_delivery_blocked: AtomicBool::new(false),
1110            signal_stack: SyncUnsafeCell::new(None),
1111            itimers: super::timer::ITimers::new(),
1112            wake_pending: AtomicBool::new(false),
1113            wake_deadline_ns: AtomicU64::new(0),
1114            trampoline_entry: AtomicU64::new(0),
1115            trampoline_stack_top: AtomicU64::new(0),
1116            trampoline_arg0: AtomicU64::new(0),
1117            ticks: AtomicU64::new(0),
1118            sched_policy: SyncUnsafeCell::new(Self::default_sched_policy(priority)),
1119            home_cpu: AtomicUsize::new(usize::MAX),
1120            last_cpu: AtomicUsize::new(usize::MAX),
1121            affinity_mask: AtomicU64::new(0),
1122            vruntime: AtomicU64::new(0),
1123            fair_rq_generation: AtomicU64::new(0),
1124            fair_on_rq: AtomicBool::new(false),
1125            clear_child_tid: AtomicU64::new(0),
1126            robust_list_head: AtomicU64::new(0),
1127            robust_list_len: AtomicUsize::new(0),
1128            user_fs_base: AtomicU64::new(0),
1129            fpu_state: SyncUnsafeCell::new(fpu_state),
1130            xcr0_mask: AtomicU64::new(xcr0_mask),
1131            rt_link: LinkedListLink::new(),
1132            rt_budget_remaining: AtomicU64::new(0),
1133            rt_budget_period_start: AtomicU64::new(0),
1134            rt_degraded: AtomicBool::new(false),
1135            fair_wait_ticks: AtomicU64::new(0),
1136        }))
1137    }
1138
1139    /// Reset signal handlers during execve.
1140    ///
1141    /// POSIX requires handlers installed by userspace to revert to SIG_DFL on
1142    /// exec, while dispositions already set to SIG_IGN remain ignored.
1143    pub fn reset_signals(&self) {
1144        // SAFETY: We have a valid reference to the task.
1145        unsafe {
1146            let actions = &mut *self.process.signal_actions.get();
1147            for action in actions.iter_mut() {
1148                if !action.is_ignore() {
1149                    *action = super::signal::SigActionData::default();
1150                }
1151            }
1152        }
1153    }
1154
1155    /// Returns true if this is a kernel task (shares the kernel address space).
1156    pub fn is_kernel(&self) -> bool {
1157        self.process.address_space_arc().is_kernel()
1158    }
1159
1160    /// Record the kernel-owned user stack mapping for this task.
1161    ///
1162    /// Called by `/thread/create` between task construction and scheduler
1163    /// registration, so the mapping is visible before the thread can run.
1164    pub fn set_kernel_stack_user(&self, base: u64, size: u64) {
1165        unsafe {
1166            *self.kernel_stack_user.get() = Some((base, size));
1167        }
1168    }
1169
1170    /// Take the kernel-owned user stack mapping (idempotent reclamation hook).
1171    pub fn take_kernel_stack_user(&self) -> Option<(u64, u64)> {
1172        unsafe { (*self.kernel_stack_user.get()).take() }
1173    }
1174
1175    /// Allocate POSIX identifiers for a new process leader.
1176    pub fn allocate_process_ids() -> (Pid, Tid, Pid) {
1177        let pid = next_pid();
1178        let tid = next_tid();
1179        (pid, tid, pid)
1180    }
1181
1182    /// Print the memory layout of Task and Process structs for debugging.
1183    ///
1184    /// Computes field offsets at runtime using addr_of! so the output is
1185    /// accurate regardless of Rust's struct reordering decisions.
1186    /// Call this early in kernel init to validate the crash-site offset analysis.
1187    pub fn debug_print_layout() {
1188        use core::mem;
1189        crate::serial_println!("[layout] === Struct Layout Debug ===");
1190        crate::serial_println!(
1191            "[layout] sizeof(Task)          = {}",
1192            mem::size_of::<Task>()
1193        );
1194        crate::serial_println!(
1195            "[layout] sizeof(ExtendedState) = {}",
1196            mem::size_of::<ExtendedState>()
1197        );
1198        crate::serial_println!(
1199            "[layout] alignof(ExtendedState)= {}",
1200            mem::align_of::<ExtendedState>()
1201        );
1202        crate::serial_println!(
1203            "[layout] sizeof(CpuContext)    = {}",
1204            mem::size_of::<CpuContext>()
1205        );
1206        crate::serial_println!(
1207            "[layout] sizeof(KernelStack)   = {}",
1208            mem::size_of::<KernelStack>()
1209        );
1210        crate::serial_println!(
1211            "[layout] sizeof(Process)       = {}",
1212            mem::size_of::<crate::process::process::Process>()
1213        );
1214        crate::serial_println!(
1215            "[layout] sizeof(FileDescriptorTable) = {}",
1216            mem::size_of::<crate::vfs::fd::FileDescriptorTable>()
1217        );
1218        crate::serial_println!(
1219            "[layout] sizeof(CapabilityTable)     = {}",
1220            mem::size_of::<crate::capability::CapabilityTable>()
1221        );
1222        crate::serial_println!(
1223            "[layout] sizeof(SigActionData)       = {}",
1224            mem::size_of::<crate::process::signal::SigActionData>()
1225        );
1226
1227        // Use heap-allocated MaybeUninit to avoid stack overflow from the ~3 KiB
1228        // ExtendedState embedded in Task. We only take *addresses* (addr_of!),
1229        // never read the uninitialized data itself, so this is sound.
1230        let task_box: alloc::boxed::Box<core::mem::MaybeUninit<Task>> =
1231            alloc::boxed::Box::new_uninit();
1232        // Cast to *const Task : we never read Task data, only compute field addresses.
1233        let task_ptr = task_box.as_ptr() as *const Task;
1234        let base = task_ptr as u64;
1235        // SAFETY: We only take addresses via addr_of!, no uninitialized reads.
1236        unsafe {
1237            let off_id = core::ptr::addr_of!((*task_ptr).id) as u64 - base;
1238            let off_pid = core::ptr::addr_of!((*task_ptr).pid) as u64 - base;
1239            let off_context = core::ptr::addr_of!((*task_ptr).context) as u64 - base;
1240            let off_kstack = core::ptr::addr_of!((*task_ptr).kernel_stack) as u64 - base;
1241            let off_process = core::ptr::addr_of!((*task_ptr).process) as u64 - base;
1242            let off_fpu = core::ptr::addr_of!((*task_ptr).fpu_state) as u64 - base;
1243            let off_xcr0 = core::ptr::addr_of!((*task_ptr).xcr0_mask) as u64 - base;
1244            let off_ticks = core::ptr::addr_of!((*task_ptr).ticks) as u64 - base;
1245            let off_name = core::ptr::addr_of!((*task_ptr).name) as u64 - base;
1246            let off_vruntime = core::ptr::addr_of!((*task_ptr).vruntime) as u64 - base;
1247            crate::serial_println!("[layout] Task field offsets (byte offset from Task data ptr):");
1248            crate::serial_println!("[layout]   id           @ +{:#x}", off_id);
1249            crate::serial_println!("[layout]   pid          @ +{:#x}", off_pid);
1250            crate::serial_println!("[layout]   context      @ +{:#x}", off_context);
1251            crate::serial_println!("[layout]   kernel_stack @ +{:#x}", off_kstack);
1252            crate::serial_println!("[layout]   process      @ +{:#x}", off_process);
1253            crate::serial_println!("[layout]   fpu_state    @ +{:#x}", off_fpu);
1254            crate::serial_println!("[layout]   xcr0_mask    @ +{:#x}", off_xcr0);
1255            crate::serial_println!("[layout]   ticks        @ +{:#x}", off_ticks);
1256            crate::serial_println!("[layout]   name         @ +{:#x}", off_name);
1257            crate::serial_println!("[layout]   vruntime     @ +{:#x}", off_vruntime);
1258        }
1259        // Arc<T> ArcInner overhead: strong(8)+weak(8)+data = data at offset 16.
1260        // So the crash at [ArcInner<Task>+0xbf8] means Task.process is at offset
1261        // 0xbf8 - 16 = 0xbe8 inside Task data. Check against off_process above.
1262        crate::serial_println!(
1263            "[layout] Expected task.process crash offset from Task data: {:#x}",
1264            0xbf8u64.saturating_sub(16)
1265        );
1266
1267        // Process field offsets
1268        let proc_box: alloc::boxed::Box<core::mem::MaybeUninit<crate::process::process::Process>> =
1269            alloc::boxed::Box::new_uninit();
1270        #[allow(unused_variables)]
1271        let proc_ptr = proc_box.as_ptr() as *const crate::process::process::Process;
1272        let proc_base = proc_ptr as u64;
1273        unsafe {
1274            let off_pid = core::ptr::addr_of!((*proc_ptr).pid) as u64 - proc_base;
1275            let off_as = core::ptr::addr_of!((*proc_ptr).address_space) as u64 - proc_base;
1276            let off_fd = core::ptr::addr_of!((*proc_ptr).fd_table) as u64 - proc_base;
1277            let off_caps = core::ptr::addr_of!((*proc_ptr).capabilities) as u64 - proc_base;
1278            let off_sigs = core::ptr::addr_of!((*proc_ptr).signal_actions) as u64 - proc_base;
1279            let off_brk = core::ptr::addr_of!((*proc_ptr).brk) as u64 - proc_base;
1280            crate::serial_println!(
1281                "[layout] Process field offsets (byte offset from Process data ptr):"
1282            );
1283            crate::serial_println!("[layout]   pid            @ +{:#x}", off_pid);
1284            crate::serial_println!("[layout]   address_space  @ +{:#x}", off_as);
1285            crate::serial_println!("[layout]   fd_table       @ +{:#x}", off_fd);
1286            crate::serial_println!("[layout]   capabilities   @ +{:#x}", off_caps);
1287            crate::serial_println!("[layout]   signal_actions @ +{:#x}", off_sigs);
1288            crate::serial_println!("[layout]   brk            @ +{:#x}", off_brk);
1289        }
1290        // The crash reads [ArcInner<Process>+0x830].
1291        // ArcInner<Process>.data is at ArcInner+16, so Process offset is 0x830-16 = 0x820.
1292        crate::serial_println!(
1293            "[layout] Expected process field crash offset from Process data: {:#x}",
1294            0x830u64.saturating_sub(16)
1295        );
1296        crate::serial_println!("[layout] ===========================");
1297    }
1298}
1299
1300/// Context switch dispatcher. Picks the xsave or fxsave path based on host
1301/// capabilities, then performs the full save/swap/restore sequence.
1302///
1303/// # Safety
1304/// Caller must ensure all pointers in `target` are valid and interrupts are disabled.
1305pub(super) unsafe fn do_switch_context(target: &super::scheduler::SwitchTarget) {
1306    // XSAVE/XRSTOR #GP on non-64B-aligned operands; FXSAVE/FXRSTOR need 16B.
1307    // The naked asm below cannot check this : enforce at the call boundary.
1308    debug_assert_eq!(
1309        (target.old_fpu_ptr as usize) % 64,
1310        0,
1311        "old FPU save area misaligned (XSAVE would #GP)"
1312    );
1313    debug_assert_eq!(
1314        (target.new_fpu_ptr as usize) % 64,
1315        0,
1316        "new FPU save area misaligned (XRSTOR would #GP)"
1317    );
1318    if crate::arch::x86_64::cpuid::host_uses_xsave() {
1319        let old_xcr0 = normalized_xcr0(target.old_xcr0);
1320        let new_xcr0 = normalized_xcr0(target.new_xcr0);
1321        switch_context_xsave(
1322            target.old_rsp_ptr,
1323            target.new_rsp_ptr,
1324            target.old_fpu_ptr,
1325            target.new_fpu_ptr,
1326            new_xcr0,
1327            old_xcr0,
1328        );
1329    } else {
1330        switch_context_fxsave(
1331            target.old_rsp_ptr,
1332            target.new_rsp_ptr,
1333            target.old_fpu_ptr,
1334            target.new_fpu_ptr,
1335        );
1336    }
1337}
1338
1339/// First-task restore dispatcher. Like `do_switch_context` but without
1340/// saving old state (there is no previous task).
1341///
1342/// # Safety
1343/// Caller must ensure pointers are valid and interrupts are disabled. Never returns.
1344pub(super) unsafe fn do_restore_first_task(
1345    frame_ptr: *const u64, // Points to the stack frame (r15, r14, r13, r12, rbp, rbx, ret)
1346    fpu_ptr: *const u8,
1347    xcr0: u64,
1348) -> ! {
1349    // Debug: verify frame pointer
1350    crate::serial_force_println!(
1351        "[task] do_restore_first_task frame_ptr={:#x} fpu_ptr={:#x}",
1352        frame_ptr as u64,
1353        fpu_ptr as u64
1354    );
1355
1356    // Verify the stack frame contains expected values
1357    crate::serial_force_println!(
1358        "[task] do_restore_first_task stack frame: r15={:#x} r14={:#x} r13={:#x} r12={:#x} rbp={:#x} rbx={:#x} ret={:#x}",
1359        *frame_ptr.add(0),
1360        *frame_ptr.add(1),
1361        *frame_ptr.add(2),
1362        *frame_ptr.add(3),
1363        *frame_ptr.add(4),
1364        *frame_ptr.add(5),
1365        *frame_ptr.add(6)
1366    );
1367
1368    // Verify canary immediately above the fake frame (frame is 7 words long).
1369    let canary_addr = frame_ptr as u64 + 56;
1370    let canary = *(canary_addr as *const u64);
1371    crate::serial_force_println!(
1372        "[task] do_restore_first_task canary at {:#x} = {:#x} (expected 0xdeadbeefcafebabe)",
1373        canary_addr,
1374        canary
1375    );
1376
1377    if crate::arch::x86_64::cpuid::host_uses_xsave() {
1378        debug_assert_eq!(
1379            (fpu_ptr as usize) % 64,
1380            0,
1381            "first-task FPU save area misaligned (XRSTOR would #GP)"
1382        );
1383        restore_first_task_xsave(frame_ptr, fpu_ptr, normalized_xcr0(xcr0));
1384    } else {
1385        let _ = xcr0;
1386        restore_first_task_fxsave(frame_ptr, fpu_ptr);
1387    }
1388}
1389
1390//  FXSAVE path (legacy, no XSAVE support)
1391
1392/// rdi=old_rsp, rsi=new_rsp, rdx=old_fpu, rcx=new_fpu
1393#[unsafe(naked)]
1394unsafe extern "C" fn switch_context_fxsave(
1395    _old_rsp_ptr: *mut u64,
1396    _new_rsp_ptr: *const u64,
1397    _old_fpu_ptr: *mut u8,
1398    _new_fpu_ptr: *const u8,
1399) {
1400    core::arch::naked_asm!(
1401        "fxsave [rdx]",
1402        "push rbx",
1403        "push rbp",
1404        "push r12",
1405        "push r13",
1406        "push r14",
1407        "push r15",
1408        "mov [rdi], rsp",
1409        "mov rsp, [rsi]",
1410        "pop r15",
1411        "pop r14",
1412        "pop r13",
1413        "pop r12",
1414        "pop rbp",
1415        "pop rbx",
1416        "fxrstor [rcx]",
1417        "ret",
1418    );
1419}
1420
1421/// rdi=frame_ptr, rsi=fpu_ptr
1422#[unsafe(naked)]
1423unsafe extern "C" fn restore_first_task_fxsave(_rsp_ptr: *const u64, _fpu_ptr: *const u8) -> ! {
1424    // Debug: output pointers before restore (will be last serial output)
1425    // We can't use serial_println in naked functions, so this is just a marker
1426    // The actual debug output is in do_restore_first_task
1427    core::arch::naked_asm!(
1428        // `do_restore_first_task` passes the frame address directly.
1429        "mov rsp, rdi",
1430        "pop r15",
1431        "pop r14",
1432        "pop r13",
1433        "pop r12",
1434        "pop rbp",
1435        "pop rbx",
1436        "fxrstor [rsi]",
1437        "ret",
1438    );
1439}
1440
1441//  XSAVE path (with XCR0 switching per-silo)
1442
1443/// rdi=old_rsp, rsi=new_rsp, rdx=old_fpu, rcx=new_fpu, r8=new_xcr0, r9=old_xcr0
1444#[unsafe(naked)]
1445unsafe extern "C" fn switch_context_xsave(
1446    _old_rsp_ptr: *mut u64,
1447    _new_rsp_ptr: *const u64,
1448    _old_fpu_ptr: *mut u8,
1449    _new_fpu_ptr: *const u8,
1450    _new_xcr0: u64,
1451    _old_xcr0: u64,
1452) {
1453    core::arch::naked_asm!(
1454        "mov r10, rdx",
1455        "mov r11, r8",
1456        "test r11, r11",
1457        "jnz 10f",
1458        "mov r11, 3",
1459        "10:",
1460        "test r9, r9",
1461        "jnz 11f",
1462        "mov r9, 3",
1463        "11:",
1464        "mov r8, r9", // Store normalized old_xcr0 in r8 before it is modified
1465        "mov eax, r9d",
1466        "shr r9, 32",
1467        "mov edx, r9d",
1468        "xsave [r10]",
1469        "push rbx",
1470        "push rbp",
1471        "push r12",
1472        "push r13",
1473        "push r14",
1474        "push r15",
1475        "mov [rdi], rsp",
1476        "mov rsp, [rsi]",
1477        "pop r15",
1478        "pop r14",
1479        "pop r13",
1480        "pop r12",
1481        "pop rbp",
1482        "pop rbx",
1483        "cmp r11, r8", // Check if new_xcr0 == old_xcr0
1484        "je 20f",      // Skip xsetbv if they are equal
1485        "push rcx",
1486        "mov ecx, 0",
1487        "mov eax, r11d",
1488        "mov r8, r11",
1489        "shr r8, 32",
1490        "mov edx, r8d",
1491        "xsetbv",
1492        "pop rcx",
1493        "20:",
1494        "mov eax, r11d",
1495        "mov r8, r11",
1496        "shr r8, 32",
1497        "mov edx, r8d",
1498        "xrstor [rcx]",
1499        "ret",
1500    );
1501}
1502
1503/// rdi=frame_ptr, rsi=fpu_ptr, rdx=xcr0
1504#[unsafe(naked)]
1505unsafe extern "C" fn restore_first_task_xsave(
1506    _rsp_ptr: *const u64,
1507    _fpu_ptr: *const u8,
1508    _xcr0: u64,
1509) -> ! {
1510    core::arch::naked_asm!(
1511        // `do_restore_first_task` passes the frame address directly.
1512        "mov rsp, rdi",
1513        "pop r15",
1514        "pop r14",
1515        "pop r13",
1516        "pop r12",
1517        "pop rbp",
1518        "pop rbx",
1519        "mov r8, rdx",
1520        "test r8, r8",
1521        "jnz 10f",
1522        "mov r8, 3",
1523        "10:",
1524        "mov r9, r8",
1525        "push rsi",
1526        "mov ecx, 0",
1527        "mov eax, r8d",
1528        "shr r8, 32",
1529        "mov edx, r8d",
1530        "xsetbv",
1531        "pop rsi",
1532        "mov eax, r9d",
1533        "shr r9, 32",
1534        "mov edx, r9d",
1535        "xrstor [rsi]",
1536        "ret",
1537    );
1538}