Skip to main content

strat9_kernel/arch/x86_64/
idt.rs

1//! Interrupt Descriptor Table (IDT) for Strat9-OS
2//!
3//! Handles CPU exceptions and hardware IRQs.
4//! Inspired by MaestroOS `idt.rs` and Redox-OS kernel.
5
6use super::{pic, tss};
7use core::sync::atomic::{AtomicBool, AtomicU32, Ordering};
8use x86_64::{
9    structures::{
10        gdt::SegmentSelector,
11        idt::{InterruptDescriptorTable, InterruptStackFrame, PageFaultErrorCode},
12    },
13    VirtAddr,
14};
15
16/// Kernel code segment selector for IDT entries.
17///
18/// At IDT init time the kernel runs under the UEFI firmware's GDT (CS=0x38),
19/// but we intentionally use the **kernel's** selector (0x08) so that once
20/// `gdt::init()` loads the real GDT the IDT entries are immediately valid.
21///
22/// Before `gdt::init()`, any exception will triple-fault because CS=0x08
23/// references a non-existent descriptor in the UEFI GDT.  That is acceptable:
24/// there should be no exceptions before the kernel GDT is live (the only
25/// known early #UD from `init_serial` / `uart_16550` is already disabled).
26const KERNEL_CODE_SELECTOR: SegmentSelector = SegmentSelector(0x08);
27
28#[repr(C, packed)]
29struct Idtr {
30    limit: u16,
31    base: u64,
32}
33
34#[derive(Clone, Copy, Debug)]
35pub struct LiveIdtGateInfo {
36    pub vector: u8,
37    pub selector: u16,
38    pub options: u16,
39    pub offset: u64,
40}
41
42/// IRQ interrupt vector numbers (PIC1_OFFSET + IRQ number)
43#[allow(dead_code)]
44pub mod irq {
45    pub const TIMER: u8 = super::pic::PIC1_OFFSET; // IRQ0 = 0x20
46    pub const KEYBOARD: u8 = super::pic::PIC1_OFFSET + 1; // IRQ1 = 0x21
47    pub const CASCADE: u8 = super::pic::PIC1_OFFSET + 2; // IRQ2 = 0x22
48    pub const MOUSE: u8 = super::pic::PIC1_OFFSET + 12; // IRQ12 = 0x2C
49    pub const COM2: u8 = super::pic::PIC1_OFFSET + 3; // IRQ3 = 0x23
50    pub const COM1: u8 = super::pic::PIC1_OFFSET + 4; // IRQ4 = 0x24
51    pub const FLOPPY: u8 = super::pic::PIC1_OFFSET + 6; // IRQ6 = 0x26
52    pub const ATA_PRIMARY: u8 = super::pic::PIC1_OFFSET + 14; // IRQ14 = 0x2E
53    pub const ATA_SECONDARY: u8 = super::pic::PIC1_OFFSET + 15; // IRQ15 = 0x2F
54}
55
56/// RAII guard: swap GS to kernel on entry if we came from Ring 3, and restore
57/// user GS automatically on drop (covers every exit path including early returns).
58///
59/// # Why this is needed
60/// After `swapgs ; iretq` in the Ring-3 trampoline:
61///   - `IA32_GS_BASE`        = 0  (user GS base, inactive)
62///   - `IA32_KERNEL_GS_BASE` = kernel per-CPU pointer
63/// When an interrupt fires from Ring 3, the CPU does NOT automatically call
64/// swapgs.  The first `gs:[0]` access (e.g. in `current_cpu_index`) would
65/// dereference virtual address 0 => page fault => double fault => triple fault.
66///
67/// # Safety
68/// Must be constructed **before** any `gs:[…]` access in the handler.
69/// `InterruptStackFrame::code_segment` is a plain memory read from the
70/// interrupt stack : it does not access GS.
71struct SwapGsGuard {
72    from_ring3: bool,
73}
74
75impl SwapGsGuard {
76    /// Construct.  If `from_ring3` is true, executes `swapgs` immediately to
77    /// restore the kernel per-CPU GS base.
78    #[inline(always)]
79    fn new(from_ring3: bool) -> Self {
80        if from_ring3 {
81            // SAFETY: We are in Ring 0 with interrupts disabled (standard for
82            // interrupt handlers).  GS_BASE currently points at user space (0);
83            // swapgs gives us the kernel per-CPU block via KERNEL_GS_BASE.
84            unsafe { core::arch::asm!("swapgs", options(nostack, preserves_flags)) };
85        }
86        Self { from_ring3 }
87    }
88}
89
90impl Drop for SwapGsGuard {
91    #[inline(always)]
92    fn drop(&mut self) {
93        if self.from_ring3 {
94            // SAFETY: Symmetric to the constructor.  Restores user GS_BASE so
95            // that iretq returns to Ring 3 with the correct GS state.
96            unsafe { core::arch::asm!("swapgs", options(nostack, preserves_flags)) };
97        }
98    }
99}
100
101/// Determine whether `swapgs` is needed at interrupt/exception entry.
102///
103/// In the normal case, `code_segment & 3 == 3` (Ring 3) means we need
104/// swapgs.  However, between `swapgs` and `iretq` in
105/// `elf_ring3_trampoline`, CS is still Ring 0 but `IA32_GS_BASE` is
106/// already the user value (0).  If `iretq` itself faults, the exception
107/// handler sees CS=Ring 0 but GS=user : the simple ring check misses
108/// this.  Reading `IA32_GS_BASE` via `rdmsr` catches both cases.
109///
110/// Cost: ~20-30 cycles for the `rdmsr` : acceptable in exception paths
111/// (not used for high-frequency IRQ handlers where IF=0 prevents
112/// firing in the swapgs=>iretq window).
113#[inline(always)]
114fn needs_swapgs(cs: u16) -> bool {
115    // Fast path: Ring 3 => always need swapgs.
116    if (cs & 3) == 3 {
117        return true;
118    }
119    // Slow path: check if GS_BASE unexpectedly points to user space.
120    // SAFETY: rdmsr is privileged but we are in Ring 0 (exception handler).
121    let gs_base: u64 = unsafe {
122        let lo: u32;
123        let hi: u32;
124        core::arch::asm!(
125            "rdmsr",
126            in("ecx") 0xC000_0101u32,  // IA32_GS_BASE
127            out("eax") lo,
128            out("edx") hi,
129            options(nostack, preserves_flags),
130        );
131        (lo as u64) | ((hi as u64) << 32)
132    };
133    gs_base < 0xFFFF_8000_0000_0000
134}
135
136/// Combined IDT storage + lock in a single struct to prevent memory corruption.
137/// The `static mut InterruptDescriptorTable` initialization was corrupting
138/// the adjacent `AtomicBool` lock.
139#[repr(C)]
140struct IdtStorage {
141    lock: AtomicBool,
142    idt: InterruptDescriptorTable,
143}
144
145static mut IDT_STORAGEWrapper: IdtStorage = IdtStorage {
146    lock: AtomicBool::new(false),
147    idt: InterruptDescriptorTable::new(),
148};
149static USER_PF_TRACE_BUDGET: AtomicU32 = AtomicU32::new(64);
150static RESCHED_IPI_TRACE_BUDGET: AtomicU32 = AtomicU32::new(32);
151
152pub fn live_gate_info(vector: u8) -> Option<LiveIdtGateInfo> {
153    let mut idtr = Idtr { limit: 0, base: 0 };
154    // SAFETY: `sidt` is a privileged register read with no side effect.
155    unsafe {
156        core::arch::asm!(
157            "sidt [{}]",
158            in(reg) &mut idtr,
159            options(nostack, preserves_flags),
160        );
161    }
162
163    let entry_offset = vector as usize * 16;
164    if entry_offset + 16 > idtr.limit as usize + 1 {
165        return None;
166    }
167
168    // SAFETY: The IDTR base/limit were read from the CPU and bounds-checked above.
169    let (low, high) = unsafe {
170        let entry_ptr = (idtr.base + entry_offset as u64) as *const u64;
171        (
172            core::ptr::read_unaligned(entry_ptr),
173            core::ptr::read_unaligned(entry_ptr.add(1)),
174        )
175    };
176
177    let offset = (low & 0xFFFF) | (((low >> 48) & 0xFFFF) << 16) | ((high & 0xFFFF_FFFF) << 32);
178    let selector = ((low >> 16) & 0xFFFF) as u16;
179    let options = ((low >> 32) & 0xFFFF) as u16;
180
181    Some(LiveIdtGateInfo {
182        vector,
183        selector,
184        options,
185        offset,
186    })
187}
188
189/// Decision returned by the raw interrupt trampolines.
190///
191/// Phase 1 of the preemptive scheduler refactor only wires the raw timer/IPI
192/// stubs and returns `next_rsp = 0`, which means "restore the current
193/// interrupt frame and return with iretq". The future interrupt-aware scheduler
194/// path will return a non-zero `next_rsp` and matching FPU buffers.
195#[repr(C)]
196#[derive(Clone, Copy, Debug, Default)]
197pub struct InterruptReturnDecision {
198    pub next_rsp: u64,
199    pub old_fpu: *mut u8,
200    pub new_fpu: *const u8,
201}
202
203const _: () = {
204    assert!(core::mem::size_of::<InterruptReturnDecision>() == 24);
205    assert!(core::mem::align_of::<InterruptReturnDecision>() == 8);
206};
207
208/// Raw Local APIC timer interrupt entry.
209///
210/// Saves registers in exactly the same order as `SyscallFrame`, calls the Rust
211/// inner handler, then restores the interrupted context and returns with
212/// `iretq`. This avoids the `extern "x86-interrupt"` ABI mismatch with the
213/// legacy `ret`-based scheduler switch path.
214#[unsafe(naked)]
215unsafe extern "C" fn lapic_timer_entry() -> ! {
216    core::arch::naked_asm!(
217        "cld",
218        // Hardware IRQ stack frame at entry:
219        //   [rsp+0]  = RIP
220        //   [rsp+8]  = CS
221        //   [rsp+16] = RFLAGS
222        //   [rsp+24] = RSP
223        //   [rsp+32] = SS
224        // If interrupted from Ring 3, restore kernel GS before any percpu use.
225        "test qword ptr [rsp + 8], 0x3",
226        "jz 2f",
227        "swapgs",
228        "2:",
229        // Save GPRs in reverse order so that final RSP points at a SyscallFrame.
230        "push rax",
231        "push rcx",
232        "push rdx",
233        "push rdi",
234        "push rsi",
235        "push r8",
236        "push r9",
237        "push r10",
238        "push r11",
239        "push rbx",
240        "push rbp",
241        "push r12",
242        "push r13",
243        "push r14",
244        "push r15",
245        // SysV large-struct return uses an implicit out-pointer in RDI.
246        // Reserve 32 bytes to keep 16-byte alignment before `call`.
247        "sub rsp, 32",
248        "mov rdi, rsp",
249        "lea rsi, [rsp + 32]",
250        "call {inner}",
251        // Load returned InterruptReturnDecision fields.
252        "mov rax, [rsp + 0]",
253        "mov rdx, [rsp + 8]",
254        "mov rcx, [rsp + 16]",
255        "add rsp, 32",
256        "test rax, rax",
257        "jz 3f",
258        // Context switch path: save old task's FPU, switch stack, restore new task's FPU.
259        "fxsave [rdx]",
260        "mov rsp, rax",
261        "fxrstor [rcx]",
262        "call {switch_finish}",
263        "3:",
264        // No context switch (rax == 0): skip FPU save/restore entirely.
265        // The interrupted task's FPU state remains unchanged.
266        // Restore current SyscallFrame.
267        "pop r15",
268        "pop r14",
269        "pop r13",
270        "pop r12",
271        "pop rbp",
272        "pop rbx",
273        "pop r11",
274        "pop r10",
275        "pop r9",
276        "pop r8",
277        "pop rsi",
278        "pop rdi",
279        "pop rdx",
280        "pop rcx",
281        "pop rax",
282        // Restore user GS iff we are returning to Ring 3.
283        "test qword ptr [rsp + 8], 0x3",
284        "jz 4f",
285        "swapgs",
286        "4:",
287        "iretq",
288        inner = sym lapic_timer_inner,
289        switch_finish = sym crate::process::scheduler::finish_interrupt_switch,
290    );
291}
292
293/// Raw reschedule IPI entry.
294///
295/// Uses the same `SyscallFrame` layout as the timer entry. Phase 1 only marks
296/// a reschedule hint and returns to the interrupted context.
297#[unsafe(naked)]
298unsafe extern "C" fn resched_ipi_entry() -> ! {
299    core::arch::naked_asm!(
300        "cld",
301        "test qword ptr [rsp + 8], 0x3",
302        "jz 2f",
303        "swapgs",
304        "2:",
305        "push rax",
306        "push rcx",
307        "push rdx",
308        "push rdi",
309        "push rsi",
310        "push r8",
311        "push r9",
312        "push r10",
313        "push r11",
314        "push rbx",
315        "push rbp",
316        "push r12",
317        "push r13",
318        "push r14",
319        "push r15",
320        "sub rsp, 32",
321        "mov rdi, rsp",
322        "lea rsi, [rsp + 32]",
323        "call {inner}",
324        "mov rax, [rsp + 0]",
325        "mov rdx, [rsp + 8]",
326        "mov rcx, [rsp + 16]",
327        "add rsp, 32",
328        "test rax, rax",
329        "jz 3f",
330        // Non-fatal diagnostic: next_rsp != 0 in reschedule IPI path.
331        // Phase 1 contract: resched_ipi_hint only, no context switch.
332        "push rax",
333        "mov al, 0x21",   /* '!' : unexpected switch decision */
334        "out 0xe9, al",
335        "pop rax",
336        "jmp 3f",
337        "3:",
338        "pop r15",
339        "pop r14",
340        "pop r13",
341        "pop r12",
342        "pop rbp",
343        "pop rbx",
344        "pop r11",
345        "pop r10",
346        "pop r9",
347        "pop r8",
348        "pop rsi",
349        "pop rdi",
350        "pop rdx",
351        "pop rcx",
352        "pop rax",
353        "test qword ptr [rsp + 8], 0x3",
354        "jz 4f",
355        "swapgs",
356        "4:",
357        "iretq",
358        inner = sym resched_ipi_inner,
359    );
360}
361
362extern "C" fn lapic_timer_inner(
363    frame: &mut crate::syscall::SyscallFrame,
364) -> InterruptReturnDecision {
365    let cpu = crate::arch::x86_64::percpu::current_cpu_index();
366    let ticks = crate::process::scheduler::ticks();
367    // Heartbeat: single byte only. e9_println!/format_args in IRQ can cause issues.
368    let from_ring3 = (frame.iret_cs & 3) == 3;
369    if from_ring3 && (ticks < 5 || ticks % 100 == 0) {
370        unsafe { core::arch::asm!("mov al, 0x48; out 0xe9, al", out("al") _) } // 'H'
371    }
372    crate::process::scheduler::timer_tick();
373    crate::arch::x86_64::speaker::speaker_tick();
374    // N3 watchdog: check for stalled migrations every tick.
375    crate::ipc::n3::n3_watchdog_tick();
376    super::apic::eoi();
377
378    // Deliver pending POSIX signals before returning to Ring 3 via iretq.
379    // On this IRQ-return path we only perform deliveries that are safe from
380    // timer interrupt context; fatal/default actions remain deferred to the
381    // normal syscall-side delivery path, which may kill/switch the current
382    // task and is not yet validated on the raw timer-iret path.
383    if from_ring3 {
384        crate::process::signal::deliver_pending_signal_on_interrupt_return(frame);
385    }
386
387    // Ring-3 preemption: a user task that spins without making syscalls
388    // never consumes a posted hint, so the CPU would spin forever and the
389    // shell would never be scheduled again. Run the full pick/switch from
390    // this raw naked timer stub via maybe_preempt_from_interrupt: it saves
391    // the outgoing task's frame pointer, picks the next task, seeds a
392    // kernel interrupt frame for first-launch tasks, and returns an
393    // iretq-compatible decision that the naked epilogue applies
394    // (fxsave -> rsp pivot -> fxrstor -> finish_interrupt_switch -> iretq).
395    //
396    // Ring-3-origin ticks take this path unconditionally. The interrupted
397    // Ring-3 frame is iretq-restorable by construction (the CPU pushed a
398    // clean IRET frame at entry, and SwapGsGuard restored kernel GS).
399    // Ring-0-origin ticks (interrupted kernel code) keep the hint path:
400    // resuming synthetic kernel frames from here is still not validated
401    // (same-CPL iretq does not restore RSP/SS).
402    if from_ring3 {
403        if let Some(decision) = crate::process::scheduler::maybe_preempt_from_interrupt(cpu, frame)
404        {
405            if decision.next_rsp != 0 {
406                // TEMP DEBUG: dump the resume iret frame (rip/rsp) before switching.
407                unsafe {
408                    let base = decision.next_rsp;
409                    let iret_rip = *(base as *const u64).add(15); // after 15 GPRs
410                    let iret_rsp = *(base as *const u64).add(18); // rip,cs,rflags,rsp
411                    let hex = b"0123456789abcdef";
412                    core::arch::asm!("out 0xe9, al", in("al") b'@', options(nomem, nostack));
413                    core::arch::asm!("out 0xe9, al", in("al") b'R', options(nomem, nostack));
414                    for sh in [28usize, 24, 20, 16, 12, 8, 4, 0] {
415                        let nib = hex[((iret_rip >> sh) & 0xF) as usize];
416                        core::arch::asm!("out 0xe9, al", in("al") nib, options(nomem, nostack));
417                    }
418                    core::arch::asm!("out 0xe9, al", in("al") b'/', options(nomem, nostack));
419                    for sh in [28usize, 24, 20, 16, 12, 8, 4, 0] {
420                        let nib = hex[((iret_rsp >> sh) & 0xF) as usize];
421                        core::arch::asm!("out 0xe9, al", in("al") nib, options(nomem, nostack));
422                    }
423                    core::arch::asm!("out 0xe9, al", in("al") b'\n', options(nomem, nostack));
424                }
425                return decision;
426            }
427        }
428    }
429    crate::process::scheduler::request_force_resched_hint(cpu);
430    InterruptReturnDecision::default()
431}
432
433#[inline(never)]
434extern "C" fn resched_ipi_inner(
435    frame: &mut crate::syscall::SyscallFrame,
436) -> InterruptReturnDecision {
437    let cpu = crate::arch::x86_64::percpu::current_cpu_index();
438    let should_trace = RESCHED_IPI_TRACE_BUDGET
439        .try_update(Ordering::AcqRel, Ordering::Relaxed, |budget| {
440            budget.checked_sub(1)
441        })
442        .is_ok();
443    if should_trace {
444        let rsp0 = crate::arch::x86_64::tss::kernel_stack_for(cpu)
445            .map(|addr| addr.as_u64())
446            .unwrap_or(0);
447        let (slot_rip, slot_cs, slot_rsp, slot_ss) = if rsp0 >= 40 {
448            // SAFETY: rsp0 points at the top of the current CPU's kernel stack.
449            // During a Ring3->Ring0 interrupt, the CPU-saved IRET frame lives at
450            // [rsp0-40 .. rsp0-8]. We only read those 5 u64 words for diagnosis.
451            unsafe {
452                let frame_base = (rsp0 - 40) as *const u64;
453                (
454                    *frame_base.add(0),
455                    *frame_base.add(1),
456                    *frame_base.add(3),
457                    *frame_base.add(4),
458                )
459            }
460        } else {
461            (0, 0, 0, 0)
462        };
463        crate::e9_println!(
464            "[ipi-rsp0] cpu={} rsp0={:#x} slot_rip={:#x} slot_cs={:#x} slot_rsp={:#x} slot_ss={:#x} frame_rip={:#x} frame_cs={:#x} frame_rsp={:#x} frame_ss={:#x}",
465            cpu,
466            rsp0,
467            slot_rip,
468            slot_cs,
469            slot_rsp,
470            slot_ss,
471            frame.iret_rip,
472            frame.iret_cs,
473            frame.iret_rsp,
474            frame.iret_ss,
475        );
476    }
477    super::apic::eoi();
478    crate::process::scheduler::request_force_resched_hint(cpu);
479    InterruptReturnDecision::default()
480}
481
482#[inline]
483fn lock_idt_storage() {
484    // SAFETY: Only the lock field of the static mut is accessed, which is
485    // an AtomicBool : concurrent access is safe by design.
486    while unsafe { &IDT_STORAGEWrapper.lock }
487        .compare_exchange(false, true, Ordering::Acquire, Ordering::Relaxed)
488        .is_err()
489    {
490        core::hint::spin_loop();
491    }
492}
493
494#[inline]
495fn unlock_idt_storage() {
496    // SAFETY: Only the lock field of the static mut is accessed.
497    unsafe { &IDT_STORAGEWrapper.lock }.store(false, Ordering::Release);
498}
499
500pub fn init() {
501    lock_idt_storage();
502    unsafe {
503        let idt = &raw mut IDT_STORAGEWrapper.idt;
504
505        // CPU exceptions
506        crate::e9_println!("IDT BP");
507        (*idt)
508            .breakpoint
509            .set_handler_fn(breakpoint_handler)
510            .set_code_selector(KERNEL_CODE_SELECTOR);
511        crate::e9_println!("IDT PF");
512        (*idt)
513            .page_fault
514            .set_handler_fn(page_fault_handler)
515            .set_code_selector(KERNEL_CODE_SELECTOR);
516        crate::e9_println!("IDT GP");
517        (*idt)
518            .general_protection_fault
519            .set_handler_fn(general_protection_fault_handler)
520            .set_code_selector(KERNEL_CODE_SELECTOR);
521        (*idt)
522            .stack_segment_fault
523            .set_handler_fn(stack_segment_fault_handler)
524            .set_code_selector(KERNEL_CODE_SELECTOR);
525        (*idt)
526            .non_maskable_interrupt
527            .set_handler_fn(non_maskable_interrupt_handler)
528            .set_code_selector(KERNEL_CODE_SELECTOR);
529        (*idt)
530            .invalid_opcode
531            .set_handler_fn(invalid_opcode_handler)
532            .set_code_selector(KERNEL_CODE_SELECTOR);
533        crate::e9_println!("IDT DF");
534        (*idt)
535            .double_fault
536            .set_handler_fn(double_fault_handler)
537            .set_code_selector(KERNEL_CODE_SELECTOR)
538            .set_stack_index(tss::DOUBLE_FAULT_IST_INDEX);
539
540        // Hardware IRQs (PIC remapped to 0x20+)
541        let idt_ref = &mut *idt;
542        idt_ref[irq::TIMER as u8]
543            .set_handler_fn(legacy_timer_handler)
544            .set_code_selector(KERNEL_CODE_SELECTOR);
545        idt_ref[irq::KEYBOARD as u8]
546            .set_handler_fn(keyboard_handler)
547            .set_code_selector(KERNEL_CODE_SELECTOR);
548        idt_ref[irq::MOUSE as u8]
549            .set_handler_fn(mouse_handler)
550            .set_code_selector(KERNEL_CODE_SELECTOR);
551
552        // Spurious interrupt handler at vector 0xFF (APIC spurious vector)
553        idt_ref[0xFF_u8]
554            .set_handler_fn(spurious_handler)
555            .set_code_selector(KERNEL_CODE_SELECTOR);
556
557        // Cross-CPU reschedule IPI (vector 0xE0)
558
559        idt_ref[super::apic::IPI_RESCHED_VECTOR as u8]
560            .set_handler_addr(VirtAddr::from_ptr(resched_ipi_entry as *const ()))
561            .set_code_selector(KERNEL_CODE_SELECTOR);
562
563        // Cross-CPU TLB shootdown IPI (vector 0xF0)
564        idt_ref[super::apic::IPI_TLB_SHOOTDOWN_VECTOR as u8]
565            .set_handler_fn(tlb_shootdown_handler)
566            .set_code_selector(KERNEL_CODE_SELECTOR);
567
568        // N3 MMU migration sync IPI (vector 0xF1) : naked handler
569        // for full sender context restore via iretq.
570        idt_ref[super::apic::IPI_N3_MIGRATE_VECTOR as u8]
571            .set_handler_addr(VirtAddr::from_ptr(
572                crate::ipc::n3::n3_migrate_ipi_entry as *const (),
573            ))
574            .set_code_selector(KERNEL_CODE_SELECTOR);
575
576        crate::e9_println!("IDT pre-load");
577        (*idt).load_unsafe();
578        crate::e9_println!("IDT loaded");
579    }
580    crate::e9_println!("IDT unlock");
581    unlock_idt_storage();
582    crate::e9_println!("IDT unlocked");
583
584    crate::e9_println!("IDT done");
585}
586
587pub fn load() {
588    lock_idt_storage();
589    unsafe {
590        let idt = &raw const IDT_STORAGEWrapper.idt;
591        (*idt).load_unsafe();
592    }
593    unlock_idt_storage();
594}
595
596/// Register the Local APIC timer IRQ vector to use the timer handler.
597pub fn register_lapic_timer_vector(vector: u8) {
598    lock_idt_storage();
599    unsafe {
600        let idt = &raw mut IDT_STORAGEWrapper.idt;
601        (&mut *idt)[vector]
602            .set_handler_addr(VirtAddr::from_ptr(lapic_timer_entry as *const ()))
603            .set_code_selector(KERNEL_CODE_SELECTOR);
604        (*idt).load_unsafe();
605    }
606    unlock_idt_storage();
607}
608
609/// Register the AHCI storage controller IRQ handler.
610///
611/// Called after AHCI initialisation once the PCI interrupt line is known.
612pub fn register_ahci_irq(irq: u8) {
613    let vector = if irq < 16 {
614        super::pic::PIC1_OFFSET + irq
615    } else {
616        irq
617    };
618
619    lock_idt_storage();
620    unsafe {
621        let idt = &raw mut IDT_STORAGEWrapper.idt;
622        (&mut *idt)[vector]
623            .set_handler_fn(ahci_handler)
624            .set_code_selector(KERNEL_CODE_SELECTOR);
625        (*idt).load_unsafe();
626    }
627    unlock_idt_storage();
628    log::info!("AHCI IRQ {} registered on vector {:#x}", irq, vector);
629}
630
631/// Register the NVMe storage controller IRQ handler for a specific vector.
632///
633/// Used when MSI/MSI-X is active : the vector comes directly from
634/// `msi::probe_and_enable()` instead of being derived from the IRQ line.
635pub fn register_nvme_irq_vector(vector: u8) {
636    lock_idt_storage();
637    unsafe {
638        let idt = &raw mut IDT_STORAGEWrapper.idt;
639        (&mut *idt)[vector]
640            .set_handler_fn(nvme_handler)
641            .set_code_selector(KERNEL_CODE_SELECTOR);
642        unlock_idt_storage();
643    }
644    log::info!("NVMe IRQ vector {:#x} registered", vector);
645}
646
647/// Register the NVMe storage controller IRQ handler.
648///
649/// Called after NVMe initialisation once the PCI interrupt line is known.
650pub fn register_nvme_irq(irq: u8) {
651    let vector = if irq < 16 {
652        super::pic::PIC1_OFFSET + irq
653    } else {
654        irq
655    };
656
657    lock_idt_storage();
658    unsafe {
659        let idt = &raw mut IDT_STORAGEWrapper.idt;
660        (&mut *idt)[vector]
661            .set_handler_fn(nvme_handler)
662            .set_code_selector(KERNEL_CODE_SELECTOR);
663        (*idt).load_unsafe();
664    }
665    unlock_idt_storage();
666    log::info!("NVMe IRQ {} registered on vector {:#x}", irq, vector);
667}
668
669/// Register the VirtIO block device IRQ handler
670///
671/// Called after VirtIO block device initialization to route the device's
672/// IRQ to the correct handler.
673pub fn register_virtio_block_irq(irq: u8) {
674    // PCI INTx gives an IRQ line number (typically 0..15), while IDT expects
675    // a vector number. Map legacy IRQ lines to the remapped interrupt vectors.
676    let vector = if irq < 16 {
677        super::pic::PIC1_OFFSET + irq
678    } else {
679        irq
680    };
681
682    lock_idt_storage();
683    unsafe {
684        let idt = &raw mut IDT_STORAGEWrapper.idt;
685        (&mut *idt)[vector]
686            .set_handler_fn(virtio_block_handler)
687            .set_code_selector(KERNEL_CODE_SELECTOR);
688        (*idt).load_unsafe();
689    }
690    unlock_idt_storage();
691    log::info!("VirtIO-blk IRQ {} registered on vector {:#x}", irq, vector);
692}
693
694/// Register the xHCI USB controller IRQ handler.
695///
696/// Called after xHCI initialization once the PCI interrupt line is known.
697/// Register the xHCI interrupt handler for a specific vector (MSI/MSI-X).
698pub fn register_xhci_irq_vector(vector: u8) {
699    lock_idt_storage();
700    unsafe {
701        let idt = &raw mut IDT_STORAGEWrapper.idt;
702        (&mut *idt)[vector]
703            .set_handler_fn(xhci_handler)
704            .set_code_selector(KERNEL_CODE_SELECTOR);
705        (*idt).load_unsafe();
706    }
707    unlock_idt_storage();
708    log::info!("xHCI IRQ vector {:#x} registered", vector);
709}
710
711pub fn register_xhci_irq(irq: u8) {
712    let vector = if irq < 16 {
713        super::pic::PIC1_OFFSET + irq
714    } else {
715        irq
716    };
717
718    lock_idt_storage();
719    unsafe {
720        let idt = &raw mut IDT_STORAGEWrapper.idt;
721        (&mut *idt)[vector]
722            .set_handler_fn(xhci_handler)
723            .set_code_selector(KERNEL_CODE_SELECTOR);
724        (*idt).load_unsafe();
725    }
726    unlock_idt_storage();
727    log::info!("xHCI IRQ {} registered on vector {:#x}", irq, vector);
728}
729
730/// Register the NIC IRQ handler.
731///
732/// Called after a NIC driver successfully initialises and has read its
733/// PCI interrupt line.  The handler reads ICR and sends EOI; actual
734/// receive processing is triggered via the driver's `receive()` method
735/// which the network stack calls when `handle_interrupt()` signals
736/// that packets are available.
737pub fn register_nic_irq(irq: u8) {
738    let vector = if irq < 16 {
739        super::pic::PIC1_OFFSET + irq
740    } else {
741        irq
742    };
743
744    lock_idt_storage();
745    unsafe {
746        let idt = &raw mut IDT_STORAGEWrapper.idt;
747        (&mut *idt)[vector]
748            .set_handler_fn(nic_handler)
749            .set_code_selector(KERNEL_CODE_SELECTOR);
750        (*idt).load_unsafe();
751    }
752    unlock_idt_storage();
753    log::info!("NIC IRQ {} registered on vector {:#x}", irq, vector);
754}
755
756// =============================================
757// CPU Exception Handlers
758// =============================================
759
760/// Performs the breakpoint handler operation.
761extern "x86-interrupt" fn breakpoint_handler(stack_frame: InterruptStackFrame) {
762    let _gs = SwapGsGuard::new(needs_swapgs(stack_frame.code_segment.0));
763    log::warn!("EXCEPTION: BREAKPOINT\n{:#?}", stack_frame);
764}
765
766/// Performs the invalid opcode handler operation.
767///
768/// SAFETY: This handler uses ONLY raw asm for diagnostic output (no
769/// `log::error!`, no `e9_println!`, no `Port::write`). The `Port::write`
770/// path from the x86_64 crate can trigger re-entrant `#UD` in exception
771/// context, causing an infinite fault loop. Raw `out 0xe9, al` is safe.
772extern "x86-interrupt" fn invalid_opcode_handler(stack_frame: InterruptStackFrame) {
773    let cs = stack_frame.code_segment.0;
774    let is_user = (cs & 3) == 3;
775    let _gs = SwapGsGuard::new(needs_swapgs(cs));
776    if is_user {
777        if let Some(tid) = crate::process::current_task_id() {
778            crate::silo::handle_user_fault(
779                tid,
780                crate::silo::SiloFaultReason::InvalidOpcode,
781                stack_frame.instruction_pointer.as_u64(),
782                0,
783                stack_frame.instruction_pointer.as_u64(),
784            );
785            return;
786        }
787    }
788    // Emit raw e9 marker so we can confirm the handler fires.
789    crate::e9_mark!(b'#');
790    crate::e9_mark!(b'U');
791    crate::e9_mark!(b'D');
792    // Emit the faulting RIP as raw bytes (little-endian u64) for diagnosis.
793    // Use ONLY inline asm : no Rust function calls, no format_args, no
794    // to_le_bytes() : to avoid identity-mapped function pointer issues.
795    let rip: u64 = stack_frame.instruction_pointer.as_u64();
796    unsafe {
797        core::arch::asm!(
798            "out 0xe9, al",
799            "shr rcx, 8",
800            "mov al, cl",
801            "out 0xe9, al",
802            "shr rcx, 8",
803            "mov al, cl",
804            "out 0xe9, al",
805            "shr rcx, 8",
806            "mov al, cl",
807            "out 0xe9, al",
808            "shr rcx, 8",
809            "mov al, cl",
810            "out 0xe9, al",
811            "shr rcx, 8",
812            "mov al, cl",
813            "out 0xe9, al",
814            "shr rcx, 8",
815            "mov al, cl",
816            "out 0xe9, al",
817            "shr rcx, 8",
818            "mov al, cl",
819            "out 0xe9, al",
820            inout("rcx") rip => _,
821            in("al") rip as u8,
822            options(nostack, nomem),
823        );
824    }
825    // Halt forever.
826    loop {
827        crate::arch::hlt();
828    }
829}
830
831extern "x86-interrupt" fn non_maskable_interrupt_handler(stack_frame: InterruptStackFrame) {
832    // NMI can fire at any point : including the swapgs=>iretq window.
833    // Use rdmsr to safely restore kernel GS if needed.
834    let _gs = SwapGsGuard::new(needs_swapgs(stack_frame.code_segment.0));
835    if crate::boot::panic::panic_in_progress() {
836        crate::arch::x86_64::cli();
837        loop {
838            crate::arch::x86_64::hlt();
839        }
840    }
841    crate::serial_force_println!(
842        "[NMI] rip={:#x} cs={:#x}",
843        stack_frame.instruction_pointer.as_u64(),
844        stack_frame.code_segment.0
845    );
846    crate::arch::x86_64::cli();
847    loop {
848        crate::arch::x86_64::hlt();
849    }
850}
851
852/// Performs the page fault handler operation.
853extern "x86-interrupt" fn page_fault_handler(
854    stack_frame: InterruptStackFrame,
855    error_code: PageFaultErrorCode,
856) {
857    use x86_64::registers::control::Cr2;
858    let cs = stack_frame.code_segment.0;
859    let is_user = (cs & 3) == 3;
860    // SAFETY: must be before any gs:[...] access – GS may point to user memory
861    // if the fault fired from Ring 3 (after swapgs in elf_ring3_trampoline),
862    // OR during the swapgs=>iretq window (CS=Ring0 but GS=user).
863    // needs_swapgs() uses rdmsr to catch both cases.
864    let swapgs_needed = needs_swapgs(cs);
865    let _gs = SwapGsGuard::new(swapgs_needed);
866
867    // Detect the swapgs=>iretq window: CS=Ring0 but GS was user (0).
868    if swapgs_needed && !is_user {
869        let fault_addr = x86_64::registers::control::Cr2::read()
870            .as_ref()
871            .map(|v| v.as_u64())
872            .unwrap_or(0);
873        crate::serial_force_println!(
874            "\x1b[31;1m[pagefault]\x1b[0m SWAPGS-WINDOW: CS={:#x} (Ring0) but GS was user! rip={:#x} addr={:#x} err={:#x}",
875            cs,
876            stack_frame.instruction_pointer.as_u64(),
877            fault_addr,
878            error_code.bits()
879        );
880
881        // Kill the faulting process instead of panicking the kernel.
882        // The page fault is unrecoverable in this window: we cannot return
883        // to Ring 3 because the IRETQ frame or a user page is invalid.
884        // Keep SwapGsGuard alive until AFTER current_task_clone to avoid
885        // a second #PF from current_cpu_index() reading gs:[0] (kernel GS
886        // is still needed for percpu access during the kill path).
887        if let Some(task) = crate::process::current_task_clone() {
888            crate::process::kill_task(task.id);
889        }
890        drop(_gs);
891        crate::process::scheduler::exit_current_task(-11); // SIGSEGV
892    }
893
894    // Get the faulting address
895    let fault_addr = Cr2::read();
896    let fault_vaddr = fault_addr.as_ref().map(|v| v.as_u64()).unwrap_or(0);
897    let rip = stack_frame.instruction_pointer.as_u64();
898    let user_rsp = stack_frame.stack_pointer.as_u64();
899
900    let mut trace_ctx = crate::trace::TraceTaskCtx::empty();
901    if is_user {
902        if let Some(task) = crate::process::current_task_clone() {
903            let as_ref = task.process.address_space_arc();
904            trace_ctx = crate::trace::TraceTaskCtx {
905                task_id: task.id.as_u64(),
906                pid: task.pid,
907                tid: task.tid,
908                cr3: as_ref.cr3().as_u64(),
909            };
910        }
911    }
912
913    let do_pf_trace = if is_user {
914        USER_PF_TRACE_BUDGET
915            .try_update(Ordering::Relaxed, Ordering::Relaxed, |v| {
916                if v > 0 {
917                    Some(v - 1)
918                } else {
919                    None
920                }
921            })
922            .is_ok()
923    } else {
924        true
925    };
926    if do_pf_trace {
927        crate::trace_mem!(
928            crate::trace::category::MEM_PF,
929            crate::trace::TraceKind::MemPageFault,
930            error_code.bits() as u64,
931            trace_ctx,
932            rip,
933            fault_vaddr,
934            user_rsp,
935            0
936        );
937    }
938
939    // Try COW only for write-protection faults on already-present pages.
940    // For not-present faults, demand paging should run first.
941    if error_code.contains(PageFaultErrorCode::PROTECTION_VIOLATION)
942        && error_code.contains(PageFaultErrorCode::CAUSED_BY_WRITE)
943        && is_user
944    {
945        if let Some(task) = crate::process::current_task_clone() {
946            let address_space = task.process.address_space_arc();
947            if let Ok(vaddr) = fault_addr {
948                match crate::syscall::fork::handle_cow_fault(vaddr.as_u64(), &address_space) {
949                    Ok(()) => {
950                        crate::trace_mem!(
951                            crate::trace::category::MEM_COW,
952                            crate::trace::TraceKind::MemCow,
953                            1,
954                            trace_ctx,
955                            rip,
956                            vaddr.as_u64(),
957                            0,
958                            0
959                        );
960                        return;
961                    }
962                    Err(reason) => {
963                        crate::trace_mem!(
964                            crate::trace::category::MEM_COW,
965                            crate::trace::TraceKind::MemCow,
966                            0,
967                            trace_ctx,
968                            rip,
969                            vaddr.as_u64(),
970                            0,
971                            0
972                        );
973                        crate::serial_println!(
974                            "\x1b[31m[pagefault] COW resolve failed\x1b[0m: task={} \x1b[36mpid={}\x1b[0m tid={} \x1b[35maddr={:#x}\x1b[0m \x1b[35mrip={:#x}\x1b[0m err={}",
975                            task.id.as_u64(),
976                            task.pid,
977                            task.tid,
978                            vaddr.as_u64(),
979                            stack_frame.instruction_pointer.as_u64(),
980                            reason
981                        );
982                    }
983                }
984            }
985        }
986    }
987
988    if is_user {
989        if let Some(task) = crate::process::current_task_clone() {
990            let address_space = task.process.address_space_arc();
991            if let Ok(vaddr) = fault_addr {
992                if do_pf_trace {
993                    // FORCE OUTPUT for the first user faults only; lazy demand paging can
994                    // legitimately fault thousands of times during boot and flood serial.
995                    crate::serial_force_println!(
996                        "\x1b[33m[pagefault] USER fault\x1b[0m: tid={} rip={:#x} addr={:#x} err={:#x}",
997                        task.tid,
998                        rip,
999                        vaddr.as_u64(),
1000                        error_code.bits()
1001                    );
1002                    // Mirror to e9 so the first handled faults stay visible in e9_debug.log.
1003                    crate::e9_println!(
1004                        "[PF] tid={} rip={:#x} addr={:#x} err={:#x}",
1005                        task.tid,
1006                        rip,
1007                        vaddr.as_u64(),
1008                        error_code.bits()
1009                    );
1010                }
1011
1012                // Demand paging only resolves missing pages. A present page
1013                // with forbidden access must not take handle_fault's
1014                // already-mapped success path and retry forever. COW was
1015                // attempted above for recoverable write-protection faults.
1016                let resolution = if error_code.contains(PageFaultErrorCode::PROTECTION_VIOLATION) {
1017                    Err("Unresolved user page protection violation")
1018                } else {
1019                    address_space.handle_fault(vaddr.as_u64())
1020                };
1021                match resolution {
1022                    Ok(()) => {
1023                        if do_pf_trace {
1024                            crate::serial_force_println!(
1025                                "\x1b[32m[pagefault] USER fault resolved\x1b[0m: tid={} addr={:#x}",
1026                                task.tid,
1027                                vaddr.as_u64()
1028                            );
1029                        }
1030                        return;
1031                    }
1032                    Err(e) => {
1033                        crate::serial_force_println!(
1034                            "\x1b[31m[pagefault] USER fault resolution FAILED\x1b[0m: tid={} addr={:#x} err={:?}",
1035                            task.tid,
1036                            vaddr.as_u64(),
1037                            e
1038                        );
1039                        crate::e9_println!(
1040                            "[PF-FAIL] tid={} rip={:#x} addr={:#x}",
1041                            task.tid,
1042                            rip,
1043                            vaddr.as_u64()
1044                        );
1045                        dump_user_pf_context(&address_space, rip, user_rsp);
1046                        // Show the actual permissions at the faulting address,
1047                        // including restrictive intermediate page-table entries.
1048                        let (active_cr3, _) = x86_64::registers::control::Cr3::read();
1049                        crate::serial_force_println!(
1050                            "[pagefault] fault mapping: active_cr3={:#x} task_cr3={:#x}",
1051                            active_cr3.start_address().as_u64(),
1052                            address_space.cr3().as_u64()
1053                        );
1054                        dump_page_table_walk(vaddr.as_u64(), active_cr3.start_address().as_u64());
1055                    }
1056                }
1057            }
1058        }
1059    }
1060
1061    if is_user {
1062        if let Some(tid) = crate::process::current_task_id() {
1063            crate::silo::handle_user_fault(
1064                tid,
1065                crate::silo::SiloFaultReason::PageFault,
1066                fault_addr.as_ref().map(|v| v.as_u64()).unwrap_or(0),
1067                error_code.bits() as u64,
1068                stack_frame.instruction_pointer.as_u64(),
1069            );
1070            return;
1071        }
1072    } else {
1073        // FORCE OUTPUT for kernel fault
1074        crate::serial_force_println!(
1075            "\x1b[31;1m[pagefault] KERNEL fault\x1b[0m: rip={:#x} addr={:#x} err={:#x}",
1076            rip,
1077            fault_addr.as_ref().map(|v| v.as_u64()).unwrap_or(0),
1078            error_code.bits()
1079        );
1080    }
1081
1082    // Capture current task (non-blocking, safe from IRQ context) for the diagnostic dump.
1083    let task_snap = crate::process::scheduler::current_task_clone_try();
1084    dump_page_fault_full(&stack_frame, error_code, fault_addr, &task_snap);
1085}
1086
1087// =============================================================================
1088// CRITICAL: Full page fault diagnostic dump
1089//
1090// Invoked for every non-recoverable page fault (kernel or unhandled user).
1091// Designed to be deadlock-safe:
1092//   - Uses serial_println! (direct UART) instead of the log framework, which
1093//     may itself allocate or acquire locks.
1094//   - All memory reads go through translate_via_raw_pt so no unmapped address
1095//     is ever dereferenced.
1096//   - The buddy allocator lock is acquired with try_lock (non-blocking) for
1097//     memory statistics.
1098//   - Uses current_task_clone_try (non-blocking) instead of current_task_clone.
1099// =============================================================================
1100
1101/// Decodes `PageFaultErrorCode` bits into a human-readable string.
1102fn decode_error_code(ec: PageFaultErrorCode) -> &'static str {
1103    let p = ec.contains(PageFaultErrorCode::PROTECTION_VIOLATION);
1104    let w = ec.contains(PageFaultErrorCode::CAUSED_BY_WRITE);
1105    let u = ec.contains(PageFaultErrorCode::USER_MODE);
1106    match (p, w, u) {
1107        (false, false, false) => "kernel read of non-present page",
1108        (false, true, false) => "kernel write to non-present page",
1109        (false, false, true) => "user read of non-present page",
1110        (false, true, true) => "user write to non-present page",
1111        (true, false, false) => "kernel read protection violation",
1112        (true, true, false) => "kernel write protection violation (COW / RO page)",
1113        (true, false, true) => "user read protection violation (NX / supervisor-only)",
1114        (true, true, true) => "user write protection violation (COW / RO page)",
1115    }
1116}
1117
1118/// Formats page table entry flags into a short human-readable byte string.
1119fn format_pte_flags(entry: u64) -> [u8; 32] {
1120    let mut buf = [b' '; 32];
1121    let mut pos = 0usize;
1122    let flags: &[(&str, u64)] = &[
1123        ("P", 1 << 0),
1124        ("RW", 1 << 1),
1125        ("US", 1 << 2),
1126        ("PWT", 1 << 3),
1127        ("PCD", 1 << 4),
1128        ("A", 1 << 5),
1129        ("D", 1 << 6),
1130        ("PS", 1 << 7),
1131        ("G", 1 << 8),
1132        ("NX", 1 << 63),
1133    ];
1134    for &(name, bit) in flags {
1135        if entry & bit != 0 {
1136            for &b in name.as_bytes() {
1137                if pos < buf.len() {
1138                    buf[pos] = b;
1139                    pos += 1;
1140                }
1141            }
1142            if pos < buf.len() {
1143                buf[pos] = b'|';
1144                pos += 1;
1145            }
1146        }
1147    }
1148    if pos > 0 && buf[pos - 1] == b'|' {
1149        buf[pos - 1] = b' ';
1150    }
1151    buf
1152}
1153
1154/// Translates a virtual address to a physical address via a manual 4-level
1155/// page table walk.  Returns `Some(phys)` or `None` if any level is absent.
1156///
1157/// # SAFETY
1158/// Read-only access to page tables through the HHDM mapping.
1159/// All intermediate addresses are derived from table entries : no pointer
1160/// originating from user-controlled data is ever dereferenced.
1161fn translate_via_raw_pt(vaddr: u64, cr3_phys: u64, hhdm: u64) -> Option<u64> {
1162    unsafe {
1163        let l4_ptr = (cr3_phys + hhdm) as *const u64;
1164        let l4e = *l4_ptr.add(((vaddr >> 39) & 0x1FF) as usize);
1165        if l4e & 1 == 0 {
1166            return None;
1167        }
1168
1169        let l3_ptr = ((l4e & 0x000F_FFFF_FFFF_F000) + hhdm) as *const u64;
1170        let l3e = *l3_ptr.add(((vaddr >> 30) & 0x1FF) as usize);
1171        if l3e & 1 == 0 {
1172            return None;
1173        }
1174        if l3e & 0x80 != 0 {
1175            return Some((l3e & 0x000F_FFFF_C000_0000) + (vaddr & 0x3FFF_FFFF));
1176        }
1177
1178        let l2_ptr = ((l3e & 0x000F_FFFF_FFFF_F000) + hhdm) as *const u64;
1179        let l2e = *l2_ptr.add(((vaddr >> 21) & 0x1FF) as usize);
1180        if l2e & 1 == 0 {
1181            return None;
1182        }
1183        if l2e & 0x80 != 0 {
1184            return Some((l2e & 0x000F_FFFF_FFE0_0000) + (vaddr & 0x1F_FFFF));
1185        }
1186
1187        let l1_ptr = ((l2e & 0x000F_FFFF_FFFF_F000) + hhdm) as *const u64;
1188        let l1e = *l1_ptr.add(((vaddr >> 12) & 0x1FF) as usize);
1189        if l1e & 1 == 0 {
1190            return None;
1191        }
1192        Some((l1e & 0x000F_FFFF_FFFF_F000) + (vaddr & 0xFFF))
1193    }
1194}
1195
1196/// Hex + ASCII dump of `count` bytes at virtual address `vaddr`.
1197/// Each page boundary is translated through the raw page tables.
1198fn dump_memory_bytes(vaddr: u64, cr3_phys: u64, count: usize, prefix: &str) {
1199    let hhdm = crate::memory::hhdm_offset();
1200    let mut offset = 0usize;
1201    while offset < count {
1202        let cur_va = vaddr.wrapping_add(offset as u64);
1203        let page_off = (cur_va & 0xFFF) as usize;
1204        let chunk = core::cmp::min(count - offset, 0x1000 - page_off);
1205        let Some(phys) = translate_via_raw_pt(cur_va, cr3_phys, hhdm) else {
1206            crate::serial_println!("{}(page {:#x} not mapped)", prefix, cur_va);
1207            offset += chunk;
1208            continue;
1209        };
1210        // SAFETY: read-only access to a valid physical page through the HHDM mapping.
1211        let src = (phys - (cur_va & 0xFFF) + hhdm) as *const u8;
1212        let mut line_off = 0usize;
1213        while line_off < chunk {
1214            let ll = core::cmp::min(16, chunk - line_off);
1215            let line_va = cur_va.wrapping_add(line_off as u64);
1216            let mut hex = [0u8; 48];
1217            let mut asc = [b'.'; 16];
1218            for i in 0..ll {
1219                let byte = unsafe { *src.add(page_off + line_off + i) };
1220                let hi = byte >> 4;
1221                let lo = byte & 0xF;
1222                hex[i * 3] = if hi < 10 { b'0' + hi } else { b'a' + hi - 10 };
1223                hex[i * 3 + 1] = if lo < 10 { b'0' + lo } else { b'a' + lo - 10 };
1224                hex[i * 3 + 2] = b' ';
1225                if byte >= 0x20 && byte < 0x7F {
1226                    asc[i] = byte;
1227                }
1228            }
1229            for i in ll..16 {
1230                hex[i * 3] = b' ';
1231                hex[i * 3 + 1] = b' ';
1232                hex[i * 3 + 2] = b' ';
1233            }
1234            crate::serial_println!(
1235                "{}{:#018x}: {} |{}|",
1236                prefix,
1237                line_va,
1238                core::str::from_utf8(&hex[..48]).unwrap_or("???"),
1239                core::str::from_utf8(&asc[..ll]).unwrap_or("???")
1240            );
1241            line_off += ll;
1242        }
1243        offset += chunk;
1244    }
1245}
1246
1247/// Detailed page table walk with flag decoding at every level.
1248fn dump_page_table_walk(vaddr: u64, cr3_phys: u64) {
1249    let hhdm = crate::memory::hhdm_offset();
1250    let l4_idx = ((vaddr >> 39) & 0x1FF) as usize;
1251    let l3_idx = ((vaddr >> 30) & 0x1FF) as usize;
1252    let l2_idx = ((vaddr >> 21) & 0x1FF) as usize;
1253    let l1_idx = ((vaddr >> 12) & 0x1FF) as usize;
1254
1255    // SAFETY: read-only access through the HHDM mapping for diagnostic purposes.
1256    unsafe {
1257        let l4_ptr = (cr3_phys + hhdm) as *const u64;
1258        let l4e = *l4_ptr.add(l4_idx);
1259        let f = format_pte_flags(l4e);
1260        crate::serial_println!(
1261            "  PML4[{:>3}] = {:#018x}  phys={:#014x}  [{}]",
1262            l4_idx,
1263            l4e,
1264            l4e & 0x000F_FFFF_FFFF_F000,
1265            core::str::from_utf8(&f).unwrap_or("?").trim()
1266        );
1267        if l4e & 1 == 0 {
1268            crate::serial_println!("  \x1b[1;31m  ==> STOP: PML4 not present !\x1b[0m");
1269            return;
1270        }
1271
1272        let l3_ptr = ((l4e & 0x000F_FFFF_FFFF_F000) + hhdm) as *const u64;
1273        let l3e = *l3_ptr.add(l3_idx);
1274        let f = format_pte_flags(l3e);
1275        crate::serial_println!(
1276            "  PDPT[{:>3}] = {:#018x}  phys={:#014x}  [{}]",
1277            l3_idx,
1278            l3e,
1279            l3e & 0x000F_FFFF_FFFF_F000,
1280            core::str::from_utf8(&f).unwrap_or("?").trim()
1281        );
1282        if l3e & 1 == 0 {
1283            crate::serial_println!("  \x1b[1;31m ==> STOP: PDPT not present !\x1b[0m");
1284            return;
1285        }
1286        if l3e & 0x80 != 0 {
1287            crate::serial_println!(
1288                "  ==> 1 GiB huge page => phys {:#x}",
1289                l3e & 0x000F_FFFF_C000_0000
1290            );
1291            return;
1292        } // 1 GiB
1293
1294        let l2_ptr = ((l3e & 0x000F_FFFF_FFFF_F000) + hhdm) as *const u64;
1295        let l2e = *l2_ptr.add(l2_idx);
1296        let f = format_pte_flags(l2e);
1297        crate::serial_println!(
1298            "  PD  [{:>3}] = {:#018x}  phys={:#014x}  [{}]",
1299            l2_idx,
1300            l2e,
1301            l2e & 0x000F_FFFF_FFFF_F000,
1302            core::str::from_utf8(&f).unwrap_or("?").trim()
1303        );
1304        if l2e & 1 == 0 {
1305            crate::serial_println!("  \x1b[1;31m ==> STOP: PD not present !\x1b[0m");
1306            return;
1307        }
1308        if l2e & 0x80 != 0 {
1309            crate::serial_println!(
1310                "  ==> 2 MiB huge page => phys {:#x}",
1311                l2e & 0x000F_FFFF_FFE0_0000
1312            );
1313            return;
1314        } // 2 MiB
1315
1316        let l1_ptr = ((l2e & 0x000F_FFFF_FFFF_F000) + hhdm) as *const u64;
1317        let l1e = *l1_ptr.add(l1_idx);
1318        let f = format_pte_flags(l1e);
1319        crate::serial_println!(
1320            "  PT  [{:>3}] = {:#018x}  phys={:#014x}  [{}]",
1321            l1_idx,
1322            l1e,
1323            l1e & 0x000F_FFFF_FFFF_F000,
1324            core::str::from_utf8(&f).unwrap_or("?").trim()
1325        );
1326        if l1e & 1 == 0 {
1327            crate::serial_println!("  \x1b[1;31m ==> STOP: PT not present !\x1b[0m");
1328        } else {
1329            crate::serial_println!(
1330                "  \x1b[1;32m ==> PAGE PRESENT\x1b[0m => phys {:#x} (check RW/US/NX flags)",
1331                l1e & 0x000F_FFFF_FFFF_F000
1332            );
1333        }
1334        // Neighbouring PT entries for context
1335        crate::serial_println!("  --- Neighbouring PT entries ---");
1336        let start = if l1_idx >= 2 { l1_idx - 2 } else { 0 };
1337        for i in start..core::cmp::min(l1_idx + 3, 512) {
1338            let e = *l1_ptr.add(i);
1339            if e != 0 {
1340                let f = format_pte_flags(e);
1341                crate::serial_println!(
1342                    "    PT[{:>3}] = {:#018x}  [{}]{}",
1343                    i,
1344                    e,
1345                    core::str::from_utf8(&f).unwrap_or("?").trim(),
1346                    if i == l1_idx { " <<<" } else { "" }
1347                );
1348            }
1349        }
1350    }
1351}
1352
1353/// Dumps VMA regions near the faulting address.
1354fn dump_nearby_vma_regions(as_ref: &crate::memory::AddressSpace, fault_vaddr: u64) {
1355    let page_start = fault_vaddr & !0xFFF;
1356    let probes = [
1357        page_start,
1358        fault_vaddr & !0x1F_FFFF,
1359        fault_vaddr & !0x3FFF_FFFF,
1360        0x0000_0001_0000_0000,
1361        0x0000_0000_0040_0000,
1362        0x0000_7FFF_F000_0000,
1363    ];
1364    let mut found_any = false;
1365    for &p in &probes {
1366        if let Some(vma) = as_ref.region_by_start(p) {
1367            let end = vma.start + (vma.page_count as u64) * vma.page_size.bytes();
1368            let hit = fault_vaddr >= vma.start && fault_vaddr < end;
1369            crate::serial_println!(
1370                "  VMA {:#014x}..{:#014x}  pages={:<5}  type={:?}  flags={:?}  pgsz={:?}{}",
1371                vma.start,
1372                end,
1373                vma.page_count,
1374                vma.vma_type,
1375                vma.flags,
1376                vma.page_size,
1377                if hit {
1378                    "  \x1b[1;32m<<< FAULT\x1b[0m"
1379                } else {
1380                    ""
1381                }
1382            );
1383            found_any = true;
1384        }
1385    }
1386    if as_ref.has_mapping_in_range(page_start, 0x1000) {
1387        crate::serial_println!(
1388            "  Note: fault page {:#x} IS within a tracked mapping range",
1389            page_start
1390        );
1391    } else {
1392        crate::serial_println!(
1393            "  Note: fault page {:#x} is NOT within any tracked mapping range",
1394            page_start
1395        );
1396    }
1397    if !found_any {
1398        crate::serial_println!("  (no VMA regions found at probed addresses)");
1399    }
1400}
1401
1402/// Full diagnostic dump for a non-recoverable page fault.
1403///
1404/// Uses `serial_println!` directly (lock-free UART) to avoid any deadlock
1405/// with the log framework or the heap allocator.
1406fn dump_page_fault_full(
1407    stack_frame: &InterruptStackFrame,
1408    error_code: PageFaultErrorCode,
1409    fault_addr: Result<x86_64::VirtAddr, x86_64::addr::VirtAddrNotValid>,
1410    task: &Option<alloc::sync::Arc<crate::process::task::Task>>,
1411) -> ! {
1412    use x86_64::registers::control::{Cr0, Cr3, Cr4};
1413
1414    let rip = stack_frame.instruction_pointer.as_u64();
1415    let rsp = stack_frame.stack_pointer.as_u64();
1416    let cs = stack_frame.code_segment.0;
1417    let ss = stack_frame.stack_segment.0;
1418    let rflags = stack_frame.cpu_flags.bits();
1419    let fault_vaddr = fault_addr.as_ref().map(|v| v.as_u64()).unwrap_or(0);
1420    let is_user = (cs & 3) == 3;
1421
1422    crate::serial_println!("\x1b[1;31m");
1423    crate::serial_println!(
1424        "****************************************************************************"
1425    );
1426    crate::serial_println!("*                  KERNEL PAGE FAULT EXCEPTiON                     *");
1427    crate::serial_println!(
1428        "********************************************************************\x1b[0m"
1429    );
1430
1431    // --- Error code ---
1432    crate::serial_println!("\x1b[1;33m--- Error code ---\x1b[0m");
1433    crate::serial_println!("  Raw         : {:#06x}", error_code.bits());
1434    crate::serial_println!(
1435        "  Diagnostic  : \x1b[1;31m{}\x1b[0m",
1436        decode_error_code(error_code)
1437    );
1438    crate::serial_println!(
1439        "  PRESENT     : {} | WRITE : {} | USER : {} | RSVD : {} | FETCH : {}",
1440        error_code.contains(PageFaultErrorCode::PROTECTION_VIOLATION) as u8,
1441        error_code.contains(PageFaultErrorCode::CAUSED_BY_WRITE) as u8,
1442        error_code.contains(PageFaultErrorCode::USER_MODE) as u8,
1443        (error_code.bits() >> 3) & 1,
1444        (error_code.bits() >> 4) & 1
1445    );
1446
1447    // --- Faulting context ---
1448    crate::serial_println!("\x1b[1;33m--- Faulting context ---\x1b[0m");
1449    crate::serial_println!("  CR2 (addr)  : \x1b[1;35m{:#018x}\x1b[0m", fault_vaddr);
1450    crate::serial_println!("  RIP         : \x1b[1;36m{:#018x}\x1b[0m", rip);
1451    crate::serial_println!("  RSP         : {:#018x}", rsp);
1452    crate::serial_println!(
1453        "  CS          : {:#06x}  (ring={}{}) | SS : {:#06x}",
1454        cs,
1455        cs & 3,
1456        if is_user { " USER" } else { " KERNEL" },
1457        ss
1458    );
1459
1460    // RFLAGS décodé
1461    let mut rf_str = [0u8; 64];
1462    let mut rfp = 0usize;
1463    for &(name, bit) in &[
1464        ("CF", 1u64),
1465        ("PF", 4),
1466        ("AF", 16),
1467        ("ZF", 64),
1468        ("SF", 128),
1469        ("TF", 256),
1470        ("IF", 512),
1471        ("DF", 1024),
1472        ("OF", 2048),
1473    ] {
1474        if rflags & bit != 0 {
1475            for &b in name.as_bytes() {
1476                if rfp < rf_str.len() {
1477                    rf_str[rfp] = b;
1478                    rfp += 1;
1479                }
1480            }
1481            if rfp < rf_str.len() {
1482                rf_str[rfp] = b' ';
1483                rfp += 1;
1484            }
1485        }
1486    }
1487    crate::serial_println!(
1488        "  RFLAGS      : {:#018x}  [{}]",
1489        rflags,
1490        core::str::from_utf8(&rf_str[..rfp]).unwrap_or("?")
1491    );
1492
1493    // --- Control registers ---
1494    crate::serial_println!("\x1b[1;33m--- Control registers ---\x1b[0m");
1495    let cr0 = Cr0::read_raw();
1496    let (cr3_frame, cr3_flags) = Cr3::read();
1497    let cr3_phys = cr3_frame.start_address().as_u64();
1498    let cr4 = Cr4::read_raw();
1499    let efer: u64 = x86_64::registers::model_specific::Efer::read_raw();
1500    crate::serial_println!("  CR0         : {:#018x}", cr0);
1501    crate::serial_println!(
1502        "  CR3         : {:#018x}  (flags={:#x})",
1503        cr3_phys,
1504        cr3_flags.bits()
1505    );
1506    crate::serial_println!("  CR4         : {:#018x}", cr4);
1507    crate::serial_println!(
1508        "  EFER        : {:#018x}  [{}{}{}]",
1509        efer,
1510        if efer & 1 != 0 { "SCE " } else { "" },
1511        if efer & (1 << 8) != 0 { "LME " } else { "" },
1512        if efer & (1 << 11) != 0 { "NXE" } else { "" }
1513    );
1514
1515    // --- CPU context ---
1516    crate::serial_println!("\x1b[1;33m--- CPU context ---\x1b[0m");
1517    crate::serial_println!("  LAPIC ID    : {}", super::apic::lapic_id());
1518    crate::serial_println!("  Ticks sched : {}", crate::process::scheduler::ticks());
1519    crate::serial_println!("  HHDM offset : {:#x}", crate::memory::hhdm_offset());
1520
1521    // --- Task context ---
1522    crate::serial_println!("\x1b[1;33m--- Task context ---\x1b[0m");
1523    if let Some(ref t) = *task {
1524        crate::serial_println!(
1525            "  ID={} PID={} TID={} TGID={} name=\"{}\" prio={:?} ticks={}",
1526            t.id.as_u64(),
1527            t.pid,
1528            t.tid,
1529            t.tgid,
1530            t.name,
1531            t.priority,
1532            t.ticks.load(core::sync::atomic::Ordering::Relaxed)
1533        );
1534        // SAFETY: Read task CR3 safely using the hardware page-table walker
1535        // (translate_via_raw_pt) to prevent recursive page faults if the
1536        // process's Arc<AddressSpace> is partially initialized or corrupted.
1537        //
1538        // Chain: &t.process => Arc<Process> data ptr (Arc::as_ptr)
1539        //      => (*process).address_space.get() => *mut Arc<AddressSpace>
1540        //      => Arc::as_ptr(arc_as) => *const AddressSpace
1541        //      => (*addr_space).cr3_phys
1542        //
1543        // Each step uses translate_via_raw_pt to verify the pointer is mapped
1544        // before dereferencing, using the hardware CR3 (cr3_phys) which always
1545        // maps the kernel's HHDM region.
1546        let task_cr3: u64 = {
1547            let hhdm = crate::memory::hhdm_offset();
1548            // Step 1: Arc<Process> data (Arc::as_ptr is always valid for a live Arc)
1549            let _proc_ptr: u64 = alloc::sync::Arc::as_ptr(&t.process) as u64;
1550            // Step 2: address_space field in Process = SyncUnsafeCell whose .get()
1551            // returns a raw ptr into the Process data : always valid for a live Process.
1552            // However, reading the Arc<AddressSpace> *value* from that pointer may
1553            // fault if the memory is unmapped, so we use translate_via_raw_pt.
1554            let as_cell_addr: u64 =
1555                unsafe { (*alloc::sync::Arc::as_ptr(&t.process)).address_space.get() as u64 };
1556            // Step 3: read the 8-byte Arc<AddressSpace> inner pointer from as_cell_addr
1557            // via raw page table walk with current hardware CR3.
1558            let as_inner_u64: u64 = match translate_via_raw_pt(as_cell_addr, cr3_phys, hhdm) {
1559                Some(phys) => unsafe { *((phys + hhdm) as *const u64) },
1560                None => 0,
1561            };
1562            if as_inner_u64 == 0 {
1563                0u64
1564            } else {
1565                // as_inner_u64 is the NonNull ptr inside Arc<AddressSpace>
1566                // = pointer to ArcInner<AddressSpace>.
1567                // ArcInner = strong(8) + weak(8) + data(AddressSpace).
1568                // So AddressSpace data is at as_inner_u64 + 16.
1569                let as_data_ptr: u64 = as_inner_u64 + 2 * core::mem::size_of::<usize>() as u64;
1570                // cr3_phys is the first field of AddressSpace (PhysAddr = u64, 8 bytes).
1571                match translate_via_raw_pt(as_data_ptr, cr3_phys, hhdm) {
1572                    Some(phys) => unsafe { *((phys + hhdm) as *const u64) },
1573                    None => 0,
1574                }
1575            }
1576        };
1577        if task_cr3 == 0 {
1578            crate::serial_println!(
1579                "  Task CR3    : <unreadable : null/unmapped Arc<AddressSpace>>"
1580            );
1581        } else {
1582            crate::serial_println!(
1583                "  Task CR3    : {:#018x}{}",
1584                task_cr3,
1585                if task_cr3 != cr3_phys {
1586                    " *** DIFFERS from hardware CR3! ***"
1587                } else {
1588                    " (matches hardware CR3)"
1589                }
1590            );
1591        }
1592    } else {
1593        crate::serial_println!("  (no current task : scheduler idle or unavailable)");
1594    }
1595
1596    // --- Memory statistics ---
1597    crate::serial_println!("\x1b[1;33m--- Memory stats ---\x1b[0m");
1598    if let Some(guard) = crate::memory::get_allocator().try_lock() {
1599        if let Some(ref alloc) = *guard {
1600            let (total, allocated) = alloc.page_totals();
1601            let free = total.saturating_sub(allocated);
1602            crate::serial_println!(
1603                "  Total={} pages ({} MiB)  Alloc={} ({} MiB)  Free={} ({} MiB)",
1604                total,
1605                total * 4 / 1024,
1606                allocated,
1607                allocated * 4 / 1024,
1608                free,
1609                free * 4 / 1024
1610            );
1611            let mut zones =
1612                [crate::memory::buddy::ZoneStats::empty(); crate::memory::zone::ZoneType::COUNT];
1613            let n = alloc.zone_snapshot(&mut zones);
1614            for i in 0..n {
1615                let zone = zones[i];
1616                let zone_ref = alloc.get_zone(i);
1617                crate::serial_println!(
1618                    "    Zone {} ({}): base={:#x} managed={} present={} spanned={} reserved={} alloc={} free={} state={:?} seg={}/{} largest={:?}",
1619                    i,
1620                    match zone.zone_type {
1621                        crate::memory::zone::ZoneType::DMA => "DMA",
1622                        crate::memory::zone::ZoneType::Normal => "Normal",
1623                        crate::memory::zone::ZoneType::HighMem => "High",
1624                    },
1625                    zone.base,
1626                    zone.managed_pages,
1627                    zone.present_pages,
1628                    zone.spanned_pages,
1629                    zone.reserved_pages,
1630                    zone.allocated_pages,
1631                    zone.free_pages,
1632                    zone.pressure(),
1633                    zone.segment_count,
1634                    zone.segment_capacity,
1635                    zone.largest_free_order
1636                );
1637                crate::serial_println!(
1638                    "      reserve={} avail={} holes={} cached[u/m]={}/{} free[u/m]={}/{} pageblocks[u/m]={}/{} total={} order={}",
1639                    zone.reserve_floor_pages(),
1640                    zone.available_after_reserve_pages(),
1641                    zone.hole_pages(),
1642                    zone.cached_unmovable_pages,
1643                    zone.cached_movable_pages,
1644                    zone.unmovable_free_pages,
1645                    zone.movable_free_pages,
1646                    zone.unmovable_pageblocks,
1647                    zone.movable_pageblocks,
1648                    zone.pageblock_count,
1649                    crate::memory::zone::PAGEBLOCK_ORDER
1650                );
1651                crate::serial_println!(
1652                    "      frag/order: o1={}%% o4={}%% o{}={}%%",
1653                    zone_ref.fragmentation_score(1, zone.cached_pages),
1654                    zone_ref.fragmentation_score(4, zone.cached_pages),
1655                    crate::memory::zone::PAGEBLOCK_ORDER,
1656                    zone_ref.fragmentation_score(
1657                        crate::memory::zone::PAGEBLOCK_ORDER as u8,
1658                        zone.cached_pages,
1659                    )
1660                );
1661            }
1662        } else {
1663            crate::serial_println!("  (allocator not initialized)");
1664        }
1665    } else {
1666        crate::serial_println!("  (allocator lock contended : skipping)");
1667    }
1668
1669    let quarantine = crate::memory::buddy::poison_quarantine_pages_snapshot();
1670    let fail_counts = crate::memory::buddy::buddy_alloc_fail_counts_snapshot();
1671    let compaction = crate::memory::buddy::compaction_stats_snapshot();
1672    crate::serial_println!("  Poison quarantine : {} pages", quarantine);
1673
1674    let mut printed_fail = false;
1675    for (order, count) in fail_counts.iter().enumerate() {
1676        if *count == 0 {
1677            continue;
1678        }
1679        printed_fail = true;
1680        crate::serial_println!("  Buddy alloc fail  : order={} count={}", order, count);
1681    }
1682    if !printed_fail {
1683        crate::serial_println!("  Buddy alloc fail  : none");
1684    }
1685
1686    if compaction.attempts == 0 {
1687        crate::serial_println!("  Compaction assist : none");
1688    } else {
1689        crate::serial_println!(
1690            "  Compaction assist : attempts={} success={} last_order={:?} migratetype={:?} zone={:?} pressure={:?}",
1691            compaction.attempts,
1692            compaction.successes,
1693            compaction.last_order,
1694            compaction.last_migratetype,
1695            compaction.last_zone,
1696            compaction.last_pressure
1697        );
1698        crate::serial_println!(
1699            "                      frag={}%% req={} avail={} usable={} cached={} drained={} pageblocks={}/{}",
1700            compaction.last_fragmentation_score,
1701            compaction.last_requested_pages,
1702            compaction.last_available_pages,
1703            compaction.last_usable_pages,
1704            compaction.last_cached_pages,
1705            compaction.last_drained_pages,
1706            compaction.last_matching_pageblocks,
1707            compaction.last_pageblock_count
1708        );
1709    }
1710
1711    // --- Code bytes at RIP ---
1712    crate::serial_println!("\x1b[1;33m--- Code at RIP ({:#x}) ---\x1b[0m", rip);
1713    dump_memory_bytes(rip, cr3_phys, 32, "  ");
1714
1715    // --- Stack dump ---
1716    crate::serial_println!("\x1b[1;33m--- Stack dump (RSP={:#x}) ---\x1b[0m", rsp);
1717    dump_memory_bytes(rsp, cr3_phys, 128, "  ");
1718
1719    // --- Page table walk ---
1720    crate::serial_println!(
1721        "\x1b[1;33m--- Page table walk (CR2={:#x}, CR3={:#x}) ---\x1b[0m",
1722        fault_vaddr,
1723        cr3_phys
1724    );
1725    if fault_addr.is_ok() {
1726        dump_page_table_walk(fault_vaddr, cr3_phys);
1727    } else {
1728        crate::serial_println!("  (CR2 is a non-canonical address: {:#x})", fault_vaddr);
1729    }
1730
1731    // --- VMA regions near fault ---
1732    if let Some(ref t) = *task {
1733        crate::serial_println!("\x1b[1;33m--- VMA regions near fault ---\x1b[0m");
1734        // SAFETY: Use the same safe ptr-chain read strategy as the Task CR3 section above:
1735        // Arc::as_ptr gives a valid *const AddressSpace if the Arc is alive, but the
1736        // Arc<AddressSpace> stored inside the SyncUnsafeCell might be corrupted.
1737        // We validate via translate_via_raw_pt before reading the inner ptr.
1738        let hhdm_vma = crate::memory::hhdm_offset();
1739        let safe_as: Option<*const crate::memory::AddressSpace> = unsafe {
1740            let as_cell_addr: u64 =
1741                (*alloc::sync::Arc::as_ptr(&t.process)).address_space.get() as u64;
1742            match translate_via_raw_pt(as_cell_addr, cr3_phys, hhdm_vma) {
1743                Some(phys) => {
1744                    // Read the Arc<AddressSpace> inner pointer (a NonNull ptr stored at this phys)
1745                    let as_inner_u64 = *((phys + hhdm_vma) as *const u64);
1746                    if as_inner_u64 == 0 {
1747                        None
1748                    } else {
1749                        // ArcInner<AddressSpace>.data at +16
1750                        let as_data_ptr = (as_inner_u64 + 2 * core::mem::size_of::<usize>() as u64)
1751                            as *const crate::memory::AddressSpace;
1752                        // Validate the AddressSpace pointer is mapped before returning it
1753                        if translate_via_raw_pt(as_data_ptr as u64, cr3_phys, hhdm_vma).is_some() {
1754                            Some(as_data_ptr)
1755                        } else {
1756                            None
1757                        }
1758                    }
1759                }
1760                None => None,
1761            }
1762        };
1763        if let Some(as_ptr) = safe_as {
1764            // SAFETY: We verified above that as_ptr is mapped and readable.
1765            let as_ref = unsafe { &*as_ptr };
1766            dump_nearby_vma_regions(as_ref, fault_vaddr);
1767        } else {
1768            crate::serial_println!("  (AddressSpace unreadable : skipping VMA dump)");
1769        }
1770    }
1771
1772    crate::serial_println!("\x1b[1;31m***********************************************************");
1773    crate::serial_println!("*                     END OF PAGE FAULT DuMP                      *");
1774    crate::serial_println!(
1775        "*******************************************************************\x1b[0m"
1776    );
1777
1778    panic!(
1779        "PAGE FAULT: {} at {:#x}, RIP={:#x}, CR3={:#x}, err={:#x}",
1780        decode_error_code(error_code),
1781        fault_vaddr,
1782        rip,
1783        cr3_phys,
1784        error_code.bits()
1785    );
1786}
1787
1788/// Performs the dump user pf context operation.
1789fn dump_user_pf_context(as_ref: &crate::memory::AddressSpace, rip: u64, rsp: u64) {
1790    use x86_64::VirtAddr;
1791
1792    let hhdm = crate::memory::hhdm_offset();
1793
1794    if let Some(phys) = as_ref.translate(VirtAddr::new(rip)) {
1795        let off = (rip & 0xfff) as usize;
1796        let mut bytes = [0u8; 8];
1797        // SAFETY: We read at most 8 bytes from a mapped user instruction page via HHDM.
1798        unsafe {
1799            let src = (phys.as_u64() - (rip & 0xfff) + hhdm + off as u64) as *const u8;
1800            core::ptr::copy_nonoverlapping(src, bytes.as_mut_ptr(), bytes.len());
1801        }
1802        crate::serial_println!(
1803            "[pagefault] ctx: rsp={:#x} rip-bytes={:02x} {:02x} {:02x} {:02x} {:02x} {:02x} {:02x} {:02x}",
1804            rsp,
1805            bytes[0],
1806            bytes[1],
1807            bytes[2],
1808            bytes[3],
1809            bytes[4],
1810            bytes[5],
1811            bytes[6],
1812            bytes[7],
1813        );
1814    } else {
1815        crate::serial_println!("[pagefault] ctx: rsp={:#x} rip page unmapped", rsp);
1816    }
1817
1818    if let Some(phys) = as_ref.translate(VirtAddr::new(rsp)) {
1819        crate::serial_println!(
1820            "[pagefault] stack-top: rsp mapped (phys={:#x})",
1821            phys.as_u64()
1822        );
1823    } else {
1824        crate::serial_println!("[pagefault] stack-top: rsp unmapped");
1825    }
1826}
1827
1828/// Performs the general protection fault handler operation.
1829extern "x86-interrupt" fn general_protection_fault_handler(
1830    stack_frame: InterruptStackFrame,
1831    error_code: u64,
1832) {
1833    let cs = stack_frame.code_segment.0;
1834    let is_user = (cs & 3) == 3;
1835    // Use rdmsr-based check: catches the swapgs=>iretq window where
1836    // CS=Ring0 but GS=user (0).
1837    // Without this, #GP from a bad iretq would escalate to double fault => triple fault.
1838    let swapgs_needed = needs_swapgs(cs);
1839    let _gs = SwapGsGuard::new(swapgs_needed);
1840    // Detect the swapgs=>iretq window case: CS says Ring 0 but GS was user.
1841    if swapgs_needed && !is_user {
1842        crate::serial_force_println!(
1843            "\x1b[31;1m[GPF]\x1b[0m SWAPGS-WINDOW: CS={:#x} (Ring0) but GS was user! rip={:#x} err={:#x} rsp={:#x}",
1844            cs,
1845            stack_frame.instruction_pointer.as_u64(),
1846            error_code,
1847            stack_frame.stack_pointer.as_u64()
1848        );
1849        // Keep SwapGsGuard alive through kill path.
1850        if let Some(task) = crate::process::current_task_clone() {
1851            crate::process::kill_task(task.id);
1852        }
1853        drop(_gs);
1854        crate::process::scheduler::exit_current_task(-11); // SIGSEGV
1855    }
1856    if is_user {
1857        if let Some(tid) = crate::process::current_task_id() {
1858            crate::serial_force_println!(
1859                "\x1b[31;1m[GPF]\x1b[0m USER tid={} rip={:#x} err={:#x}",
1860                tid,
1861                stack_frame.instruction_pointer.as_u64(),
1862                error_code
1863            );
1864            crate::silo::handle_user_fault(
1865                tid,
1866                crate::silo::SiloFaultReason::GeneralProtection,
1867                stack_frame.instruction_pointer.as_u64(),
1868                error_code,
1869                stack_frame.instruction_pointer.as_u64(),
1870            );
1871            return;
1872        }
1873    }
1874    crate::serial_force_println!(
1875        "\x1b[31;1m[GPF]\x1b[0m KERNEL rip={:#x} err={:#x} cs={:#x} rsp={:#x}",
1876        stack_frame.instruction_pointer.as_u64(),
1877        error_code,
1878        stack_frame.code_segment.0,
1879        stack_frame.stack_pointer.as_u64()
1880    );
1881    panic!("General protection fault");
1882}
1883
1884/// Performs the stack segment fault handler operation.
1885extern "x86-interrupt" fn stack_segment_fault_handler(
1886    stack_frame: InterruptStackFrame,
1887    error_code: u64,
1888) {
1889    // Use rdmsr-based check: iretq can trigger #SS if the user SS is bad,
1890    // and at that point GS is already swapped to user.
1891    let _gs = SwapGsGuard::new(needs_swapgs(stack_frame.code_segment.0));
1892    crate::serial_force_println!(
1893        "\x1b[31;1m[STACK_FAULT]\x1b[0m rip={:#x} err={:#x} cs={:#x} rsp={:#x}",
1894        stack_frame.instruction_pointer.as_u64(),
1895        error_code,
1896        stack_frame.code_segment.0,
1897        stack_frame.stack_pointer.as_u64()
1898    );
1899    panic!("Stack segment fault");
1900}
1901
1902/// Performs the double fault handler operation.
1903///
1904/// Uses IST stack so the handler always runs on a known-good stack, even
1905/// when RSP0 is corrupt.  We must still do `swapgs` if the fault originated
1906/// from Ring 3 (or from Ring 0 code that already did `swapgs`, e.g. the
1907/// `iretq` path in `elf_ring3_trampoline`).
1908///
1909/// # Note on divergent handler
1910/// This handler is `-> !`, so `SwapGsGuard::drop` will never run.  That is
1911/// fine because we never return to the interrupted context.
1912extern "x86-interrupt" fn double_fault_handler(
1913    stack_frame: InterruptStackFrame,
1914    error_code: u64,
1915) -> ! {
1916    // Best-effort swapgs: if GS currently points at user space (address 0)
1917    // we need to swap to kernel GS so that any code below that touches
1918    // `gs:[0]` (e.g. via `current_cpu_index`) does not page-fault again.
1919    // We use a raw read of IA32_GS_BASE via rdmsr to decide.
1920    //
1921    // During `elf_ring3_trampoline`, `swapgs` is executed *before* `iretq`.
1922    // If `iretq` itself faults, `code_segment` is still Ring 0 (0x08) but
1923    // GS_BASE is already the user value (0).  The normal `cs & 3 == 3` test
1924    // would miss this case.  Reading the MSR catches it.
1925    unsafe {
1926        let lo: u32;
1927        let hi: u32;
1928        core::arch::asm!(
1929            "rdmsr",
1930            in("ecx") 0xC000_0101u32,  // IA32_GS_BASE
1931            out("eax") lo,
1932            out("edx") hi,
1933            options(nostack, preserves_flags),
1934        );
1935        let gs_base = (lo as u64) | ((hi as u64) << 32);
1936        // If GS_BASE is in the low half (user space) or zero, swap to kernel.
1937        if gs_base < 0xFFFF_8000_0000_0000 {
1938            core::arch::asm!("swapgs", options(nostack, preserves_flags));
1939        }
1940    }
1941    crate::serial_force_println!(
1942        "\x1b[31;1m[DOUBLE_FAULT]\x1b[0m rip={:#x} err={:#x} cs={:#x} rsp={:#x}",
1943        stack_frame.instruction_pointer.as_u64(),
1944        error_code,
1945        stack_frame.code_segment.0,
1946        stack_frame.stack_pointer.as_u64()
1947    );
1948    panic!(
1949        "EXCEPTION: DOUBLE FAULT (error code: {:#x})\n{:#?}",
1950        error_code, stack_frame
1951    );
1952}
1953
1954// =============================================
1955// Hardware IRQ handlers
1956// =============================================
1957
1958/// Legacy external timer IRQ handler (PIC/IOAPIC IRQ0 path, vector 0x20).
1959///
1960/// When the LAPIC timer is active, we ignore this source to avoid double-ticking.
1961extern "x86-interrupt" fn legacy_timer_handler(stack_frame: InterruptStackFrame) {
1962    // Restore kernel GS if the timer fired while Ring 3 was running.
1963    let _gs = SwapGsGuard::new((stack_frame.code_segment.0 & 3) == 3);
1964    if crate::arch::x86_64::timer::is_apic_timer_active() {
1965        // Ignore legacy timer source once LAPIC timer is running.
1966        if super::apic::is_initialized() {
1967            super::apic::eoi();
1968        } else {
1969            pic::end_of_interrupt(0);
1970        }
1971        return;
1972    }
1973
1974    // NOTE: serial_force_println! (formatted format_args!) can hang this IRQ
1975    // handler (known vtable issue) : a hung timer handler kills all
1976    // preemption. The tick counter itself is the trace: E9 raw pulses only.
1977    let ticks = crate::process::scheduler::ticks();
1978    unsafe {
1979        core::arch::asm!("out 0xe9, al", in("al") b't', options(nomem, nostack));
1980    }
1981
1982    // Increment tick counter
1983    crate::process::scheduler::timer_tick();
1984    // NOTE: avoid complex rendering/allocation work in IRQ context.
1985    // Status bar refresh is currently done from non-IRQ paths.
1986
1987    // Send EOI first so the timer can fire again on the new task
1988    if super::apic::is_initialized() {
1989        super::apic::eoi();
1990    } else {
1991        pic::end_of_interrupt(0);
1992    }
1993
1994    // Mirror the LAPIC timer policy: do not run maybe_preempt() directly
1995    // from a Ring-3-origin timer IRQ. The extern "x86-interrupt" frame
1996    // must unwind via iretq; switching away from it can corrupt the
1997    // interrupt return state.
1998    //
1999    // A posted hint is NOT enough: a Ring-3 task that spins without making
2000    // syscalls never consumes it (the hint is only taken in maybe_preempt,
2001    // called from syscall paths and kernel-side preemption), so nothing
2002    // else on this CPU ever schedules again. Send a SELF resched IPI
2003    // instead: the IPI handler has its own full context-save frame and can
2004    // run the scheduler safely, then iretq back to whatever was running.
2005    let cpl = stack_frame.code_segment.0 & 3;
2006    if cpl == 3 {
2007        crate::arch::x86_64::apic::self_ipi_resched();
2008    } else {
2009        crate::process::scheduler::maybe_preempt();
2010    }
2011}
2012
2013/// Local APIC timer handler (dedicated vector, e.g. 0xD2).
2014extern "x86-interrupt" fn lapic_timer_handler(stack_frame: InterruptStackFrame) {
2015    // Restore kernel GS if the timer fired while Ring 3 was running.
2016    let cs = stack_frame.code_segment.0;
2017    let _gs = SwapGsGuard::new((cs & 3) == 3);
2018    let cpu = crate::arch::x86_64::percpu::current_cpu_index();
2019    let ticks = crate::process::scheduler::ticks();
2020    // Trace first 10 ticks per CPU unconditionally to confirm timer fires
2021    // after Ring-3 entry, then one-per-100 heartbeat to avoid flooding.
2022    unsafe {
2023        // Raw pulse only : formatted prints hang the IRQ handler.
2024        core::arch::asm!("out 0xe9, al", in("al") b'T', options(nomem, nostack));
2025    }
2026
2027    // serial_force_println holds FORCE_LOCK (IRQ-disabled spinlock) while writing
2028    // to the UART. At 115200 baud each byte takes ~87 µs; a 60-char message is
2029    // ~5 ms of IRQs-off time : long enough to miss ticks and corrupt scheduling.
2030    // Keep serial output out of the hot IRQ path; use e9 port (µs-range) instead.
2031    crate::process::scheduler::timer_tick();
2032    unsafe { core::arch::asm!("mov al, '1'; out 0xe9, al", out("al") _) };
2033    super::apic::eoi();
2034    // IMPORTANT:
2035    // Do not run `maybe_preempt()` directly from a Ring-3-origin timer IRQ.
2036    //
2037    // Current scheduler switch path (`do_switch_context` + `ret`) is built for
2038    // task context frames, while this function is an `extern "x86-interrupt"`
2039    // frame that the compiler expects to unwind with iretq.
2040    //
2041    // On first user-mode preemption (CPU1), switching away from this frame can
2042    // corrupt the interrupt return state and trigger #DF/#TF. Instead, mark a
2043    // lock-free resched hint and return through the normal interrupt epilogue.
2044    // The scheduler will consume the hint on a safe path.
2045    if (cs & 3) == 3 {
2046        crate::process::scheduler::request_force_resched_hint(cpu);
2047    } else {
2048        crate::process::scheduler::maybe_preempt();
2049    }
2050}
2051
2052/// PS/2 Mouse IRQ12 handler.
2053extern "x86-interrupt" fn mouse_handler(_stack_frame: InterruptStackFrame) {
2054    // TEMP DEBUG: mouse IRQ pulse on E9.
2055    unsafe {
2056        core::arch::asm!("out 0xe9, al", in("al") b'M', options(nomem, nostack));
2057        core::arch::asm!("out 0xe9, al", in("al") b'\n', options(nomem, nostack));
2058    }
2059    crate::arch::x86_64::mouse::handle_irq();
2060    // PS/2 mouse IRQ12 is intentionally kept on the remapped legacy PIC path.
2061    // Even when LAPIC/IOAPIC are active for timer/IPI traffic, this source must
2062    // still be acknowledged via the 8259 PIC.
2063    pic::end_of_interrupt(12);
2064}
2065
2066/// Performs the keyboard handler operation.
2067extern "x86-interrupt" fn keyboard_handler(_stack_frame: InterruptStackFrame) {
2068    let raw = unsafe { super::io::inb(0x60) };
2069    // TEMP DEBUG: echo every scancode byte on the E9 port.
2070    unsafe {
2071        core::arch::asm!("out 0xe9, al", in("al") b'K', options(nomem, nostack));
2072        core::arch::asm!("out 0xe9, al", in("al") raw, options(nomem, nostack));
2073        core::arch::asm!("out 0xe9, al", in("al") b'\n', options(nomem, nostack));
2074    }
2075
2076    // Feed scancode + TSC low bits into the entropy pool.
2077    crate::entropy::add_entropy(1, (raw as u64) ^ super::rdtsc());
2078
2079    if let Some(ch) = super::keyboard_layout::handle_scancode_raw(raw) {
2080        crate::arch::x86_64::keyboard::add_to_buffer(ch);
2081    }
2082
2083    // PS/2 keyboard IRQ1 is intentionally kept on the remapped legacy PIC path.
2084    // A LAPIC EOI here leaves the PIC request in service and stalls keyboard
2085    // delivery after the first edge.
2086    pic::end_of_interrupt(1);
2087}
2088
2089/// Spurious interrupt handler (APIC vector 0xFF).
2090/// Per Intel SDM: do NOT send EOI for spurious interrupts.
2091extern "x86-interrupt" fn spurious_handler(_stack_frame: InterruptStackFrame) {
2092    // Intentionally empty : no EOI per Intel SDM
2093}
2094
2095/// AHCI storage controller IRQ handler.
2096///
2097/// Reads `HBA_IS`, processes per-port completions, wakes waiting tasks, then
2098/// sends EOI.  Must not call any function that may block or allocate.
2099extern "x86-interrupt" fn ahci_handler(_stack_frame: InterruptStackFrame) {
2100    // Feed IRQ timing into entropy pool.
2101    crate::entropy::add_entropy(3, super::rdtsc());
2102
2103    crate::hardware::storage::ahci::handle_interrupt();
2104
2105    if super::apic::is_initialized() {
2106        super::apic::eoi();
2107    } else {
2108        let irq = crate::hardware::storage::ahci::AHCI_IRQ_LINE
2109            .load(core::sync::atomic::Ordering::Relaxed);
2110        pic::end_of_interrupt(irq);
2111    }
2112}
2113
2114/// NVMe storage controller IRQ handler.
2115///
2116/// Processes I/O completion queue entries and wakes waiting tasks.
2117extern "x86-interrupt" fn nvme_handler(_stack_frame: InterruptStackFrame) {
2118    crate::entropy::add_entropy(3, super::rdtsc());
2119
2120    crate::hardware::storage::nvme::handle_interrupt();
2121
2122    if super::apic::is_initialized() {
2123        super::apic::eoi();
2124    } else {
2125        let irq = crate::hardware::storage::nvme::NVME_IRQ_LINE
2126            .load(core::sync::atomic::Ordering::Relaxed);
2127        pic::end_of_interrupt(irq);
2128    }
2129}
2130
2131/// VirtIO Block device IRQ handler
2132///
2133/// Handles interrupts from the VirtIO block device.
2134/// The IRQ line is determined at runtime from PCI config.
2135extern "x86-interrupt" fn virtio_block_handler(_stack_frame: InterruptStackFrame) {
2136    // Handle the VirtIO block interrupt
2137    crate::hardware::storage::virtio_block::handle_interrupt();
2138
2139    // Send EOI
2140    if super::apic::is_initialized() {
2141        super::apic::eoi();
2142    } else {
2143        // Get the IRQ number from the device
2144        let irq = crate::hardware::storage::virtio_block::get_irq();
2145        pic::end_of_interrupt(irq);
2146    }
2147}
2148
2149/// xHCI USB controller IRQ handler
2150///
2151/// Handles interrupts from the xHCI host controller.
2152/// Processes event ring completions for control transfers and HID reports.
2153extern "x86-interrupt" fn xhci_handler(_stack_frame: InterruptStackFrame) {
2154    crate::hardware::usb::xhci::handle_interrupt();
2155
2156    if super::apic::is_initialized() {
2157        super::apic::eoi();
2158    } else {
2159        let irq =
2160            crate::hardware::usb::xhci::XHCI_IRQ_LINE.load(core::sync::atomic::Ordering::Relaxed);
2161        pic::end_of_interrupt(irq);
2162    }
2163}
2164
2165/// NIC IRQ handler
2166///
2167/// Dispatches to the registered NIC device's `handle_interrupt()` which
2168/// reads ICR, tracks link state, and reclaims completed TX buffers.
2169/// The network stack is expected to call `receive()` in response (or a
2170/// future NAPI-style poll can be driven from here).
2171extern "x86-interrupt" fn nic_handler(stack_frame: InterruptStackFrame) {
2172    // SAFETY: SwapGsGuard restores kernel GS if we interrupted Ring 3.
2173    // Without this guard, SpinLock::lock => current_cpu_index() reads gs:[0]
2174    // and #PF if GS is still set to the user value (swapgs=>iretq window).
2175    let _gs = SwapGsGuard::new((stack_frame.code_segment.0 & 3) == 3);
2176
2177    crate::hardware::nic::handle_interrupt();
2178
2179    if super::apic::is_initialized() {
2180        super::apic::eoi();
2181    } else {
2182        let irq = crate::hardware::nic::NIC_IRQ_LINE.load(core::sync::atomic::Ordering::Relaxed);
2183        pic::end_of_interrupt(irq);
2184    }
2185}
2186
2187/// Cross-CPU reschedule IPI handler (vector 0xF0).
2188///
2189/// Sent by another CPU (via `apic::send_resched_ipi`) to request that this
2190/// CPU preempts its current task immediately rather than waiting for the next
2191/// timer tick. This is used when a task running on this CPU is killed or
2192/// suspended by a different CPU.
2193///
2194/// EOI is sent ***before*** ` maybe_preempt()` so the APIC can accept further
2195/// IPIs before the potentially long context-switch path runs.
2196extern "x86-interrupt" fn resched_ipi_handler(stack_frame: InterruptStackFrame) {
2197    // Restore kernel GS if the IPI arrived while Ring 3 was running.
2198    let _gs = SwapGsGuard::new((stack_frame.code_segment.0 & 3) == 3);
2199    super::apic::eoi();
2200    crate::process::scheduler::maybe_preempt();
2201}
2202
2203/// Cross-CPU TLB shootdown IPI handler (vector 0xF0).
2204extern "x86-interrupt" fn tlb_shootdown_handler(stack_frame: InterruptStackFrame) {
2205    // Restore kernel GS if the IPI arrived while Ring 3 was running.
2206    let _gs = SwapGsGuard::new((stack_frame.code_segment.0 & 3) == 3);
2207    // Note: EOI is sent by the architecture-independent handler.
2208    super::tlb::tlb_shootdown_ipi_handler();
2209}