Skip to main content

strat9_kernel/arch/x86_64/
smp.rs

1//! SMP (Symmetric Multi-Processing) boot for x86_64.
2//!
3//! Boots Application Processors (APs) using the legacy INIT+SIPI sequence.
4//! Inspired by Redox-OS's approach: a minimal trampoline does the 16=>64 bit
5//! mode switch, then jumps directly to `smp_main` in Rust.
6//!
7//! Data layout after the trampoline code (written by BSP, read by AP):
8//!   offset +0: PML4 physical address (CR3)
9//!   offset +8: kernel stack top virtual address (RSP)
10//!
11//! Synchronization: AP increments `BOOTED_CORES` after finishing per-CPU init.
12//! BSP spins until all expected APs are online.
13
14use core::{
15    arch::global_asm,
16    sync::atomic::{AtomicBool, AtomicUsize, Ordering},
17};
18
19use alloc::{vec, vec::Vec};
20use x86_64::{
21    structures::paging::{Page, PageTableFlags, PhysFrame, Size4KiB},
22    PhysAddr, VirtAddr,
23};
24
25use crate::{
26    acpi::madt,
27    arch::x86_64::{apic, idt, io::io_wait, percpu, timer},
28    memory,
29};
30
31/// Physical address where the SMP trampoline is copied.
32pub const TRAMPOLINE_PHYS_ADDR: u64 = 0x8000;
33
34/// Number of booted cores (starts at 1 for BSP).
35static BOOTED_CORES: AtomicUsize = AtomicUsize::new(1);
36/// Counter for synchronization barriers.
37static SYNC_BARRIER: AtomicUsize = AtomicUsize::new(0);
38/// Target count for the rendezvous barrier (set by BSP before barrier).
39static BARRIER_TARGET: AtomicUsize = AtomicUsize::new(0);
40/// Gate used by BSP to release APs into scheduler/timer start.
41static AP_SCHED_GATE_OPEN: AtomicBool = AtomicBool::new(false);
42/// AP ACK: set to apic_id by each AP at start of smp_main(), BSP waits for it.
43/// Sentinel usize::MAX means "no ACK yet" (avoids collision with APIC ID 0).
44static AP_REACHED_RUST: AtomicUsize = AtomicUsize::new(usize::MAX);
45
46// ---------------------------------------------------------------------------
47// Trampoline: 16-bit => 32-bit => 64-bit mode switch.
48//
49// The AP starts in real mode at the SIPI vector (0x8000).  This stub:
50//   1. Loads a GDT embedded at known physical offsets.
51//   2. Enables protected mode, then PAE + long mode + paging.
52//   3. Loads the kernel PML4 (CR3) and kernel stack (RSP) from the data area.
53//   4. Jumps to smp_main (64-bit Rust code).
54//
55// LAYOUT (physical addresses, copied to 0x8000):
56//   0x8000:  cli ; cld ; ljmp 0, 0x8040         (16-bit)
57//   0x8010:  _gdt_table  (32 bytes: null, code64, data, code32)
58//   0x8030:  _gdt        (GDTR: limit=31, base=0x8010)
59//   0x8040:  real-mode setup (xor ax,ax; lgdt; enter PM)
60//   SMP_PM_ADDR:   32-bit code (PAT check, PAE + NXE + LME + paging)
61//   SMP_LONG_ADDR: 64-bit code (load stack, jump to smp_main)
62//
63// CRITICAL: The GDT table and descriptor MUST occupy these exact offsets.
64// The `lgdt [0x8030]` instruction reads physical 0x8030 which contains the
65// GDTR. Without this embedded GDT data, the AP loads garbage and triple-faults.
66//
67// Data area (at smp_trampoline_end, written by BSP via HHDM):
68//   +0 u64: CR3 (PML4 physical address)
69//   +8 u64: RSP (kernel stack top virtual address)
70// ---------------------------------------------------------------------------
71#[cfg(target_arch = "x86_64")]
72global_asm!(
73    r#"
74.section .text
75.code16
76
77.global smp_trampoline
78.global smp_trampoline_end
79
80.set SMP_VAR_ADDR, 0x8000 + (smp_trampoline_end - smp_trampoline)
81.set SMP_PM_ADDR, 0x8000 + (smp_trampoline_32 - smp_trampoline)
82.set SMP_LONG_ADDR, 0x8000 + (smp_trampoline_64 - smp_trampoline)
83
84smp_trampoline:
85    cli
86    cld
87    # Jump over the GDT data : code continues at 0x8040.
88    ljmp 0, 0x8040
89
90# -------------------------------------------------------------------
91# GDT : must be at physical offset 0x10 so that:
92#   _gdt_table starts at 0x8010, _gdt (GDTR) is at 0x8030.
93# -------------------------------------------------------------------
94.align 16
95_gdt_table:
96    .long 0, 0                       # null  (selector 0)
97    .long 0x0000ffff, 0x00af9b00     # code64, accessed (identity page is RX)
98    .long 0x0000ffff, 0x00cf9300     # data, accessed
99    .long 0x0000ffff, 0x00cf9b00     # code32, accessed
100_gdt:
101    .word _gdt - _gdt_table - 1      # limit = 31 (4 entries × 8 - 1)
102    .long 0x8010                     # base  = 0x8010
103    .long 0, 0                       # padding
104.align 64
105
106# -------------------------------------------------------------------
107# Real-mode setup continues at 0x8040.
108# -------------------------------------------------------------------
109    xor ax, ax
110    mov ds, ax
111    lgdt [0x8030]                    # loads GDTR from 0x8030
112
113    # Enter protected mode
114    mov eax, cr0
115    or eax, 1
116    mov cr0, eax
117    # Far JMP ptr16:16 (EA iw iw). LLVM's Intel-syntax `ljmp` rejects
118    # this forward expression; data directives resolve it after layout.
119    # The trampoline lives entirely in 0x8000..0x8fff, so IP fits in u16.
120    .byte 0xea
121    .word SMP_PM_ADDR
122    .word 24                        # => code32 segment
123
124.align 32
125.code32
126smp_trampoline_32:
127    mov ax, 16
128    mov ds, ax
129    mov ss, ax
130
131    # INIT does not reset PAT. Check the selectors used by BSP's new tables
132    # before enabling paging; stop this AP explicitly if firmware disagrees.
133    mov ecx, 0x277
134    rdmsr
135    and eax, 0xff0000ff
136    cmp eax, 6                       # PAT[0] = WB, PAT[3] = UC
137    je 3f
138    mov al, 0x50                     # E9: P (PAT mismatch)
139    out 0xe9, al
1402:  hlt
141    jmp 2b
1423:
143    mov eax, cr0
144    or eax, 0x40000000              # CD = 1
145    and eax, 0xdfffffff             # NW = 0
146    mov cr0, eax
147    wbinvd
148
149    # Enable PAE + PSE + OSFXSR + OSXMMEXCPT
150    # NOTE: do NOT force SMEP/SMAP (CR4 bits 20/21) here. qemu64 (and many
151    # older hosts) lack them: `mov cr4` then raises #GP and the AP dies
152    # silently (no IDT yet -> triple fault -> "waiting for APs" hang).
153    # Feature-gated bits (OSXSAVE, SMEP, SMAP) are enabled conditionally
154    # in Rust once CPUID detection has run.
155    mov eax, cr4
156    or eax, 0x630
157    mov cr4, eax
158
159    # Enable Long Mode (EFER.LME)
160    mov ecx, 0xc0000080
161    xor edx, edx
162    rdmsr
163    or eax, 0x901                    # NXE + LME + SCE
164    wrmsr
165
166    # Load kernel PML4 from data area
167    mov eax, [SMP_VAR_ADDR]
168    mov cr3, eax
169
170    # Enable paging (activates long mode)
171    mov eax, cr0
172    and eax, 0x9FFFFFF3              # Clear CD/NW/EM/TS
173    or eax, 0x80010002               # PG + WP + MP
174    mov cr0, eax
175
176    # Far JMP ptr16:32 (EA id iw), decoded while CS is still 32-bit.
177    # Keep the destination symbolic when the preceding stub changes size.
178    .byte 0xea
179    .long SMP_LONG_ADDR
180    .word 8                         # => code64 segment
181
182.align 32
183.code64
184smp_trampoline_64:
185    # Load kernel stack pointer
186    mov rsp, [SMP_VAR_ADDR + 8]
187
188    # Clear RFLAGS.IF only, preserve architectural default state.
189    # pushfq reads RFLAGS; AND clears IF (bit 9); popfq restores.
190    pushfq
191    pop rax
192    btr rax, 9             # IF = 0
193    push rax
194    popfq
195
196    # Jump to smp_main (Rust)
197    # NOTE: must use movabs (64-bit absolute), NOT RIP-relative lea.
198    # The trampoline is copied to 0x8000, so RIP-relative would compute
199    # an offset based on the wrong base address (the linker's original
200    # placement in the kernel .text section, not 0x8000).
201    movabs rax, offset smp_main
202    jmp rax
203
204.align 8
205smp_trampoline_end:
206"#
207);
208
209unsafe extern "C" {
210    fn smp_trampoline();
211    fn smp_trampoline_end();
212}
213
214/// Busy-wait for the given number of microseconds (very rough).
215fn udelay(us: u32) {
216    for _ in 0..us {
217        io_wait();
218    }
219}
220
221/// Identity-map the trampoline physical pages so the AP can execute the
222/// trampoline code in real mode / protected mode before paging is enabled.
223fn ensure_identity_mapping(phys_start: u64, length: usize) -> Result<(), &'static str> {
224    let start = phys_start & !0xFFFu64;
225    let end = (phys_start + length as u64 + 0xFFF) & !0xFFFu64;
226    let flags = PageTableFlags::PRESENT;
227
228    let mut addr = start;
229    while addr < end {
230        let virt = VirtAddr::new(addr);
231        if let Some(mapped) = crate::memory::paging::translate(virt) {
232            if mapped.as_u64() != addr {
233                return Err("SMP: trampoline identity map collision");
234            }
235            crate::memory::paging::set_trampoline_execution(addr, true)?;
236        } else {
237            let page = Page::<Size4KiB>::containing_address(virt);
238            let frame = PhysFrame::<Size4KiB>::containing_address(PhysAddr::new(addr));
239            crate::memory::paging::map_page(page, frame, flags)?;
240        }
241        addr += 0x1000;
242    }
243    Ok(())
244}
245
246/// Copy the trampoline to physical address 0x8000 and write the data area.
247///
248/// Data area layout (at smp_trampoline_end):
249///   +0: CR3 (PML4 physical address)
250///   +8: RSP (kernel stack top virtual address)
251///
252/// After writing, performs WBINVD to flush the cache hierarchy to RAM.
253/// This is essential on real hardware: the BSP writes the trampoline via
254/// HHDM (WB cacheable), but the AP boots in real mode where the effective
255/// memory type is determined by MTRRs. If MTRRs mark the region as UC, or
256/// if platform firmware does not guarantee cache coherency, the AP would
257/// read stale data from RAM without this flush.
258fn copy_trampoline(cr3_phys: u64, stack_top_virt: u64) -> Result<(), &'static str> {
259    let tramp_len = (smp_trampoline_end as *const u8 as usize)
260        .saturating_sub(smp_trampoline as *const u8 as usize);
261
262    if tramp_len + 16 > 4096 {
263        return Err("SMP: trampoline exceeds its reserved page");
264    }
265    ensure_identity_mapping(TRAMPOLINE_PHYS_ADDR, tramp_len + 16)?;
266
267    let tramp_virt = memory::phys_to_virt(TRAMPOLINE_PHYS_ADDR) as *mut u8;
268
269    // SAFETY: trampoline destination is mapped and writable in HHDM.
270    unsafe {
271        core::ptr::copy_nonoverlapping(smp_trampoline as *const u8, tramp_virt, tramp_len);
272        let data = tramp_virt.add(tramp_len) as *mut u64;
273        // +0: CR3
274        core::ptr::write_volatile(data, cr3_phys);
275        // +8: RSP (stack top virtual address)
276        core::ptr::write_volatile(data.add(1), stack_top_virt);
277
278        // SFENCE + WBINVD: ensure all stores reach RAM before the AP starts.
279        core::arch::asm!("sfence");
280        core::arch::asm!("wbinvd");
281    }
282    Ok(())
283}
284
285/// Wait for ICR delivery to complete.
286///
287/// In x2APIC mode, the ICR MSR write is synchronous : the CPU blocks until
288/// the IPI is dispatched, so there is nothing to wait for.  In xAPIC mode,
289/// we poll the delivery-pending bit (ICR_LOW bit 12) until it clears.
290fn wait_delivery() {
291    // x2APIC: MSR write is synchronous, no pending bit to poll.
292    if apic::is_x2apic_enabled() {
293        return;
294    }
295
296    const DELIVERY_PENDING: u32 = 1 << 12;
297    for i in 0..1_000_000 {
298        let val = unsafe { apic::read_reg(apic::REG_ICR_LOW) };
299        if val & DELIVERY_PENDING == 0 {
300            return;
301        }
302        if i > 0 && i % 200_000 == 0 {
303            crate::serial_println!(
304                "[smp] wait_delivery: still pending (iter={}, icr={:#x})",
305                i,
306                val,
307            );
308        }
309        core::hint::spin_loop();
310    }
311    let final_val = unsafe { apic::read_reg(apic::REG_ICR_LOW) };
312    crate::serial_println!("[smp] wait_delivery: TIMEOUT, final icr={:#x}", final_val,);
313    log::warn!("SMP: IPI delivery timeout");
314}
315
316/// Send an IPI and wait for delivery.
317fn send_ipi(apic_id: u32, value: u32) {
318    apic::send_ipi_raw(apic_id, value);
319    wait_delivery();
320}
321
322/// Send INIT + SIPI×2 to an AP per Intel SDM Volume 3, Section 10.6.7.1.
323///
324/// The Intel SDM recommends sending SIPI twice (200 µs apart) so that if
325/// the first SIPI is missed, the second one still catches the AP. Some
326/// platforms (Redox-OS) succeed with a single SIPI, but sending two is
327/// more robust and matches Linux/FreeBSD behaviour.
328///
329///   Step 1: INIT level-assert (0xC500) : wait ≥10 ms
330///   Step 2: INIT level-de-assert (0x8500) : wait ≥200 µs
331///   Step 3: SIPI (0x4608, vector=0x08 => trampoline at 0x8000) : 200 µs
332///   Step 4: SIPI again : 200 µs
333///
334/// ICR values (xAPIC format, delivery-mode bit-fields):
335///   INIT:  delivery=INIT(101), level=1(assert), trigger=1(level)
336///   Deassert: delivery=INIT(101), level=0(de-assert), trigger=1(level)
337///   SIPI:  delivery=STARTUP(110), level=1, vector=0x8
338fn send_init_sipi(apic_id: u32) {
339    crate::serial_println!("[smp] send_init_sipi: apic_id={}", apic_id);
340
341    // Step 1: INIT level-assert
342    crate::serial_println!("[smp]   INIT assert -> {}", apic_id);
343    send_ipi(apic_id, 0xC500);
344    udelay(10_000);
345
346    // Step 2: INIT level-de-assert
347    crate::serial_println!("[smp]   INIT de-assert -> {}", apic_id);
348    send_ipi(apic_id, 0x8500);
349    udelay(200);
350
351    // Step 3: SIPI #1
352    crate::serial_println!("[smp]   SIPI (1/2) -> {} (vector=0x8)", apic_id);
353    send_ipi(apic_id, 0x0608);
354    udelay(200);
355
356    // Step 4: SIPI #2  (SDM recommends two SIPIs)
357    crate::serial_println!("[smp]   SIPI (2/2) -> {} (vector=0x8)", apic_id);
358    send_ipi(apic_id, 0x0608);
359    udelay(200);
360
361    crate::serial_println!("[smp]   INIT+SIPI complete for {}", apic_id);
362}
363
364/// Broadcast a halt command to all other CPUs.
365///
366/// Used during panic to stop the system and prevent log corruption.
367pub fn broadcast_panic_halt() {
368    if !apic::is_initialized() {
369        return;
370    }
371    // SMI delivery mode (0b100 << 8), all-excluding-self shorthand (0b11 << 18).
372    // Level bit = 0 (reserved for SMI mode per Intel SDM Vol. 3A Table 10-1).
373    let icr_low = (0b11 << 18) | (0b100 << 8);
374    apic::send_ipi_raw(0, icr_low);
375}
376
377/// Wait at a synchronization barrier until all expected CPUs arrive.
378fn rendezvous_barrier() {
379    let expected = BARRIER_TARGET.load(Ordering::Acquire);
380    SYNC_BARRIER.fetch_add(1, Ordering::AcqRel);
381    while SYNC_BARRIER.load(Ordering::Acquire) < expected {
382        core::hint::spin_loop();
383    }
384    // BSP may have revoked the trampoline before publishing BARRIER_TARGET.
385    // Each AP drops its formerly executable translation before normal work.
386    unsafe {
387        core::arch::asm!(
388            "invlpg [{}]", in(reg) TRAMPOLINE_PHYS_ADDR,
389            options(nostack, preserves_flags),
390        );
391    }
392}
393
394/// Boot Application Processors.
395///
396/// AP kernel stack frames are allocated from the buddy allocator during this
397/// function and intentionally leaked (never freed). The allocation happens
398/// once at boot while only the BSP is online, so there is no lock contention
399/// and no risk of heap exhaustion under concurrent pressure.
400///
401/// If CPU hotplug is added in the future, stack allocation must be moved to
402/// a sleepable context (e.g. `Mutex`-protected) to avoid heap allocation
403/// under any spinlock that could be held during hot-add.
404pub fn init() -> Result<usize, &'static str> {
405    crate::serial_println!("[smp] init: entering SMP initialization");
406    if !apic::is_initialized() {
407        crate::serial_println!("[smp] init: ERROR - APIC not initialized");
408        return Err("APIC not initialized");
409    }
410
411    BOOTED_CORES.store(1, Ordering::Release);
412    SYNC_BARRIER.store(0, Ordering::Release);
413    BARRIER_TARGET.store(0, Ordering::Release);
414
415    let madt_info = madt::parse_madt().ok_or("MADT not available")?;
416    let bsp_apic_id = apic::lapic_id();
417
418    crate::serial_println!(
419        "[smp] init: MADT parsed, {} local APICs, BSP apic_id={}",
420        madt_info.local_apic_count,
421        bsp_apic_id,
422    );
423
424    if madt_info.local_apic_count <= 1 {
425        crate::serial_println!("[smp] init: single CPU system, returning");
426        log::info!("SMP: single CPU system");
427        return Ok(1);
428    }
429
430    let mut max_apic_id: usize = 0;
431    for i in 0..madt_info.local_apic_count {
432        if let Some(ref entry) = madt_info.local_apics[i] {
433            crate::serial_println!(
434                "[smp]   APIC[{}]: id={} proc={} flags={:#x} {}",
435                i,
436                entry.apic_id,
437                entry.processor,
438                entry.flags,
439                if entry.flags & 1 == 0 {
440                    "(DISABLED)"
441                } else {
442                    ""
443                },
444            );
445            max_apic_id = max_apic_id.max(entry.apic_id as usize);
446        }
447    }
448
449    let mut stack_tops: Vec<u64> = vec![0; max_apic_id + 1];
450    let cr3_phys = crate::memory::paging::kernel_l4_phys().as_u64();
451    let mut targets: Vec<u32> = Vec::new();
452    let mut expected: usize = 1;
453
454    for i in 0..madt_info.local_apic_count {
455        let Some(ref entry) = madt_info.local_apics[i] else {
456            continue;
457        };
458
459        let apic_id = entry.apic_id as u32;
460        if apic_id == bsp_apic_id {
461            continue;
462        }
463
464        // Allocate AP kernel stack from the buddy allocator.
465        let stack_size = crate::process::task::Task::DEFAULT_STACK_SIZE;
466        let pages = (stack_size + 4095) / 4096;
467        let order = pages.next_power_of_two().trailing_zeros() as u8;
468        let frame = crate::sync::with_irqs_disabled(|token| {
469            crate::memory::allocate_phys_contiguous(token, order)
470        })
471        .map_err(|_| "SMP: failed to allocate AP stack from buddy")?;
472        let stack_phys = frame.start_address.as_u64();
473        let stack_virt = crate::memory::phys_to_virt(stack_phys);
474
475        // Zero the stack.
476        unsafe { core::ptr::write_bytes(stack_virt as *mut u8, 0, stack_size) };
477
478        // Stack grows downward: top = base_virt + size.
479        let stack_top = stack_virt.saturating_add(stack_size as u64);
480
481        if apic_id as usize >= stack_tops.len() {
482            log::warn!("SMP: APIC id {} out of stack array range", apic_id);
483            continue;
484        }
485
486        stack_tops[apic_id as usize] = stack_top;
487
488        let cpu_index =
489            percpu::register_cpu(apic_id).ok_or("SMP: exceeded MAX_CPUS for per-CPU data")?;
490        percpu::set_kernel_stack_top(cpu_index, stack_top);
491
492        targets.push(apic_id);
493        expected += 1;
494    }
495
496    // Copy trampoline and write data area (CR3 + first AP's stack top).
497    // All APs share the same CR3; each gets its own stack from stack_tops.
498    // We write the first target's stack top; subsequent APs will use their
499    // own stack (set up in smp_main via percpu::kernel_stack_top).
500    let first_stack_top = targets
501        .first()
502        .and_then(|id| stack_tops.get(*id as usize))
503        .copied()
504        .unwrap_or(0);
505    copy_trampoline(cr3_phys, first_stack_top)?;
506    crate::serial_println!(
507        "[smp] init: trampoline at {:#x}, cr3={:#x}, stack={:#x}",
508        TRAMPOLINE_PHYS_ADDR,
509        cr3_phys,
510        first_stack_top,
511    );
512
513    // Send INIT + single SIPI to each AP, one at a time.
514    // Critical: each AP must consume the trampoline RSP before we overwrite
515    // it for the next AP. We wait for AP_REACHED_RUST ack after each SIPI.
516    crate::serial_println!(
517        "[smp] init: sending INIT+SIPI to {} APs (sequential)",
518        targets.len(),
519    );
520    for apic_id in &targets {
521        AP_REACHED_RUST.store(usize::MAX, Ordering::Release);
522
523        // Write this AP's stack pointer into the trampoline data area.
524        if let Some(stack_top) = stack_tops.get(*apic_id as usize) {
525            let tramp_len = (smp_trampoline_end as *const u8 as usize)
526                .saturating_sub(smp_trampoline as *const u8 as usize);
527            let data = memory::phys_to_virt(TRAMPOLINE_PHYS_ADDR) as *mut u64;
528            unsafe {
529                // +8: RSP
530                core::ptr::write_volatile(data.add(tramp_len / 8 + 1), *stack_top);
531            }
532        }
533
534        send_init_sipi(*apic_id);
535
536        // Wait for this AP to reach Rust and consume the stack pointer
537        // before we modify the slot for the next AP.
538        let mut spin: u64 = 0;
539        const ACK_TIMEOUT: u64 = 500_000_000;
540        while AP_REACHED_RUST.load(Ordering::Acquire) != *apic_id as usize && spin < ACK_TIMEOUT {
541            core::hint::spin_loop();
542            spin = spin.saturating_add(1);
543        }
544        if spin >= ACK_TIMEOUT {
545            crate::serial_println!(
546                "[smp] init: WARNING AP {} did not ACK within timeout",
547                apic_id
548            );
549        } else {
550            crate::serial_println!("[smp] init: AP {} ACKed (rust reached)", apic_id);
551        }
552    }
553
554    // Wait for APs to come online (they increment BOOTED_CORES in smp_main).
555    let mut spins: u64 = 0;
556    const MAX_SPINS: u64 = 200_000_000;
557    crate::serial_println!("[smp] init: waiting for APs (expected={})...", expected,);
558    while BOOTED_CORES.load(Ordering::Acquire) < expected && spins < MAX_SPINS {
559        if spins > 0 && spins % 50_000_000 == 0 {
560            crate::serial_println!(
561                "[smp] init: waiting... online={} expected={} spins={}",
562                BOOTED_CORES.load(Ordering::Acquire),
563                expected,
564                spins,
565            );
566        }
567        core::hint::spin_loop();
568        spins = spins.saturating_add(1);
569    }
570    let online = BOOTED_CORES.load(Ordering::Acquire);
571    crate::serial_println!(
572        "[smp] init: AP wait done: online={} expected={} spins={}",
573        online,
574        expected,
575        spins,
576    );
577    if online < expected {
578        crate::serial_println!(
579            "[smp] init: WARNING only {}/{} APs online",
580            online,
581            expected,
582        );
583        log::warn!(
584            "SMP: timeout waiting APs (online={} expected={}), continuing",
585            online,
586            expected
587        );
588    }
589
590    log::info!("SMP: {} cores online (expected {})", online, expected);
591
592    if online == expected {
593        if let Err(error) =
594            crate::memory::paging::set_trampoline_execution(TRAMPOLINE_PHYS_ADDR, false)
595        {
596            // A legacy boot path can provide a huge identity leaf. Never revoke
597            // execution of unrelated addresses when retiring the trampoline.
598            log::warn!("SMP: cannot retire trampoline execution: {}", error);
599        }
600    }
601
602    // Publish barrier target so APs can proceed.
603    BARRIER_TARGET.store(online, Ordering::Release);
604    rendezvous_barrier();
605
606    Ok(online)
607}
608
609/// First Rust function executed on APs after the trampoline.
610///
611/// The trampoline enables paging with the kernel's PML4, sets up the stack,
612/// and jumps here. All virtual addresses are valid at this point.
613#[unsafe(no_mangle)]
614pub extern "C" fn smp_main() -> ! {
615    // Read APIC ID first : needed to find our per-CPU state.
616    // Use raw port output since serial mutex isn't initialized yet.
617    let apic_id: u32;
618    {
619        use core::fmt::Write;
620        let mut port = unsafe { uart_16550::SerialPort::new(0x3F8) };
621        port.init();
622        let _ = port.write_fmt(format_args!("[smp][ap] entered smp_main\n"));
623        // CPUID leaf 1 EBX[31:24] = initial APIC ID.
624        // rbx is LLVM-reserved, so save/restore it manually.
625        let apic_id_raw: u32;
626        unsafe {
627            core::arch::asm!(
628                "push rbx",
629                "cpuid",
630                "mov {val:e}, ebx",
631                "pop rbx",
632                val = out(reg) apic_id_raw,
633                in("eax") 1u32,
634                out("ecx") _,
635                out("edx") _,
636            );
637        }
638        apic_id = apic_id_raw >> 24;
639        let _ = port.write_fmt(format_args!("[smp][ap] cpuid apic_id={}\n", apic_id));
640    }
641
642    // Signal BSP that this AP has reached Rust and consumed the trampoline RSP.
643    // Must happen immediately : before any per-CPU init that might fail.
644    AP_REACHED_RUST.store(apic_id as usize, Ordering::Release);
645
646    let cpu_index = match percpu::cpu_index_by_apic(apic_id) {
647        Some(idx) => idx,
648        None => {
649            {
650                use core::fmt::Write;
651                let mut port = unsafe { uart_16550::SerialPort::new(0x3F8) };
652                let _ = port.write_fmt(format_args!(
653                    "[smp][ap] ERROR: apic_id={} not registered, halting\n",
654                    apic_id
655                ));
656            }
657            loop {
658                core::hint::spin_loop();
659            }
660        }
661    };
662
663    // ── Phase 1: per-CPU invariants (before any interrupt can fire) ──
664    // Order matters: GS base must be set before any percpu access,
665    // GDT before IDT (CS selector must be valid), TSS before any
666    // ring-0 → ring-3 transition.
667
668    crate::arch::x86_64::percpu::init_gs_base(cpu_index);
669
670    crate::arch::x86_64::tss::init_cpu(cpu_index);
671    crate::arch::x86_64::gdt::init_cpu(cpu_index);
672
673    crate::arch::x86_64::syscall::init();
674    crate::arch::x86_64::init_cpu_extensions();
675
676    // Verify XCR0 was programmed correctly on this AP.
677    if crate::arch::x86_64::cpuid::host_uses_xsave() {
678        let expected = crate::arch::x86_64::cpuid::host_default_xcr0_fast();
679        let actual = crate::arch::x86_64::xgetbv(0);
680        if actual != expected {
681            use core::fmt::Write;
682            let mut port = unsafe { uart_16550::SerialPort::new(0x3F8) };
683            let _ = port.write_fmt(format_args!(
684                "[smp][ap] WARNING: XCR0 mismatch! expected={:#x} actual={:#x}\n",
685                expected, actual
686            ));
687        }
688    }
689
690    if let Some(stack_top) = percpu::kernel_stack_top(cpu_index) {
691        crate::arch::x86_64::tss::set_kernel_stack_for(cpu_index, x86_64::VirtAddr::new(stack_top));
692    }
693
694    // ── Phase 2: interrupt infrastructure ──
695    // IDT is loaded LAST after all per-CPU state is established.
696    // This prevents any exception/IPI from firing with incomplete state.
697    idt::load();
698
699    // Re-initialize Local APIC for this core.
700    apic::init_ap();
701
702    // ── Phase 3: signal online ──
703    let _ = percpu::mark_online_by_apic(apic_id);
704    BOOTED_CORES.fetch_add(1, Ordering::Release);
705
706    {
707        use core::fmt::Write;
708        let mut port = unsafe { uart_16550::SerialPort::new(0x3F8) };
709        let _ = port.write_fmt(format_args!(
710            "[smp][ap] cpu_index={} online, waiting for barrier\n",
711            cpu_index
712        ));
713    }
714
715    // AP spins until BSP publishes the barrier target.
716    while BARRIER_TARGET.load(Ordering::Acquire) == 0 {
717        core::hint::spin_loop();
718    }
719    rendezvous_barrier();
720
721    crate::serial_println!("[trace][ap] cpu_index={} entering scheduler", cpu_index);
722
723    // Wait until BSP has finished scheduler initialization.
724    while !AP_SCHED_GATE_OPEN.load(Ordering::Acquire) {
725        core::hint::spin_loop();
726    }
727
728    // Start APIC timer on this CPU.
729    timer::start_apic_timer_cached();
730
731    // Start per-CPU scheduler (never returns).
732    crate::process::scheduler::schedule_on_cpu(cpu_index)
733}
734
735/// Return the number of online CPUs.
736pub fn cpu_count() -> usize {
737    BOOTED_CORES.load(Ordering::Acquire)
738}
739
740/// Allow APs to start their local timer and enter the scheduler.
741pub fn open_ap_scheduler_gate() {
742    AP_SCHED_GATE_OPEN.store(true, Ordering::Release);
743}