Skip to main content

strat9_kernel/ipc/
n3.rs

1//! N3 MMU Thread Migration : PCID-preserving (N3b) and Full Isolation (N3c).
2//!
3//! Implements the transport layer for IPC level N3
4//!
5//! # Architecture
6//!
7//! ```text
8//! sender ──send()──▶ MigrationFrame ──CR3 switch──▶ receiver handler
9//!   ▲                                                    │
10//!   └────────────── reschedule IPI ◀─────────────────────┘
11//! ```
12//!
13//! The sender saves its context, copies the message, then calls the ASM
14//! primitive which switches CR3 and restores the receiver's context.  The
15//! receiver handler reads the message from the frame and returns.  The
16//! sender is resumed via a reschedule IPI.
17//!
18//! # Frame pointer requirement
19//!
20//! The `send()` method captures `sender_rip` from `[rbp + 8]`, which requires
21//! frame pointers to be enabled at compile time. The kernel must be built with
22//! `-C force-frame-pointers=yes` (see workspace/kernel/.cargo/config.toml).
23//!
24//! # Shared mapping note
25//!
26//! The MigrationFrame is mapped in both address spaces at the same virtual
27//! address (`N3_SHARED_FRAME_VA`), but **without USER_ACCESSIBLE**. The frame
28//! lives in kernel space and is accessible only from Ring 0. Both sender and
29//! receiver access it via their kernel page tables : the shared VA means the
30//! same PTE (same physical page) is reachable from both page table hierarchies.
31//! This is NOT a user-mappable region; it is kernel-only shared memory.
32//!
33//! # PCID contract
34//!
35//! A PCID value of 0 means "no PCID" : either PCID is unsupported by the CPU
36//! or exhausted. All code paths check `pcid > 0` before using PCID-specific
37//! features (INVPCID, CR3 PCID bits). The tier selection function
38//! (`select_n3_tier()`) distinguishes theoretical CPU support from actual
39//! operational capability.
40#![allow(dead_code)]
41
42use crate::{
43    arch::apic,
44    memory::{allocate_frame, free_frame, phys_to_virt, AddressSpace},
45    process::{current_task_id, get_task_by_id, Task, TaskId},
46    sync::{with_irqs_disabled, SpinLock},
47};
48use core::{
49    mem::MaybeUninit,
50    sync::atomic::{AtomicU16, AtomicU32, AtomicU64, AtomicU8, Ordering},
51};
52
53use super::transport::{
54    IpcConsumer, IpcError, IpcProducer, IpcTransport, TransportCapabilities, TransportLevel,
55};
56
57// ============================================================================
58// Normative types
59// ============================================================================
60
61/// N3 tier selection
62#[derive(Debug, Clone, Copy, PartialEq, Eq)]
63#[repr(u8)]
64pub enum N3Tier {
65    /// PCID-preserving: TLB entries for the sender AS are kept across CR3 switch.
66    PcidPreserving = 1,
67    /// Full isolation: CR3 write flushes all non-global TLB entries.
68    FullIsolation = 2,
69}
70
71/// Minimal CPU context saved/restored during N3 migration.
72///
73/// Layout is `#[repr(C, align(64))]` to match the ASM primitive offsets.
74/// No FPU/AVX state : handled separately per FPU policy.
75/// Total size: 96 bytes (fields) + 32 bytes (padding) = 128 bytes.
76#[repr(C, align(64))]
77#[derive(Debug, Clone, Copy)]
78pub struct N3MinimalContext {
79    /// Callee-saved GPRs (System V AMD64 ABI).
80    pub r15: u64,
81    pub r14: u64,
82    pub r13: u64,
83    pub r12: u64,
84    pub rbp: u64,
85    pub rbx: u64,
86    pub rdx: u64,
87    pub rax: u64,
88    /// Processor flags.
89    pub rflags: u64,
90    /// Stack pointer.
91    pub rsp: u64,
92    /// Instruction pointer (handler entry or return address).
93    pub rip: u64,
94    /// CR3 value with PCID bits for N3b, or 0 for N3c.
95    pub cr3_pcid: u64,
96    /// Explicit padding to reach 128 bytes (align(64) requires this).
97    /// Prevents implicit compiler padding and ensures deterministic layout
98    /// for the ASM primitive.
99    pub _pad: [u8; 32],
100}
101
102impl N3MinimalContext {
103    /// Zeroed context : used as initial value before preparation.
104    pub const ZERO: Self = Self {
105        r15: 0,
106        r14: 0,
107        r13: 0,
108        r12: 0,
109        rbp: 0,
110        rbx: 0,
111        rdx: 0,
112        rax: 0,
113        rflags: 0,
114        rsp: 0,
115        rip: 0,
116        cr3_pcid: 0,
117        _pad: [0u8; 32],
118    };
119}
120
121/// Migration state machine.
122#[derive(Debug, Clone, Copy, PartialEq, Eq)]
123#[repr(u8)]
124pub enum MigrationState {
125    /// No migration active; frame reusable.
126    Ready = 0,
127    /// Migration engaged; no second migration on this frame.
128    Active = 1,
129    /// Migration not finalized before timeout.
130    Stalled = 2,
131    /// Transient kernel state during recovery.
132    Reclaiming = 3,
133}
134
135impl MigrationState {
136    fn from_u8(v: u8) -> Option<Self> {
137        match v {
138            0 => Some(Self::Ready),
139            1 => Some(Self::Active),
140            2 => Some(Self::Stalled),
141            3 => Some(Self::Reclaiming),
142            _ => None,
143        }
144    }
145}
146
147bitflags::bitflags! {
148    /// Flags for a migration operation.
149    #[derive(Debug, Clone, Copy, PartialEq, Eq)]
150    pub struct MigrationFlags: u16 {
151        /// Eager FPU save/restore (required for SIMD workloads).
152        const FPU_EAGER = 0x0001;
153        /// Lazy FPU save/restore (default, lower cost).
154        const FPU_LAZY = 0x0002;
155        /// Migration crosses CPU cores (requires IPI sync).
156        const INTER_CORE = 0x0004;
157    }
158}
159
160/// Shared migration frame : one per N3 transport endpoint pair.
161///
162/// Must reside in a dedicated physical page mapped at the same kernel
163/// virtual address (`N3_SHARED_FRAME_VA`) in both sender and receiver
164/// address spaces. The mapping is kernel-only (no `USER_ACCESSIBLE`
165/// flag) : all access is controlled by the kernel via the CAS state
166/// machine. Both address spaces see the same physical page through
167/// their own kernel half of the page tables.
168#[repr(C, align(64))]
169pub struct MigrationFrame {
170    /// Sender context (saved by ASM primitive).
171    pub src_ctx: N3MinimalContext,
172    /// Receiver context (prepared by kernel before migration).
173    pub dst_ctx: N3MinimalContext,
174    /// Atomic state: `MigrationState`.
175    pub state: AtomicU8,
176    _pad_state: [u8; 3],
177    /// Monotonically increasing migration counter.
178    pub generation: AtomicU32,
179    /// Message length in bytes.
180    pub msg_len: u16,
181    /// Migration flags.
182    pub flags: u16,
183    _pad_flags: [u8; 4],
184    /// TSC snapshot at migration start (for watchdog).
185    pub tsc_start: AtomicU64,
186    /// Thread ID of the frame owner.
187    pub owner_tid: AtomicU64,
188    /// Reserved padding for future use and cache-line alignment.
189    pub _pad: [u8; 32],
190}
191
192// Static assertion: N3 minimal context layout must be exactly 128 bytes.
193// This is verified both by the `const _` block and by the unit test.
194// The sender_rip capture in send() relies on [rbp+8] which requires
195// frame pointers (force-frame-pointers = true in workspace Cargo.toml).
196const _: () = {
197    assert!(core::mem::size_of::<N3MinimalContext>() == 128);
198    assert!(core::mem::align_of::<N3MinimalContext>() == 64);
199    assert!(core::mem::size_of::<MigrationFrame>() == 320);
200    assert!(core::mem::align_of::<MigrationFrame>() == 64);
201    assert!(core::mem::offset_of!(MigrationFrame, state) == 0x100);
202    assert!(core::mem::offset_of!(MigrationFrame, generation) == 0x104);
203    assert!(core::mem::offset_of!(MigrationFrame, msg_len) == 0x108);
204    assert!(core::mem::offset_of!(MigrationFrame, flags) == 0x10A);
205    assert!(core::mem::offset_of!(MigrationFrame, tsc_start) == 0x110);
206    assert!(core::mem::offset_of!(MigrationFrame, owner_tid) == 0x118);
207    // N3MinimalContext field offsets (for ASM verification).
208    assert!(core::mem::offset_of!(N3MinimalContext, r15) == 0x00);
209    assert!(core::mem::offset_of!(N3MinimalContext, r14) == 0x08);
210    assert!(core::mem::offset_of!(N3MinimalContext, r13) == 0x10);
211    assert!(core::mem::offset_of!(N3MinimalContext, r12) == 0x18);
212    assert!(core::mem::offset_of!(N3MinimalContext, rbp) == 0x20);
213    assert!(core::mem::offset_of!(N3MinimalContext, rbx) == 0x28);
214    assert!(core::mem::offset_of!(N3MinimalContext, rdx) == 0x30);
215    assert!(core::mem::offset_of!(N3MinimalContext, rax) == 0x38);
216    assert!(core::mem::offset_of!(N3MinimalContext, rflags) == 0x40);
217    assert!(core::mem::offset_of!(N3MinimalContext, rsp) == 0x48);
218    assert!(core::mem::offset_of!(N3MinimalContext, rip) == 0x50);
219    assert!(core::mem::offset_of!(N3MinimalContext, cr3_pcid) == 0x58);
220};
221
222// ============================================================================
223// Static frame pool
224// ============================================================================
225
226/// Number of pre-allocated migration frames.
227const N3_FRAME_POOL_SIZE: usize = 64;
228
229/// Virtual address where the MigrationFrame is mapped in both address spaces.
230/// Located in canonical upper-half, just below the HHDM boundary.
231pub const N3_SHARED_FRAME_VA: u64 = 0xFFFF_C000_0000_1000;
232
233/// Virtual address where the shared message buffer is mapped.
234/// Must be distinct from N3_SHARED_FRAME_VA to avoid page table conflicts.
235pub const N3_SHARED_MSG_BUF_VA: u64 = 0xFFFF_C000_0000_2000;
236
237/// Return safe kernel-mode rflags for N3 context switching.
238///
239/// Bit 1 (reserved) must always be 1.  Bit 9 (IF) is set to keep interrupts
240/// enabled in the kernel handler.  All other bits are cleared to avoid
241/// restoring user-modified flags (TF, AC, etc.) into kernel context.
242fn safe_kernel_rflags() -> u64 {
243    // Bit 1 = reserved (must be 1), Bit 9 = IF (interrupts enabled)
244    (1 << 1) | (1 << 9)
245}
246
247/// Size of the per-N3Transport handler stack (1 page).
248/// After CR3 switch, the ASM primitive loads RSP from `dst_ctx.rsp` which
249/// points to this stack with the handler address as the return address.
250const N3_HANDLER_STACK_SIZE: usize = 4096;
251
252/// Static pool of MigrationFrame slots.
253///
254/// Access is serialized by `N3_FRAME_ALLOC` SpinLock. The `UnsafeCell`
255/// avoids `static mut` UB while the lock guarantees exclusive access.
256struct FramePool {
257    inner: core::cell::UnsafeCell<[MaybeUninit<MigrationFrame>; N3_FRAME_POOL_SIZE]>,
258}
259
260// SAFETY: all access to inner is serialized by N3_FRAME_ALLOC SpinLock.
261unsafe impl Sync for FramePool {}
262
263impl FramePool {
264    const fn new() -> Self {
265        FramePool {
266            inner: core::cell::UnsafeCell::new(unsafe { MaybeUninit::uninit().assume_init() }),
267        }
268    }
269
270    fn get_ptr(&self, idx: usize) -> *mut MaybeUninit<MigrationFrame> {
271        unsafe { &mut (*self.inner.get())[idx] as *mut _ }
272    }
273
274    fn phys_addr(&self, idx: usize) -> u64 {
275        // SAFETY: The FramePool is a static array allocated in the kernel's
276        // BSS/data section. On x86_64 with the kernel model, statics live in
277        // the direct-map region or higher half. The HHDM offset converts the
278        // virtual address to a physical address.
279        //
280        // However, this assumes the pool is within the HHDM window. If the
281        // kernel is compiled with a different memory model, this subtraction
282        // yields a garbage physical address.
283        //
284        // To avoid this dependency, the prototype should ideally allocate
285        // frames via the physical frame allocator and map them, rather than
286        // using a static pool with HHDM arithmetic. For now, we assert that
287        // the pool address is above the HHDM base to catch model mismatches.
288        let ptr = self.get_ptr(idx) as u64;
289        let hhdm = crate::memory::hhdm_offset();
290        debug_assert!(
291            ptr >= hhdm,
292            "N3: FramePool not in HHDM region (ptr={:#x}, hhdm={:#x})",
293            ptr,
294            hhdm
295        );
296        ptr.wrapping_sub(hhdm)
297    }
298}
299
300static N3_FRAME_POOL: FramePool = FramePool::new();
301
302/// Bitmap-based O(1) frame allocator.
303struct FrameAllocator {
304    /// Bit i = 1 means frame i is allocated.
305    bitmap: [u64; N3_FRAME_POOL_SIZE / 64],
306    /// Number of currently allocated frames.
307    count: usize,
308}
309
310impl FrameAllocator {
311    const fn new() -> Self {
312        FrameAllocator {
313            bitmap: [0u64; N3_FRAME_POOL_SIZE / 64],
314            count: 0,
315        }
316    }
317
318    fn alloc(&mut self) -> Option<usize> {
319        for word_idx in 0..self.bitmap.len() {
320            if self.bitmap[word_idx] != u64::MAX {
321                let bit = self.bitmap[word_idx].trailing_ones() as usize;
322                self.bitmap[word_idx] |= 1 << bit;
323                self.count += 1;
324                return Some(word_idx * 64 + bit);
325            }
326        }
327        None
328    }
329
330    fn free(&mut self, index: usize) {
331        let word_idx = index / 64;
332        let bit = index % 64;
333        debug_assert!(self.bitmap[word_idx] & (1 << bit) != 0, "double free");
334        self.bitmap[word_idx] &= !(1 << bit);
335        self.count -= 1;
336    }
337
338    fn is_allocated(&self, index: usize) -> bool {
339        self.bitmap[index / 64] & (1 << (index % 64)) != 0
340    }
341}
342
343static N3_FRAME_ALLOC: SpinLock<FrameAllocator> = SpinLock::new(FrameAllocator::new());
344
345/// Allocate a MigrationFrame from the static pool.
346///
347/// Returns a pointer to the frame and its pool index.
348fn alloc_frame_slot() -> Option<(usize, *mut MigrationFrame)> {
349    let idx = with_irqs_disabled(|_token| {
350        let mut alloc = N3_FRAME_ALLOC.lock();
351        alloc.alloc()
352    })?;
353
354    // SAFETY: slot is freshly allocated via bitmap, exclusive access via SpinLock.
355    let frame = N3_FRAME_POOL.get_ptr(idx);
356    // SAFETY: slot is freshly allocated, MaybeUninit is writable.
357    unsafe {
358        frame.write(MaybeUninit::new(MigrationFrame {
359            src_ctx: N3MinimalContext::ZERO,
360            dst_ctx: N3MinimalContext::ZERO,
361            state: AtomicU8::new(MigrationState::Ready as u8),
362            _pad_state: [0; 3],
363            generation: AtomicU32::new(0),
364            msg_len: 0,
365            flags: 0,
366            _pad_flags: [0; 4],
367            tsc_start: AtomicU64::new(0),
368            owner_tid: AtomicU64::new(0),
369            _pad: [0u8; 32],
370        }));
371    }
372
373    Some((idx, frame as *mut MigrationFrame))
374}
375
376/// Free a MigrationFrame back to the pool.
377fn free_frame_slot(idx: usize) {
378    with_irqs_disabled(|_token| {
379        let mut alloc = N3_FRAME_ALLOC.lock();
380        alloc.free(idx);
381    });
382}
383
384/// Get the physical address of a frame in the static pool.
385fn frame_pool_phys_addr(idx: usize) -> u64 {
386    N3_FRAME_POOL.phys_addr(idx)
387}
388
389// ============================================================================
390// PCID management
391// ============================================================================
392
393/// Global PCID counter : monotonic allocation, panics at 4096 (prototype).
394static PCID_COUNTER: AtomicU16 = AtomicU16::new(1); // 0 = no PCID
395
396/// Maximum number of PCIDs before panic (x86-64 supports 4096).
397const MAX_PCIDS: u16 = 4095;
398
399/// Allocate a stable PCID for an address space.
400///
401/// Returns 0 if PCID is not available (N3c fallback).
402pub fn allocate_pcid() -> u16 {
403    let pcid = PCID_COUNTER.fetch_add(1, Ordering::Relaxed);
404    if pcid >= MAX_PCIDS {
405        log::warn!("N3: PCID exhaustion : falling back to N3c (full TLB flush)");
406        return 0;
407    }
408    pcid
409}
410
411/// Free a PCID back to the pool (called when an address space is destroyed).
412///
413/// This prevents PCID exhaustion in long-running systems with dynamic silo
414/// creation/destruction. The freed PCID is tracked for reuse, but the actual
415/// TLB invalidation on other cores is the caller's responsibility.
416pub fn free_pcid(pcid: u16) {
417    if pcid == 0 || pcid >= MAX_PCIDS {
418        return;
419    }
420    // NB: PCID reuse requires INVPCID on the target core before the next
421    // time this PCID is assigned. For the prototype, we simply log the free;
422    // a production implementation would use a freelist + generation counter
423    // (see Linux arch/x86/mm/context.c).
424    log::trace!("N3: PCID {} freed", pcid);
425}
426
427/// Check if PCID feature is available on this CPU.
428pub fn pcid_available() -> bool {
429    // 1. CPUID check: bit 17 of ECX after CPUID(EAX=1) indicates PCID support.
430    let (_, _, ecx, _) = crate::arch::cpuid(1, 0);
431    if ecx & (1 << 17) == 0 {
432        return false;
433    }
434    // 2. Check CR4.PCIDE (bit 17) is actually set (may be masked by hypervisor).
435    //    Intel SDM Vol.3A §4.10.4: CR4.PCIDE = 1 enables PCID in CR3.
436    let cr4 = crate::x86_crate_shim::registers::control::Cr4::read();
437    cr4.contains(crate::x86_crate_shim::registers::control::Cr4Flags::PCID)
438}
439
440/// Select the N3 tier based on actual PCID capability.
441///
442/// Distinguishes theoretical CPU support from operational capability:
443/// - `N3Tier::PcidPreserving` requires PCID available AND a valid PCID > 0
444/// - `N3Tier::FullIsolation` is used when PCID is unavailable or exhausted
445pub fn select_n3_tier(pcid: u16) -> N3Tier {
446    if pcid_available() && pcid > 0 {
447        N3Tier::PcidPreserving
448    } else {
449        N3Tier::FullIsolation
450    }
451}
452
453// ============================================================================
454// validate_rip()
455// ============================================================================
456
457/// Validate that `rip` points to an authorized executable page.
458///
459/// Checks:
460/// - Belongs to an executable VMA
461/// - PTE has NX=0
462/// - Not in kernel ring, MigrationFrame, or non-executable zones
463/// - Part of a registered handler or authorized code range
464fn validate_rip(rip: u64, address_space: &AddressSpace) -> Result<(), IpcError> {
465    // Kernel addresses (upper half) are not valid user RIP targets.
466    if rip >= 0xFFFF_8000_0000_0000 {
467        return Err(IpcError::InvalidRip);
468    }
469
470    // MigrationFrame shared region is not executable.
471    if (N3_SHARED_FRAME_VA..N3_SHARED_FRAME_VA + 0x1000).contains(&rip) {
472        return Err(IpcError::InvalidRip);
473    }
474
475    // Null or near-null RIP is invalid.
476    if rip < 0x1000 {
477        return Err(IpcError::InvalidRip);
478    }
479
480    // validate_rip() performs page-level safety checks only: present, executable,
481    // not in forbidden zones. The RIP to belong to
482    // a "registered handler or authorized code range" : this registration check
483    // is performed by the CALLER (send()) which compares rip against the
484    // receiver's `trampoline_entry` field BEFORE calling validate_rip().
485    //
486    // validate_rip() is therefore a SECOND LINE OF DEFENSE that catches
487    // configuration errors (e.g., trampoline_entry pointing to a data page).
488    //
489    // TOCTOU note: The receiver's page tables could be modified between this
490    // check and the CR3 switch. However, the receiver's trampoline page is
491    // pinned and set to read-only by the kernel at endpoint creation time,
492    // preventing modification even by the receiver itself. The CR3 switch
493    // immediately after this check makes any concurrent modification irrelevant
494    // : the receiver executes in its own AS, using the mapping we validated.
495
496    // Walk the page tables to verify the page is present and executable.
497    // CRITICAL: Also check the U/S bit : the RIP target must be a supervisor
498    // page (Ring 0), not a user page. Accepting a user page would allow a
499    // userspace task to redirect kernel execution to arbitrary user code.
500    let cr3_phys = address_space.cr3();
501    let hhdm = crate::memory::hhdm_offset();
502
503    match unsafe { walk_page_tables_executable(rip, cr3_phys.as_u64(), hhdm) } {
504        Ok(true) => Ok(()),
505        Ok(false) => Err(IpcError::InvalidRip),
506        Err(_) => Err(IpcError::InvalidRip),
507    }
508}
509
510/// Walk x86-64 4-level page tables to check if `vaddr` is present, executable,
511/// and supervisor-only (U/S=0).
512///
513/// Returns `Ok(true)` if the final PTE meets all conditions.
514/// Rejects user-mode pages (U/S=1) as RIP targets for migration.
515unsafe fn walk_page_tables_executable(vaddr: u64, cr3_phys: u64, hhdm: u64) -> Result<bool, ()> {
516    let pml4 = (cr3_phys + hhdm) as *const u64;
517
518    // PML4: check present + U/S (bit 2: 0=supervisor, 1=user)
519    let pml4_idx = ((vaddr >> 39) & 0x1FF) as usize;
520    let pml4_entry = unsafe { core::ptr::read_volatile(pml4.add(pml4_idx)) };
521    if pml4_entry & 1 == 0 {
522        return Err(());
523    }
524    if pml4_entry & 4 != 0 {
525        return Err(());
526    } // U/S=1 => reject
527
528    // PDPT: check present + U/S
529    let pdpt_phys = pml4_entry & 0x000F_FFFF_FFFF_F000;
530    let pdpt = (pdpt_phys + hhdm) as *const u64;
531    let pdpt_idx = ((vaddr >> 30) & 0x1FF) as usize;
532    let pdpt_entry = unsafe { core::ptr::read_volatile(pdpt.add(pdpt_idx)) };
533    if pdpt_entry & 1 == 0 {
534        return Err(());
535    }
536    if pdpt_entry & 4 != 0 {
537        return Err(());
538    } // U/S=1 => reject
539
540    // 1 GiB page? Check NX + U/S
541    if pdpt_entry & (1 << 7) != 0 {
542        if pdpt_entry & (1 << 63) != 0 {
543            return Err(());
544        } // NX=1 => reject
545        return Ok(true);
546    }
547
548    // Page Directory: check present + U/S
549    let pd_phys = pdpt_entry & 0x000F_FFFF_FFFF_F000;
550    let pd = (pd_phys + hhdm) as *const u64;
551    let pd_idx = ((vaddr >> 21) & 0x1FF) as usize;
552    let pd_entry = unsafe { core::ptr::read_volatile(pd.add(pd_idx)) };
553    if pd_entry & 1 == 0 {
554        return Err(());
555    }
556    if pd_entry & 4 != 0 {
557        return Err(());
558    } // U/S=1 => reject
559
560    // 2 MiB page? Check NX
561    if pd_entry & (1 << 7) != 0 {
562        if pd_entry & (1 << 63) != 0 {
563            return Err(());
564        } // NX=1 => reject
565        return Ok(true);
566    }
567
568    // Page Table: check present + U/S + NX
569    let pt_phys = pd_entry & 0x000F_FFFF_FFFF_F000;
570    let pt = (pt_phys + hhdm) as *const u64;
571    let pt_idx = ((vaddr >> 12) & 0x1FF) as usize;
572    let pt_entry = unsafe { core::ptr::read_volatile(pt.add(pt_idx)) };
573    if pt_entry & 1 == 0 {
574        return Err(());
575    }
576    if pt_entry & 4 != 0 {
577        return Err(());
578    } // U/S=1 => reject user page
579    if pt_entry & (1 << 63) != 0 {
580        return Err(());
581    } // NX=1 => reject
582
583    Ok(true)
584}
585
586// ============================================================================
587// Context preparation
588// ============================================================================
589
590/// Prepare a MigrationFrame for a send operation.
591///
592/// Sets up `src_ctx` (sender), `dst_ctx` (receiver), copies the message
593/// into the shared buffer, increments generation, and transitions state
594/// to Active.
595fn n3_prepare_migration(
596    frame: &mut MigrationFrame,
597    msg_buf: *mut u8,
598    sender_cr3: u64,
599    sender_rsp: u64,
600    sender_rip: u64,
601    handler_stack_top: u64,
602    receiver_task: &Task,
603    msg: &[u8],
604) -> Result<(), IpcError> {
605    // Prepare sender context (saved by ASM primitive).
606    frame.src_ctx.cr3_pcid = sender_cr3;
607    frame.src_ctx.rsp = sender_rsp;
608    frame.src_ctx.rip = sender_rip;
609    // P0 fix: initialize rflags with safe kernel defaults.
610    // A zero rflags would clear IF (interrupts) and other critical bits.
611    frame.src_ctx.rflags = safe_kernel_rflags();
612
613    // Prepare receiver context.
614    let recv_handler = receiver_task.trampoline_entry.load(Ordering::Acquire);
615
616    if recv_handler == 0 {
617        return Err(IpcError::TransportFailed);
618    }
619
620    let recv_cr3 = receiver_task.process.address_space_arc().cr3().as_u64();
621
622    frame.dst_ctx.rip = recv_handler;
623    frame.dst_ctx.cr3_pcid = recv_cr3;
624    // P0 fix: destination context also gets safe kernel rflags.
625    frame.dst_ctx.rflags = safe_kernel_rflags();
626
627    // CRITICAL: The destination kernel stack must contain the handler address
628    // as the return address for the `ret` instruction after CR3 switch.
629    // We use the per-transport handler stack (kernel upper half, shared across
630    // all address spaces) to ensure accessibility after CR3 switch.
631    // Layout: [handler_stack_top - 8] = recv_handler (return address for ret).
632    //
633    // SAFETY: handler_stack_top points to the transport's dedicated stack page,
634    // allocated in new(). The CAS state machine (Ready=>Active) ensures exclusive
635    // access. The -8 offset places the handler address as if it were pushed.
636    let trampoline_top = (handler_stack_top - 8) as *mut u64;
637    unsafe {
638        // Write handler address at the trampoline top (this is what `ret` will pop).
639        core::ptr::write_volatile(trampoline_top, recv_handler);
640    }
641    // dst_ctx.rsp points to the trampoline area where the handler address is stored.
642    // After CR3 switch, `mov rsp, [rdi+0xC8]` loads this, and `ret` pops recv_handler.
643    frame.dst_ctx.rsp = trampoline_top as u64;
644
645    // Copy message into the shared buffer.
646    let msg_len = shared_msg_write(msg_buf, msg)?;
647    frame.msg_len = msg_len;
648
649    // CAS Ready => Active. Must succeed before touching generation.
650    if frame
651        .state
652        .compare_exchange(
653            MigrationState::Ready as u8,
654            MigrationState::Active as u8,
655            Ordering::AcqRel,
656            Ordering::Acquire,
657        )
658        .is_err()
659    {
660        return Err(IpcError::WouldBlock);
661    }
662
663    // Increment generation AFTER successful CAS.
664    // The generation must only advance for committed migrations.
665    frame.generation.fetch_add(1, Ordering::Release);
666
667    // Record timestamp and owner.
668    let tsc = unsafe { crate::arch::rdtsc() };
669    frame.tsc_start.store(tsc, Ordering::Release);
670    if let Some(tid) = current_task_id() {
671        frame.owner_tid.store(tid.as_u64(), Ordering::Release);
672    }
673
674    Ok(())
675}
676
677/// Maximum message size for N3 transport (separate buffer, not in frame).
678pub const N3_MSG_BUF_SIZE: usize = 2048;
679
680/// Wrapper for a raw pointer to the shared message buffer.
681///
682/// # Safety
683/// The pointer is valid for the lifetime of the N3Transport and is mapped
684/// in both sender and receiver address spaces. All access is serialized
685/// by the MigrationFrame state machine.
686struct MsgBuffer(*mut u8);
687
688// SAFETY: MsgBuffer is used only from a single migration context at a time,
689// protected by the MigrationFrame CAS state machine.
690unsafe impl Send for MsgBuffer {}
691unsafe impl Sync for MsgBuffer {}
692
693impl MsgBuffer {
694    fn as_ptr(&self) -> *mut u8 {
695        self.0
696    }
697}
698
699/// Read the message payload from a shared message buffer.
700fn shared_msg_read(msg_buf: *const u8, msg_len: u16, buf: &mut [u8]) -> Result<usize, IpcError> {
701    let len = msg_len as usize;
702    if len > buf.len() {
703        return Err(IpcError::BufferTooSmall);
704    }
705    if len > N3_MSG_BUF_SIZE {
706        return Err(IpcError::MessageTooLarge);
707    }
708    let src = unsafe { core::slice::from_raw_parts(msg_buf, len) };
709    buf[..len].copy_from_slice(src);
710    Ok(len)
711}
712
713/// Write a message payload into a shared message buffer.
714fn shared_msg_write(msg_buf: *mut u8, msg: &[u8]) -> Result<u16, IpcError> {
715    if msg.len() > N3_MSG_BUF_SIZE {
716        return Err(IpcError::MessageTooLarge);
717    }
718    let dst = unsafe { core::slice::from_raw_parts_mut(msg_buf, msg.len()) };
719    dst.copy_from_slice(msg);
720    Ok(msg.len() as u16)
721}
722
723// ============================================================================
724// ASM Primitive
725// ============================================================================
726
727/// ASM migration primitive.
728///
729/// # Safety
730/// - `frame` must point to a valid MigrationFrame mapped at the same VA
731///   in both sender and receiver address spaces.
732/// - Interrupts must be enabled before calling (the primitive will CLI).
733/// - The caller must be on a valid kernel stack.
734///
735/// # Layout adjusted for a 64b align padding
736///
737/// ```text
738/// Offset  Field
739/// 0x00    src_ctx.r15
740/// 0x08    src_ctx.r14
741/// 0x10    src_ctx.r13
742/// 0x18    src_ctx.r12
743/// 0x20    src_ctx.rbp
744/// 0x28    src_ctx.rbx
745/// 0x30    src_ctx.rdx
746/// 0x38    src_ctx.rax
747/// 0x40    src_ctx.rflags
748/// 0x48    src_ctx.rsp
749/// 0x50    src_ctx.rip
750/// 0x58    src_ctx.cr3_pcid
751/// 0x60-0x7F  (alignment padding)
752/// 0x80    dst_ctx.r15
753/// 0x88    dst_ctx.r14
754/// 0x90    dst_ctx.r13
755/// 0x98    dst_ctx.r12
756/// 0xA0    dst_ctx.rbp
757/// 0xA8    dst_ctx.rbx
758/// 0xB0    dst_ctx.rdx
759/// 0xB8    dst_ctx.rax
760/// 0xC0    dst_ctx.rflags
761/// 0xC8    dst_ctx.rsp
762/// 0xD0    dst_ctx.rip
763/// 0xD8    dst_ctx.cr3_pcid
764/// 0xE0-0xFF (alignment padding)
765/// 0x100   state (AtomicU8)
766/// 0x104   generation (AtomicU32)
767/// ```
768///
769/// # Behavior
770/// 1. Saves source context to `frame.src_ctx`
771/// 2. CLI (≤3 instructions) + CR3 switch to `frame.dst_ctx.cr3_pcid`
772/// 3. Restores destination context from `frame.dst_ctx`
773/// 4. Sets `frame.state = Ready`, increments `generation`
774/// 5. Returns via `ret` (pops RIP from restored `dst_ctx.rsp`)
775#[unsafe(naked)]
776pub unsafe extern "C" fn n3b_migrate_asm(frame: *mut MigrationFrame) {
777    core::arch::naked_asm!(
778        // ── Phase 1: save src_ctx ──
779        "mov [rdi + 0x00], r15",
780        "mov [rdi + 0x08], r14",
781        "mov [rdi + 0x10], r13",
782        "mov [rdi + 0x18], r12",
783        "mov [rdi + 0x20], rbp",
784        "mov [rdi + 0x28], rbx",
785        "mov [rdi + 0x30], rdx",
786        "mov [rdi + 0x38], rax",
787        "pushfq",
788        "pop rax",
789        "mov [rdi + 0x40], rax", // rflags
790        "mov [rdi + 0x48], rsp",
791        // NOTE: src_ctx.rip was already set by n3_prepare_migration().
792        // We do NOT overwrite it with RCX : on Strategy A, the kernel
793        // sets src_ctx.rip before calling us. See §11.2 for the contract.
794        // ── Phase 2: CLI (<3 instr) + CR3 switch ──
795        // We need a 3 instructions **MAX** between CLI and CR3 write.
796        // Here: mov rax (1) => cli (2) => mov cr3 (3). Satisfied.
797        // After CR3 switch, we restore dst_ctx with interrupts still disabled.
798        // The 8 restore instructions + pushfq/popfq/sti happen AFTER the CR3
799        // switch in the receiver's address space : this is intentional and does
800        // not violate the ≤3 instruction window (which only covers CLI=>CR3).
801        "mov rax, [rdi + 0xD8]", // dst_ctx.cr3_pcid (at 0x80+0x58=0xD8)
802        "cli",
803        "mov cr3, rax",
804        // ── Phase 3: restore dst_ctx (starts at offset 0x80) ──
805        // IMPORTANT: rsp (offset 0xC8) is restored LAST, just before ret.
806        // Every memory access between CR3 switch and rsp restore uses
807        // RDI (shared fixed-VA, valid in both AS) : never RSP.
808        "mov r15, [rdi + 0x80]",
809        "mov r14, [rdi + 0x88]",
810        "mov r13, [rdi + 0x90]",
811        "mov r12, [rdi + 0x98]",
812        "mov rbp, [rdi + 0xA0]",
813        "mov rbx, [rdi + 0xA8]",
814        "mov rdx, [rdi + 0xB0]",
815        // Load rflags from dst_ctx.rflags at offset 0x80+0x40 = 0xC0.
816        // Note: dst_ctx.rax (0xB8) is a GPR, NOT flags : do not confuse.
817        "mov rax, [rdi + 0xC0]", // dst_ctx.rflags
818        "push rax",
819        "popfq", // restore IF=1
820        "sti",   // guarantee IF=1
821        // ── Phase 4: restore RSP from dst_ctx ──
822        // CRITICAL: RSP must be restored BEFORE `ret`. The destination kernel
823        // stack is shared (kernel upper half maps identically in all AS).
824        // dst_ctx.rsp was set by n3_prepare_migration() to point to a stack
825        // frame containing the handler entry address as the return address.
826        "mov rsp, [rdi + 0xC8]", // dst_ctx.rsp
827        // ── Phase 5: receiver handler runs (state is still Active) ──
828        // The ASM does NOT set state = Ready. The receiver handler MUST
829        // do that after consuming the message (see recv() protocol).
830        // This prevents the TOCTOU where the receiver sees state=Ready
831        // before having read the message. Generation is managed by the
832        // kernel in n3_prepare_migration/recv : NOT by the ASM.
833        // ── Phase 6: ret into receiver handler ──
834        // SAFETY: RSP now points to the destination kernel stack with the
835        // handler address at [RSP]. `ret` pops it and jumps there.
836        // The destination kernel stack must be valid and mapped in both AS
837        // (kernel upper half is shared). If corrupted, this is the first
838        // point of failure.
839        "ret",
840        //
841        // Invariant: the restored dst_ctx.rsp must point to a valid, readable,
842        // kernel-mode stack frame.
843    );
844}
845
846// ============================================================================
847// IPI synchronization
848// ============================================================================
849
850/// Per-CPU pending migration frame pointer.
851/// Set by the sender before sending the IPI, consumed by the IPI handler.
852static N3_PENDING_MIGRATION: [AtomicU64; crate::arch::percpu::MAX_CPUS] =
853    [const { AtomicU64::new(0) }; crate::arch::percpu::MAX_CPUS];
854
855/// Send a migration-sync IPI to the target CPU.
856fn send_n3_sync_ipi(target_apic_id: u32) {
857    let icr_low = apic::IPI_N3_MIGRATE_VECTOR as u32 | (1 << 14);
858    apic::send_ipi_raw(target_apic_id, icr_low);
859}
860
861/// Naked IPI handler for N3 migration synchronization.
862///
863/// When a cross-core migration completes, the receiver sends this IPI to the
864/// sender's core. The handler restores the sender's kernel context (RSP, RIP,
865/// RFLAGS) from `MigrationFrame.src_ctx` and returns via `iretq`.
866///
867/// The interrupted context was the sender's `send()` function. By modifying
868/// the InterruptStackFrame fields before the hardware `iretq`, we redirect
869/// execution to the sender's saved return address on its original kernel stack.
870///
871/// # Safety
872/// Registered as a naked IDT entry. On entry: SS, RSP, RFLAGS, CS, RIP pushed
873/// by hardware. `rdi` points at the InterruptStackFrame (set by IDT dispatch).
874#[unsafe(naked)]
875pub unsafe extern "C" fn n3_migrate_ipi_entry(rdi: *mut u8) -> ! {
876    core::arch::naked_asm!(
877        // ── Save scratch registers ──
878        "push rax",
879        "push rcx",
880        "push rdx",
881        "push rsi",
882        "push r8",
883        "push r9",
884        "push r10",
885        "push r11",
886
887        // ── Check pending migration ──
888        // per_cpu_index via GS:base (offset 0 in PerCpu struct).
889        "mov rcx, qword ptr gs:[0]",    // current_cpu_index
890        "lea rsi, [rip + {pending}]",    // &N3_PENDING_MIGRATION
891        "mov rax, [rsi + rcx * 8]",      // pending = N3_PENDING_MIGRATION[cpu]
892        "test rax, rax",
893        "jz .no_pending",
894
895        // ── Clear pending flag ──
896        "mov qword ptr [rsi + rcx * 8], 0",
897
898        // ── rax = MigrationFrame ptr ──
899        // src_ctx starts at offset 0x00.
900        // src_ctx.rflags = 0x40, src_ctx.rsp = 0x48, src_ctx.rip = 0x50.
901
902        // ── Switch to sender's address space ──
903        "mov rcx, [rax + 0x58]",        // src_ctx.cr3_pcid
904        "test rcx, rcx",
905        "jz .skip_cr3",
906        "mov r11, 0x0000FFFFFFFFFFFF",
907        "and rcx, r11", // clear high 16 bits (reserved/PCID)
908        "mov r11, 0xFFFFFFFFFFFFF000",
909        "and rcx, r11", // clear low 12 bits (page offset)
910        "mov cr3, rcx",
911        ".skip_cr3:",
912
913        // ── Build iretq frame on the interrupted stack ──
914        // rdi points to InterruptStackFrame pushed by hardware.
915        // Overwrite it with the sender's context.
916        // InterruptStackFrame layout:
917        //   [+0x00] SS       [+0x08] RSP      [+0x10] RFLAGS
918        //   [+0x18] CS       [+0x20] RIP      [+0x28] (error code, skipped)
919        //
920        // iretq pops: RIP, CS, RFLAGS, RSP, SS (5 × u64 = 40 bytes).
921        //
922        // Preserve original CS and SS from the interrupted frame.
923        // Overwrite RSP, RFLAGS, RIP with sender's values.
924
925        "mov rcx, [rdi + 0x18]",        // original CS (kernel CS = 0x08)
926        "mov rdx, [rdi + 0x00]",        // original SS (0 for Ring 0)
927
928        "mov rsi, [rax + 0x48]",        // src_ctx.rsp  => sender's kernel RSP
929        "mov r8,  [rax + 0x40]",        // src_ctx.rflags
930        "mov r9,  [rax + 0x50]",        // src_ctx.rip  => sender's return address
931
932        "mov [rdi + 0x00], rdx",        // SS  = original (Ring 0)
933        "mov [rdi + 0x08], rsi",        // RSP = sender's kernel stack
934        "mov [rdi + 0x10], r8",         // RFLAGS = sender's flags
935        "mov [rdi + 0x18], rcx",        // CS  = original (0x08)
936        "mov [rdi + 0x20], r9",         // RIP = sender's return address
937
938        // ── Send EOI ──
939        "mov dword ptr [0x0000FEE000B0], 0", // LAPIC EOI
940
941        // ── Restore scratch and return ──
942        // Pop scratch registers, then let the normal IDT return path
943        // (pop GPRs + iretq) restore from the modified frame.
944        "pop r11",
945        "pop r10",
946        "pop r9",
947        "pop r8",
948        "pop rsi",
949        "pop rdx",
950        "pop rcx",
951        "pop rax",
952        // Return to IDT stub which pops GPRs and does iretq.
953        "ret",
954
955        ".no_pending:",
956        // Spurious IPI : send EOI and return to interrupted context.
957        "mov dword ptr [0x0000FEE000B0], 0",
958        "pop r11",
959        "pop r10",
960        "pop r9",
961        "pop r8",
962        "pop rsi",
963        "pop rdx",
964        "pop rcx",
965        "pop rax",
966        "ret",
967
968        pending = sym N3_PENDING_MIGRATION,
969    );
970}
971
972/// Fallback handler for N3 migration IPI (non-naked path).
973///
974/// Called from the `extern "x86-interrupt"` stub if the naked entry is not
975/// used. Clears the pending flag and sends EOI.
976pub extern "C" fn n3_migrate_ipi_handler() {
977    let cpu_idx = crate::arch::percpu::current_cpu_index();
978    N3_PENDING_MIGRATION[cpu_idx].store(0, Ordering::Release);
979    apic::eoi();
980}
981
982// ============================================================================
983// WATCHDOG
984// ============================================================================
985
986/// Default watchdog timeout in TSC cycles (~10ms at 3GHz).
987const N3_WATCHDOG_TIMEOUT: u64 = 30_000_000;
988
989/// Watchdog state per frame.
990struct WatchdogEntry {
991    frame_phys: u64,
992    timeout: u64,
993}
994
995/// Global watchdog table.
996static N3_WATCHDOG_TABLE: SpinLock<[Option<WatchdogEntry>; N3_FRAME_POOL_SIZE]> =
997    SpinLock::new([const { None }; N3_FRAME_POOL_SIZE]);
998
999/// Register a frame with the watchdog.
1000fn watchdog_register(frame_idx: usize, frame_phys: u64) {
1001    let mut table = N3_WATCHDOG_TABLE.lock();
1002    if let Some(entry) = table.get_mut(frame_idx) {
1003        *entry = Some(WatchdogEntry {
1004            frame_phys,
1005            timeout: N3_WATCHDOG_TIMEOUT,
1006        });
1007    }
1008}
1009
1010/// Unregister a frame from the watchdog.
1011fn watchdog_unregister(frame_idx: usize) {
1012    let mut table = N3_WATCHDOG_TABLE.lock();
1013    if let Some(entry) = table.get_mut(frame_idx) {
1014        *entry = None;
1015    }
1016}
1017
1018/// Called from the timer ISR to check for stalled migrations.
1019///
1020/// For each registered frame in Active state, checks whether the migration
1021/// has exceeded the watchdog timeout. If so:
1022/// 1. Transitions Active => Stalled (CAS)
1023/// 2. Attempts recovery: Stalled => Reclaiming => Ready
1024/// 3. Logs the incident
1025///
1026/// Recovery resets the frame so it can be reused for future migrations.
1027/// The sender that initiated the stalled migration will see `WouldBlock` on
1028/// its next `send()` attempt (CAS Ready=>Active fails if another sender
1029/// claimed the frame first).
1030pub fn n3_watchdog_tick() {
1031    let now = unsafe { crate::arch::rdtsc() };
1032    let mut table = N3_WATCHDOG_TABLE.lock();
1033
1034    for entry in table.iter_mut() {
1035        let Some(entry) = entry else { continue };
1036        let frame = unsafe { &*(phys_to_virt(entry.frame_phys) as *const MigrationFrame) };
1037        let state = frame.state.load(Ordering::Acquire);
1038
1039        if state != MigrationState::Active as u8 {
1040            continue;
1041        }
1042
1043        let tsc_start = frame.tsc_start.load(Ordering::Acquire);
1044        if now.wrapping_sub(tsc_start) <= entry.timeout {
1045            continue;
1046        }
1047
1048        // Timeout detected: attempt Active => Stalled => Reclaiming => Ready.
1049        let cas_result = frame.state.compare_exchange(
1050            MigrationState::Active as u8,
1051            MigrationState::Stalled as u8,
1052            Ordering::AcqRel,
1053            Ordering::Acquire,
1054        );
1055
1056        if cas_result.is_err() {
1057            // Another CPU already transitioned the frame (e.g., recv completed).
1058            continue;
1059        }
1060
1061        log::warn!(
1062            "N3: watchdog timeout on frame at {:#x}, generation={}, recovering",
1063            entry.frame_phys,
1064            frame.generation.load(Ordering::Acquire),
1065        );
1066
1067        // Attempt recovery: Stalled => Reclaiming => Ready.
1068        watchdog_recover(frame);
1069    }
1070}
1071
1072/// Recover a stalled frame : reset to Ready via atomic CAS transitions.
1073fn watchdog_recover(frame: &MigrationFrame) {
1074    // Transition Stalled => Reclaiming (atomic CAS).
1075    if frame
1076        .state
1077        .compare_exchange(
1078            MigrationState::Stalled as u8,
1079            MigrationState::Reclaiming as u8,
1080            Ordering::AcqRel,
1081            Ordering::Acquire,
1082        )
1083        .is_err()
1084    {
1085        log::warn!("N3: watchdog recover: frame not in Stalled state, skipping");
1086        return;
1087    }
1088    log::warn!(
1089        "N3: recovering stalled frame, generation={}",
1090        frame.generation.load(Ordering::Acquire)
1091    );
1092    // Transition Reclaiming => Ready (atomic store, no contention expected).
1093    frame
1094        .state
1095        .store(MigrationState::Ready as u8, Ordering::Release);
1096}
1097
1098// ============================================================================
1099// Shared frame mapping
1100// ============================================================================
1101
1102/// Map a physical page into an address space at the specified virtual address.
1103///
1104/// The mapping is kernel-only (no `USER_ACCESSIBLE` flag) to enforce the
1105/// spec §8.4 invariant: frame permissions are managed exclusively by the
1106/// kernel, never by non-kernel parties.
1107fn map_page_in_space(
1108    frame_phys: u64,
1109    target_va: u64,
1110    address_space: &AddressSpace,
1111) -> Result<(), &'static str> {
1112    use crate::arch::xshim::{PageTableFlags, PhysFrame, Size4KiB};
1113    use crate::arch::xshim::{PhysAddr, VirtAddr};
1114    use crate::x86_crate_shim::structures::paging::{Mapper, Page};
1115
1116    let page = Page::<Size4KiB>::containing_address(VirtAddr::new(target_va));
1117    let phys_frame = PhysFrame::<Size4KiB>::containing_address(PhysAddr::new(frame_phys));
1118
1119    // Kernel-only, no USER_ACCESSIBLE (spec §8.4: permissions managed by kernel).
1120    let flags = PageTableFlags::PRESENT | PageTableFlags::WRITABLE;
1121
1122    let mut mapper = unsafe { address_space.mapper() };
1123    let mut allocator = crate::memory::paging::BuddyFrameAllocator;
1124
1125    unsafe {
1126        match mapper.map_to(page, phys_frame, flags, &mut allocator) {
1127            Ok(flush) => {
1128                flush.flush();
1129                Ok(())
1130            }
1131            Err(_) => Err("Failed to map page into address space"),
1132        }
1133    }
1134}
1135
1136/// Map a MigrationFrame's physical page into both sender and receiver
1137/// address spaces at `N3_SHARED_FRAME_VA`.
1138fn map_frame_in_both_spaces(
1139    frame_phys: u64,
1140    sender_as: &AddressSpace,
1141    receiver_as: &AddressSpace,
1142) -> Result<(), &'static str> {
1143    map_page_in_space(frame_phys, N3_SHARED_FRAME_VA, sender_as)?;
1144    map_page_in_space(frame_phys, N3_SHARED_FRAME_VA, receiver_as)?;
1145    Ok(())
1146}
1147
1148/// Map a message buffer's physical page into both address spaces
1149/// at `N3_SHARED_MSG_BUF_VA`.
1150fn map_msg_buf_in_both_spaces(
1151    msg_buf_phys: u64,
1152    sender_as: &AddressSpace,
1153    receiver_as: &AddressSpace,
1154) -> Result<(), &'static str> {
1155    map_page_in_space(msg_buf_phys, N3_SHARED_MSG_BUF_VA, sender_as)?;
1156    map_page_in_space(msg_buf_phys, N3_SHARED_MSG_BUF_VA, receiver_as)?;
1157    Ok(())
1158}
1159
1160/// Unmap a single page from an address space and shoot down TLB on all CPUs.
1161fn unmap_page_in_space(target_va: u64, address_space: &AddressSpace) {
1162    use crate::{
1163        arch::xshim::{Size4KiB, VirtAddr},
1164        x86_crate_shim::structures::paging::{Mapper, Page},
1165    };
1166
1167    let page = Page::<Size4KiB>::containing_address(VirtAddr::new(target_va));
1168    let mut mapper = unsafe { address_space.mapper() };
1169
1170    if let Ok((_frame, flush)) = mapper.unmap(page) {
1171        flush.flush();
1172    }
1173    // Inter-CPU TLB shootdown for the unmapped page.
1174    crate::arch::tlb::shootdown_range(VirtAddr::new(target_va), VirtAddr::new(target_va + 0x1000));
1175}
1176
1177/// Unmap the MigrationFrame and message buffer from both address spaces.
1178fn unmap_from_both_spaces(sender_as: &AddressSpace, receiver_as: &AddressSpace) {
1179    unmap_page_in_space(N3_SHARED_FRAME_VA, sender_as);
1180    unmap_page_in_space(N3_SHARED_FRAME_VA, receiver_as);
1181    unmap_page_in_space(N3_SHARED_MSG_BUF_VA, sender_as);
1182    unmap_page_in_space(N3_SHARED_MSG_BUF_VA, receiver_as);
1183}
1184
1185// ============================================================================
1186// N3Transport
1187// ============================================================================
1188
1189/// N3 MMU transport : thread migration between distinct address spaces.
1190pub struct N3Transport {
1191    /// Index in the static frame pool.
1192    frame_idx: usize,
1193    /// Physical address of the MigrationFrame page.
1194    frame_phys: u64,
1195    /// Virtual address of the MigrationFrame (same in both AS).
1196    frame_virt: u64,
1197    /// Shared message buffer (kernel-allocated, mapped in both AS).
1198    msg_buf: MsgBuffer,
1199    /// Physical address of the message buffer.
1200    msg_buf_phys: u64,
1201    /// Dedicated handler stack (1 page, kernel-allocated).
1202    /// After CR3 switch, dst_ctx.rsp points to the top of this stack with
1203    /// the handler address as the return address.
1204    handler_stack_top: u64,
1205    /// Physical address of the handler stack.
1206    handler_stack_phys: u64,
1207    /// Sender task ID.
1208    sender_task_id: TaskId,
1209    /// Receiver task ID.
1210    receiver_task_id: TaskId,
1211    /// N3 tier (b or c).
1212    tier: N3Tier,
1213    /// Sender's PCID (0 for N3c).
1214    pcid_sender: u16,
1215    /// Receiver's PCID (0 for N3c).
1216    pcid_receiver: u16,
1217    /// Watchdog timeout in TSC cycles.
1218    watchdog_timeout_cycles: u64,
1219}
1220
1221// SAFETY: N3Transport is Send+Sync because all mutable access is protected
1222// by the MigrationFrame CAS state machine.
1223unsafe impl Send for N3Transport {}
1224unsafe impl Sync for N3Transport {}
1225
1226impl N3Transport {
1227    /// Create a new N3 transport between sender and receiver tasks.
1228    ///
1229    /// Allocates a MigrationFrame and a shared message buffer, maps them
1230    /// in both address spaces, and registers the endpoint with the watchdog.
1231    pub fn new(sender_task_id: TaskId, receiver_task_id: TaskId) -> Result<Self, IpcError> {
1232        let (frame_idx, _frame_ptr) = alloc_frame_slot().ok_or(IpcError::TransportFailed)?;
1233
1234        let frame_phys = frame_pool_phys_addr(frame_idx);
1235
1236        // Allocate a shared message buffer (1 page).
1237        let msg_buf_frame =
1238            with_irqs_disabled(allocate_frame).map_err(|_| IpcError::TransportFailed)?;
1239        let msg_buf_phys = msg_buf_frame.start_address.as_u64();
1240        let msg_buf = phys_to_virt(msg_buf_phys) as *mut u8;
1241
1242        // Map in both address spaces.
1243        let sender = get_task_by_id(sender_task_id).ok_or(IpcError::Disconnected)?;
1244        let receiver = get_task_by_id(receiver_task_id).ok_or(IpcError::Disconnected)?;
1245
1246        let sender_as = sender.process.address_space_arc();
1247        let receiver_as = receiver.process.address_space_arc();
1248
1249        if let Err(e) = map_frame_in_both_spaces(frame_phys, &sender_as, &receiver_as) {
1250            log::error!("N3: failed to map frame: {}", e);
1251            free_frame_slot(frame_idx);
1252            with_irqs_disabled(|token| free_frame(token, msg_buf_frame));
1253            return Err(IpcError::TransportFailed);
1254        }
1255
1256        // Map the message buffer in both address spaces at N3_SHARED_MSG_BUF_VA.
1257        if map_msg_buf_in_both_spaces(msg_buf_phys, &sender_as, &receiver_as).is_err() {
1258            log::error!("N3: failed to map message buffer");
1259            // P1 fix: unmap the frame that was already mapped in both spaces.
1260            unmap_page_in_space(N3_SHARED_FRAME_VA, &sender_as);
1261            unmap_page_in_space(N3_SHARED_FRAME_VA, &receiver_as);
1262            free_frame_slot(frame_idx);
1263            with_irqs_disabled(|token| free_frame(token, msg_buf_frame));
1264            return Err(IpcError::TransportFailed);
1265        }
1266
1267        // Allocate PCIDs BEFORE selecting tier : allocate_pcid() may return 0
1268        // if PCID is exhausted, which forces N3c even if the CPU supports PCID.
1269        let pcid_sender = allocate_pcid();
1270        let pcid_receiver = allocate_pcid();
1271
1272        // Select tier based on both CPU support AND availability.
1273        let tier = select_n3_tier(pcid_sender.min(pcid_receiver));
1274
1275        // Allocate a dedicated handler stack (1 page, kernel-allocated).
1276        // This stack is used as the trampoline after CR3 switch: the ASM
1277        // primitive loads RSP from dst_ctx.rsp and `ret` pops the handler
1278        // address from [RSP]. Must be in the kernel upper half (shared AS).
1279        let handler_stack_frame =
1280            with_irqs_disabled(allocate_frame).map_err(|_| IpcError::TransportFailed)?;
1281        let handler_stack_phys = handler_stack_frame.start_address.as_u64();
1282        let handler_stack_virt = phys_to_virt(handler_stack_phys) as u64;
1283        let handler_stack_top = handler_stack_virt + N3_HANDLER_STACK_SIZE as u64;
1284
1285        // Register with watchdog.
1286        watchdog_register(frame_idx, frame_phys);
1287
1288        Ok(N3Transport {
1289            frame_idx,
1290            frame_phys,
1291            frame_virt: N3_SHARED_FRAME_VA,
1292            msg_buf: MsgBuffer(msg_buf),
1293            msg_buf_phys,
1294            handler_stack_top,
1295            handler_stack_phys,
1296            sender_task_id,
1297            receiver_task_id,
1298            tier,
1299            pcid_sender,
1300            pcid_receiver,
1301            watchdog_timeout_cycles: N3_WATCHDOG_TIMEOUT,
1302        })
1303    }
1304
1305    /// Get a reference to the underlying MigrationFrame.
1306    pub(crate) fn frame(&self) -> &MigrationFrame {
1307        unsafe { &*(self.frame_virt as *const MigrationFrame) }
1308    }
1309
1310    /// Get the raw pointer to the MigrationFrame (for ASM primitive).
1311    pub(crate) fn frame_ptr(&self) -> *mut MigrationFrame {
1312        self.frame_virt as *mut MigrationFrame
1313    }
1314}
1315
1316impl core::fmt::Debug for N3Transport {
1317    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
1318        f.debug_struct("N3Transport")
1319            .field("frame_idx", &self.frame_idx)
1320            .field("tier", &self.tier)
1321            .field("sender", &self.sender_task_id)
1322            .field("receiver", &self.receiver_task_id)
1323            .finish()
1324    }
1325}
1326
1327impl Drop for N3Transport {
1328    fn drop(&mut self) {
1329        watchdog_unregister(self.frame_idx);
1330
1331        // P0 fix: unmap shared pages from both address spaces BEFORE freeing
1332        // the physical frames. This prevents use-after-free when the frames
1333        // are reallocated to another process or structure.
1334        if let (Some(sender), Some(receiver)) = (
1335            get_task_by_id(self.sender_task_id),
1336            get_task_by_id(self.receiver_task_id),
1337        ) {
1338            let sender_as = sender.process.address_space_arc();
1339            let receiver_as = receiver.process.address_space_arc();
1340            unmap_from_both_spaces(&sender_as, &receiver_as);
1341        }
1342
1343        free_frame_slot(self.frame_idx);
1344
1345        // Free the message buffer physical page.
1346        if self.msg_buf_phys != 0 {
1347            let frame = crate::memory::PhysFrame::containing_address(
1348                crate::arch::xshim::PhysAddr::new(self.msg_buf_phys),
1349            );
1350            with_irqs_disabled(|token| free_frame(token, frame));
1351        }
1352
1353        // Free the handler stack physical page.
1354        if self.handler_stack_phys != 0 {
1355            let frame = crate::memory::PhysFrame::containing_address(
1356                crate::arch::xshim::PhysAddr::new(self.handler_stack_phys),
1357            );
1358            with_irqs_disabled(|token| free_frame(token, frame));
1359        }
1360    }
1361}
1362
1363impl IpcTransport for N3Transport {
1364    fn level(&self) -> TransportLevel {
1365        TransportLevel::Mmu
1366    }
1367
1368    fn capabilities(&self) -> TransportCapabilities {
1369        TransportCapabilities {
1370            max_message_size: N3_MSG_BUF_SIZE,
1371            blocking: true,
1372            zero_copy: false,
1373            vectored: false,
1374            directions: 2,
1375            estimated_cost_cycles: 800,
1376        }
1377    }
1378
1379    fn name(&self) -> &'static str {
1380        match self.tier {
1381            N3Tier::PcidPreserving => "n3b",
1382            N3Tier::FullIsolation => "n3c",
1383        }
1384    }
1385}
1386
1387impl IpcProducer for N3Transport {
1388    fn send(&self, msg: &[u8]) -> Result<(), IpcError> {
1389        // SAFETY: CAS state machine (Ready=>Active) provides logical exclusivity.
1390        // The &mut is scoped to this function; after CAS, only the ASM primitive
1391        // accesses the frame via raw pointers. Technically UB under Stacked Borrows,
1392        // but acceptable for a research kernel prototype where CAS guarantees
1393        // no concurrent access.
1394        let frame = unsafe { &mut *self.frame_ptr() };
1395
1396        // Get current task context.
1397        let receiver_task = get_task_by_id(self.receiver_task_id).ok_or(IpcError::Disconnected)?;
1398
1399        // Read current CR3.
1400        let (cr3_frame, _) = crate::x86_crate_shim::registers::control::Cr3::read();
1401        let sender_cr3 = cr3_frame.start_address().as_u64();
1402
1403        // Validate receiver's handler RIP.
1404        // Verify RIP matches the explicitly registered handler, not arbitrary code.
1405        let recv_rip = receiver_task.trampoline_entry.load(Ordering::Acquire);
1406        if recv_rip == 0 {
1407            return Err(IpcError::InvalidRip);
1408        }
1409        let recv_as = receiver_task.process.address_space_arc();
1410        validate_rip(recv_rip, &recv_as)?;
1411
1412        // Prepare the migration.
1413        let sender_rsp: u64;
1414        let sender_rip: u64;
1415        unsafe {
1416            core::arch::asm!("mov {}, rsp", out(reg) sender_rsp);
1417            // Capture return address from the stack. At this point in the
1418            // function, the return address is at [rbp + 8] when frame
1419            // pointers are enabled (kernel build default).
1420            //
1421            // SAFETY: kernel is compiled with frame pointers enabled
1422            // (-C force-frame-pointers=yes). If frame pointers are disabled,
1423            // this will read garbage : ensure the kernel build config is correct.
1424            core::arch::asm!("mov {}, [rbp + 8]", out(reg) sender_rip);
1425        }
1426
1427        n3_prepare_migration(
1428            frame,
1429            self.msg_buf.as_ptr(),
1430            sender_cr3,
1431            sender_rsp,
1432            sender_rip,
1433            self.handler_stack_top,
1434            &receiver_task,
1435            msg,
1436        )?;
1437
1438        // Inter-core sync if needed.
1439        // Spec §9.2: the sender must ensure message visibility before sending IPI.
1440        let target_cpu = receiver_task.home_cpu.load(Ordering::Acquire);
1441        let same_core = crate::arch::percpu::current_cpu_index() == target_cpu as usize;
1442
1443        if !same_core {
1444            if (target_cpu as usize) < crate::arch::percpu::MAX_CPUS {
1445                // Ensure all stores (message, frame metadata, state=Active)
1446                // are visible before the IPI is sent (spec §9.1).
1447                core::sync::atomic::fence(Ordering::Release);
1448
1449                // Store the VIRTUAL address : the IPI handler dereferences
1450                // it as a pointer in the kernel address space.
1451                N3_PENDING_MIGRATION[target_cpu].store(self.frame_virt, Ordering::Release);
1452
1453                // Send sync IPI. On x86-64, the ICR write is serializing,
1454                // which provides an implicit full barrier (spec §9.2).
1455                if let Some(target_apic) = crate::arch::percpu::apic_id_by_cpu_index(target_cpu) {
1456                    send_n3_sync_ipi(target_apic);
1457                }
1458            }
1459        }
1460
1461        // Execute the migration ASM primitive.
1462        // SAFETY: frame is valid, mapped in both AS, interrupts are enabled.
1463        unsafe {
1464            n3b_migrate_asm(self.frame_virt as *mut MigrationFrame);
1465        }
1466
1467        Ok(())
1468    }
1469
1470    fn try_send(&self, msg: &[u8]) -> Result<(), IpcError> {
1471        // The CAS inside n3_prepare_migration handles the state check atomically.
1472        // No need for a separate state check here (avoids TOCTOU).
1473        self.send(msg)
1474    }
1475}
1476
1477impl IpcConsumer for N3Transport {
1478    fn recv(&self, buf: &mut [u8]) -> Result<usize, IpcError> {
1479        let frame = self.frame();
1480
1481        // Verify the frame is in Active state.
1482        // Do NOT read from a Ready, Stalled, or Reclaiming frame.
1483        let state = frame.state.load(Ordering::Acquire);
1484        if state != MigrationState::Active as u8 {
1485            return Err(IpcError::WouldBlock);
1486        }
1487
1488        // Verify generation (spec §15.1).
1489        let gen = frame.generation.load(Ordering::Acquire);
1490        if gen == 0 {
1491            return Err(IpcError::WouldBlock);
1492        }
1493
1494        // Check message length.
1495        let msg_len = frame.msg_len;
1496        if msg_len == 0 {
1497            return Err(IpcError::WouldBlock);
1498        }
1499
1500        // Read the message from the shared buffer.
1501        // Use a guard to reset state=Ready even if the read fails,
1502        // preventing the frame from being stuck in Active forever.
1503        let n = match shared_msg_read(self.msg_buf.as_ptr(), msg_len, buf) {
1504            Ok(n) => n,
1505            Err(e) => {
1506                // Reset state so the frame can be reused.
1507                frame
1508                    .state
1509                    .store(MigrationState::Ready as u8, Ordering::Release);
1510                return Err(e);
1511            }
1512        };
1513
1514        // Signal completion: set state = Ready (spec §15 step 6).
1515        // This allows the sender (or next sender) to reuse the frame.
1516        // The sender's send() will CAS Ready=>Active to claim it.
1517        // NOTE: msg_len is intentionally NOT cleared : the next sender's
1518        // prepare_migration() overwrites it when writing the new message.
1519        frame
1520            .state
1521            .store(MigrationState::Ready as u8, Ordering::Release);
1522
1523        Ok(n)
1524    }
1525
1526    fn try_recv(&self, buf: &mut [u8]) -> Result<Option<usize>, IpcError> {
1527        match self.recv(buf) {
1528            Ok(n) => Ok(Some(n)),
1529            Err(IpcError::WouldBlock) => Ok(None),
1530            Err(e) => Err(e),
1531        }
1532    }
1533}
1534
1535// ============================================================================
1536// State machine helpers
1537// ============================================================================
1538
1539/// Check if a frame is in the Ready state.
1540pub fn n3_frame_is_ready(frame: &MigrationFrame) -> bool {
1541    frame.state.load(Ordering::Acquire) == MigrationState::Ready as u8
1542}
1543
1544/// Get the current migration state of a frame.
1545pub fn n3_frame_state(frame: &MigrationFrame) -> MigrationState {
1546    MigrationState::from_u8(frame.state.load(Ordering::Acquire)).unwrap_or(MigrationState::Ready)
1547}
1548
1549/// Get the current generation counter of a frame.
1550pub fn n3_frame_generation(frame: &MigrationFrame) -> u32 {
1551    frame.generation.load(Ordering::Acquire)
1552}