strat9_kernel/ipc/n3.rs
1//! N3 MMU Thread Migration : PCID-preserving (N3b) and Full Isolation (N3c).
2//!
3//! Implements the transport layer for IPC level N3
4//!
5//! # Architecture
6//!
7//! ```text
8//! sender ──send()──▶ MigrationFrame ──CR3 switch──▶ receiver handler
9//! ▲ │
10//! └────────────── reschedule IPI ◀─────────────────────┘
11//! ```
12//!
13//! The sender saves its context, copies the message, then calls the ASM
14//! primitive which switches CR3 and restores the receiver's context. The
15//! receiver handler reads the message from the frame and returns. The
16//! sender is resumed via a reschedule IPI.
17//!
18//! # Frame pointer requirement
19//!
20//! The `send()` method captures `sender_rip` from `[rbp + 8]`, which requires
21//! frame pointers to be enabled at compile time. The kernel must be built with
22//! `-C force-frame-pointers=yes` (see workspace/kernel/.cargo/config.toml).
23//!
24//! # Shared mapping note
25//!
26//! The MigrationFrame is mapped in both address spaces at the same virtual
27//! address (`N3_SHARED_FRAME_VA`), but **without USER_ACCESSIBLE**. The frame
28//! lives in kernel space and is accessible only from Ring 0. Both sender and
29//! receiver access it via their kernel page tables : the shared VA means the
30//! same PTE (same physical page) is reachable from both page table hierarchies.
31//! This is NOT a user-mappable region; it is kernel-only shared memory.
32//!
33//! # PCID contract
34//!
35//! A PCID value of 0 means "no PCID" : either PCID is unsupported by the CPU
36//! or exhausted. All code paths check `pcid > 0` before using PCID-specific
37//! features (INVPCID, CR3 PCID bits). The tier selection function
38//! (`select_n3_tier()`) distinguishes theoretical CPU support from actual
39//! operational capability.
40#![allow(dead_code)]
41
42use crate::{
43 arch::apic,
44 memory::{allocate_frame, free_frame, phys_to_virt, AddressSpace},
45 process::{current_task_id, get_task_by_id, Task, TaskId},
46 sync::{with_irqs_disabled, SpinLock},
47};
48use core::{
49 mem::MaybeUninit,
50 sync::atomic::{AtomicU16, AtomicU32, AtomicU64, AtomicU8, Ordering},
51};
52
53use super::transport::{
54 IpcConsumer, IpcError, IpcProducer, IpcTransport, TransportCapabilities, TransportLevel,
55};
56
57// ============================================================================
58// Normative types
59// ============================================================================
60
61/// N3 tier selection
62#[derive(Debug, Clone, Copy, PartialEq, Eq)]
63#[repr(u8)]
64pub enum N3Tier {
65 /// PCID-preserving: TLB entries for the sender AS are kept across CR3 switch.
66 PcidPreserving = 1,
67 /// Full isolation: CR3 write flushes all non-global TLB entries.
68 FullIsolation = 2,
69}
70
71/// Minimal CPU context saved/restored during N3 migration.
72///
73/// Layout is `#[repr(C, align(64))]` to match the ASM primitive offsets.
74/// No FPU/AVX state : handled separately per FPU policy.
75/// Total size: 96 bytes (fields) + 32 bytes (padding) = 128 bytes.
76#[repr(C, align(64))]
77#[derive(Debug, Clone, Copy)]
78pub struct N3MinimalContext {
79 /// Callee-saved GPRs (System V AMD64 ABI).
80 pub r15: u64,
81 pub r14: u64,
82 pub r13: u64,
83 pub r12: u64,
84 pub rbp: u64,
85 pub rbx: u64,
86 pub rdx: u64,
87 pub rax: u64,
88 /// Processor flags.
89 pub rflags: u64,
90 /// Stack pointer.
91 pub rsp: u64,
92 /// Instruction pointer (handler entry or return address).
93 pub rip: u64,
94 /// CR3 value with PCID bits for N3b, or 0 for N3c.
95 pub cr3_pcid: u64,
96 /// Explicit padding to reach 128 bytes (align(64) requires this).
97 /// Prevents implicit compiler padding and ensures deterministic layout
98 /// for the ASM primitive.
99 pub _pad: [u8; 32],
100}
101
102impl N3MinimalContext {
103 /// Zeroed context : used as initial value before preparation.
104 pub const ZERO: Self = Self {
105 r15: 0,
106 r14: 0,
107 r13: 0,
108 r12: 0,
109 rbp: 0,
110 rbx: 0,
111 rdx: 0,
112 rax: 0,
113 rflags: 0,
114 rsp: 0,
115 rip: 0,
116 cr3_pcid: 0,
117 _pad: [0u8; 32],
118 };
119}
120
121/// Migration state machine.
122#[derive(Debug, Clone, Copy, PartialEq, Eq)]
123#[repr(u8)]
124pub enum MigrationState {
125 /// No migration active; frame reusable.
126 Ready = 0,
127 /// Migration engaged; no second migration on this frame.
128 Active = 1,
129 /// Migration not finalized before timeout.
130 Stalled = 2,
131 /// Transient kernel state during recovery.
132 Reclaiming = 3,
133}
134
135impl MigrationState {
136 fn from_u8(v: u8) -> Option<Self> {
137 match v {
138 0 => Some(Self::Ready),
139 1 => Some(Self::Active),
140 2 => Some(Self::Stalled),
141 3 => Some(Self::Reclaiming),
142 _ => None,
143 }
144 }
145}
146
147bitflags::bitflags! {
148 /// Flags for a migration operation.
149 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
150 pub struct MigrationFlags: u16 {
151 /// Eager FPU save/restore (required for SIMD workloads).
152 const FPU_EAGER = 0x0001;
153 /// Lazy FPU save/restore (default, lower cost).
154 const FPU_LAZY = 0x0002;
155 /// Migration crosses CPU cores (requires IPI sync).
156 const INTER_CORE = 0x0004;
157 }
158}
159
160/// Shared migration frame : one per N3 transport endpoint pair.
161///
162/// Must reside in a dedicated physical page mapped at the same kernel
163/// virtual address (`N3_SHARED_FRAME_VA`) in both sender and receiver
164/// address spaces. The mapping is kernel-only (no `USER_ACCESSIBLE`
165/// flag) : all access is controlled by the kernel via the CAS state
166/// machine. Both address spaces see the same physical page through
167/// their own kernel half of the page tables.
168#[repr(C, align(64))]
169pub struct MigrationFrame {
170 /// Sender context (saved by ASM primitive).
171 pub src_ctx: N3MinimalContext,
172 /// Receiver context (prepared by kernel before migration).
173 pub dst_ctx: N3MinimalContext,
174 /// Atomic state: `MigrationState`.
175 pub state: AtomicU8,
176 _pad_state: [u8; 3],
177 /// Monotonically increasing migration counter.
178 pub generation: AtomicU32,
179 /// Message length in bytes.
180 pub msg_len: u16,
181 /// Migration flags.
182 pub flags: u16,
183 _pad_flags: [u8; 4],
184 /// TSC snapshot at migration start (for watchdog).
185 pub tsc_start: AtomicU64,
186 /// Thread ID of the frame owner.
187 pub owner_tid: AtomicU64,
188 /// Reserved padding for future use and cache-line alignment.
189 pub _pad: [u8; 32],
190}
191
192// Static assertion: N3 minimal context layout must be exactly 128 bytes.
193// This is verified both by the `const _` block and by the unit test.
194// The sender_rip capture in send() relies on [rbp+8] which requires
195// frame pointers (force-frame-pointers = true in workspace Cargo.toml).
196const _: () = {
197 assert!(core::mem::size_of::<N3MinimalContext>() == 128);
198 assert!(core::mem::align_of::<N3MinimalContext>() == 64);
199 assert!(core::mem::size_of::<MigrationFrame>() == 320);
200 assert!(core::mem::align_of::<MigrationFrame>() == 64);
201 assert!(core::mem::offset_of!(MigrationFrame, state) == 0x100);
202 assert!(core::mem::offset_of!(MigrationFrame, generation) == 0x104);
203 assert!(core::mem::offset_of!(MigrationFrame, msg_len) == 0x108);
204 assert!(core::mem::offset_of!(MigrationFrame, flags) == 0x10A);
205 assert!(core::mem::offset_of!(MigrationFrame, tsc_start) == 0x110);
206 assert!(core::mem::offset_of!(MigrationFrame, owner_tid) == 0x118);
207 // N3MinimalContext field offsets (for ASM verification).
208 assert!(core::mem::offset_of!(N3MinimalContext, r15) == 0x00);
209 assert!(core::mem::offset_of!(N3MinimalContext, r14) == 0x08);
210 assert!(core::mem::offset_of!(N3MinimalContext, r13) == 0x10);
211 assert!(core::mem::offset_of!(N3MinimalContext, r12) == 0x18);
212 assert!(core::mem::offset_of!(N3MinimalContext, rbp) == 0x20);
213 assert!(core::mem::offset_of!(N3MinimalContext, rbx) == 0x28);
214 assert!(core::mem::offset_of!(N3MinimalContext, rdx) == 0x30);
215 assert!(core::mem::offset_of!(N3MinimalContext, rax) == 0x38);
216 assert!(core::mem::offset_of!(N3MinimalContext, rflags) == 0x40);
217 assert!(core::mem::offset_of!(N3MinimalContext, rsp) == 0x48);
218 assert!(core::mem::offset_of!(N3MinimalContext, rip) == 0x50);
219 assert!(core::mem::offset_of!(N3MinimalContext, cr3_pcid) == 0x58);
220};
221
222// ============================================================================
223// Static frame pool
224// ============================================================================
225
226/// Number of pre-allocated migration frames.
227const N3_FRAME_POOL_SIZE: usize = 64;
228
229/// Virtual address where the MigrationFrame is mapped in both address spaces.
230/// Located in canonical upper-half, just below the HHDM boundary.
231pub const N3_SHARED_FRAME_VA: u64 = 0xFFFF_C000_0000_1000;
232
233/// Virtual address where the shared message buffer is mapped.
234/// Must be distinct from N3_SHARED_FRAME_VA to avoid page table conflicts.
235pub const N3_SHARED_MSG_BUF_VA: u64 = 0xFFFF_C000_0000_2000;
236
237/// Return safe kernel-mode rflags for N3 context switching.
238///
239/// Bit 1 (reserved) must always be 1. Bit 9 (IF) is set to keep interrupts
240/// enabled in the kernel handler. All other bits are cleared to avoid
241/// restoring user-modified flags (TF, AC, etc.) into kernel context.
242fn safe_kernel_rflags() -> u64 {
243 // Bit 1 = reserved (must be 1), Bit 9 = IF (interrupts enabled)
244 (1 << 1) | (1 << 9)
245}
246
247/// Size of the per-N3Transport handler stack (1 page).
248/// After CR3 switch, the ASM primitive loads RSP from `dst_ctx.rsp` which
249/// points to this stack with the handler address as the return address.
250const N3_HANDLER_STACK_SIZE: usize = 4096;
251
252/// Static pool of MigrationFrame slots.
253///
254/// Access is serialized by `N3_FRAME_ALLOC` SpinLock. The `UnsafeCell`
255/// avoids `static mut` UB while the lock guarantees exclusive access.
256struct FramePool {
257 inner: core::cell::UnsafeCell<[MaybeUninit<MigrationFrame>; N3_FRAME_POOL_SIZE]>,
258}
259
260// SAFETY: all access to inner is serialized by N3_FRAME_ALLOC SpinLock.
261unsafe impl Sync for FramePool {}
262
263impl FramePool {
264 const fn new() -> Self {
265 FramePool {
266 inner: core::cell::UnsafeCell::new(unsafe { MaybeUninit::uninit().assume_init() }),
267 }
268 }
269
270 fn get_ptr(&self, idx: usize) -> *mut MaybeUninit<MigrationFrame> {
271 unsafe { &mut (*self.inner.get())[idx] as *mut _ }
272 }
273
274 fn phys_addr(&self, idx: usize) -> u64 {
275 // SAFETY: The FramePool is a static array allocated in the kernel's
276 // BSS/data section. On x86_64 with the kernel model, statics live in
277 // the direct-map region or higher half. The HHDM offset converts the
278 // virtual address to a physical address.
279 //
280 // However, this assumes the pool is within the HHDM window. If the
281 // kernel is compiled with a different memory model, this subtraction
282 // yields a garbage physical address.
283 //
284 // To avoid this dependency, the prototype should ideally allocate
285 // frames via the physical frame allocator and map them, rather than
286 // using a static pool with HHDM arithmetic. For now, we assert that
287 // the pool address is above the HHDM base to catch model mismatches.
288 let ptr = self.get_ptr(idx) as u64;
289 let hhdm = crate::memory::hhdm_offset();
290 debug_assert!(
291 ptr >= hhdm,
292 "N3: FramePool not in HHDM region (ptr={:#x}, hhdm={:#x})",
293 ptr,
294 hhdm
295 );
296 ptr.wrapping_sub(hhdm)
297 }
298}
299
300static N3_FRAME_POOL: FramePool = FramePool::new();
301
302/// Bitmap-based O(1) frame allocator.
303struct FrameAllocator {
304 /// Bit i = 1 means frame i is allocated.
305 bitmap: [u64; N3_FRAME_POOL_SIZE / 64],
306 /// Number of currently allocated frames.
307 count: usize,
308}
309
310impl FrameAllocator {
311 const fn new() -> Self {
312 FrameAllocator {
313 bitmap: [0u64; N3_FRAME_POOL_SIZE / 64],
314 count: 0,
315 }
316 }
317
318 fn alloc(&mut self) -> Option<usize> {
319 for word_idx in 0..self.bitmap.len() {
320 if self.bitmap[word_idx] != u64::MAX {
321 let bit = self.bitmap[word_idx].trailing_ones() as usize;
322 self.bitmap[word_idx] |= 1 << bit;
323 self.count += 1;
324 return Some(word_idx * 64 + bit);
325 }
326 }
327 None
328 }
329
330 fn free(&mut self, index: usize) {
331 let word_idx = index / 64;
332 let bit = index % 64;
333 debug_assert!(self.bitmap[word_idx] & (1 << bit) != 0, "double free");
334 self.bitmap[word_idx] &= !(1 << bit);
335 self.count -= 1;
336 }
337
338 fn is_allocated(&self, index: usize) -> bool {
339 self.bitmap[index / 64] & (1 << (index % 64)) != 0
340 }
341}
342
343static N3_FRAME_ALLOC: SpinLock<FrameAllocator> = SpinLock::new(FrameAllocator::new());
344
345/// Allocate a MigrationFrame from the static pool.
346///
347/// Returns a pointer to the frame and its pool index.
348fn alloc_frame_slot() -> Option<(usize, *mut MigrationFrame)> {
349 let idx = with_irqs_disabled(|_token| {
350 let mut alloc = N3_FRAME_ALLOC.lock();
351 alloc.alloc()
352 })?;
353
354 // SAFETY: slot is freshly allocated via bitmap, exclusive access via SpinLock.
355 let frame = N3_FRAME_POOL.get_ptr(idx);
356 // SAFETY: slot is freshly allocated, MaybeUninit is writable.
357 unsafe {
358 frame.write(MaybeUninit::new(MigrationFrame {
359 src_ctx: N3MinimalContext::ZERO,
360 dst_ctx: N3MinimalContext::ZERO,
361 state: AtomicU8::new(MigrationState::Ready as u8),
362 _pad_state: [0; 3],
363 generation: AtomicU32::new(0),
364 msg_len: 0,
365 flags: 0,
366 _pad_flags: [0; 4],
367 tsc_start: AtomicU64::new(0),
368 owner_tid: AtomicU64::new(0),
369 _pad: [0u8; 32],
370 }));
371 }
372
373 Some((idx, frame as *mut MigrationFrame))
374}
375
376/// Free a MigrationFrame back to the pool.
377fn free_frame_slot(idx: usize) {
378 with_irqs_disabled(|_token| {
379 let mut alloc = N3_FRAME_ALLOC.lock();
380 alloc.free(idx);
381 });
382}
383
384/// Get the physical address of a frame in the static pool.
385fn frame_pool_phys_addr(idx: usize) -> u64 {
386 N3_FRAME_POOL.phys_addr(idx)
387}
388
389// ============================================================================
390// PCID management
391// ============================================================================
392
393/// Global PCID counter : monotonic allocation, panics at 4096 (prototype).
394static PCID_COUNTER: AtomicU16 = AtomicU16::new(1); // 0 = no PCID
395
396/// Maximum number of PCIDs before panic (x86-64 supports 4096).
397const MAX_PCIDS: u16 = 4095;
398
399/// Allocate a stable PCID for an address space.
400///
401/// Returns 0 if PCID is not available (N3c fallback).
402pub fn allocate_pcid() -> u16 {
403 let pcid = PCID_COUNTER.fetch_add(1, Ordering::Relaxed);
404 if pcid >= MAX_PCIDS {
405 log::warn!("N3: PCID exhaustion : falling back to N3c (full TLB flush)");
406 return 0;
407 }
408 pcid
409}
410
411/// Free a PCID back to the pool (called when an address space is destroyed).
412///
413/// This prevents PCID exhaustion in long-running systems with dynamic silo
414/// creation/destruction. The freed PCID is tracked for reuse, but the actual
415/// TLB invalidation on other cores is the caller's responsibility.
416pub fn free_pcid(pcid: u16) {
417 if pcid == 0 || pcid >= MAX_PCIDS {
418 return;
419 }
420 // NB: PCID reuse requires INVPCID on the target core before the next
421 // time this PCID is assigned. For the prototype, we simply log the free;
422 // a production implementation would use a freelist + generation counter
423 // (see Linux arch/x86/mm/context.c).
424 log::trace!("N3: PCID {} freed", pcid);
425}
426
427/// Check if PCID feature is available on this CPU.
428pub fn pcid_available() -> bool {
429 // 1. CPUID check: bit 17 of ECX after CPUID(EAX=1) indicates PCID support.
430 let (_, _, ecx, _) = crate::arch::cpuid(1, 0);
431 if ecx & (1 << 17) == 0 {
432 return false;
433 }
434 // 2. Check CR4.PCIDE (bit 17) is actually set (may be masked by hypervisor).
435 // Intel SDM Vol.3A §4.10.4: CR4.PCIDE = 1 enables PCID in CR3.
436 let cr4 = crate::x86_crate_shim::registers::control::Cr4::read();
437 cr4.contains(crate::x86_crate_shim::registers::control::Cr4Flags::PCID)
438}
439
440/// Select the N3 tier based on actual PCID capability.
441///
442/// Distinguishes theoretical CPU support from operational capability:
443/// - `N3Tier::PcidPreserving` requires PCID available AND a valid PCID > 0
444/// - `N3Tier::FullIsolation` is used when PCID is unavailable or exhausted
445pub fn select_n3_tier(pcid: u16) -> N3Tier {
446 if pcid_available() && pcid > 0 {
447 N3Tier::PcidPreserving
448 } else {
449 N3Tier::FullIsolation
450 }
451}
452
453// ============================================================================
454// validate_rip()
455// ============================================================================
456
457/// Validate that `rip` points to an authorized executable page.
458///
459/// Checks:
460/// - Belongs to an executable VMA
461/// - PTE has NX=0
462/// - Not in kernel ring, MigrationFrame, or non-executable zones
463/// - Part of a registered handler or authorized code range
464fn validate_rip(rip: u64, address_space: &AddressSpace) -> Result<(), IpcError> {
465 // Kernel addresses (upper half) are not valid user RIP targets.
466 if rip >= 0xFFFF_8000_0000_0000 {
467 return Err(IpcError::InvalidRip);
468 }
469
470 // MigrationFrame shared region is not executable.
471 if (N3_SHARED_FRAME_VA..N3_SHARED_FRAME_VA + 0x1000).contains(&rip) {
472 return Err(IpcError::InvalidRip);
473 }
474
475 // Null or near-null RIP is invalid.
476 if rip < 0x1000 {
477 return Err(IpcError::InvalidRip);
478 }
479
480 // validate_rip() performs page-level safety checks only: present, executable,
481 // not in forbidden zones. The RIP to belong to
482 // a "registered handler or authorized code range" : this registration check
483 // is performed by the CALLER (send()) which compares rip against the
484 // receiver's `trampoline_entry` field BEFORE calling validate_rip().
485 //
486 // validate_rip() is therefore a SECOND LINE OF DEFENSE that catches
487 // configuration errors (e.g., trampoline_entry pointing to a data page).
488 //
489 // TOCTOU note: The receiver's page tables could be modified between this
490 // check and the CR3 switch. However, the receiver's trampoline page is
491 // pinned and set to read-only by the kernel at endpoint creation time,
492 // preventing modification even by the receiver itself. The CR3 switch
493 // immediately after this check makes any concurrent modification irrelevant
494 // : the receiver executes in its own AS, using the mapping we validated.
495
496 // Walk the page tables to verify the page is present and executable.
497 // CRITICAL: Also check the U/S bit : the RIP target must be a supervisor
498 // page (Ring 0), not a user page. Accepting a user page would allow a
499 // userspace task to redirect kernel execution to arbitrary user code.
500 let cr3_phys = address_space.cr3();
501 let hhdm = crate::memory::hhdm_offset();
502
503 match unsafe { walk_page_tables_executable(rip, cr3_phys.as_u64(), hhdm) } {
504 Ok(true) => Ok(()),
505 Ok(false) => Err(IpcError::InvalidRip),
506 Err(_) => Err(IpcError::InvalidRip),
507 }
508}
509
510/// Walk x86-64 4-level page tables to check if `vaddr` is present, executable,
511/// and supervisor-only (U/S=0).
512///
513/// Returns `Ok(true)` if the final PTE meets all conditions.
514/// Rejects user-mode pages (U/S=1) as RIP targets for migration.
515unsafe fn walk_page_tables_executable(vaddr: u64, cr3_phys: u64, hhdm: u64) -> Result<bool, ()> {
516 let pml4 = (cr3_phys + hhdm) as *const u64;
517
518 // PML4: check present + U/S (bit 2: 0=supervisor, 1=user)
519 let pml4_idx = ((vaddr >> 39) & 0x1FF) as usize;
520 let pml4_entry = unsafe { core::ptr::read_volatile(pml4.add(pml4_idx)) };
521 if pml4_entry & 1 == 0 {
522 return Err(());
523 }
524 if pml4_entry & 4 != 0 {
525 return Err(());
526 } // U/S=1 => reject
527
528 // PDPT: check present + U/S
529 let pdpt_phys = pml4_entry & 0x000F_FFFF_FFFF_F000;
530 let pdpt = (pdpt_phys + hhdm) as *const u64;
531 let pdpt_idx = ((vaddr >> 30) & 0x1FF) as usize;
532 let pdpt_entry = unsafe { core::ptr::read_volatile(pdpt.add(pdpt_idx)) };
533 if pdpt_entry & 1 == 0 {
534 return Err(());
535 }
536 if pdpt_entry & 4 != 0 {
537 return Err(());
538 } // U/S=1 => reject
539
540 // 1 GiB page? Check NX + U/S
541 if pdpt_entry & (1 << 7) != 0 {
542 if pdpt_entry & (1 << 63) != 0 {
543 return Err(());
544 } // NX=1 => reject
545 return Ok(true);
546 }
547
548 // Page Directory: check present + U/S
549 let pd_phys = pdpt_entry & 0x000F_FFFF_FFFF_F000;
550 let pd = (pd_phys + hhdm) as *const u64;
551 let pd_idx = ((vaddr >> 21) & 0x1FF) as usize;
552 let pd_entry = unsafe { core::ptr::read_volatile(pd.add(pd_idx)) };
553 if pd_entry & 1 == 0 {
554 return Err(());
555 }
556 if pd_entry & 4 != 0 {
557 return Err(());
558 } // U/S=1 => reject
559
560 // 2 MiB page? Check NX
561 if pd_entry & (1 << 7) != 0 {
562 if pd_entry & (1 << 63) != 0 {
563 return Err(());
564 } // NX=1 => reject
565 return Ok(true);
566 }
567
568 // Page Table: check present + U/S + NX
569 let pt_phys = pd_entry & 0x000F_FFFF_FFFF_F000;
570 let pt = (pt_phys + hhdm) as *const u64;
571 let pt_idx = ((vaddr >> 12) & 0x1FF) as usize;
572 let pt_entry = unsafe { core::ptr::read_volatile(pt.add(pt_idx)) };
573 if pt_entry & 1 == 0 {
574 return Err(());
575 }
576 if pt_entry & 4 != 0 {
577 return Err(());
578 } // U/S=1 => reject user page
579 if pt_entry & (1 << 63) != 0 {
580 return Err(());
581 } // NX=1 => reject
582
583 Ok(true)
584}
585
586// ============================================================================
587// Context preparation
588// ============================================================================
589
590/// Prepare a MigrationFrame for a send operation.
591///
592/// Sets up `src_ctx` (sender), `dst_ctx` (receiver), copies the message
593/// into the shared buffer, increments generation, and transitions state
594/// to Active.
595fn n3_prepare_migration(
596 frame: &mut MigrationFrame,
597 msg_buf: *mut u8,
598 sender_cr3: u64,
599 sender_rsp: u64,
600 sender_rip: u64,
601 handler_stack_top: u64,
602 receiver_task: &Task,
603 msg: &[u8],
604) -> Result<(), IpcError> {
605 // Prepare sender context (saved by ASM primitive).
606 frame.src_ctx.cr3_pcid = sender_cr3;
607 frame.src_ctx.rsp = sender_rsp;
608 frame.src_ctx.rip = sender_rip;
609 // P0 fix: initialize rflags with safe kernel defaults.
610 // A zero rflags would clear IF (interrupts) and other critical bits.
611 frame.src_ctx.rflags = safe_kernel_rflags();
612
613 // Prepare receiver context.
614 let recv_handler = receiver_task.trampoline_entry.load(Ordering::Acquire);
615
616 if recv_handler == 0 {
617 return Err(IpcError::TransportFailed);
618 }
619
620 let recv_cr3 = receiver_task.process.address_space_arc().cr3().as_u64();
621
622 frame.dst_ctx.rip = recv_handler;
623 frame.dst_ctx.cr3_pcid = recv_cr3;
624 // P0 fix: destination context also gets safe kernel rflags.
625 frame.dst_ctx.rflags = safe_kernel_rflags();
626
627 // CRITICAL: The destination kernel stack must contain the handler address
628 // as the return address for the `ret` instruction after CR3 switch.
629 // We use the per-transport handler stack (kernel upper half, shared across
630 // all address spaces) to ensure accessibility after CR3 switch.
631 // Layout: [handler_stack_top - 8] = recv_handler (return address for ret).
632 //
633 // SAFETY: handler_stack_top points to the transport's dedicated stack page,
634 // allocated in new(). The CAS state machine (Ready=>Active) ensures exclusive
635 // access. The -8 offset places the handler address as if it were pushed.
636 let trampoline_top = (handler_stack_top - 8) as *mut u64;
637 unsafe {
638 // Write handler address at the trampoline top (this is what `ret` will pop).
639 core::ptr::write_volatile(trampoline_top, recv_handler);
640 }
641 // dst_ctx.rsp points to the trampoline area where the handler address is stored.
642 // After CR3 switch, `mov rsp, [rdi+0xC8]` loads this, and `ret` pops recv_handler.
643 frame.dst_ctx.rsp = trampoline_top as u64;
644
645 // Copy message into the shared buffer.
646 let msg_len = shared_msg_write(msg_buf, msg)?;
647 frame.msg_len = msg_len;
648
649 // CAS Ready => Active. Must succeed before touching generation.
650 if frame
651 .state
652 .compare_exchange(
653 MigrationState::Ready as u8,
654 MigrationState::Active as u8,
655 Ordering::AcqRel,
656 Ordering::Acquire,
657 )
658 .is_err()
659 {
660 return Err(IpcError::WouldBlock);
661 }
662
663 // Increment generation AFTER successful CAS.
664 // The generation must only advance for committed migrations.
665 frame.generation.fetch_add(1, Ordering::Release);
666
667 // Record timestamp and owner.
668 let tsc = unsafe { crate::arch::rdtsc() };
669 frame.tsc_start.store(tsc, Ordering::Release);
670 if let Some(tid) = current_task_id() {
671 frame.owner_tid.store(tid.as_u64(), Ordering::Release);
672 }
673
674 Ok(())
675}
676
677/// Maximum message size for N3 transport (separate buffer, not in frame).
678pub const N3_MSG_BUF_SIZE: usize = 2048;
679
680/// Wrapper for a raw pointer to the shared message buffer.
681///
682/// # Safety
683/// The pointer is valid for the lifetime of the N3Transport and is mapped
684/// in both sender and receiver address spaces. All access is serialized
685/// by the MigrationFrame state machine.
686struct MsgBuffer(*mut u8);
687
688// SAFETY: MsgBuffer is used only from a single migration context at a time,
689// protected by the MigrationFrame CAS state machine.
690unsafe impl Send for MsgBuffer {}
691unsafe impl Sync for MsgBuffer {}
692
693impl MsgBuffer {
694 fn as_ptr(&self) -> *mut u8 {
695 self.0
696 }
697}
698
699/// Read the message payload from a shared message buffer.
700fn shared_msg_read(msg_buf: *const u8, msg_len: u16, buf: &mut [u8]) -> Result<usize, IpcError> {
701 let len = msg_len as usize;
702 if len > buf.len() {
703 return Err(IpcError::BufferTooSmall);
704 }
705 if len > N3_MSG_BUF_SIZE {
706 return Err(IpcError::MessageTooLarge);
707 }
708 let src = unsafe { core::slice::from_raw_parts(msg_buf, len) };
709 buf[..len].copy_from_slice(src);
710 Ok(len)
711}
712
713/// Write a message payload into a shared message buffer.
714fn shared_msg_write(msg_buf: *mut u8, msg: &[u8]) -> Result<u16, IpcError> {
715 if msg.len() > N3_MSG_BUF_SIZE {
716 return Err(IpcError::MessageTooLarge);
717 }
718 let dst = unsafe { core::slice::from_raw_parts_mut(msg_buf, msg.len()) };
719 dst.copy_from_slice(msg);
720 Ok(msg.len() as u16)
721}
722
723// ============================================================================
724// ASM Primitive
725// ============================================================================
726
727/// ASM migration primitive.
728///
729/// # Safety
730/// - `frame` must point to a valid MigrationFrame mapped at the same VA
731/// in both sender and receiver address spaces.
732/// - Interrupts must be enabled before calling (the primitive will CLI).
733/// - The caller must be on a valid kernel stack.
734///
735/// # Layout adjusted for a 64b align padding
736///
737/// ```text
738/// Offset Field
739/// 0x00 src_ctx.r15
740/// 0x08 src_ctx.r14
741/// 0x10 src_ctx.r13
742/// 0x18 src_ctx.r12
743/// 0x20 src_ctx.rbp
744/// 0x28 src_ctx.rbx
745/// 0x30 src_ctx.rdx
746/// 0x38 src_ctx.rax
747/// 0x40 src_ctx.rflags
748/// 0x48 src_ctx.rsp
749/// 0x50 src_ctx.rip
750/// 0x58 src_ctx.cr3_pcid
751/// 0x60-0x7F (alignment padding)
752/// 0x80 dst_ctx.r15
753/// 0x88 dst_ctx.r14
754/// 0x90 dst_ctx.r13
755/// 0x98 dst_ctx.r12
756/// 0xA0 dst_ctx.rbp
757/// 0xA8 dst_ctx.rbx
758/// 0xB0 dst_ctx.rdx
759/// 0xB8 dst_ctx.rax
760/// 0xC0 dst_ctx.rflags
761/// 0xC8 dst_ctx.rsp
762/// 0xD0 dst_ctx.rip
763/// 0xD8 dst_ctx.cr3_pcid
764/// 0xE0-0xFF (alignment padding)
765/// 0x100 state (AtomicU8)
766/// 0x104 generation (AtomicU32)
767/// ```
768///
769/// # Behavior
770/// 1. Saves source context to `frame.src_ctx`
771/// 2. CLI (≤3 instructions) + CR3 switch to `frame.dst_ctx.cr3_pcid`
772/// 3. Restores destination context from `frame.dst_ctx`
773/// 4. Sets `frame.state = Ready`, increments `generation`
774/// 5. Returns via `ret` (pops RIP from restored `dst_ctx.rsp`)
775#[unsafe(naked)]
776pub unsafe extern "C" fn n3b_migrate_asm(frame: *mut MigrationFrame) {
777 core::arch::naked_asm!(
778 // ── Phase 1: save src_ctx ──
779 "mov [rdi + 0x00], r15",
780 "mov [rdi + 0x08], r14",
781 "mov [rdi + 0x10], r13",
782 "mov [rdi + 0x18], r12",
783 "mov [rdi + 0x20], rbp",
784 "mov [rdi + 0x28], rbx",
785 "mov [rdi + 0x30], rdx",
786 "mov [rdi + 0x38], rax",
787 "pushfq",
788 "pop rax",
789 "mov [rdi + 0x40], rax", // rflags
790 "mov [rdi + 0x48], rsp",
791 // NOTE: src_ctx.rip was already set by n3_prepare_migration().
792 // We do NOT overwrite it with RCX : on Strategy A, the kernel
793 // sets src_ctx.rip before calling us. See §11.2 for the contract.
794 // ── Phase 2: CLI (<3 instr) + CR3 switch ──
795 // We need a 3 instructions **MAX** between CLI and CR3 write.
796 // Here: mov rax (1) => cli (2) => mov cr3 (3). Satisfied.
797 // After CR3 switch, we restore dst_ctx with interrupts still disabled.
798 // The 8 restore instructions + pushfq/popfq/sti happen AFTER the CR3
799 // switch in the receiver's address space : this is intentional and does
800 // not violate the ≤3 instruction window (which only covers CLI=>CR3).
801 "mov rax, [rdi + 0xD8]", // dst_ctx.cr3_pcid (at 0x80+0x58=0xD8)
802 "cli",
803 "mov cr3, rax",
804 // ── Phase 3: restore dst_ctx (starts at offset 0x80) ──
805 // IMPORTANT: rsp (offset 0xC8) is restored LAST, just before ret.
806 // Every memory access between CR3 switch and rsp restore uses
807 // RDI (shared fixed-VA, valid in both AS) : never RSP.
808 "mov r15, [rdi + 0x80]",
809 "mov r14, [rdi + 0x88]",
810 "mov r13, [rdi + 0x90]",
811 "mov r12, [rdi + 0x98]",
812 "mov rbp, [rdi + 0xA0]",
813 "mov rbx, [rdi + 0xA8]",
814 "mov rdx, [rdi + 0xB0]",
815 // Load rflags from dst_ctx.rflags at offset 0x80+0x40 = 0xC0.
816 // Note: dst_ctx.rax (0xB8) is a GPR, NOT flags : do not confuse.
817 "mov rax, [rdi + 0xC0]", // dst_ctx.rflags
818 "push rax",
819 "popfq", // restore IF=1
820 "sti", // guarantee IF=1
821 // ── Phase 4: restore RSP from dst_ctx ──
822 // CRITICAL: RSP must be restored BEFORE `ret`. The destination kernel
823 // stack is shared (kernel upper half maps identically in all AS).
824 // dst_ctx.rsp was set by n3_prepare_migration() to point to a stack
825 // frame containing the handler entry address as the return address.
826 "mov rsp, [rdi + 0xC8]", // dst_ctx.rsp
827 // ── Phase 5: receiver handler runs (state is still Active) ──
828 // The ASM does NOT set state = Ready. The receiver handler MUST
829 // do that after consuming the message (see recv() protocol).
830 // This prevents the TOCTOU where the receiver sees state=Ready
831 // before having read the message. Generation is managed by the
832 // kernel in n3_prepare_migration/recv : NOT by the ASM.
833 // ── Phase 6: ret into receiver handler ──
834 // SAFETY: RSP now points to the destination kernel stack with the
835 // handler address at [RSP]. `ret` pops it and jumps there.
836 // The destination kernel stack must be valid and mapped in both AS
837 // (kernel upper half is shared). If corrupted, this is the first
838 // point of failure.
839 "ret",
840 //
841 // Invariant: the restored dst_ctx.rsp must point to a valid, readable,
842 // kernel-mode stack frame.
843 );
844}
845
846// ============================================================================
847// IPI synchronization
848// ============================================================================
849
850/// Per-CPU pending migration frame pointer.
851/// Set by the sender before sending the IPI, consumed by the IPI handler.
852static N3_PENDING_MIGRATION: [AtomicU64; crate::arch::percpu::MAX_CPUS] =
853 [const { AtomicU64::new(0) }; crate::arch::percpu::MAX_CPUS];
854
855/// Send a migration-sync IPI to the target CPU.
856fn send_n3_sync_ipi(target_apic_id: u32) {
857 let icr_low = apic::IPI_N3_MIGRATE_VECTOR as u32 | (1 << 14);
858 apic::send_ipi_raw(target_apic_id, icr_low);
859}
860
861/// Naked IPI handler for N3 migration synchronization.
862///
863/// When a cross-core migration completes, the receiver sends this IPI to the
864/// sender's core. The handler restores the sender's kernel context (RSP, RIP,
865/// RFLAGS) from `MigrationFrame.src_ctx` and returns via `iretq`.
866///
867/// The interrupted context was the sender's `send()` function. By modifying
868/// the InterruptStackFrame fields before the hardware `iretq`, we redirect
869/// execution to the sender's saved return address on its original kernel stack.
870///
871/// # Safety
872/// Registered as a naked IDT entry. On entry: SS, RSP, RFLAGS, CS, RIP pushed
873/// by hardware. `rdi` points at the InterruptStackFrame (set by IDT dispatch).
874#[unsafe(naked)]
875pub unsafe extern "C" fn n3_migrate_ipi_entry(rdi: *mut u8) -> ! {
876 core::arch::naked_asm!(
877 // ── Save scratch registers ──
878 "push rax",
879 "push rcx",
880 "push rdx",
881 "push rsi",
882 "push r8",
883 "push r9",
884 "push r10",
885 "push r11",
886
887 // ── Check pending migration ──
888 // per_cpu_index via GS:base (offset 0 in PerCpu struct).
889 "mov rcx, qword ptr gs:[0]", // current_cpu_index
890 "lea rsi, [rip + {pending}]", // &N3_PENDING_MIGRATION
891 "mov rax, [rsi + rcx * 8]", // pending = N3_PENDING_MIGRATION[cpu]
892 "test rax, rax",
893 "jz .no_pending",
894
895 // ── Clear pending flag ──
896 "mov qword ptr [rsi + rcx * 8], 0",
897
898 // ── rax = MigrationFrame ptr ──
899 // src_ctx starts at offset 0x00.
900 // src_ctx.rflags = 0x40, src_ctx.rsp = 0x48, src_ctx.rip = 0x50.
901
902 // ── Switch to sender's address space ──
903 "mov rcx, [rax + 0x58]", // src_ctx.cr3_pcid
904 "test rcx, rcx",
905 "jz .skip_cr3",
906 "mov r11, 0x0000FFFFFFFFFFFF",
907 "and rcx, r11", // clear high 16 bits (reserved/PCID)
908 "mov r11, 0xFFFFFFFFFFFFF000",
909 "and rcx, r11", // clear low 12 bits (page offset)
910 "mov cr3, rcx",
911 ".skip_cr3:",
912
913 // ── Build iretq frame on the interrupted stack ──
914 // rdi points to InterruptStackFrame pushed by hardware.
915 // Overwrite it with the sender's context.
916 // InterruptStackFrame layout:
917 // [+0x00] SS [+0x08] RSP [+0x10] RFLAGS
918 // [+0x18] CS [+0x20] RIP [+0x28] (error code, skipped)
919 //
920 // iretq pops: RIP, CS, RFLAGS, RSP, SS (5 × u64 = 40 bytes).
921 //
922 // Preserve original CS and SS from the interrupted frame.
923 // Overwrite RSP, RFLAGS, RIP with sender's values.
924
925 "mov rcx, [rdi + 0x18]", // original CS (kernel CS = 0x08)
926 "mov rdx, [rdi + 0x00]", // original SS (0 for Ring 0)
927
928 "mov rsi, [rax + 0x48]", // src_ctx.rsp => sender's kernel RSP
929 "mov r8, [rax + 0x40]", // src_ctx.rflags
930 "mov r9, [rax + 0x50]", // src_ctx.rip => sender's return address
931
932 "mov [rdi + 0x00], rdx", // SS = original (Ring 0)
933 "mov [rdi + 0x08], rsi", // RSP = sender's kernel stack
934 "mov [rdi + 0x10], r8", // RFLAGS = sender's flags
935 "mov [rdi + 0x18], rcx", // CS = original (0x08)
936 "mov [rdi + 0x20], r9", // RIP = sender's return address
937
938 // ── Send EOI ──
939 "mov dword ptr [0x0000FEE000B0], 0", // LAPIC EOI
940
941 // ── Restore scratch and return ──
942 // Pop scratch registers, then let the normal IDT return path
943 // (pop GPRs + iretq) restore from the modified frame.
944 "pop r11",
945 "pop r10",
946 "pop r9",
947 "pop r8",
948 "pop rsi",
949 "pop rdx",
950 "pop rcx",
951 "pop rax",
952 // Return to IDT stub which pops GPRs and does iretq.
953 "ret",
954
955 ".no_pending:",
956 // Spurious IPI : send EOI and return to interrupted context.
957 "mov dword ptr [0x0000FEE000B0], 0",
958 "pop r11",
959 "pop r10",
960 "pop r9",
961 "pop r8",
962 "pop rsi",
963 "pop rdx",
964 "pop rcx",
965 "pop rax",
966 "ret",
967
968 pending = sym N3_PENDING_MIGRATION,
969 );
970}
971
972/// Fallback handler for N3 migration IPI (non-naked path).
973///
974/// Called from the `extern "x86-interrupt"` stub if the naked entry is not
975/// used. Clears the pending flag and sends EOI.
976pub extern "C" fn n3_migrate_ipi_handler() {
977 let cpu_idx = crate::arch::percpu::current_cpu_index();
978 N3_PENDING_MIGRATION[cpu_idx].store(0, Ordering::Release);
979 apic::eoi();
980}
981
982// ============================================================================
983// WATCHDOG
984// ============================================================================
985
986/// Default watchdog timeout in TSC cycles (~10ms at 3GHz).
987const N3_WATCHDOG_TIMEOUT: u64 = 30_000_000;
988
989/// Watchdog state per frame.
990struct WatchdogEntry {
991 frame_phys: u64,
992 timeout: u64,
993}
994
995/// Global watchdog table.
996static N3_WATCHDOG_TABLE: SpinLock<[Option<WatchdogEntry>; N3_FRAME_POOL_SIZE]> =
997 SpinLock::new([const { None }; N3_FRAME_POOL_SIZE]);
998
999/// Register a frame with the watchdog.
1000fn watchdog_register(frame_idx: usize, frame_phys: u64) {
1001 let mut table = N3_WATCHDOG_TABLE.lock();
1002 if let Some(entry) = table.get_mut(frame_idx) {
1003 *entry = Some(WatchdogEntry {
1004 frame_phys,
1005 timeout: N3_WATCHDOG_TIMEOUT,
1006 });
1007 }
1008}
1009
1010/// Unregister a frame from the watchdog.
1011fn watchdog_unregister(frame_idx: usize) {
1012 let mut table = N3_WATCHDOG_TABLE.lock();
1013 if let Some(entry) = table.get_mut(frame_idx) {
1014 *entry = None;
1015 }
1016}
1017
1018/// Called from the timer ISR to check for stalled migrations.
1019///
1020/// For each registered frame in Active state, checks whether the migration
1021/// has exceeded the watchdog timeout. If so:
1022/// 1. Transitions Active => Stalled (CAS)
1023/// 2. Attempts recovery: Stalled => Reclaiming => Ready
1024/// 3. Logs the incident
1025///
1026/// Recovery resets the frame so it can be reused for future migrations.
1027/// The sender that initiated the stalled migration will see `WouldBlock` on
1028/// its next `send()` attempt (CAS Ready=>Active fails if another sender
1029/// claimed the frame first).
1030pub fn n3_watchdog_tick() {
1031 let now = unsafe { crate::arch::rdtsc() };
1032 let mut table = N3_WATCHDOG_TABLE.lock();
1033
1034 for entry in table.iter_mut() {
1035 let Some(entry) = entry else { continue };
1036 let frame = unsafe { &*(phys_to_virt(entry.frame_phys) as *const MigrationFrame) };
1037 let state = frame.state.load(Ordering::Acquire);
1038
1039 if state != MigrationState::Active as u8 {
1040 continue;
1041 }
1042
1043 let tsc_start = frame.tsc_start.load(Ordering::Acquire);
1044 if now.wrapping_sub(tsc_start) <= entry.timeout {
1045 continue;
1046 }
1047
1048 // Timeout detected: attempt Active => Stalled => Reclaiming => Ready.
1049 let cas_result = frame.state.compare_exchange(
1050 MigrationState::Active as u8,
1051 MigrationState::Stalled as u8,
1052 Ordering::AcqRel,
1053 Ordering::Acquire,
1054 );
1055
1056 if cas_result.is_err() {
1057 // Another CPU already transitioned the frame (e.g., recv completed).
1058 continue;
1059 }
1060
1061 log::warn!(
1062 "N3: watchdog timeout on frame at {:#x}, generation={}, recovering",
1063 entry.frame_phys,
1064 frame.generation.load(Ordering::Acquire),
1065 );
1066
1067 // Attempt recovery: Stalled => Reclaiming => Ready.
1068 watchdog_recover(frame);
1069 }
1070}
1071
1072/// Recover a stalled frame : reset to Ready via atomic CAS transitions.
1073fn watchdog_recover(frame: &MigrationFrame) {
1074 // Transition Stalled => Reclaiming (atomic CAS).
1075 if frame
1076 .state
1077 .compare_exchange(
1078 MigrationState::Stalled as u8,
1079 MigrationState::Reclaiming as u8,
1080 Ordering::AcqRel,
1081 Ordering::Acquire,
1082 )
1083 .is_err()
1084 {
1085 log::warn!("N3: watchdog recover: frame not in Stalled state, skipping");
1086 return;
1087 }
1088 log::warn!(
1089 "N3: recovering stalled frame, generation={}",
1090 frame.generation.load(Ordering::Acquire)
1091 );
1092 // Transition Reclaiming => Ready (atomic store, no contention expected).
1093 frame
1094 .state
1095 .store(MigrationState::Ready as u8, Ordering::Release);
1096}
1097
1098// ============================================================================
1099// Shared frame mapping
1100// ============================================================================
1101
1102/// Map a physical page into an address space at the specified virtual address.
1103///
1104/// The mapping is kernel-only (no `USER_ACCESSIBLE` flag) to enforce the
1105/// spec §8.4 invariant: frame permissions are managed exclusively by the
1106/// kernel, never by non-kernel parties.
1107fn map_page_in_space(
1108 frame_phys: u64,
1109 target_va: u64,
1110 address_space: &AddressSpace,
1111) -> Result<(), &'static str> {
1112 use crate::arch::xshim::{PageTableFlags, PhysFrame, Size4KiB};
1113 use crate::arch::xshim::{PhysAddr, VirtAddr};
1114 use crate::x86_crate_shim::structures::paging::{Mapper, Page};
1115
1116 let page = Page::<Size4KiB>::containing_address(VirtAddr::new(target_va));
1117 let phys_frame = PhysFrame::<Size4KiB>::containing_address(PhysAddr::new(frame_phys));
1118
1119 // Kernel-only, no USER_ACCESSIBLE (spec §8.4: permissions managed by kernel).
1120 let flags = PageTableFlags::PRESENT | PageTableFlags::WRITABLE;
1121
1122 let mut mapper = unsafe { address_space.mapper() };
1123 let mut allocator = crate::memory::paging::BuddyFrameAllocator;
1124
1125 unsafe {
1126 match mapper.map_to(page, phys_frame, flags, &mut allocator) {
1127 Ok(flush) => {
1128 flush.flush();
1129 Ok(())
1130 }
1131 Err(_) => Err("Failed to map page into address space"),
1132 }
1133 }
1134}
1135
1136/// Map a MigrationFrame's physical page into both sender and receiver
1137/// address spaces at `N3_SHARED_FRAME_VA`.
1138fn map_frame_in_both_spaces(
1139 frame_phys: u64,
1140 sender_as: &AddressSpace,
1141 receiver_as: &AddressSpace,
1142) -> Result<(), &'static str> {
1143 map_page_in_space(frame_phys, N3_SHARED_FRAME_VA, sender_as)?;
1144 map_page_in_space(frame_phys, N3_SHARED_FRAME_VA, receiver_as)?;
1145 Ok(())
1146}
1147
1148/// Map a message buffer's physical page into both address spaces
1149/// at `N3_SHARED_MSG_BUF_VA`.
1150fn map_msg_buf_in_both_spaces(
1151 msg_buf_phys: u64,
1152 sender_as: &AddressSpace,
1153 receiver_as: &AddressSpace,
1154) -> Result<(), &'static str> {
1155 map_page_in_space(msg_buf_phys, N3_SHARED_MSG_BUF_VA, sender_as)?;
1156 map_page_in_space(msg_buf_phys, N3_SHARED_MSG_BUF_VA, receiver_as)?;
1157 Ok(())
1158}
1159
1160/// Unmap a single page from an address space and shoot down TLB on all CPUs.
1161fn unmap_page_in_space(target_va: u64, address_space: &AddressSpace) {
1162 use crate::{
1163 arch::xshim::{Size4KiB, VirtAddr},
1164 x86_crate_shim::structures::paging::{Mapper, Page},
1165 };
1166
1167 let page = Page::<Size4KiB>::containing_address(VirtAddr::new(target_va));
1168 let mut mapper = unsafe { address_space.mapper() };
1169
1170 if let Ok((_frame, flush)) = mapper.unmap(page) {
1171 flush.flush();
1172 }
1173 // Inter-CPU TLB shootdown for the unmapped page.
1174 crate::arch::tlb::shootdown_range(VirtAddr::new(target_va), VirtAddr::new(target_va + 0x1000));
1175}
1176
1177/// Unmap the MigrationFrame and message buffer from both address spaces.
1178fn unmap_from_both_spaces(sender_as: &AddressSpace, receiver_as: &AddressSpace) {
1179 unmap_page_in_space(N3_SHARED_FRAME_VA, sender_as);
1180 unmap_page_in_space(N3_SHARED_FRAME_VA, receiver_as);
1181 unmap_page_in_space(N3_SHARED_MSG_BUF_VA, sender_as);
1182 unmap_page_in_space(N3_SHARED_MSG_BUF_VA, receiver_as);
1183}
1184
1185// ============================================================================
1186// N3Transport
1187// ============================================================================
1188
1189/// N3 MMU transport : thread migration between distinct address spaces.
1190pub struct N3Transport {
1191 /// Index in the static frame pool.
1192 frame_idx: usize,
1193 /// Physical address of the MigrationFrame page.
1194 frame_phys: u64,
1195 /// Virtual address of the MigrationFrame (same in both AS).
1196 frame_virt: u64,
1197 /// Shared message buffer (kernel-allocated, mapped in both AS).
1198 msg_buf: MsgBuffer,
1199 /// Physical address of the message buffer.
1200 msg_buf_phys: u64,
1201 /// Dedicated handler stack (1 page, kernel-allocated).
1202 /// After CR3 switch, dst_ctx.rsp points to the top of this stack with
1203 /// the handler address as the return address.
1204 handler_stack_top: u64,
1205 /// Physical address of the handler stack.
1206 handler_stack_phys: u64,
1207 /// Sender task ID.
1208 sender_task_id: TaskId,
1209 /// Receiver task ID.
1210 receiver_task_id: TaskId,
1211 /// N3 tier (b or c).
1212 tier: N3Tier,
1213 /// Sender's PCID (0 for N3c).
1214 pcid_sender: u16,
1215 /// Receiver's PCID (0 for N3c).
1216 pcid_receiver: u16,
1217 /// Watchdog timeout in TSC cycles.
1218 watchdog_timeout_cycles: u64,
1219}
1220
1221// SAFETY: N3Transport is Send+Sync because all mutable access is protected
1222// by the MigrationFrame CAS state machine.
1223unsafe impl Send for N3Transport {}
1224unsafe impl Sync for N3Transport {}
1225
1226impl N3Transport {
1227 /// Create a new N3 transport between sender and receiver tasks.
1228 ///
1229 /// Allocates a MigrationFrame and a shared message buffer, maps them
1230 /// in both address spaces, and registers the endpoint with the watchdog.
1231 pub fn new(sender_task_id: TaskId, receiver_task_id: TaskId) -> Result<Self, IpcError> {
1232 let (frame_idx, _frame_ptr) = alloc_frame_slot().ok_or(IpcError::TransportFailed)?;
1233
1234 let frame_phys = frame_pool_phys_addr(frame_idx);
1235
1236 // Allocate a shared message buffer (1 page).
1237 let msg_buf_frame =
1238 with_irqs_disabled(allocate_frame).map_err(|_| IpcError::TransportFailed)?;
1239 let msg_buf_phys = msg_buf_frame.start_address.as_u64();
1240 let msg_buf = phys_to_virt(msg_buf_phys) as *mut u8;
1241
1242 // Map in both address spaces.
1243 let sender = get_task_by_id(sender_task_id).ok_or(IpcError::Disconnected)?;
1244 let receiver = get_task_by_id(receiver_task_id).ok_or(IpcError::Disconnected)?;
1245
1246 let sender_as = sender.process.address_space_arc();
1247 let receiver_as = receiver.process.address_space_arc();
1248
1249 if let Err(e) = map_frame_in_both_spaces(frame_phys, &sender_as, &receiver_as) {
1250 log::error!("N3: failed to map frame: {}", e);
1251 free_frame_slot(frame_idx);
1252 with_irqs_disabled(|token| free_frame(token, msg_buf_frame));
1253 return Err(IpcError::TransportFailed);
1254 }
1255
1256 // Map the message buffer in both address spaces at N3_SHARED_MSG_BUF_VA.
1257 if map_msg_buf_in_both_spaces(msg_buf_phys, &sender_as, &receiver_as).is_err() {
1258 log::error!("N3: failed to map message buffer");
1259 // P1 fix: unmap the frame that was already mapped in both spaces.
1260 unmap_page_in_space(N3_SHARED_FRAME_VA, &sender_as);
1261 unmap_page_in_space(N3_SHARED_FRAME_VA, &receiver_as);
1262 free_frame_slot(frame_idx);
1263 with_irqs_disabled(|token| free_frame(token, msg_buf_frame));
1264 return Err(IpcError::TransportFailed);
1265 }
1266
1267 // Allocate PCIDs BEFORE selecting tier : allocate_pcid() may return 0
1268 // if PCID is exhausted, which forces N3c even if the CPU supports PCID.
1269 let pcid_sender = allocate_pcid();
1270 let pcid_receiver = allocate_pcid();
1271
1272 // Select tier based on both CPU support AND availability.
1273 let tier = select_n3_tier(pcid_sender.min(pcid_receiver));
1274
1275 // Allocate a dedicated handler stack (1 page, kernel-allocated).
1276 // This stack is used as the trampoline after CR3 switch: the ASM
1277 // primitive loads RSP from dst_ctx.rsp and `ret` pops the handler
1278 // address from [RSP]. Must be in the kernel upper half (shared AS).
1279 let handler_stack_frame =
1280 with_irqs_disabled(allocate_frame).map_err(|_| IpcError::TransportFailed)?;
1281 let handler_stack_phys = handler_stack_frame.start_address.as_u64();
1282 let handler_stack_virt = phys_to_virt(handler_stack_phys) as u64;
1283 let handler_stack_top = handler_stack_virt + N3_HANDLER_STACK_SIZE as u64;
1284
1285 // Register with watchdog.
1286 watchdog_register(frame_idx, frame_phys);
1287
1288 Ok(N3Transport {
1289 frame_idx,
1290 frame_phys,
1291 frame_virt: N3_SHARED_FRAME_VA,
1292 msg_buf: MsgBuffer(msg_buf),
1293 msg_buf_phys,
1294 handler_stack_top,
1295 handler_stack_phys,
1296 sender_task_id,
1297 receiver_task_id,
1298 tier,
1299 pcid_sender,
1300 pcid_receiver,
1301 watchdog_timeout_cycles: N3_WATCHDOG_TIMEOUT,
1302 })
1303 }
1304
1305 /// Get a reference to the underlying MigrationFrame.
1306 pub(crate) fn frame(&self) -> &MigrationFrame {
1307 unsafe { &*(self.frame_virt as *const MigrationFrame) }
1308 }
1309
1310 /// Get the raw pointer to the MigrationFrame (for ASM primitive).
1311 pub(crate) fn frame_ptr(&self) -> *mut MigrationFrame {
1312 self.frame_virt as *mut MigrationFrame
1313 }
1314}
1315
1316impl core::fmt::Debug for N3Transport {
1317 fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
1318 f.debug_struct("N3Transport")
1319 .field("frame_idx", &self.frame_idx)
1320 .field("tier", &self.tier)
1321 .field("sender", &self.sender_task_id)
1322 .field("receiver", &self.receiver_task_id)
1323 .finish()
1324 }
1325}
1326
1327impl Drop for N3Transport {
1328 fn drop(&mut self) {
1329 watchdog_unregister(self.frame_idx);
1330
1331 // P0 fix: unmap shared pages from both address spaces BEFORE freeing
1332 // the physical frames. This prevents use-after-free when the frames
1333 // are reallocated to another process or structure.
1334 if let (Some(sender), Some(receiver)) = (
1335 get_task_by_id(self.sender_task_id),
1336 get_task_by_id(self.receiver_task_id),
1337 ) {
1338 let sender_as = sender.process.address_space_arc();
1339 let receiver_as = receiver.process.address_space_arc();
1340 unmap_from_both_spaces(&sender_as, &receiver_as);
1341 }
1342
1343 free_frame_slot(self.frame_idx);
1344
1345 // Free the message buffer physical page.
1346 if self.msg_buf_phys != 0 {
1347 let frame = crate::memory::PhysFrame::containing_address(
1348 crate::arch::xshim::PhysAddr::new(self.msg_buf_phys),
1349 );
1350 with_irqs_disabled(|token| free_frame(token, frame));
1351 }
1352
1353 // Free the handler stack physical page.
1354 if self.handler_stack_phys != 0 {
1355 let frame = crate::memory::PhysFrame::containing_address(
1356 crate::arch::xshim::PhysAddr::new(self.handler_stack_phys),
1357 );
1358 with_irqs_disabled(|token| free_frame(token, frame));
1359 }
1360 }
1361}
1362
1363impl IpcTransport for N3Transport {
1364 fn level(&self) -> TransportLevel {
1365 TransportLevel::Mmu
1366 }
1367
1368 fn capabilities(&self) -> TransportCapabilities {
1369 TransportCapabilities {
1370 max_message_size: N3_MSG_BUF_SIZE,
1371 blocking: true,
1372 zero_copy: false,
1373 vectored: false,
1374 directions: 2,
1375 estimated_cost_cycles: 800,
1376 }
1377 }
1378
1379 fn name(&self) -> &'static str {
1380 match self.tier {
1381 N3Tier::PcidPreserving => "n3b",
1382 N3Tier::FullIsolation => "n3c",
1383 }
1384 }
1385}
1386
1387impl IpcProducer for N3Transport {
1388 fn send(&self, msg: &[u8]) -> Result<(), IpcError> {
1389 // SAFETY: CAS state machine (Ready=>Active) provides logical exclusivity.
1390 // The &mut is scoped to this function; after CAS, only the ASM primitive
1391 // accesses the frame via raw pointers. Technically UB under Stacked Borrows,
1392 // but acceptable for a research kernel prototype where CAS guarantees
1393 // no concurrent access.
1394 let frame = unsafe { &mut *self.frame_ptr() };
1395
1396 // Get current task context.
1397 let receiver_task = get_task_by_id(self.receiver_task_id).ok_or(IpcError::Disconnected)?;
1398
1399 // Read current CR3.
1400 let (cr3_frame, _) = crate::x86_crate_shim::registers::control::Cr3::read();
1401 let sender_cr3 = cr3_frame.start_address().as_u64();
1402
1403 // Validate receiver's handler RIP.
1404 // Verify RIP matches the explicitly registered handler, not arbitrary code.
1405 let recv_rip = receiver_task.trampoline_entry.load(Ordering::Acquire);
1406 if recv_rip == 0 {
1407 return Err(IpcError::InvalidRip);
1408 }
1409 let recv_as = receiver_task.process.address_space_arc();
1410 validate_rip(recv_rip, &recv_as)?;
1411
1412 // Prepare the migration.
1413 let sender_rsp: u64;
1414 let sender_rip: u64;
1415 unsafe {
1416 core::arch::asm!("mov {}, rsp", out(reg) sender_rsp);
1417 // Capture return address from the stack. At this point in the
1418 // function, the return address is at [rbp + 8] when frame
1419 // pointers are enabled (kernel build default).
1420 //
1421 // SAFETY: kernel is compiled with frame pointers enabled
1422 // (-C force-frame-pointers=yes). If frame pointers are disabled,
1423 // this will read garbage : ensure the kernel build config is correct.
1424 core::arch::asm!("mov {}, [rbp + 8]", out(reg) sender_rip);
1425 }
1426
1427 n3_prepare_migration(
1428 frame,
1429 self.msg_buf.as_ptr(),
1430 sender_cr3,
1431 sender_rsp,
1432 sender_rip,
1433 self.handler_stack_top,
1434 &receiver_task,
1435 msg,
1436 )?;
1437
1438 // Inter-core sync if needed.
1439 // Spec §9.2: the sender must ensure message visibility before sending IPI.
1440 let target_cpu = receiver_task.home_cpu.load(Ordering::Acquire);
1441 let same_core = crate::arch::percpu::current_cpu_index() == target_cpu as usize;
1442
1443 if !same_core {
1444 if (target_cpu as usize) < crate::arch::percpu::MAX_CPUS {
1445 // Ensure all stores (message, frame metadata, state=Active)
1446 // are visible before the IPI is sent (spec §9.1).
1447 core::sync::atomic::fence(Ordering::Release);
1448
1449 // Store the VIRTUAL address : the IPI handler dereferences
1450 // it as a pointer in the kernel address space.
1451 N3_PENDING_MIGRATION[target_cpu].store(self.frame_virt, Ordering::Release);
1452
1453 // Send sync IPI. On x86-64, the ICR write is serializing,
1454 // which provides an implicit full barrier (spec §9.2).
1455 if let Some(target_apic) = crate::arch::percpu::apic_id_by_cpu_index(target_cpu) {
1456 send_n3_sync_ipi(target_apic);
1457 }
1458 }
1459 }
1460
1461 // Execute the migration ASM primitive.
1462 // SAFETY: frame is valid, mapped in both AS, interrupts are enabled.
1463 unsafe {
1464 n3b_migrate_asm(self.frame_virt as *mut MigrationFrame);
1465 }
1466
1467 Ok(())
1468 }
1469
1470 fn try_send(&self, msg: &[u8]) -> Result<(), IpcError> {
1471 // The CAS inside n3_prepare_migration handles the state check atomically.
1472 // No need for a separate state check here (avoids TOCTOU).
1473 self.send(msg)
1474 }
1475}
1476
1477impl IpcConsumer for N3Transport {
1478 fn recv(&self, buf: &mut [u8]) -> Result<usize, IpcError> {
1479 let frame = self.frame();
1480
1481 // Verify the frame is in Active state.
1482 // Do NOT read from a Ready, Stalled, or Reclaiming frame.
1483 let state = frame.state.load(Ordering::Acquire);
1484 if state != MigrationState::Active as u8 {
1485 return Err(IpcError::WouldBlock);
1486 }
1487
1488 // Verify generation (spec §15.1).
1489 let gen = frame.generation.load(Ordering::Acquire);
1490 if gen == 0 {
1491 return Err(IpcError::WouldBlock);
1492 }
1493
1494 // Check message length.
1495 let msg_len = frame.msg_len;
1496 if msg_len == 0 {
1497 return Err(IpcError::WouldBlock);
1498 }
1499
1500 // Read the message from the shared buffer.
1501 // Use a guard to reset state=Ready even if the read fails,
1502 // preventing the frame from being stuck in Active forever.
1503 let n = match shared_msg_read(self.msg_buf.as_ptr(), msg_len, buf) {
1504 Ok(n) => n,
1505 Err(e) => {
1506 // Reset state so the frame can be reused.
1507 frame
1508 .state
1509 .store(MigrationState::Ready as u8, Ordering::Release);
1510 return Err(e);
1511 }
1512 };
1513
1514 // Signal completion: set state = Ready (spec §15 step 6).
1515 // This allows the sender (or next sender) to reuse the frame.
1516 // The sender's send() will CAS Ready=>Active to claim it.
1517 // NOTE: msg_len is intentionally NOT cleared : the next sender's
1518 // prepare_migration() overwrites it when writing the new message.
1519 frame
1520 .state
1521 .store(MigrationState::Ready as u8, Ordering::Release);
1522
1523 Ok(n)
1524 }
1525
1526 fn try_recv(&self, buf: &mut [u8]) -> Result<Option<usize>, IpcError> {
1527 match self.recv(buf) {
1528 Ok(n) => Ok(Some(n)),
1529 Err(IpcError::WouldBlock) => Ok(None),
1530 Err(e) => Err(e),
1531 }
1532 }
1533}
1534
1535// ============================================================================
1536// State machine helpers
1537// ============================================================================
1538
1539/// Check if a frame is in the Ready state.
1540pub fn n3_frame_is_ready(frame: &MigrationFrame) -> bool {
1541 frame.state.load(Ordering::Acquire) == MigrationState::Ready as u8
1542}
1543
1544/// Get the current migration state of a frame.
1545pub fn n3_frame_state(frame: &MigrationFrame) -> MigrationState {
1546 MigrationState::from_u8(frame.state.load(Ordering::Acquire)).unwrap_or(MigrationState::Ready)
1547}
1548
1549/// Get the current generation counter of a frame.
1550pub fn n3_frame_generation(frame: &MigrationFrame) -> u32 {
1551 frame.generation.load(Ordering::Acquire)
1552}