Skip to main content

strat9_kernel/memory/
address_space.rs

1//! Per-process address spaces for Strat9-OS.
2//!
3//! Each task owns an `AddressSpace` backed by a PML4 page table.
4//! Kernel tasks share a single kernel address space. User tasks get a fresh
5//! PML4 with the kernel half (entries 256..512) cloned from the kernel's table.
6//!
7//! PML4 is the Page Map Level 4, the top-level page table in x86_64's 4-level paging scheme. It
8//! contains 512 entries, each covering a 512 GiB region of the virtual address space. By cloning the
9//! kernel half of the PML4, each user address space automatically shares the kernel mappings without needing to duplicate them.
10//! Source : https://wiki.osdev.org/Memory_Management
11//!
12//!
13//! x86_64 virtual address space layout:
14//!   - PML4[0..256]   => user space (per-process, zeroed for new AS)
15//!   - PML4[256..512] => kernel space (shared, cloned from kernel L4)
16//!
17
18use alloc::{collections::BTreeMap, sync::Arc, vec::Vec};
19use core::sync::atomic::{AtomicU32, Ordering};
20
21use crate::{
22    arch::xshim::{
23        PageTableFlags, PhysAddr, PhysFrame as X86PhysFrame, Size2MiB, Size4KiB, VirtAddr,
24    },
25    x86_crate_shim::{
26        registers::control::{Cr3, Cr3Flags},
27        structures::paging::{
28            mapper::TranslateResult, Mapper, OffsetPageTable, Page, PageTable, Translate,
29        },
30    },
31};
32use spin::Once;
33
34use crate::{
35    capability::CapId,
36    memory::{
37        allocate_mapping_cap_id, mapping_index, paging::BuddyFrameAllocator, release_owned_block,
38        resolve_handle, try_register_mapping_identity, unregister_mapping_identity, BlockHandle,
39        MappingRef,
40    },
41    process::task::Pid,
42    sync::SpinLock,
43};
44
45/// Flags describing permissions for a virtual memory region.
46#[derive(Debug, Clone, Copy, PartialEq, Eq)]
47pub struct VmaFlags {
48    pub readable: bool,
49    pub writable: bool,
50    pub executable: bool,
51    pub user_accessible: bool,
52}
53
54impl VmaFlags {
55    /// Convert to x86_64 page table flags.
56    pub fn to_page_flags(self) -> PageTableFlags {
57        let mut flags = PageTableFlags::PRESENT;
58        if self.writable {
59            flags |= PageTableFlags::WRITABLE;
60        }
61        if !self.executable {
62            flags |= PageTableFlags::NO_EXECUTE;
63        }
64        if self.user_accessible {
65            flags |= PageTableFlags::USER_ACCESSIBLE;
66        }
67        flags
68    }
69}
70
71/// Type/purpose of a virtual memory region.
72#[derive(Debug, Clone, Copy, PartialEq, Eq)]
73pub enum VmaType {
74    /// Zero-filled anonymous memory (heap, mmap).
75    Anonymous,
76    /// Stack region (grows downward).
77    Stack,
78    /// Code/text segment (typically RX).
79    Code,
80    /// Kernel-internal mapping.
81    Kernel,
82}
83
84/// Supported page sizes for VMAs.
85#[derive(Debug, Clone, Copy, PartialEq, Eq)]
86pub enum VmaPageSize {
87    /// Standard 4 KiB page.
88    Small,
89    /// Huge 2 MiB page.
90    Huge,
91}
92
93impl VmaPageSize {
94    /// Performs the bytes operation.
95    pub fn bytes(self) -> u64 {
96        match self {
97            VmaPageSize::Small => 4096,
98            VmaPageSize::Huge => 2 * 1024 * 1024,
99        }
100    }
101}
102
103/// A tracked virtual memory region within an address space.
104#[derive(Debug, Clone)]
105pub struct VirtualMemoryRegion {
106    /// Start virtual address (page-aligned).
107    pub start: u64,
108    /// Number of pages in this region (size depends on `page_size`).
109    pub page_count: usize,
110    /// Access permissions.
111    pub flags: VmaFlags,
112    /// Purpose of this region.
113    pub vma_type: VmaType,
114    /// Size of each page in this region.
115    pub page_size: VmaPageSize,
116}
117
118/// An effective mapping currently installed in the page tables.
119#[derive(Debug, Clone, Copy, PartialEq, Eq)]
120pub struct EffectiveMapping {
121    /// Start virtual address of the mapping.
122    pub start: u64,
123    /// Internal capability identifier associated with this mapping.
124    pub cap_id: CapId,
125    /// Physical block currently backing this mapping.
126    pub handle: BlockHandle,
127    /// Hardware page-table flags currently installed for this mapping.
128    pub flags: PageTableFlags,
129    /// Page size of the mapping.
130    pub page_size: VmaPageSize,
131}
132
133/// A per-process address space backed by a PML4 page table.
134///
135/// Kernel tasks share a single `AddressSpace` (the kernel AS).
136/// User tasks each get their own, with kernel entries (PML4[256..512]) cloned
137/// so that the kernel is always mapped regardless of which AS is active.
138///
139/// # Lock ordering
140///
141/// `regions` and `effective_mappings` are `SpinLock<BTreeMap<...>>`.
142/// Both BTreeMap insert/remove may allocate from the heap allocator.
143/// This is safe because:
144///
145///   1. **No IRQ-reachable path.** These locks are only taken from process
146///      context (syscall handlers, fork, munmap, page-fault). No interrupt
147///      handler or softirq acquires either lock, so IRQ-disabling is not
148///      required and cannot deadlock with the heap allocator.
149///
150///   2. **Heap lock order is consistent.** The implicit heap lock is always
151///      innermost (address_space => heap), never outermost. No other lock is
152///      acquired between the SpinLock hold and the potential heap alloc,
153///      so no ABBA cycle exists.
154///
155///   3. **Bounded contention.** Each lock protects a per-process map; only
156///      one task at a time operates on a given address space, so contention
157///      is low and hold times are short.
158///
159/// If an IRQ or NMI path ever needs to query these maps in the future,
160/// the `SpinLock` must be replaced with a lock-free snapshot or a `Mutex`
161/// (sleepable, IRQ-safe via `spin::Mutex`).
162pub struct AddressSpace {
163    /// Physical address of the PML4 table (loaded into CR3).
164    cr3_phys: PhysAddr,
165    /// Virtual address of the PML4 table (via HHDM, for reading/modifying).
166    l4_table_virt: VirtAddr,
167    /// Whether this is the kernel address space (never freed).
168    is_kernel: bool,
169    /// Tracked virtual memory regions (key = start address).
170    regions: SpinLock<BTreeMap<u64, VirtualMemoryRegion>>,
171    /// Tracked effective mappings (key = mapping start address).
172    effective_mappings: SpinLock<BTreeMap<u64, EffectiveMapping>>,
173    /// Process identifier owning this address space, when bound to a process.
174    owner_pid: AtomicU32,
175}
176
177// SAFETY: AddressSpace is protected by the scheduler lock and per-task ownership.
178// The PML4 table is accessed through HHDM virtual addresses which are valid on all CPUs.
179unsafe impl Send for AddressSpace {}
180unsafe impl Sync for AddressSpace {}
181
182impl AddressSpace {
183    /// Create the kernel address space by wrapping the current (boot) CR3.
184    ///
185    /// # Safety
186    /// Must be called exactly once, during single-threaded init, after paging is initialized.
187    pub unsafe fn new_kernel() -> Self {
188        let (level_4_frame, _flags) = Cr3::read();
189        let cr3_phys = level_4_frame.start_address();
190        let l4_table_virt = VirtAddr::new(crate::memory::phys_to_virt(cr3_phys.as_u64()));
191
192        log::info!(
193            "Kernel address space initialized: CR3={:#x}",
194            cr3_phys.as_u64()
195        );
196
197        AddressSpace {
198            cr3_phys,
199            l4_table_virt,
200            is_kernel: true,
201            regions: SpinLock::new(BTreeMap::new()),
202            effective_mappings: SpinLock::new(BTreeMap::new()),
203            owner_pid: AtomicU32::new(0),
204        }
205    }
206
207    /// Create a new user address space with the kernel half cloned.
208    ///
209    /// Allocates a fresh PML4 frame, zeroes it, then copies entries 256..512
210    /// from the kernel PML4. This shares the kernel's L3/L2/L1 subtrees so
211    /// kernel mapping changes propagate automatically.
212    pub fn new_user() -> Result<Self, &'static str> {
213        // Allocate a frame for the new PML4 table.
214        let new_l4_phys =
215            crate::sync::with_irqs_disabled(|token| crate::memory::allocate_frame(token))
216                .map_err(|_| "Failed to allocate PML4 frame")?
217                .start_address;
218
219        let new_l4_virt = VirtAddr::new(crate::memory::phys_to_virt(new_l4_phys.as_u64()));
220
221        // Zero the entire table first (clears user-half entries 0..256).
222        // SAFETY: new_l4_virt points to a freshly allocated, HHDM-mapped frame.
223        unsafe {
224            core::ptr::write_bytes(new_l4_virt.as_mut_ptr::<u8>(), 0, 4096);
225        }
226
227        // Clone kernel entries (PML4[256..512]) from the kernel's L4 table.
228        let kernel_l4_phys = crate::memory::paging::kernel_l4_phys();
229        let kernel_l4_virt = VirtAddr::new(crate::memory::phys_to_virt(kernel_l4_phys.as_u64()));
230
231        // SAFETY: Both pointers are valid HHDM-mapped page tables. We only read
232        // from the kernel table and write to the freshly allocated table.
233        unsafe {
234            let kernel_l4 = &*(kernel_l4_virt.as_ptr::<PageTable>());
235            let new_l4 = &mut *(new_l4_virt.as_mut_ptr::<PageTable>());
236            for i in 256..512 {
237                new_l4[i] = kernel_l4[i].clone();
238            }
239
240            // NOTE: the kernel image stays SHARED (PML4[511] subtree) : the
241            // kernel must remain executable while a user CR3 is active (the
242            // scheduler, syscall entry and IRQ paths run in kernel mode under
243            // the user page tables). Isolation comes from the U/S page bits:
244            // kernel pages are supervisor-only, so CPL3 code cannot touch
245            // them. User binaries are linked in the LOW half (0x400000, see
246            // workspace/components/user-linker.ld), so there is no address
247            // conflict with the kernel image window.
248        }
249
250        // ---------- LAPIC low-half mapping (HHDM=0 workaround) ----------
251        //
252        // When the bootloader provides a non-zero HHDM offset the LAPIC is mapped in
253        // PML4[256..512] (kernel half) and is already shared above.
254        //
255        // When HHDM=0 the LAPIC is identity-mapped at its physical address
256        // (0xFEE00000) in the low half (PML4[0]).  Every Ring-0 interrupt
257        // handler calls apic::eoi() which writes to this address.  If the
258        // handler fires while a user CR3 is active the write faults because
259        // PML4[0] is absent in the user page tables.
260        //
261        // Fix: map just the LAPIC 4KiB MMIO page into every new user AS using
262        // a fresh private L3/L2/L1 hierarchy (no sharing with the kernel's
263        // page table subtrees at the LAPIC virtual address).
264        {
265            let lapic_phys = crate::arch::apic::lapic_phys();
266            if lapic_phys != 0 {
267                let lapic_virt = crate::memory::phys_to_virt(lapic_phys);
268                // Only needed when LAPIC is in the low half.
269                if lapic_virt < 0xFFFF_8000_0000_0000 {
270                    let phys_offset = VirtAddr::new(crate::memory::hhdm_offset());
271                    // SAFETY: new_l4_virt is the freshly allocated user PML4.
272                    let l4 = unsafe { &mut *new_l4_virt.as_mut_ptr::<PageTable>() };
273                    let mut mapper = unsafe { OffsetPageTable::new(l4, phys_offset) };
274                    let mut buddy = crate::memory::paging::BuddyFrameAllocator;
275                    let mmio_flags = PageTableFlags::PRESENT
276                        | PageTableFlags::WRITABLE
277                        | PageTableFlags::NO_CACHE;
278                    let lapic_page =
279                        Page::<Size4KiB>::containing_address(VirtAddr::new(lapic_virt));
280                    let lapic_frame =
281                        X86PhysFrame::<Size4KiB>::containing_address(PhysAddr::new(lapic_phys));
282                    // Use map_to_with_table_flags to avoid USER_ACCESSIBLE on
283                    // intermediate tables so user code cannot reach LAPIC MMIO.
284                    match unsafe { mapper.map_to(lapic_page, lapic_frame, mmio_flags, &mut buddy) }
285                    {
286                        Ok(flush) => flush.flush(),
287                        Err(e) => {
288                            crate::serial_println!(
289                                "[as] WARN: failed to map LAPIC ({:#x}) in user AS: {:?}",
290                                lapic_phys,
291                                e
292                            );
293                        }
294                    }
295                }
296            }
297        }
298
299        log::debug!(
300            "User address space created: CR3={:#x} (kernel entries cloned from {:#x})",
301            new_l4_phys.as_u64(),
302            kernel_l4_phys.as_u64()
303        );
304
305        Ok(AddressSpace {
306            cr3_phys: new_l4_phys,
307            l4_table_virt: new_l4_virt,
308            is_kernel: false,
309            regions: SpinLock::new(BTreeMap::new()),
310            effective_mappings: SpinLock::new(BTreeMap::new()),
311            owner_pid: AtomicU32::new(0),
312        })
313    }
314
315    /// Registers an effective mapping in the address space tracking table.
316    pub fn register_effective_mapping(
317        &self,
318        mapping: EffectiveMapping,
319    ) -> Result<(), &'static str> {
320        let previous_at_start = self.effective_mapping_by_start(mapping.start);
321        if let Some(previous) = previous_at_start {
322            if previous.handle == mapping.handle && previous.cap_id == mapping.cap_id {
323                self.effective_mappings
324                    .lock()
325                    .insert(mapping.start, mapping);
326                if let Some(pid) = self.owner_pid() {
327                    mapping_index().unregister(mapping.cap_id, pid, VirtAddr::new(mapping.start));
328                    mapping_index().register(
329                        mapping.cap_id,
330                        MappingRef {
331                            pid,
332                            vaddr: VirtAddr::new(mapping.start),
333                            page_size: mapping.page_size,
334                        },
335                    );
336                }
337                return Ok(());
338            }
339        }
340
341        if let Err(error) = try_register_mapping_identity(mapping.handle, mapping.cap_id) {
342            if error != crate::memory::OwnerError::CapAlreadyPresent {
343                log::warn!(
344                    "memory: failed to register effective mapping identity cap={} block={:#x}/{} vaddr={:#x}: {:?}",
345                    mapping.cap_id.as_u64(),
346                    mapping.handle.base.as_u64(),
347                    mapping.handle.order,
348                    mapping.start,
349                    error
350                );
351                return Err("Failed to register effective mapping identity");
352            }
353        }
354
355        let replaced = self
356            .effective_mappings
357            .lock()
358            .insert(mapping.start, mapping);
359        if let Some(previous) = replaced {
360            if let Some(block) = unregister_mapping_identity(previous.handle, previous.cap_id) {
361                release_owned_block(block);
362            }
363            if let Some(pid) = self.owner_pid() {
364                mapping_index().unregister(previous.cap_id, pid, VirtAddr::new(previous.start));
365            }
366        }
367
368        if let Some(pid) = self.owner_pid() {
369            mapping_index().register(
370                mapping.cap_id,
371                MappingRef {
372                    pid,
373                    vaddr: VirtAddr::new(mapping.start),
374                    page_size: mapping.page_size,
375                },
376            );
377        }
378        Ok(())
379    }
380
381    /// Removes an effective mapping from the address space tracking table.
382    pub fn unregister_effective_mapping(&self, start: u64) -> Option<EffectiveMapping> {
383        let mapping = self.effective_mappings.lock().remove(&start);
384        if let Some(mapping) = mapping {
385            if let Some(block) = unregister_mapping_identity(mapping.handle, mapping.cap_id) {
386                release_owned_block(block);
387            }
388            if let Some(pid) = self.owner_pid() {
389                mapping_index().unregister(mapping.cap_id, pid, VirtAddr::new(mapping.start));
390            }
391            Some(mapping)
392        } else {
393            None
394        }
395    }
396
397    /// Updates the hardware flags recorded for an effective mapping.
398    pub fn update_effective_mapping_flags(&self, start: u64, flags: PageTableFlags) -> bool {
399        if let Some(mapping) = self.effective_mappings.lock().get_mut(&start) {
400            mapping.flags = flags;
401            true
402        } else {
403            false
404        }
405    }
406
407    /// Returns the effective mapping that starts exactly at `start`.
408    pub fn effective_mapping_by_start(&self, start: u64) -> Option<EffectiveMapping> {
409        self.effective_mappings.lock().get(&start).copied()
410    }
411
412    /// Unmaps the effective mapping that starts at `start`.
413    pub fn unmap_effective_mapping(&self, start: u64) -> Result<(), &'static str> {
414        let mapping = self
415            .effective_mapping_by_start(start)
416            .ok_or("Mapping not found")?;
417        self.unmap_range(start, mapping.page_size.bytes())
418    }
419
420    /// Returns the effective mapping covering `addr`, if any.
421    pub fn effective_mapping_containing(&self, addr: u64) -> Option<EffectiveMapping> {
422        let mappings = self.effective_mappings.lock();
423        if let Some(mapping) = mappings.get(&(addr & !(VmaPageSize::Small.bytes() - 1))) {
424            if mapping.page_size == VmaPageSize::Small {
425                return Some(*mapping);
426            }
427        }
428        mappings
429            .get(&(addr & !(VmaPageSize::Huge.bytes() - 1)))
430            .copied()
431    }
432
433    /// Binds this address space to the given process identifier.
434    pub fn set_owner_pid(&self, pid: Pid) {
435        let previous = self.owner_pid.swap(pid, Ordering::Relaxed);
436        let mappings: Vec<EffectiveMapping> = {
437            let guard = self.effective_mappings.lock();
438            guard.values().copied().collect()
439        };
440
441        if previous != 0 && previous != pid {
442            for mapping in mappings.iter().copied() {
443                mapping_index().unregister(mapping.cap_id, previous, VirtAddr::new(mapping.start));
444            }
445        }
446
447        if pid != 0 {
448            for mapping in mappings {
449                mapping_index().register(
450                    mapping.cap_id,
451                    MappingRef {
452                        pid,
453                        vaddr: VirtAddr::new(mapping.start),
454                        page_size: mapping.page_size,
455                    },
456                );
457            }
458        }
459    }
460
461    /// Returns the owning process identifier, if one has been assigned.
462    pub fn owner_pid(&self) -> Option<Pid> {
463        match self.owner_pid.load(Ordering::Relaxed) {
464            0 => None,
465            pid => Some(pid),
466        }
467    }
468
469    /// Construct a temporary `OffsetPageTable` mapper for this address space.
470    ///
471    /// # Safety
472    /// The caller must ensure exclusive access to the page tables (e.g. via
473    /// the scheduler lock or single-threaded context).
474    pub(crate) unsafe fn mapper(&self) -> OffsetPageTable<'_> {
475        let phys_offset = VirtAddr::new(crate::memory::hhdm_offset());
476        // SAFETY: l4_table_virt is the HHDM-mapped address of our PML4.
477        // The caller guarantees exclusive access.
478        unsafe {
479            OffsetPageTable::new(
480                &mut *self.l4_table_virt.as_mut_ptr::<PageTable>(),
481                phys_offset,
482            )
483        }
484    }
485
486    /// Reserve a contiguous region of virtual pages without allocating physical frames.
487    ///
488    /// The pages will be mapped lazily during page faults (Demand Paging).
489    pub fn reserve_region(
490        &self,
491        start: u64,
492        page_count: usize,
493        flags: VmaFlags,
494        vma_type: VmaType,
495        page_size: VmaPageSize,
496    ) -> Result<(), &'static str> {
497        let page_bytes = page_size.bytes();
498        if page_count == 0 || start % page_bytes != 0 {
499            return Err("Invalid region arguments");
500        }
501        let len = (page_count as u64)
502            .checked_mul(page_bytes)
503            .ok_or("Region length overflow")?;
504        let end = start.checked_add(len).ok_or("Region end overflow")?;
505        const USER_SPACE_END: u64 = crate::memory::userslice::USER_SPACE_END;
506        if end > USER_SPACE_END {
507            return Err("Region out of user-space range");
508        }
509
510        // Reject overlapping VMAs
511        {
512            let regions = self.regions.lock();
513            if regions.iter().any(|(&vma_start, vma)| {
514                let vma_end = vma_start
515                    .saturating_add((vma.page_count as u64).saturating_mul(vma.page_size.bytes()));
516                vma_start < end && vma_end > start
517            }) {
518                return Err("Region overlaps existing mapping");
519            }
520        }
521
522        // Enforce per-silo memory quota (best effort; non-silo tasks are ignored).
523        crate::silo::charge_current_task_memory(len).map_err(|_| "Silo memory quota exceeded")?;
524
525        // Track the region, attempting to merge with previous.
526        let mut regions = self.regions.lock();
527        let mut merged = false;
528
529        if let Some((&prev_start, prev_vma)) = regions.range(..start).next_back() {
530            let prev_end = prev_start + (prev_vma.page_count as u64) * prev_vma.page_size.bytes();
531            if prev_end == start
532                && prev_vma.flags == flags
533                && prev_vma.vma_type == vma_type
534                && prev_vma.page_size == page_size
535            {
536                let new_count = prev_vma
537                    .page_count
538                    .checked_add(page_count)
539                    .ok_or("Region page_count overflow")?;
540                let updated_vma = VirtualMemoryRegion {
541                    start: prev_start,
542                    page_count: new_count,
543                    flags,
544                    vma_type,
545                    page_size,
546                };
547                regions.insert(prev_start, updated_vma);
548                merged = true;
549            }
550        }
551
552        if !merged {
553            let region = VirtualMemoryRegion {
554                start,
555                page_count,
556                flags,
557                vma_type,
558                page_size,
559            };
560            regions.insert(start, region);
561        }
562
563        log::trace!(
564            "Reserved lazy region: {:#x} ({} pages, size={:?})",
565            start,
566            page_count,
567            page_size
568        );
569        Ok(())
570    }
571
572    /// Handle a page fault by checking if the address falls within a reserved VMA.
573    ///
574    /// If it does, allocates a physical frame and maps it.
575    /// Only call this for non-present faults; existing mappings are accepted
576    /// for concurrent demand faults, without changing their permissions.
577    pub fn handle_fault(&self, fault_addr: u64) -> Result<(), &'static str> {
578        use crate::x86_crate_shim::structures::paging::mapper::MapToError;
579
580        // 1. Find the VMA covering this address
581        let vma = {
582            let regions = self.regions.lock();
583            let mut iter = regions.range(..=fault_addr);
584            let (&start, vma) = iter.next_back().ok_or("No VMA found for address")?;
585            let end = start + (vma.page_count as u64) * vma.page_size.bytes();
586            if fault_addr >= end {
587                return Err("Address outside VMA bounds");
588            }
589            vma.clone()
590        };
591
592        // Align fault address to the page size used by this VMA.
593        let page_bytes = vma.page_size.bytes();
594        let page_addr = fault_addr & !(page_bytes - 1);
595
596        // 2. Only Anonymous/Stack regions support demand paging for now
597        match vma.vma_type {
598            VmaType::Anonymous | VmaType::Stack | VmaType::Code => {}
599            _ => return Err("VMA type does not support demand paging"),
600        }
601
602        // 3. If already mapped (race/re-fault), treat as handled.
603        if self.translate(VirtAddr::new(page_addr)).is_some() {
604            return Ok(());
605        }
606
607        // 4. Allocate and map a single page of the required size.
608        //
609        // IMPORTANT: `allocate_frame` (order-0) now goes through
610        // `FrameAllocOptions::new()` which zeroes by default.  For order > 0
611        // (huge pages) we still need a manual zero via the HHDM.
612        //
613        // The zero MUST go through phys_to_virt (HHDM), NOT through the user
614        // virtual address, because the user address space is not necessarily
615        // the currently active CR3.  Writing through `page_addr as *mut u8`
616        // would either write into a different process's memory or fault.
617        let mut frame_allocator = crate::memory::paging::BuddyFrameAllocator;
618        let order = match vma.page_size {
619            VmaPageSize::Small => 0,
620            VmaPageSize::Huge => 9,
621        };
622
623        let frame = crate::sync::with_irqs_disabled(|token| {
624            if order == 0 {
625                crate::memory::allocate_frame(token)
626            } else {
627                let f = crate::memory::allocate_phys_contiguous(token, order)?;
628                // SAFETY: phys_to_virt gives a valid HHDM pointer for this
629                // frame; we have exclusive ownership from the buddy allocator.
630                unsafe {
631                    core::ptr::write_bytes(
632                        crate::memory::phys_to_virt(f.start_address.as_u64()) as *mut u8,
633                        0,
634                        page_bytes as usize,
635                    );
636                }
637                Ok(f)
638            }
639        })
640        .map_err(|_| "OOM during demand paging")?;
641
642        let mut page_flags = vma.flags.to_page_flags();
643
644        // SAFETY: We own the address space.
645        unsafe {
646            let mut mapper = self.mapper();
647            match vma.page_size {
648                VmaPageSize::Small => {
649                    let page =
650                        Page::<Size4KiB>::from_start_address(VirtAddr::new(page_addr)).unwrap();
651                    let phys_frame = crate::arch::xshim::PhysFrame::<Size4KiB>::containing_address(
652                        frame.start_address,
653                    );
654                    match mapper.map_to(page, phys_frame, page_flags, &mut frame_allocator) {
655                        Ok(flush) => {
656                            flush.flush();
657                        }
658                        Err(MapToError::PageAlreadyMapped(_)) => {
659                            crate::sync::with_irqs_disabled(|token| {
660                                crate::memory::free_phys_contiguous(token, frame, order);
661                            });
662                            return Ok(());
663                        }
664                        Err(_) => {
665                            crate::sync::with_irqs_disabled(|token| {
666                                crate::memory::free_phys_contiguous(token, frame, order);
667                            });
668                            return Err("Failed to map demand page (4K)");
669                        }
670                    }
671                }
672                VmaPageSize::Huge => {
673                    let page =
674                        Page::<Size2MiB>::from_start_address(VirtAddr::new(page_addr)).unwrap();
675                    let phys_frame = crate::arch::xshim::PhysFrame::<Size2MiB>::containing_address(
676                        frame.start_address,
677                    );
678                    page_flags |= PageTableFlags::HUGE_PAGE;
679                    match mapper.map_to(page, phys_frame, page_flags, &mut frame_allocator) {
680                        Ok(flush) => {
681                            flush.flush();
682                        }
683                        Err(MapToError::PageAlreadyMapped(_)) => {
684                            crate::sync::with_irqs_disabled(|token| {
685                                crate::memory::free_phys_contiguous(token, frame, order);
686                            });
687                            return Ok(());
688                        }
689                        Err(_) => {
690                            crate::sync::with_irqs_disabled(|token| {
691                                crate::memory::free_phys_contiguous(token, frame, order);
692                            });
693                            return Err("Failed to map demand page (2M)");
694                        }
695                    }
696                }
697            }
698        }
699
700        if self
701            .register_effective_mapping(EffectiveMapping {
702                start: page_addr,
703                cap_id: allocate_mapping_cap_id(),
704                handle: resolve_handle(frame.start_address),
705                flags: page_flags,
706                page_size: vma.page_size,
707            })
708            .is_err()
709        {
710            unsafe {
711                let mut mapper = self.mapper();
712                match vma.page_size {
713                    VmaPageSize::Small => {
714                        let page =
715                            Page::<Size4KiB>::from_start_address(VirtAddr::new(page_addr)).unwrap();
716                        if let Ok((_, flush)) = mapper.unmap(page) {
717                            flush.flush();
718                        }
719                    }
720                    VmaPageSize::Huge => {
721                        let page =
722                            Page::<Size2MiB>::from_start_address(VirtAddr::new(page_addr)).unwrap();
723                        if let Ok((_, flush)) = mapper.unmap(page) {
724                            flush.flush();
725                        }
726                    }
727                }
728            }
729            crate::sync::with_irqs_disabled(|token| {
730                crate::memory::free_phys_contiguous(token, frame, order);
731            });
732            return Err("Failed to track demand page mapping");
733        }
734
735        // Initialize COW refcount.
736        //
737        // Order-0 frames come from FrameAllocOptions which stamps refcount=1
738        // via CAS(REFCOUNT_UNUSED => 1) : the frame is already "sole owner".
739        // Huge pages (order > 0) are raw-allocated with REFCOUNT_UNUSED still
740        // in the metadata; initialise explicitly to 1 here.
741        //
742        // Do NOT call frame_inc_ref for fresh allocations: that would push the
743        // count to 2, breaking the COW semantics (refcount==1 means sole owner).
744        // frame_inc_ref is correct only when sharing an existing frame (fork).
745        if order != 0 {
746            crate::memory::cow::handle_init_ref(resolve_handle(frame.start_address));
747        }
748
749        Ok(())
750    }
751
752    /// Map a contiguous region of pages backed by newly allocated physical frames.
753    ///
754    /// Frames are allocated from the buddy allocator and zero-filled.
755    /// The region is tracked in the VMA list.
756    pub fn map_region(
757        &self,
758        start: u64,
759        page_count: usize,
760        flags: VmaFlags,
761        vma_type: VmaType,
762        page_size: VmaPageSize,
763    ) -> Result<(), &'static str> {
764        let page_bytes = page_size.bytes();
765        if page_count == 0 || start % page_bytes != 0 {
766            return Err("Invalid region arguments");
767        }
768        let len = (page_count as u64)
769            .checked_mul(page_bytes)
770            .ok_or("Region length overflow")?;
771        let end = start.checked_add(len).ok_or("Region end overflow")?;
772        const USER_SPACE_END: u64 = crate::memory::userslice::USER_SPACE_END;
773        if end > USER_SPACE_END {
774            return Err("Region out of user-space range");
775        }
776
777        // Reject overlapping VMAs early
778        {
779            let regions = self.regions.lock();
780            if regions.iter().any(|(&vma_start, vma)| {
781                let vma_end = vma_start
782                    .saturating_add((vma.page_count as u64).saturating_mul(vma.page_size.bytes()));
783                vma_start < end && vma_end > start
784            }) {
785                return Err("Region overlaps existing mapping");
786            }
787        }
788
789        // Enforce per-silo memory quota for eagerly mapped regions.
790        crate::silo::charge_current_task_memory(len).map_err(|_| "Silo memory quota exceeded")?;
791
792        let page_flags = flags.to_page_flags();
793        let mut frame_allocator = BuddyFrameAllocator;
794
795        // SAFETY: we have logical ownership of this address space.
796        let mut mapper = unsafe { self.mapper() };
797        let mut mapped_pages = 0usize;
798
799        for i in 0..page_count {
800            let page_addr = start
801                .checked_add((i as u64).saturating_mul(page_bytes))
802                .ok_or("Page address overflow")?;
803
804            // Allocate a physical frame of appropriate size.
805            //
806            // order-0 frames go through FrameAllocOptions (zeroed + metadata
807            // stamped).  order > 0 (huge pages) are zeroed manually via HHDM.
808            let order = match page_size {
809                VmaPageSize::Small => 0,
810                VmaPageSize::Huge => 9,
811            };
812
813            let frame = crate::sync::with_irqs_disabled(|token| {
814                if order == 0 {
815                    crate::memory::allocate_frame(token)
816                } else {
817                    let f = crate::memory::allocate_phys_contiguous(token, order)?;
818                    unsafe {
819                        let virt = crate::memory::phys_to_virt(f.start_address.as_u64());
820                        core::ptr::write_bytes(virt as *mut u8, 0, page_bytes as usize);
821                    }
822                    Ok(f)
823                }
824            })
825            .map_err(|_| "Failed to allocate frame")?;
826
827            // Map the page.
828            let map_ok = match page_size {
829                VmaPageSize::Small => {
830                    use crate::arch::xshim::Size4KiB;
831                    let page = Page::<Size4KiB>::from_start_address(VirtAddr::new(page_addr))
832                        .map_err(|_| "Map 4K: invalid page address")?;
833                    let phys_frame = crate::arch::xshim::PhysFrame::<Size4KiB>::containing_address(
834                        frame.start_address,
835                    );
836                    unsafe {
837                        mapper
838                            .map_to(page, phys_frame, page_flags, &mut frame_allocator)
839                            .map(|flush| flush.flush())
840                            .is_ok()
841                    }
842                }
843                VmaPageSize::Huge => {
844                    use crate::arch::xshim::Size2MiB;
845                    let page = Page::<Size2MiB>::from_start_address(VirtAddr::new(page_addr))
846                        .map_err(|_| "Map 2M: invalid page address")?;
847                    let phys_frame = crate::arch::xshim::PhysFrame::<Size2MiB>::containing_address(
848                        frame.start_address,
849                    );
850                    let mut huge_flags = page_flags;
851                    huge_flags |= PageTableFlags::HUGE_PAGE;
852                    unsafe {
853                        mapper
854                            .map_to(page, phys_frame, huge_flags, &mut frame_allocator)
855                            .map(|flush| flush.flush())
856                            .is_ok()
857                    }
858                }
859            };
860
861            if !map_ok {
862                log::error!(
863                    "map_region: map_to failed at page {} vaddr={:#x} size={:?}",
864                    i,
865                    page_addr,
866                    page_size
867                );
868                // Free frame for this page that failed to map.
869                crate::sync::with_irqs_disabled(|token| {
870                    crate::memory::free_phys_contiguous(token, frame, order);
871                });
872
873                // Roll back already mapped pages to keep state consistent.
874                for j in (0..mapped_pages).rev() {
875                    let rb_addr = start + (j as u64) * page_bytes;
876                    match page_size {
877                        VmaPageSize::Small => {
878                            use crate::arch::xshim::Size4KiB;
879                            let rb_page =
880                                Page::<Size4KiB>::from_start_address(VirtAddr::new(rb_addr))
881                                    .map_err(|_| "Rollback: invalid 4K page address")?;
882                            if let Ok((_, rb_flush)) = mapper.unmap(rb_page) {
883                                rb_flush.flush();
884                                let _ = self.unregister_effective_mapping(rb_addr);
885                            }
886                        }
887                        VmaPageSize::Huge => {
888                            use crate::arch::xshim::Size2MiB;
889                            let rb_page =
890                                Page::<Size2MiB>::from_start_address(VirtAddr::new(rb_addr))
891                                    .map_err(|_| "Rollback: invalid 2M page address")?;
892                            if let Ok((_, rb_flush)) = mapper.unmap(rb_page) {
893                                rb_flush.flush();
894                                let _ = self.unregister_effective_mapping(rb_addr);
895                            }
896                        }
897                    }
898                }
899
900                crate::silo::release_current_task_memory(len);
901                return Err("Failed to map page");
902            }
903
904            // Initialize COW refcount (same logic as demand_page above).
905            let effective_flags = match page_size {
906                VmaPageSize::Small => page_flags,
907                VmaPageSize::Huge => page_flags | PageTableFlags::HUGE_PAGE,
908            };
909            if self
910                .register_effective_mapping(EffectiveMapping {
911                    start: page_addr,
912                    cap_id: allocate_mapping_cap_id(),
913                    handle: resolve_handle(frame.start_address),
914                    flags: effective_flags,
915                    page_size,
916                })
917                .is_err()
918            {
919                match page_size {
920                    VmaPageSize::Small => {
921                        use crate::arch::xshim::Size4KiB;
922                        let page = Page::<Size4KiB>::from_start_address(VirtAddr::new(page_addr))
923                            .map_err(|_| "Rollback: invalid 4K page address")?;
924                        if let Ok((_, flush)) = mapper.unmap(page) {
925                            flush.flush();
926                        }
927                    }
928                    VmaPageSize::Huge => {
929                        use crate::arch::xshim::Size2MiB;
930                        let page = Page::<Size2MiB>::from_start_address(VirtAddr::new(page_addr))
931                            .map_err(|_| "Rollback: invalid 2M page address")?;
932                        if let Ok((_, flush)) = mapper.unmap(page) {
933                            flush.flush();
934                        }
935                    }
936                }
937                crate::sync::with_irqs_disabled(|token| {
938                    crate::memory::free_phys_contiguous(token, frame, order);
939                });
940                for j in (0..mapped_pages).rev() {
941                    let rb_addr = start + (j as u64) * page_bytes;
942                    match page_size {
943                        VmaPageSize::Small => {
944                            use crate::arch::xshim::Size4KiB;
945                            let rb_page =
946                                Page::<Size4KiB>::from_start_address(VirtAddr::new(rb_addr))
947                                    .map_err(|_| "Rollback: invalid 4K page address")?;
948                            if let Ok((_, rb_flush)) = mapper.unmap(rb_page) {
949                                rb_flush.flush();
950                                let _ = self.unregister_effective_mapping(rb_addr);
951                            }
952                        }
953                        VmaPageSize::Huge => {
954                            use crate::arch::xshim::Size2MiB;
955                            let rb_page =
956                                Page::<Size2MiB>::from_start_address(VirtAddr::new(rb_addr))
957                                    .map_err(|_| "Rollback: invalid 2M page address")?;
958                            if let Ok((_, rb_flush)) = mapper.unmap(rb_page) {
959                                rb_flush.flush();
960                                let _ = self.unregister_effective_mapping(rb_addr);
961                            }
962                        }
963                    }
964                }
965                crate::silo::release_current_task_memory(len);
966                return Err("Failed to track mapped region page");
967            }
968
969            mapped_pages += 1;
970        }
971
972        // Track the region
973        let mut regions = self.regions.lock();
974        let region = VirtualMemoryRegion {
975            start,
976            page_count,
977            flags,
978            vma_type,
979            page_size,
980        };
981        regions.insert(start, region);
982
983        let end = start + (page_count as u64) * page_bytes;
984        crate::trace_mem!(
985            crate::trace::category::MEM_MAP,
986            crate::trace::TraceKind::MemMap,
987            page_size.bytes(),
988            crate::trace::TraceTaskCtx {
989                task_id: 0,
990                pid: 0,
991                tid: 0,
992                cr3: self.cr3_phys.as_u64(),
993            },
994            0,
995            start,
996            end,
997            page_count as u64
998        );
999
1000        Ok(())
1001    }
1002
1003    /// Maps shared frames.
1004    pub fn map_shared_frames(
1005        &self,
1006        start: u64,
1007        frame_phys_addrs: &[u64],
1008        flags: VmaFlags,
1009        vma_type: VmaType,
1010    ) -> Result<(), &'static str> {
1011        self.map_shared_frames_with_cap_ids(start, frame_phys_addrs, None, flags, vma_type)
1012    }
1013
1014    /// Maps shared physical blocks with optional stable mapping identities.
1015    pub fn map_shared_handles_with_cap_ids(
1016        &self,
1017        start: u64,
1018        handles: &[BlockHandle],
1019        mapping_cap_ids: Option<&[CapId]>,
1020        flags: VmaFlags,
1021        vma_type: VmaType,
1022        page_size: VmaPageSize,
1023    ) -> Result<(), &'static str> {
1024        let page_count = handles.len();
1025        let page_bytes = page_size.bytes();
1026        if page_count == 0 || start % page_bytes != 0 {
1027            return Err("Invalid shared region arguments");
1028        }
1029        if mapping_cap_ids.is_some_and(|cap_ids| cap_ids.len() != page_count) {
1030            return Err("Shared mapping identity count mismatch");
1031        }
1032        let len = (page_count as u64)
1033            .checked_mul(page_bytes)
1034            .ok_or("Shared region length overflow")?;
1035        let end = start.checked_add(len).ok_or("Shared region end overflow")?;
1036        const USER_SPACE_END: u64 = crate::memory::userslice::USER_SPACE_END;
1037        if end > USER_SPACE_END {
1038            return Err("Shared region out of user-space range");
1039        }
1040
1041        {
1042            let regions = self.regions.lock();
1043            if regions.iter().any(|(&vma_start, vma)| {
1044                let vma_end = vma_start
1045                    .saturating_add((vma.page_count as u64).saturating_mul(vma.page_size.bytes()));
1046                vma_start < end && vma_end > start
1047            }) {
1048                return Err("Shared region overlaps existing mapping");
1049            }
1050        }
1051
1052        let mut page_flags = flags.to_page_flags();
1053        if page_size == VmaPageSize::Huge {
1054            page_flags |= PageTableFlags::HUGE_PAGE;
1055        }
1056        let mut frame_allocator = BuddyFrameAllocator;
1057        let mut mapper = unsafe { self.mapper() };
1058        let mut mapped_pages = 0usize;
1059
1060        for (index, handle) in handles.iter().copied().enumerate() {
1061            let page_addr = start
1062                .checked_add((index as u64) * page_bytes)
1063                .ok_or("Shared page address overflow")?;
1064
1065            let map_ok = match page_size {
1066                VmaPageSize::Small => {
1067                    let page = Page::<Size4KiB>::from_start_address(VirtAddr::new(page_addr))
1068                        .map_err(|_| "Map shared: invalid 4K page address")?;
1069                    let frame = X86PhysFrame::<Size4KiB>::containing_address(handle.base);
1070                    unsafe {
1071                        mapper
1072                            .map_to(page, frame, page_flags, &mut frame_allocator)
1073                            .map(|flush| flush.flush())
1074                            .is_ok()
1075                    }
1076                }
1077                VmaPageSize::Huge => {
1078                    let page = Page::<Size2MiB>::from_start_address(VirtAddr::new(page_addr))
1079                        .map_err(|_| "Map shared: invalid 2M page address")?;
1080                    let frame = X86PhysFrame::<Size2MiB>::containing_address(handle.base);
1081                    unsafe {
1082                        mapper
1083                            .map_to(page, frame, page_flags, &mut frame_allocator)
1084                            .map(|flush| flush.flush())
1085                            .is_ok()
1086                    }
1087                }
1088            };
1089
1090            if !map_ok {
1091                for rollback in (0..mapped_pages).rev() {
1092                    let rb_addr = start + (rollback as u64) * page_bytes;
1093                    match page_size {
1094                        VmaPageSize::Small => {
1095                            if let Ok(rb_page) =
1096                                Page::<Size4KiB>::from_start_address(VirtAddr::new(rb_addr))
1097                            {
1098                                if let Ok((_, rb_flush)) = mapper.unmap(rb_page) {
1099                                    rb_flush.flush();
1100                                    let _ = self.unregister_effective_mapping(rb_addr);
1101                                }
1102                            }
1103                        }
1104                        VmaPageSize::Huge => {
1105                            if let Ok(rb_page) =
1106                                Page::<Size2MiB>::from_start_address(VirtAddr::new(rb_addr))
1107                            {
1108                                if let Ok((_, rb_flush)) = mapper.unmap(rb_page) {
1109                                    rb_flush.flush();
1110                                    let _ = self.unregister_effective_mapping(rb_addr);
1111                                }
1112                            }
1113                        }
1114                    }
1115                }
1116                return Err("Failed to map shared page");
1117            }
1118
1119            if self
1120                .register_effective_mapping(EffectiveMapping {
1121                    start: page_addr,
1122                    cap_id: mapping_cap_ids
1123                        .and_then(|cap_ids| cap_ids.get(index).copied())
1124                        .unwrap_or_else(allocate_mapping_cap_id),
1125                    handle,
1126                    flags: page_flags,
1127                    page_size,
1128                })
1129                .is_err()
1130            {
1131                match page_size {
1132                    VmaPageSize::Small => {
1133                        if let Ok(page) =
1134                            Page::<Size4KiB>::from_start_address(VirtAddr::new(page_addr))
1135                        {
1136                            if let Ok((_, flush)) = mapper.unmap(page) {
1137                                flush.flush();
1138                            }
1139                        }
1140                    }
1141                    VmaPageSize::Huge => {
1142                        if let Ok(page) =
1143                            Page::<Size2MiB>::from_start_address(VirtAddr::new(page_addr))
1144                        {
1145                            if let Ok((_, flush)) = mapper.unmap(page) {
1146                                flush.flush();
1147                            }
1148                        }
1149                    }
1150                }
1151                for rollback in (0..mapped_pages).rev() {
1152                    let rb_addr = start + (rollback as u64) * page_bytes;
1153                    match page_size {
1154                        VmaPageSize::Small => {
1155                            if let Ok(rb_page) =
1156                                Page::<Size4KiB>::from_start_address(VirtAddr::new(rb_addr))
1157                            {
1158                                if let Ok((_, rb_flush)) = mapper.unmap(rb_page) {
1159                                    rb_flush.flush();
1160                                    let _ = self.unregister_effective_mapping(rb_addr);
1161                                }
1162                            }
1163                        }
1164                        VmaPageSize::Huge => {
1165                            if let Ok(rb_page) =
1166                                Page::<Size2MiB>::from_start_address(VirtAddr::new(rb_addr))
1167                            {
1168                                if let Ok((_, rb_flush)) = mapper.unmap(rb_page) {
1169                                    rb_flush.flush();
1170                                    let _ = self.unregister_effective_mapping(rb_addr);
1171                                }
1172                            }
1173                        }
1174                    }
1175                }
1176                return Err("Failed to track shared mapping");
1177            }
1178            mapped_pages += 1;
1179        }
1180
1181        self.regions.lock().insert(
1182            start,
1183            VirtualMemoryRegion {
1184                start,
1185                page_count,
1186                flags,
1187                vma_type,
1188                page_size,
1189            },
1190        );
1191        Ok(())
1192    }
1193
1194    /// Maps shared frames with optional stable mapping identities.
1195    pub fn map_shared_frames_with_cap_ids(
1196        &self,
1197        start: u64,
1198        frame_phys_addrs: &[u64],
1199        mapping_cap_ids: Option<&[CapId]>,
1200        flags: VmaFlags,
1201        vma_type: VmaType,
1202    ) -> Result<(), &'static str> {
1203        let handles = frame_phys_addrs
1204            .iter()
1205            .copied()
1206            .map(|phys_addr| resolve_handle(PhysAddr::new(phys_addr)))
1207            .collect::<Vec<_>>();
1208        self.map_shared_handles_with_cap_ids(
1209            start,
1210            &handles,
1211            mapping_cap_ids,
1212            flags,
1213            vma_type,
1214            VmaPageSize::Small,
1215        )
1216    }
1217
1218    /// Unmap a previously mapped region and free the backing frames.
1219    pub fn unmap_region(
1220        &self,
1221        start: u64,
1222        page_count: usize,
1223        page_size: VmaPageSize,
1224    ) -> Result<(), &'static str> {
1225        let page_bytes = page_size.bytes();
1226        // SAFETY: We have logical ownership of this address space.
1227        let mut mapper = unsafe { self.mapper() };
1228
1229        for i in 0..page_count {
1230            let page_addr = start + (i as u64) * page_bytes;
1231
1232            let _frame_addr = match page_size {
1233                VmaPageSize::Small => {
1234                    use crate::arch::xshim::Size4KiB;
1235                    let page = Page::<Size4KiB>::from_start_address(VirtAddr::new(page_addr))
1236                        .map_err(|_| "Failed to unmap: invalid 4K page address")?;
1237                    let (frame, flush) =
1238                        mapper.unmap(page).map_err(|_| "Failed to unmap 4K page")?;
1239                    flush.flush();
1240                    frame.start_address()
1241                }
1242                VmaPageSize::Huge => {
1243                    use crate::arch::xshim::Size2MiB;
1244                    let page = Page::<Size2MiB>::from_start_address(VirtAddr::new(page_addr))
1245                        .map_err(|_| "Failed to unmap: invalid 2M page address")?;
1246                    let (frame, flush) =
1247                        mapper.unmap(page).map_err(|_| "Failed to unmap 2M page")?;
1248                    flush.flush();
1249                    frame.start_address()
1250                }
1251            };
1252
1253            // COW-aware refcount decrement: free only when last mapping disappears.
1254            let _ = self.unregister_effective_mapping(page_addr);
1255        }
1256
1257        // Remove from VMA tracking.
1258        self.regions.lock().remove(&start);
1259
1260        let end = start + (page_count as u64) * page_bytes;
1261
1262        // P0 fix: inter-CPU TLB shootdown after unmapping.
1263        crate::arch::tlb::shootdown_range(VirtAddr::new(start), VirtAddr::new(end));
1264
1265        log::trace!(
1266            "Unmapped region: {:#x}..{:#x} ({} pages, size={:?})",
1267            start,
1268            end,
1269            page_count,
1270            page_size
1271        );
1272
1273        crate::trace_mem!(
1274            crate::trace::category::MEM_UNMAP,
1275            crate::trace::TraceKind::MemUnmap,
1276            page_size.bytes(),
1277            crate::trace::TraceTaskCtx {
1278                task_id: 0,
1279                pid: 0,
1280                tid: 0,
1281                cr3: self.cr3_phys.as_u64(),
1282            },
1283            0,
1284            start,
1285            end,
1286            page_count as u64
1287        );
1288
1289        let released = (page_count as u64).saturating_mul(page_bytes);
1290        crate::silo::release_current_task_memory(released);
1291
1292        Ok(())
1293    }
1294
1295    /// Find a free virtual address range of `n_pages` pages of `page_size` starting at or after `hint`.
1296    pub fn find_free_vma_range(
1297        &self,
1298        hint: u64,
1299        n_pages: usize,
1300        page_size: VmaPageSize,
1301    ) -> Option<u64> {
1302        if n_pages == 0 {
1303            return None;
1304        }
1305        let page_bytes = page_size.bytes();
1306        let length = (n_pages as u64).checked_mul(page_bytes)?;
1307        let upper_limit: u64 = crate::memory::userslice::USER_SPACE_END;
1308
1309        // Round hint up to a page boundary
1310        let mut candidate = (hint.saturating_add(page_bytes - 1)) & !(page_bytes - 1);
1311        if candidate == 0 {
1312            candidate = page_bytes;
1313        }
1314
1315        let regions = self.regions.lock();
1316        for (&vma_start, vma) in regions.iter() {
1317            let vma_end = vma_start + vma.page_count as u64 * vma.page_size.bytes();
1318
1319            // A gap exists before this VMA : candidate fits.
1320            if candidate.saturating_add(length) <= vma_start {
1321                break;
1322            }
1323
1324            // Candidate overlaps this VMA; skip past it.
1325            if vma_end > candidate {
1326                candidate = (vma_end.saturating_add(page_bytes - 1)) & !(page_bytes - 1);
1327            }
1328        }
1329
1330        // Final bounds check.
1331        if candidate.checked_add(length)? <= upper_limit {
1332            Some(candidate)
1333        } else {
1334            None
1335        }
1336    }
1337
1338    /// Return true if any tracked VMA overlaps `[addr, addr + len)`.
1339    pub fn has_mapping_in_range(&self, addr: u64, len: u64) -> bool {
1340        let end = match addr.checked_add(len) {
1341            Some(v) => v,
1342            None => return true,
1343        };
1344        let regions = self.regions.lock();
1345        regions.iter().any(|(&vma_start, vma)| {
1346            let vma_end = vma_start
1347                .saturating_add((vma.page_count as u64).saturating_mul(vma.page_size.bytes()));
1348            vma_start < end && vma_end > addr
1349        })
1350    }
1351
1352    /// Return the tracked VMA that starts exactly at `start`.
1353    pub fn region_by_start(&self, start: u64) -> Option<VirtualMemoryRegion> {
1354        let regions = self.regions.lock();
1355        regions.get(&start).cloned()
1356    }
1357
1358    /// Returns true if any page in `[addr, addr + len)` is currently mapped.
1359    pub fn any_mapped_in_range(
1360        &self,
1361        addr: u64,
1362        len: u64,
1363        page_size: VmaPageSize,
1364    ) -> Result<bool, &'static str> {
1365        if len == 0 {
1366            return Ok(false);
1367        }
1368        let end = addr
1369            .checked_add(len)
1370            .ok_or("any_mapped_in_range: address overflow")?;
1371        let step = page_size.bytes();
1372        let mut cur = addr;
1373        while cur < end {
1374            if self.translate(VirtAddr::new(cur)).is_some() {
1375                return Ok(true);
1376            }
1377            cur = cur
1378                .checked_add(step)
1379                .ok_or("any_mapped_in_range: loop overflow")?;
1380        }
1381        Ok(false)
1382    }
1383
1384    /// Performs the protect range operation.
1385    pub fn protect_range(&self, addr: u64, len: u64, flags: VmaFlags) -> Result<(), &'static str> {
1386        if len == 0 {
1387            return Ok(());
1388        }
1389        let end = addr
1390            .checked_add(len)
1391            .ok_or("protect_range: address overflow")?;
1392        let mut cursor = addr;
1393
1394        {
1395            let regions = self.regions.lock();
1396            for (&vma_start, vma) in regions.iter() {
1397                let vma_end = vma_start + vma.page_count as u64 * vma.page_size.bytes();
1398                if vma_start >= end || vma_end <= addr {
1399                    continue;
1400                }
1401                if vma.page_size == VmaPageSize::Huge {
1402                    let range_start = core::cmp::max(vma_start, addr);
1403                    let range_end = core::cmp::min(vma_end, end);
1404                    if range_start % vma.page_size.bytes() != 0
1405                        || range_end % vma.page_size.bytes() != 0
1406                    {
1407                        return Err(
1408                            "protect_range: partial mprotect of 2MiB pages is not supported",
1409                        );
1410                    }
1411                }
1412            }
1413        }
1414
1415        let mut touched = false;
1416        while cursor < end {
1417            let region_info = {
1418                let regions = self.regions.lock();
1419                regions
1420                    .iter()
1421                    .find(|(&vma_start, vma)| {
1422                        let vma_end = vma_start + vma.page_count as u64 * vma.page_size.bytes();
1423                        vma_start < end && vma_end > cursor
1424                    })
1425                    .map(|(&k, v)| (k, v.clone()))
1426            };
1427
1428            let Some((vma_start, vma)) = region_info else {
1429                break;
1430            };
1431            touched = true;
1432
1433            let vma_end = vma_start + vma.page_count as u64 * vma.page_size.bytes();
1434            let range_start = core::cmp::max(vma_start, cursor);
1435            let range_end = core::cmp::min(vma_end, end);
1436            let page_bytes = vma.page_size.bytes();
1437            let new_pt_flags = flags.to_page_flags();
1438
1439            let mut mapper = unsafe { self.mapper() };
1440            let mut page_addr = range_start;
1441            while page_addr < range_end {
1442                if mapper.translate_addr(VirtAddr::new(page_addr)).is_none() {
1443                    page_addr += page_bytes;
1444                    continue;
1445                }
1446                unsafe {
1447                    match vma.page_size {
1448                        VmaPageSize::Small => {
1449                            let page =
1450                                Page::<Size4KiB>::from_start_address(VirtAddr::new(page_addr))
1451                                    .map_err(|_| "protect_range: invalid 4K page address")?;
1452                            mapper
1453                                .update_flags(page, new_pt_flags)
1454                                .map(|f| f.ignore())
1455                                .map_err(|_| "protect_range: update 4K flags failed")?;
1456                            let _ = self.update_effective_mapping_flags(page_addr, new_pt_flags);
1457                        }
1458                        VmaPageSize::Huge => {
1459                            let mut huge_flags = new_pt_flags;
1460                            huge_flags |= PageTableFlags::HUGE_PAGE;
1461                            let page =
1462                                Page::<Size2MiB>::from_start_address(VirtAddr::new(page_addr))
1463                                    .map_err(|_| "protect_range: invalid 2M page address")?;
1464                            mapper
1465                                .update_flags(page, huge_flags)
1466                                .map(|f| f.ignore())
1467                                .map_err(|_| "protect_range: update 2M flags failed")?;
1468                            let _ = self.update_effective_mapping_flags(page_addr, huge_flags);
1469                        }
1470                    }
1471                }
1472                page_addr += page_bytes;
1473            }
1474
1475            {
1476                let mut regions = self.regions.lock();
1477                regions.remove(&vma_start);
1478
1479                if range_start > vma_start {
1480                    let leading_pages = ((range_start - vma_start) / page_bytes) as usize;
1481                    regions.insert(
1482                        vma_start,
1483                        VirtualMemoryRegion {
1484                            start: vma_start,
1485                            page_count: leading_pages,
1486                            flags: vma.flags,
1487                            vma_type: vma.vma_type,
1488                            page_size: vma.page_size,
1489                        },
1490                    );
1491                }
1492
1493                let middle_pages = ((range_end - range_start) / page_bytes) as usize;
1494                if middle_pages > 0 {
1495                    regions.insert(
1496                        range_start,
1497                        VirtualMemoryRegion {
1498                            start: range_start,
1499                            page_count: middle_pages,
1500                            flags,
1501                            vma_type: vma.vma_type,
1502                            page_size: vma.page_size,
1503                        },
1504                    );
1505                }
1506
1507                if range_end < vma_end {
1508                    let trailing_pages = ((vma_end - range_end) / page_bytes) as usize;
1509                    regions.insert(
1510                        range_end,
1511                        VirtualMemoryRegion {
1512                            start: range_end,
1513                            page_count: trailing_pages,
1514                            flags: vma.flags,
1515                            vma_type: vma.vma_type,
1516                            page_size: vma.page_size,
1517                        },
1518                    );
1519                }
1520            }
1521
1522            cursor = range_end;
1523        }
1524
1525        // P0 fix: inter-CPU TLB shootdown after all PTE flag changes.
1526        // Other CPUs may still hold stale RW translations in their TLB.
1527        if touched {
1528            crate::arch::tlb::shootdown_range(VirtAddr::new(addr), VirtAddr::new(end));
1529        }
1530
1531        if !touched {
1532            return Err("protect_range: no mapped region in range");
1533        }
1534        Ok(())
1535    }
1536
1537    /// Unmaps range.
1538    pub fn unmap_range(&self, addr: u64, len: u64) -> Result<(), &'static str> {
1539        if len == 0 {
1540            return Ok(());
1541        }
1542        let end = addr
1543            .checked_add(len)
1544            .ok_or("unmap_range: address overflow")?;
1545
1546        // Pre-validate huge-page overlaps: partial unmap of 2MiB mappings is
1547        // not supported yet. Callers must unmap on huge-page boundaries.
1548        {
1549            let regions = self.regions.lock();
1550            for (&vma_start, vma) in regions.iter() {
1551                let vma_end = vma_start + vma.page_count as u64 * vma.page_size.bytes();
1552                if vma_start >= end || vma_end <= addr {
1553                    continue;
1554                }
1555                if vma.page_size == VmaPageSize::Huge {
1556                    let range_start = core::cmp::max(vma_start, addr);
1557                    let range_end = core::cmp::min(vma_end, end);
1558                    if range_start % vma.page_size.bytes() != 0
1559                        || range_end % vma.page_size.bytes() != 0
1560                    {
1561                        return Err("unmap_range: partial unmap of 2MiB pages is not supported");
1562                    }
1563                }
1564            }
1565        }
1566
1567        // Process regions one by one to avoid heap allocation (Vec)
1568        let mut released_bytes = 0u64;
1569        loop {
1570            // Find the first overlapping region
1571            let region_info = {
1572                let regions = self.regions.lock();
1573                regions
1574                    .iter()
1575                    .find(|(&vma_start, vma)| {
1576                        let vma_end = vma_start + vma.page_count as u64 * vma.page_size.bytes();
1577                        vma_start < end && vma_end > addr
1578                    })
1579                    .map(|(&k, v)| (k, v.clone()))
1580            };
1581
1582            let Some((vma_start, vma)) = region_info else {
1583                break; // No more overlapping regions
1584            };
1585
1586            let vma_end = vma_start + vma.page_count as u64 * vma.page_size.bytes();
1587            let range_start = core::cmp::max(vma_start, addr);
1588            let range_end = core::cmp::min(vma_end, end);
1589            released_bytes = released_bytes.saturating_add(range_end.saturating_sub(range_start));
1590
1591            // 1. Hardware unmap
1592            // SAFETY: Logical ownership of address space.
1593            let mut mapper = unsafe { self.mapper() };
1594            let mut page_addr = range_start;
1595            let page_bytes = vma.page_size.bytes();
1596            while page_addr < range_end {
1597                // Lazy VMAs can contain unfaulted pages (no PTE). In that case
1598                // there is nothing to unmap in hardware; just update VMA metadata.
1599                if mapper.translate_addr(VirtAddr::new(page_addr)).is_none() {
1600                    page_addr += page_bytes;
1601                    continue;
1602                }
1603
1604                let _frame_addr = match vma.page_size {
1605                    VmaPageSize::Small => {
1606                        use crate::arch::xshim::Size4KiB;
1607                        let page = Page::<Size4KiB>::from_start_address(VirtAddr::new(page_addr))
1608                            .map_err(|_| "unmap_range: invalid 4K page address")?;
1609                        let (frame, flush) = mapper
1610                            .unmap(page)
1611                            .map_err(|_| "unmap_range: unmap 4K failed")?;
1612                        flush.flush();
1613                        frame.start_address()
1614                    }
1615                    VmaPageSize::Huge => {
1616                        use crate::arch::xshim::Size2MiB;
1617                        let page = Page::<Size2MiB>::from_start_address(VirtAddr::new(page_addr))
1618                            .map_err(|_| "unmap_range: invalid 2M page address")?;
1619                        let (frame, flush) = mapper
1620                            .unmap(page)
1621                            .map_err(|_| "unmap_range: unmap 2M failed")?;
1622                        flush.flush();
1623                        frame.start_address()
1624                    }
1625                };
1626
1627                let _ = self.unregister_effective_mapping(page_addr);
1628                page_addr += page_bytes;
1629            }
1630
1631            // 2. Update tracking: remove and re-insert fragments
1632            {
1633                let mut regions = self.regions.lock();
1634                regions.remove(&vma_start);
1635
1636                if range_start > vma_start {
1637                    let leading_pages =
1638                        ((range_start - vma_start) / vma.page_size.bytes()) as usize;
1639                    regions.insert(
1640                        vma_start,
1641                        VirtualMemoryRegion {
1642                            start: vma_start,
1643                            page_count: leading_pages,
1644                            flags: vma.flags,
1645                            vma_type: vma.vma_type,
1646                            page_size: vma.page_size,
1647                        },
1648                    );
1649                }
1650
1651                if range_end < vma_end {
1652                    let trailing_pages = ((vma_end - range_end) / vma.page_size.bytes()) as usize;
1653                    regions.insert(
1654                        range_end,
1655                        VirtualMemoryRegion {
1656                            start: range_end,
1657                            page_count: trailing_pages,
1658                            flags: vma.flags,
1659                            vma_type: vma.vma_type,
1660                            page_size: vma.page_size,
1661                        },
1662                    );
1663                }
1664            }
1665        }
1666
1667        // P0 fix: inter-CPU TLB shootdown after unmapping.
1668        // Other CPUs may still have stale translations for the unmapped pages.
1669        crate::arch::tlb::shootdown_range(VirtAddr::new(addr), VirtAddr::new(end));
1670
1671        crate::silo::release_current_task_memory(released_bytes);
1672        Ok(())
1673    }
1674
1675    /// Translate a virtual address to its mapped physical address.
1676    pub fn translate(&self, vaddr: VirtAddr) -> Option<PhysAddr> {
1677        // SAFETY: Read-only access to the page tables.
1678        let mapper = unsafe { self.mapper() };
1679        mapper.translate_addr(vaddr)
1680    }
1681
1682    /// Translate a virtual address to the current block handle and page-table flags.
1683    pub fn translate_to_handle(&self, vaddr: VirtAddr) -> Option<(BlockHandle, PageTableFlags)> {
1684        // SAFETY: Read-only access to the page tables.
1685        let mapper = unsafe { self.mapper() };
1686        let translated = mapper.translate(vaddr);
1687        match translated {
1688            TranslateResult::Mapped { frame, flags, .. } => {
1689                Some((resolve_handle(frame.start_address()), flags))
1690            }
1691            TranslateResult::NotMapped | TranslateResult::InvalidFrameAddress(_) => None,
1692        }
1693    }
1694
1695    /// Get the physical address of this address space's PML4 table.
1696    pub fn cr3(&self) -> PhysAddr {
1697        self.cr3_phys
1698    }
1699
1700    /// Switch the CPU to this address space by writing CR3.
1701    ///
1702    /// Skips the write if CR3 already points to this address space (avoids
1703    /// unnecessary TLB flush).
1704    ///
1705    /// # Safety
1706    /// The caller must ensure this address space's page tables are valid and
1707    /// that the kernel half is correctly mapped.
1708    pub unsafe fn switch_to(&self) {
1709        let (current_frame, _) = Cr3::read();
1710        if current_frame.start_address() == self.cr3_phys {
1711            return; // Already active : skip to avoid TLB flush.
1712        }
1713
1714        // SAFETY: cr3_phys points to a valid, 4KiB-aligned PML4 table with
1715        // the kernel half correctly populated.
1716        unsafe {
1717            let frame =
1718                X86PhysFrame::from_start_address(self.cr3_phys).expect("CR3 address not aligned");
1719            crate::e9_println!("C");
1720            Cr3::write(frame, Cr3Flags::empty());
1721            crate::e9_println!("c");
1722        }
1723    }
1724
1725    /// Whether this is the kernel address space.
1726    pub fn is_kernel(&self) -> bool {
1727        self.is_kernel
1728    }
1729
1730    /// Check if this address space has any user-space memory mappings.
1731    pub fn has_user_mappings(&self) -> bool {
1732        if self.is_kernel {
1733            return false;
1734        }
1735        let regions = self.regions.lock();
1736        // Check for any non-kernel mappings.
1737        regions.values().any(|vma| vma.vma_type != VmaType::Kernel)
1738    }
1739
1740    fn teardown_effective_mapping_for_drop(&self, mapping: EffectiveMapping) {
1741        // SAFETY: the address space is being torn down; any remaining user mapping
1742        // must be detached from ownership tracking before page-table reclamation.
1743        unsafe {
1744            let mut mapper = self.mapper();
1745            if mapper
1746                .translate_addr(VirtAddr::new(mapping.start))
1747                .is_some()
1748            {
1749                match mapping.page_size {
1750                    VmaPageSize::Small => {
1751                        let page =
1752                            Page::<Size4KiB>::from_start_address(VirtAddr::new(mapping.start))
1753                                .unwrap();
1754                        if let Err(error) = mapper.unmap(page) {
1755                            log::warn!(
1756                                "memory: drop cleanup failed to unmap 4K page at {:#x}: {:?}",
1757                                mapping.start,
1758                                error
1759                            );
1760                        }
1761                    }
1762                    VmaPageSize::Huge => {
1763                        let page =
1764                            Page::<Size2MiB>::from_start_address(VirtAddr::new(mapping.start))
1765                                .unwrap();
1766                        if let Err(error) = mapper.unmap(page) {
1767                            log::warn!(
1768                                "memory: drop cleanup failed to unmap 2M page at {:#x}: {:?}",
1769                                mapping.start,
1770                                error
1771                            );
1772                        }
1773                    }
1774                }
1775            }
1776        }
1777
1778        let _ = self.unregister_effective_mapping(mapping.start);
1779    }
1780
1781    fn teardown_region_for_drop(&self, start: u64, region: &VirtualMemoryRegion) {
1782        let len = (region.page_count as u64).saturating_mul(region.page_size.bytes());
1783        let mut page_addr = start;
1784        let page_bytes = region.page_size.bytes();
1785
1786        while page_addr < start.saturating_add(len) {
1787            if let Some(mapping) = self.effective_mapping_by_start(page_addr) {
1788                self.teardown_effective_mapping_for_drop(mapping);
1789            }
1790            page_addr += page_bytes;
1791        }
1792
1793        let _ = self.regions.lock().remove(&start);
1794        crate::silo::release_current_task_memory(len);
1795    }
1796
1797    /// Unmap all tracked user regions (best-effort).
1798    ///
1799    /// This frees user frames and clears the VMA list. Kernel mappings are untouched.
1800    /// Does not allocate memory.
1801    pub fn unmap_all_user_regions(&self) {
1802        if self.is_kernel {
1803            return;
1804        }
1805
1806        loop {
1807            let first = {
1808                let guard = self.regions.lock();
1809                guard
1810                    .iter()
1811                    .next()
1812                    .map(|(&start, region)| (start, region.clone()))
1813            };
1814
1815            let Some((start, region)) = first else {
1816                break;
1817            };
1818
1819            let len = (region.page_count as u64).saturating_mul(region.page_size.bytes());
1820            if self.unmap_range(region.start, len).is_err() {
1821                log::warn!(
1822                    "memory: unmap_all_user_regions fallback cleanup for {:#x}..{:#x}",
1823                    start,
1824                    start.saturating_add(len)
1825                );
1826                self.teardown_region_for_drop(start, &region);
1827            }
1828        }
1829
1830        let residual_mappings: Vec<EffectiveMapping> = {
1831            let guard = self.effective_mappings.lock();
1832            guard.values().copied().collect()
1833        };
1834        for mapping in residual_mappings {
1835            log::warn!(
1836                "memory: drop cleanup removing orphan effective mapping at {:#x} cap={}",
1837                mapping.start,
1838                mapping.cap_id.as_u64()
1839            );
1840            self.teardown_effective_mapping_for_drop(mapping);
1841            crate::silo::release_current_task_memory(mapping.page_size.bytes());
1842        }
1843    }
1844
1845    /// Performs the clone cow operation.
1846    pub fn clone_cow(&self) -> Result<Arc<AddressSpace>, &'static str> {
1847        if self.is_kernel {
1848            return Err("Cannot fork kernel address space");
1849        }
1850
1851        let child = Arc::new(AddressSpace::new_user()?);
1852
1853        let regions: Vec<VirtualMemoryRegion> = {
1854            let guard = self.regions.lock();
1855            guard.values().cloned().collect()
1856        };
1857        let effective_mappings: Vec<EffectiveMapping> = {
1858            let guard = self.effective_mappings.lock();
1859            guard.values().copied().collect()
1860        };
1861
1862        let mut tlb_flush_needed = false;
1863        let mut processed_pages = Vec::new();
1864
1865        let res: Result<(), &'static str> = (|| {
1866            let mut parent_mapper = unsafe { self.mapper() };
1867            let mut child_mapper = unsafe { child.mapper() };
1868            let mut frame_allocator = BuddyFrameAllocator;
1869
1870            for region in regions.iter() {
1871                // Register VMA in child.
1872                {
1873                    let mut child_regions = child.regions.lock();
1874                    child_regions.insert(region.start, region.clone());
1875                }
1876            }
1877
1878            for mapping in effective_mappings.iter().copied() {
1879                let vaddr = VirtAddr::new(mapping.start);
1880                let phys_frame_addr = mapping.handle.base;
1881                let mut new_flags = mapping.flags;
1882                let is_writable = mapping.flags.contains(PageTableFlags::WRITABLE);
1883                const COW_BIT: PageTableFlags = PageTableFlags::BIT_9;
1884
1885                if is_writable {
1886                    new_flags.remove(PageTableFlags::WRITABLE);
1887                    new_flags.insert(COW_BIT);
1888
1889                    unsafe {
1890                        let res: Result<(), &'static str> = match mapping.page_size {
1891                            VmaPageSize::Small => parent_mapper
1892                                .update_flags(
1893                                    Page::<Size4KiB>::from_start_address(vaddr).unwrap(),
1894                                    new_flags,
1895                                )
1896                                .map(|f| f.ignore())
1897                                .map_err(|_| "Failed to update parent 4K flags"),
1898                            VmaPageSize::Huge => parent_mapper
1899                                .update_flags(
1900                                    Page::<Size2MiB>::from_start_address(vaddr).unwrap(),
1901                                    new_flags,
1902                                )
1903                                .map(|f| f.ignore())
1904                                .map_err(|_| "Failed to update parent 2M flags"),
1905                        };
1906                        if let Err(e) = res {
1907                            return Err(e);
1908                        }
1909                    }
1910                    let _ = self.update_effective_mapping_flags(vaddr.as_u64(), new_flags);
1911                    tlb_flush_needed = true;
1912                    processed_pages.push((vaddr.as_u64(), mapping.flags, mapping.page_size));
1913                }
1914
1915                let handle = mapping.handle;
1916                crate::memory::cow::handle_inc_ref(handle).map_err(|error| {
1917                    log::warn!(
1918                        "clone_cow: failed to pin source handle {:#x}/{} for vaddr={:#x}: {:?}",
1919                        handle.base.as_u64(),
1920                        handle.order,
1921                        vaddr.as_u64(),
1922                        error
1923                    );
1924                    "Failed to pin source COW frame"
1925                })?;
1926
1927                // Map in child. We map it as WRITABLE first to ensure intermediate
1928                // page tables (PDPT, PD) are created with WRITABLE bit set.
1929                // If we mapped directly as COW (Read-only), some Mapper implementations
1930                // might create Read-Only intermediate tables, blocking future COW resolution.
1931                let map_flags = new_flags | PageTableFlags::WRITABLE;
1932
1933                unsafe {
1934                    let map_res: Result<(), &'static str> = match mapping.page_size {
1935                        VmaPageSize::Small => {
1936                            let page = Page::<Size4KiB>::from_start_address(vaddr).unwrap();
1937                            let frame =
1938                                crate::arch::xshim::PhysFrame::<Size4KiB>::containing_address(
1939                                    phys_frame_addr,
1940                                );
1941                            child_mapper
1942                                .map_to(page, frame, map_flags, &mut frame_allocator)
1943                                .map(|f| f.ignore())
1944                                .map_err(|_| "Failed to map 4K in child")
1945                        }
1946                        VmaPageSize::Huge => {
1947                            let page = Page::<Size2MiB>::from_start_address(vaddr).unwrap();
1948                            let frame =
1949                                crate::arch::xshim::PhysFrame::<Size2MiB>::containing_address(
1950                                    phys_frame_addr,
1951                                );
1952                            child_mapper
1953                                .map_to(page, frame, map_flags, &mut frame_allocator)
1954                                .map(|f| f.ignore())
1955                                .map_err(|_| "Failed to map 2M in child")
1956                        }
1957                    };
1958
1959                    if let Err(e) = map_res {
1960                        crate::memory::cow::handle_dec_ref(handle);
1961                        return Err(e);
1962                    }
1963
1964                    // Now downgrade to the actual COW flags (which may be Read-Only).
1965                    if !new_flags.contains(PageTableFlags::WRITABLE) {
1966                        let downgrade_res: Result<(), &'static str> = match mapping.page_size {
1967                            VmaPageSize::Small => {
1968                                let page = Page::<Size4KiB>::from_start_address(vaddr).unwrap();
1969                                child_mapper
1970                                    .update_flags(page, new_flags)
1971                                    .map(|f| f.ignore())
1972                                    .map_err(|_| "Failed to update child 4K flags")
1973                            }
1974                            VmaPageSize::Huge => {
1975                                let page = Page::<Size2MiB>::from_start_address(vaddr).unwrap();
1976                                child_mapper
1977                                    .update_flags(page, new_flags)
1978                                    .map(|f| f.ignore())
1979                                    .map_err(|_| "Failed to update child 2M flags")
1980                            }
1981                        };
1982                        if let Err(e) = downgrade_res {
1983                            let unmapped = match mapping.page_size {
1984                                VmaPageSize::Small => {
1985                                    let page = Page::<Size4KiB>::from_start_address(vaddr).unwrap();
1986                                    child_mapper.unmap(page).map(|(_, f)| f.ignore()).is_ok()
1987                                }
1988                                VmaPageSize::Huge => {
1989                                    let page = Page::<Size2MiB>::from_start_address(vaddr).unwrap();
1990                                    child_mapper.unmap(page).map(|(_, f)| f.ignore()).is_ok()
1991                                }
1992                            };
1993                            if unmapped {
1994                                crate::memory::cow::handle_dec_ref(handle);
1995                            }
1996                            return Err(e);
1997                        }
1998                    }
1999                }
2000
2001                if child
2002                    .register_effective_mapping(EffectiveMapping {
2003                        start: vaddr.as_u64(),
2004                        cap_id: allocate_mapping_cap_id(),
2005                        handle,
2006                        flags: new_flags,
2007                        page_size: mapping.page_size,
2008                    })
2009                    .is_err()
2010                {
2011                    match mapping.page_size {
2012                        VmaPageSize::Small => {
2013                            let page = Page::<Size4KiB>::from_start_address(vaddr).unwrap();
2014                            if let Ok((_, flush)) = child_mapper.unmap(page) {
2015                                flush.ignore();
2016                            }
2017                        }
2018                        VmaPageSize::Huge => {
2019                            let page = Page::<Size2MiB>::from_start_address(vaddr).unwrap();
2020                            if let Ok((_, flush)) = child_mapper.unmap(page) {
2021                                flush.ignore();
2022                            }
2023                        }
2024                    }
2025                    crate::memory::cow::handle_dec_ref(handle);
2026                    return Err("Failed to track child COW mapping");
2027                }
2028
2029                crate::memory::cow::handle_dec_ref(handle);
2030            }
2031            Ok(())
2032        })();
2033
2034        let tlb_flush_range = if tlb_flush_needed && !processed_pages.is_empty() {
2035            let mut range_start = u64::MAX;
2036            let mut range_end = 0u64;
2037            for (vaddr, _, page_size) in &processed_pages {
2038                range_start = range_start.min(*vaddr);
2039                range_end = range_end.max(vaddr.saturating_add(page_size.bytes()));
2040            }
2041            if range_start < range_end {
2042                Some((range_start, range_end))
2043            } else {
2044                None
2045            }
2046        } else {
2047            None
2048        };
2049
2050        if let Err(e) = res {
2051            log::error!("clone_cow error: {}. Rolling back...", e);
2052            let mut parent_mapper = unsafe { self.mapper() };
2053            for &(vaddr, original_flags, page_size) in processed_pages.iter().rev() {
2054                if original_flags.contains(PageTableFlags::WRITABLE) {
2055                    unsafe {
2056                        match page_size {
2057                            VmaPageSize::Small => {
2058                                let _ = parent_mapper.update_flags(
2059                                    Page::<Size4KiB>::from_start_address(VirtAddr::new(vaddr))
2060                                        .unwrap(),
2061                                    original_flags,
2062                                );
2063                            }
2064                            VmaPageSize::Huge => {
2065                                let _ = parent_mapper.update_flags(
2066                                    Page::<Size2MiB>::from_start_address(VirtAddr::new(vaddr))
2067                                        .unwrap(),
2068                                    original_flags,
2069                                );
2070                            }
2071                        };
2072                    }
2073                    let _ = self.update_effective_mapping_flags(vaddr, original_flags);
2074                }
2075            }
2076            if let Some((range_start, range_end)) = tlb_flush_range {
2077                crate::arch::tlb::shootdown_range(
2078                    VirtAddr::new(range_start),
2079                    VirtAddr::new(range_end),
2080                );
2081            }
2082            return Err(e);
2083        }
2084
2085        if let Some((range_start, range_end)) = tlb_flush_range {
2086            crate::arch::tlb::shootdown_range(VirtAddr::new(range_start), VirtAddr::new(range_end));
2087        }
2088        Ok(child)
2089    }
2090
2091    /// Releases user page tables.
2092    fn free_user_page_tables(&self) {
2093        if self.is_kernel {
2094            return;
2095        }
2096
2097        // SAFETY: We have logical ownership of this address space during drop.
2098        let l4 = unsafe { &mut *self.l4_table_virt.as_mut_ptr::<PageTable>() };
2099
2100        for i in 0..256 {
2101            if !l4[i].flags().contains(PageTableFlags::PRESENT) {
2102                continue;
2103            }
2104            let l3_frame = match l4[i].frame() {
2105                Ok(f) => f,
2106                Err(_) => {
2107                    l4[i].set_unused();
2108                    continue;
2109                }
2110            };
2111
2112            free_l3_table(l3_frame);
2113            l4[i].set_unused();
2114        }
2115    }
2116}
2117
2118impl Drop for AddressSpace {
2119    /// Performs the drop operation.
2120    fn drop(&mut self) {
2121        if self.is_kernel {
2122            return; // Never free the kernel address space.
2123        }
2124
2125        log::trace!("AddressSpace::drop begin CR3={:#x}", self.cr3_phys.as_u64());
2126
2127        // Best-effort cleanup of user mappings.
2128        self.unmap_all_user_regions();
2129        #[cfg(not(feature = "selftest"))]
2130        self.free_user_page_tables();
2131        #[cfg(feature = "selftest")]
2132        {
2133            // Runtime selftests create/destroy many temporary address spaces and
2134            // currently expose instability in recursive page-table teardown.
2135            // Keep tests deterministic by skipping deep PT reclaim in this mode.
2136            log::trace!(
2137                "AddressSpace::drop selftest mode: skipping deep page-table free for CR3={:#x}",
2138                self.cr3_phys.as_u64()
2139            );
2140        }
2141
2142        // Free the PML4 frame itself.
2143        // NOTE: Recursive freeing of intermediate page tables (L3/L2/L1) that
2144        // belong exclusively to the user half is deferred to P2.
2145        let phys_frame = crate::memory::PhysFrame {
2146            start_address: self.cr3_phys,
2147        };
2148        crate::sync::with_irqs_disabled(|token| {
2149            crate::memory::free_frame(token, phys_frame);
2150        });
2151
2152        log::trace!("AddressSpace::drop end CR3={:#x}", self.cr3_phys.as_u64());
2153        log::debug!(
2154            "User address space dropped: CR3={:#x}",
2155            self.cr3_phys.as_u64()
2156        );
2157    }
2158}
2159
2160// ---------------------------------------------------------------------------
2161// Page table cleanup helpers (user half only)
2162// ---------------------------------------------------------------------------
2163
2164/// Releases frame.
2165fn free_frame(phys: PhysAddr) {
2166    let phys_frame = crate::memory::PhysFrame {
2167        start_address: phys,
2168    };
2169    crate::sync::with_irqs_disabled(|token| {
2170        crate::memory::free_frame(token, phys_frame);
2171    });
2172}
2173
2174/// Releases l1 table.
2175fn free_l1_table(frame: X86PhysFrame<Size4KiB>) {
2176    let l1_virt = VirtAddr::new(crate::memory::phys_to_virt(frame.start_address().as_u64()));
2177    // SAFETY: l1_virt points to a valid page table frame in HHDM.
2178    let l1 = unsafe { &mut *l1_virt.as_mut_ptr::<PageTable>() };
2179    for entry in l1.iter_mut() {
2180        if entry.flags().contains(PageTableFlags::PRESENT) {
2181            // Mapped frames are already freed via unmap_all_user_regions.
2182            entry.set_unused();
2183        }
2184    }
2185    free_frame(frame.start_address());
2186}
2187
2188/// Releases l2 table.
2189fn free_l2_table(frame: X86PhysFrame<Size4KiB>) {
2190    let l2_virt = VirtAddr::new(crate::memory::phys_to_virt(frame.start_address().as_u64()));
2191    let l2 = unsafe { &mut *l2_virt.as_mut_ptr::<PageTable>() };
2192    for entry in l2.iter_mut() {
2193        if !entry.flags().contains(PageTableFlags::PRESENT) {
2194            continue;
2195        }
2196        if entry.flags().contains(PageTableFlags::HUGE_PAGE) {
2197            // 2 MiB pages are not expected in user space today.
2198            entry.set_unused();
2199            continue;
2200        }
2201        if let Ok(l1_frame) = entry.frame() {
2202            free_l1_table(l1_frame);
2203        }
2204        entry.set_unused();
2205    }
2206    free_frame(frame.start_address());
2207}
2208
2209/// Releases l3 table.
2210fn free_l3_table(frame: X86PhysFrame<Size4KiB>) {
2211    let l3_virt = VirtAddr::new(crate::memory::phys_to_virt(frame.start_address().as_u64()));
2212    let l3 = unsafe { &mut *l3_virt.as_mut_ptr::<PageTable>() };
2213    for entry in l3.iter_mut() {
2214        if !entry.flags().contains(PageTableFlags::PRESENT) {
2215            continue;
2216        }
2217        if entry.flags().contains(PageTableFlags::HUGE_PAGE) {
2218            // 1 GiB pages are not expected in user space today.
2219            entry.set_unused();
2220            continue;
2221        }
2222        if let Ok(l2_frame) = entry.frame() {
2223            free_l2_table(l2_frame);
2224        }
2225        entry.set_unused();
2226    }
2227    free_frame(frame.start_address());
2228}
2229
2230// ---------------------------------------------------------------------------
2231// Kernel address space singleton
2232// ---------------------------------------------------------------------------
2233
2234static KERNEL_ADDRESS_SPACE: Once<Arc<AddressSpace>> = Once::new();
2235
2236/// Initialize the kernel address space singleton.
2237///
2238/// Must be called once during boot, after paging is initialized, before the
2239/// scheduler creates any tasks.
2240///
2241/// # Safety
2242/// Must be called in single-threaded init context.
2243pub unsafe fn init_kernel_address_space() {
2244    KERNEL_ADDRESS_SPACE.call_once(|| {
2245        // SAFETY: Called once, single-threaded, paging initialized.
2246        Arc::new(unsafe { AddressSpace::new_kernel() })
2247    });
2248}
2249
2250/// Get a reference to the kernel address space.
2251///
2252/// Panics if called before `init_kernel_address_space()`.
2253pub fn kernel_address_space() -> &'static Arc<AddressSpace> {
2254    KERNEL_ADDRESS_SPACE
2255        .get()
2256        .expect("Kernel address space not initialized")
2257}