Skip to main content

strat9_kernel/memory/
heap.rs

1// Heap allocator: slab sub-allocator + VM-backed large-allocation path.
2//
3// Small allocations (effective size <= 2048 B) come from per-size-class slab
4// free lists.  Each slab class draws whole pages from the buddy allocator and
5// carves them into fixed-size blocks.  Freed blocks return to the slab free
6// list, so the buddy's page counter stabilises after warm-up instead of
7// growing on every tiny allocation.
8//
9// Large allocations (> 2048 B) go through the kernel vmalloc backend:
10// virtually contiguous, physically fragmented, and independent from
11// high-order physically contiguous buddy blocks.
12//
13// Lock ordering : SLAB_ALLOC (outer) may call the frame-allocation helpers.
14// Those helpers can hit a CPU-local cache (no global buddy lock) or fall back
15// to the global buddy lock as needed.
16
17use crate::{arch::xshim::PhysAddr, memory, sync::SpinLock};
18use core::{
19    alloc::{GlobalAlloc, Layout},
20    ptr,
21    sync::atomic::{AtomicUsize, Ordering as AtomicOrdering},
22};
23
24// ---------------------------------------------------------------------------
25// Slab size classes
26// ---------------------------------------------------------------------------
27
28/// Slab block sizes chosen to bound internal fragmentation to ~25% worst-case
29/// (average ~12%) instead of 50% with pure power-of-two classes.
30///
31/// The progression follows a roughly 1.25× step above 64 bytes.  Below 64
32/// bytes the absolute waste of a 2× jump is small enough (max 32 bytes) to
33/// keep power-of-two boundaries, avoiding an explosion of size classes.
34///
35/// | Class range | Step      | Max waste |
36/// |-------------|-----------|-----------|
37/// | 8 to  64 B  | x2 / 1,5× | ≤ 32 B    |
38/// |64 to 256 B  | ~1.25×    | ≤ 64 B    |
39/// |256 to 2048 B| 1.25×     | ≤ 512 B   |
40
41const SLAB_SIZES: [usize; 26] = [
42    8, 16, 24, 32, 48, 64, 80, 96, 112, 128, 160, 192, 224, 256, 320, 384, 448, 512, 640, 768, 896,
43    1024, 1280, 1536, 1792, 2048,
44];
45const NUM_SLABS: usize = SLAB_SIZES.len();
46/// Allocations with effective size above this threshold bypass the slab.
47/// Maximum payload size that can be served from the slab allocator.
48/// Must account for REDZONE_TAIL: the actual slab class must be >= payload + redzone.
49const MAX_SLAB_SIZE: usize = SLAB_SIZES[NUM_SLABS - 1] - REDZONE_TAIL;
50
51#[derive(Clone, Copy, Debug, Eq, PartialEq)]
52pub enum KernelHeapBackend {
53    Slab,
54    Vmalloc,
55}
56
57#[derive(Clone, Copy, Debug, Eq, PartialEq)]
58pub enum KernelHeapAllocError {
59    InvalidLayout,
60    /// [`GlobalAlloc`] large path uses vmalloc, which only guarantees 4 KiB alignment.
61    AlignmentExceedsKernelPage {
62        align: usize,
63    },
64    SlabRefillFailed {
65        effective: usize,
66        class_size: usize,
67    },
68    Vmalloc(memory::vmalloc::VmallocError),
69}
70
71#[derive(Clone, Copy, Debug, Eq, PartialEq)]
72pub struct KernelHeapFailureSnapshot {
73    pub backend: KernelHeapBackend,
74    pub requested_size: usize,
75    pub align: usize,
76    pub effective_size: usize,
77    pub error: KernelHeapAllocError,
78}
79
80#[derive(Clone, Copy, Debug, Eq, PartialEq)]
81pub struct SlabDiagSnapshot {
82    pub pages_allocated: usize,
83    pub pages_reclaimed: usize,
84    pub pages_live: usize,
85}
86
87#[inline]
88pub(crate) fn classify_kernel_heap_backend(layout: Layout) -> KernelHeapBackend {
89    let effective = layout.size().max(layout.align());
90    if effective <= MAX_SLAB_SIZE {
91        KernelHeapBackend::Slab
92    } else {
93        KernelHeapBackend::Vmalloc
94    }
95}
96
97// =============================================================================
98// CRITICAL: slab corruption detection
99//
100// Set HEAP_POISON_ENABLED to true during debugging of heap-corruption crashes.
101// When enabled:
102//   - Every block carved by refill() is filled with POISON_BYTE in bytes [8..N-4]
103//     and stamped with SLAB_CANARY in the last 4 bytes.
104//   - dealloc_block() restores the canary and re-poisons before linking.
105//   - alloc_block() verifies poison and canary before handing the block out;
106//     a mismatch is logged immediately via serial_println! (non-allocating).
107//
108// This detects:
109//   - Use-after-free: a write to a freed slab block overwrites poison bytes.
110//   - Buffer overflow: a write past the end overwrites the canary or the next
111//     block's free-list pointer.
112//
113// Cost: one memset + canary write per alloc/dealloc for slab classes.
114// =============================================================================
115const HEAP_POISON_ENABLED: bool = true;
116/// Byte pattern written to the body of freed slab blocks.
117const POISON_BYTE: u8 = 0xDE;
118/// Canary word placed at the last 4 bytes of each slab block.
119const SLAB_CANARY: u32 = 0xDEAD_BEEF;
120/// Bytes reserved at the end of every slab block for the canary + padding.
121/// The user-visible allocation size is `slab_size - REDZONE_TAIL`, ensuring
122/// writes to the full user area cannot overwrite the canary.
123const REDZONE_TAIL: usize = 8;
124
125// ---------------------------------------------------------------------------
126// Slab page header : embedded at byte 0 of every buddy page used by a class.
127// Blocks start at offset SLAB_HEADER_SIZE within the page.
128// ---------------------------------------------------------------------------
129
130/// Header at the base of each 4 KiB page dedicated to a slab class.
131///
132/// Page layout:
133/// ```text
134/// [0 .. SLAB_HEADER_SIZE)   SlabPageHeader  (24 bytes)
135/// [SLAB_HEADER_SIZE .. 4096) slab blocks, each SLAB_SIZES[ci] bytes
136/// ```
137///
138/// A page lives in `partial_pages[ci]` while `0 < free_count < total_blocks`.
139/// It is removed when all blocks are allocated (`free_count == 0`), and is
140/// reclaimed to the buddy allocator when it becomes fully empty again
141/// (`free_count == total_blocks`).
142#[repr(C)]
143struct SlabPageHeader {
144    /// Next page in the partial list for this class (null = end of list).
145    next_partial: *mut SlabPageHeader,
146    /// Head of the intra-page free-block chain (null = page is full).
147    free_head: *mut u8,
148    /// Free blocks currently in this page.
149    free_count: u32,
150    /// Total blocks this page can hold (constant per class after refill).
151    total_blocks: u32,
152}
153
154// SAFETY: only accessed under SLAB_ALLOC spinlock.
155unsafe impl Send for SlabPageHeader {}
156unsafe impl Sync for SlabPageHeader {}
157
158/// Byte offset at which slab blocks begin within each slab page.
159const SLAB_HEADER_SIZE: usize = core::mem::size_of::<SlabPageHeader>();
160
161// Compile-time invariants.
162const _: () = assert!(
163    SLAB_HEADER_SIZE == 24,
164    "SlabPageHeader size changed : update docs"
165);
166const _: () = assert!(
167    (4096 - SLAB_HEADER_SIZE) / SLAB_SIZES[NUM_SLABS - 1] >= 1,
168    "SlabPageHeader too large: largest slab class gets 0 blocks per page"
169);
170
171// ---------------------------------------------------------------------------
172// SlabState
173// ---------------------------------------------------------------------------
174
175/// Per-size-class partial-page lists.
176///
177/// `partial_pages[ci]` is the head of a singly-linked list of `SlabPageHeader`
178/// nodes for class `ci`.  A page enters the list on `refill` and on the first
179/// `dealloc` after going full.  It leaves the list when all its blocks are
180/// allocated (it silently becomes "full") or when it becomes completely empty
181/// (it is then returned to the buddy allocator).
182struct SlabState {
183    partial_pages: [*mut SlabPageHeader; NUM_SLABS],
184}
185
186// SAFETY: protected exclusively through `SLAB_ALLOC: SpinLock<SlabState>`.
187unsafe impl Send for SlabState {}
188unsafe impl Sync for SlabState {}
189
190impl SlabState {
191    const fn new() -> Self {
192        SlabState {
193            partial_pages: [ptr::null_mut(); NUM_SLABS],
194        }
195    }
196
197    /// Return the slab class index for `layout`.
198    ///
199    /// The chosen class must be large enough for the payload and guarantee the
200    /// requested alignment for every block carved from that class.
201    #[inline]
202    fn class_index_for_layout(layout: Layout) -> usize {
203        for (i, &s) in SLAB_SIZES.iter().enumerate() {
204            // P0 fix: require REDZONE_TAIL bytes beyond the user payload so the
205            // canary (at offset s-4) is always outside the user's writable area.
206            if layout.size() + REDZONE_TAIL <= s && layout.align() <= slab_class_alignment(i) {
207                return i;
208            }
209        }
210        unreachable!("class_index_for_layout called for unsupported slab layout")
211    }
212
213    /// Allocate one buddy page, write a `SlabPageHeader` at its base, carve
214    /// the remaining space into blocks, and prepend the page to `partial_pages[ci]`.
215    unsafe fn refill(&mut self, ci: usize, token: &crate::sync::IrqDisabledToken) {
216        debug_assert!(
217            !crate::arch::interrupts_enabled(),
218            "refill: IRQs must be disabled (IrqDisabledToken contract)"
219        );
220        let slab_size = SLAB_SIZES[ci];
221        let slab_align = slab_class_alignment(ci);
222        let blocks_offset = (SLAB_HEADER_SIZE + slab_align - 1) & !(slab_align - 1);
223        let num_blocks = (4096 - blocks_offset) / slab_size;
224        debug_assert!(
225            num_blocks >= 1,
226            "refill: slab_size {} yields 0 blocks",
227            slab_size
228        );
229
230        let frame = match memory::allocate_frame(token) {
231            Ok(f) => f,
232            Err(_) => return, // OOM : alloc_block will see null partial and return null
233        };
234        SLAB_PAGES_ALLOCATED.fetch_add(1, AtomicOrdering::Relaxed);
235
236        let page_virt = super::phys_to_virt(frame.start_address.as_u64()) as *mut u8;
237
238        // Initialise page header at byte 0.
239        let header = page_virt as *mut SlabPageHeader;
240        (*header).next_partial = ptr::null_mut();
241        (*header).free_head = ptr::null_mut();
242        (*header).free_count = 0;
243        (*header).total_blocks = num_blocks as u32;
244
245        // Carve blocks starting at an alignment-respecting offset, highest index first so
246        // the lowest-address block ends up at the head (cosmetic only).
247        let blocks_start = page_virt.add(blocks_offset);
248        for i in (0..num_blocks).rev() {
249            let block = blocks_start.add(i * slab_size);
250            debug_assert_eq!(
251                (block as usize) & (slab_align - 1),
252                0,
253                "slab block alignment invariant broken for class {}",
254                slab_size
255            );
256            *(block as *mut *mut u8) = (*header).free_head;
257            if HEAP_POISON_ENABLED {
258                let end = slab_size.saturating_sub(4);
259                for off in 8..end {
260                    *block.add(off) = POISON_BYTE;
261                }
262                if slab_size >= 12 {
263                    let cp = block.add(slab_size - 4) as *mut u32;
264                    *cp = SLAB_CANARY;
265                }
266            }
267            (*header).free_head = block;
268            (*header).free_count += 1;
269        }
270
271        // Prepend to partial list.
272        (*header).next_partial = self.partial_pages[ci];
273        self.partial_pages[ci] = header;
274    }
275
276    /// Pop one block from the first partial page for class `ci`.
277    /// Calls `refill` when the partial list is empty.  Returns null on OOM.
278    unsafe fn alloc_block(&mut self, ci: usize, token: &crate::sync::IrqDisabledToken) -> *mut u8 {
279        if self.partial_pages[ci].is_null() {
280            self.refill(ci, token);
281        }
282        let header = self.partial_pages[ci];
283        if header.is_null() {
284            return ptr::null_mut();
285        }
286
287        let block = (*header).free_head;
288        debug_assert!(
289            !block.is_null(),
290            "alloc_block: partial page has null free_head"
291        );
292
293        (*header).free_head = *(block as *const *mut u8);
294        (*header).free_count -= 1;
295
296        // Remove page from partial list when it is now full (free_count == 0).
297        if (*header).free_count == 0 {
298            self.partial_pages[ci] = (*header).next_partial;
299            (*header).next_partial = ptr::null_mut();
300        }
301
302        if HEAP_POISON_ENABLED {
303            let slab_size = SLAB_SIZES[ci];
304            let end = slab_size.saturating_sub(4);
305            let mut bad_off: Option<usize> = None;
306            for off in 8..end {
307                if *block.add(off) != POISON_BYTE {
308                    bad_off = Some(off);
309                    break;
310                }
311            }
312            if let Some(off) = bad_off {
313                let b0 = *block.add(off);
314                let b1 = if off + 1 < slab_size {
315                    *block.add(off + 1)
316                } else {
317                    0
318                };
319                let b2 = if off + 2 < slab_size {
320                    *block.add(off + 2)
321                } else {
322                    0
323                };
324                let b3 = if off + 3 < slab_size {
325                    *block.add(off + 3)
326                } else {
327                    0
328                };
329                crate::serial_println!(
330                    "\x1b[1;31m[HEAP] USE-AFTER-FREE: slab[{}] block={:#x} off={} bytes=[{:02x} {:02x} {:02x} {:02x}]\x1b[0m",
331                    slab_size,
332                    block as u64,
333                    off,
334                    b0,
335                    b1,
336                    b2,
337                    b3
338                );
339            }
340            if slab_size >= 12 {
341                let canary = *(block.add(slab_size - 4) as *const u32);
342                if canary != SLAB_CANARY {
343                    crate::serial_println!(
344                        "\x1b[1;31m[HEAP] CANARY OVERFLOW: slab[{}] block={:#x} expected={:#x} got={:#x}\x1b[0m",
345                        slab_size,
346                        block as u64,
347                        SLAB_CANARY,
348                        canary
349                    );
350                }
351            }
352        }
353
354        block
355    }
356
357    /// Return `ptr` to its slab page and reclaim the page to the buddy
358    /// allocator if it becomes fully empty.
359    unsafe fn dealloc_block(
360        &mut self,
361        ptr: *mut u8,
362        ci: usize,
363        token: &crate::sync::IrqDisabledToken,
364    ) {
365        let slab_size = SLAB_SIZES[ci];
366
367        if HEAP_POISON_ENABLED {
368            if slab_size >= 12 {
369                let cp = ptr.add(slab_size - 4) as *mut u32;
370                *cp = SLAB_CANARY;
371            }
372            let end = slab_size.saturating_sub(4);
373            for off in 8..end {
374                *ptr.add(off) = POISON_BYTE;
375            }
376        }
377
378        // Locate the page header: round ptr down to 4 KiB boundary.
379        let page_base = (ptr as usize) & !0xFFF;
380        let header = page_base as *mut SlabPageHeader;
381
382        let was_full = (*header).free_count == 0;
383
384        // Push block onto the page's intra-page free list.
385        *(ptr as *mut *mut u8) = (*header).free_head;
386        (*header).free_head = ptr;
387        (*header).free_count += 1;
388
389        if was_full {
390            // Page went full -> partial: re-insert at list head.
391            (*header).next_partial = self.partial_pages[ci];
392            self.partial_pages[ci] = header;
393        }
394
395        // Reclaim fully-empty pages to the buddy allocator.
396        if (*header).free_count == (*header).total_blocks {
397            self.remove_from_partial(header, ci);
398            let phys = super::virt_to_phys(page_base as u64);
399            // Zero the header before freeing to catch accidental reuse.
400            core::ptr::write_bytes(header as *mut u8, 0, SLAB_HEADER_SIZE);
401            let frame = memory::frame::PhysFrame {
402                start_address: PhysAddr::new(phys),
403            };
404            memory::free_frame(token, frame);
405            SLAB_PAGES_RECLAIMED.fetch_add(1, AtomicOrdering::Relaxed);
406        }
407    }
408
409    /// Unlink `page` from `partial_pages[ci]`.  O(n) in partial-list length.
410    unsafe fn remove_from_partial(&mut self, page: *mut SlabPageHeader, ci: usize) {
411        if self.partial_pages[ci] == page {
412            self.partial_pages[ci] = (*page).next_partial;
413            (*page).next_partial = ptr::null_mut();
414            return;
415        }
416        let mut cur = self.partial_pages[ci];
417        while !cur.is_null() {
418            let next = (*cur).next_partial;
419            if next == page {
420                (*cur).next_partial = (*page).next_partial;
421                (*page).next_partial = ptr::null_mut();
422                return;
423            }
424            cur = next;
425        }
426        debug_assert!(
427            false,
428            "remove_from_partial: page {:p} not found in class {} list",
429            page, ci
430        );
431    }
432}
433
434static SLAB_ALLOC: SpinLock<SlabState> = SpinLock::new(SlabState::new());
435static LAST_HEAP_FAILURE: SpinLock<Option<KernelHeapFailureSnapshot>> = SpinLock::new(None);
436
437/// Total buddy pages ever handed to the slab allocator.
438static SLAB_PAGES_ALLOCATED: AtomicUsize = AtomicUsize::new(0);
439/// Total buddy pages ever returned from the slab allocator (fully-empty reclaim).
440static SLAB_PAGES_RECLAIMED: AtomicUsize = AtomicUsize::new(0);
441
442/// Returns the slab lock address for deadlock tracing.
443pub fn debug_slab_lock_addr() -> usize {
444    &SLAB_ALLOC as *const _ as usize
445}
446
447/// Register slab lock for E9 trace (call from init).
448pub fn debug_register_slab_trace() {
449    crate::sync::debug_set_trace_slab_addr(debug_slab_lock_addr());
450}
451
452// ---------------------------------------------------------------------------
453// GlobalAlloc implementation
454// ---------------------------------------------------------------------------
455
456pub struct LockedHeap;
457
458unsafe impl GlobalAlloc for LockedHeap {
459    /// Performs the alloc operation.
460    unsafe fn alloc(&self, layout: Layout) -> *mut u8 {
461        try_alloc_kernel_heap(layout).unwrap_or(ptr::null_mut())
462    }
463
464    /// Performs the dealloc operation.
465    unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) {
466        let effective = layout.size().max(layout.align());
467
468        match classify_kernel_heap_backend(layout) {
469            KernelHeapBackend::Slab => {
470                // --- slab path: return block to free list ---
471                let ci = SlabState::class_index_for_layout(layout);
472                let _cpu = crate::arch::percpu::current_cpu_index();
473                let _irq_enabled = crate::arch::interrupts_enabled();
474                #[cfg(debug_assertions)]
475                if !_irq_enabled {
476                    use core::sync::atomic::{AtomicUsize, Ordering};
477                    static HEAP_D_COUNT: AtomicUsize = AtomicUsize::new(0);
478                    let n = HEAP_D_COUNT.fetch_add(1, Ordering::Relaxed);
479                    if n % 100 == 0 {
480                        crate::e9_println!(
481                            "HEAP-D cpu={} irq=0 size={} ci={} n={}",
482                            _cpu,
483                            effective,
484                            ci,
485                            n
486                        );
487                    }
488                }
489                // Catch layout mismatches where a vmalloc pointer is freed with
490                // a small layout (classify_kernel_heap_backend routes to Slab).
491                // This means the caller passed a different layout to dealloc than
492                // was used for alloc : a GlobalAlloc contract violation.
493                #[cfg(debug_assertions)]
494                {
495                    let addr = ptr as u64;
496                    if addr >= crate::memory::vmalloc::VMALLOC_VIRT_START
497                        && addr < crate::memory::vmalloc::VMALLOC_VIRT_END
498                    {
499                        crate::serial_println!(
500                            "[heap][bug] slab dealloc: ptr {:#x} is in vmalloc range : layout mismatch",
501                            addr
502                        );
503                        debug_assert!(
504                            false,
505                            "slab dealloc with vmalloc pointer : alloc/dealloc layout mismatch"
506                        );
507                    }
508                }
509                let mut slab = SLAB_ALLOC.lock();
510                slab.with_mut_and_token(|s, token| s.dealloc_block(ptr, ci, token));
511            }
512            KernelHeapBackend::Vmalloc => {
513                // vmalloc path: free via the vmalloc arena
514                let addr = ptr as u64;
515                if addr >= crate::memory::vmalloc::VMALLOC_VIRT_START
516                    && addr < crate::memory::vmalloc::VMALLOC_VIRT_END
517                {
518                    let ok = crate::sync::with_irqs_disabled(|token| {
519                        crate::memory::free_kernel_virtual(ptr, token)
520                    });
521                    if !ok {
522                        crate::serial_println!(
523                            "[heap][leak] vmalloc free: no live mapping at {:#x} (wrong base or double-free?)",
524                            addr
525                        );
526                    }
527                } else {
528                    // Pointer is outside the vmalloc arena with a large-allocation
529                    // layout : nothing is freed (GlobalAlloc contract violation).
530                    crate::serial_println!(
531                        "[heap][leak] vmalloc dealloc: ptr {:#x} outside vmalloc arena [{:#x}..{:#x}]",
532                        addr,
533                        crate::memory::vmalloc::VMALLOC_VIRT_START,
534                        crate::memory::vmalloc::VMALLOC_VIRT_END,
535                    );
536                    #[cfg(debug_assertions)]
537                    debug_assert!(
538                        false,
539                        "vmalloc dealloc with out-of-range pointer : memory leaked"
540                    );
541                }
542            }
543        }
544    }
545}
546
547fn record_heap_failure(
548    layout: Layout,
549    effective: usize,
550    backend: KernelHeapBackend,
551    error: KernelHeapAllocError,
552) -> KernelHeapAllocError {
553    *LAST_HEAP_FAILURE.lock() = Some(KernelHeapFailureSnapshot {
554        backend,
555        requested_size: layout.size(),
556        align: layout.align(),
557        effective_size: effective,
558        error,
559    });
560    error
561}
562
563pub fn last_heap_failure_snapshot() -> Option<KernelHeapFailureSnapshot> {
564    *LAST_HEAP_FAILURE.lock()
565}
566
567pub fn slab_diag_snapshot() -> SlabDiagSnapshot {
568    let allocated = SLAB_PAGES_ALLOCATED.load(AtomicOrdering::Relaxed);
569    let reclaimed = SLAB_PAGES_RECLAIMED.load(AtomicOrdering::Relaxed);
570    SlabDiagSnapshot {
571        pages_allocated: allocated,
572        pages_reclaimed: reclaimed,
573        pages_live: allocated.saturating_sub(reclaimed),
574    }
575}
576
577/// Number of slab size classes.
578pub const SLAB_NUM_CLASSES: usize = NUM_SLABS;
579
580/// Block size in bytes for slab class `ci`.
581///
582/// Panics in debug if `ci >= SLAB_NUM_CLASSES`.
583#[inline]
584pub fn slab_class_size(ci: usize) -> usize {
585    SLAB_SIZES[ci]
586}
587
588/// Guaranteed alignment in bytes for slab class `ci`.
589///
590/// This is the largest power-of-two divisor of the class size.
591#[inline]
592pub fn slab_class_alignment(ci: usize) -> usize {
593    1usize << SLAB_SIZES[ci].trailing_zeros()
594}
595
596/// Number of blocks that fit in one buddy page for slab class `ci`.
597///
598/// Accounts for the `SlabPageHeader` at the base of each page.
599#[inline]
600pub fn slab_blocks_per_page(ci: usize) -> usize {
601    let align = slab_class_alignment(ci);
602    let blocks_offset = (SLAB_HEADER_SIZE + align - 1) & !(align - 1);
603    (4096 - blocks_offset) / SLAB_SIZES[ci]
604}
605
606/// Returns whether the slab page at `page_base` is currently present in the
607/// partial-page list for class `ci`.
608///
609/// Returns `None` if the slab allocator lock is contended. Intended for
610/// shell/debug validation, not for allocator hot paths.
611pub fn slab_page_in_partial_list(ci: usize, page_base: u64) -> Option<bool> {
612    let mut guard = SLAB_ALLOC.try_lock()?;
613    Some(guard.with_mut_and_token(|s, _| unsafe {
614        let mut cur = s.partial_pages[ci];
615        while !cur.is_null() {
616            if cur as u64 == page_base {
617                return true;
618            }
619            cur = (*cur).next_partial;
620        }
621        false
622    }))
623}
624
625/// Fallible heap entry point with explicit backend-aware errors.
626///
627/// Kernel code that can recover from allocation failure should prefer this API
628/// over `Box`/`Vec`/`GlobalAlloc`, which eventually route to
629/// [`alloc_error_handler`] and remain fatal by language contract.
630#[inline]
631pub unsafe fn try_alloc_kernel_heap(layout: Layout) -> Result<*mut u8, KernelHeapAllocError> {
632    // Effective size must satisfy both the size and alignment requirements.
633    let effective = layout.size().max(layout.align());
634    // `Layout` constructors guarantee a non-zero alignment; keep the power-of-two
635    // check as a defensive guard for any malformed caller input.
636    if !layout.align().is_power_of_two() {
637        return Err(record_heap_failure(
638            layout,
639            effective,
640            classify_kernel_heap_backend(layout),
641            KernelHeapAllocError::InvalidLayout,
642        ));
643    }
644    // Large heap path uses vmalloc, which only aligns to 4 KiB pages.
645    if effective > MAX_SLAB_SIZE && layout.align() > 4096 {
646        return Err(record_heap_failure(
647            layout,
648            effective,
649            KernelHeapBackend::Vmalloc,
650            KernelHeapAllocError::AlignmentExceedsKernelPage {
651                align: layout.align(),
652            },
653        ));
654    }
655    let boot_reg = crate::silo::debug_boot_reg_active();
656    if boot_reg {
657        crate::serial_println!(
658            "[trace][heap] alloc enter effective={} size={} align={}",
659            effective,
660            layout.size(),
661            layout.align()
662        );
663    }
664
665    let result = match classify_kernel_heap_backend(layout) {
666        KernelHeapBackend::Slab => {
667            // --- slab path ---
668            let ci = SlabState::class_index_for_layout(layout);
669            // Race/corruption diagnostic: log alloc when IRQs disabled (rate-limited).
670            let _cpu = crate::arch::percpu::current_cpu_index();
671            let _irq_enabled = crate::arch::interrupts_enabled();
672            #[cfg(debug_assertions)]
673            if !_irq_enabled {
674                use core::sync::atomic::{AtomicUsize, Ordering};
675                static HEAP_A_COUNT: AtomicUsize = AtomicUsize::new(0);
676                let n = HEAP_A_COUNT.fetch_add(1, Ordering::Relaxed);
677                if n % 100 == 0 {
678                    crate::e9_println!(
679                        "HEAP-A cpu={} irq=0 size={} ci={} n={}",
680                        _cpu,
681                        effective,
682                        ci,
683                        n
684                    );
685                }
686            }
687            if boot_reg {
688                crate::serial_println!(
689                    "[trace][heap] alloc slab ci={} slab_size={} lock={:#x}",
690                    ci,
691                    SLAB_SIZES[ci],
692                    &SLAB_ALLOC as *const _ as usize
693                );
694            }
695            let mut slab = SLAB_ALLOC.lock();
696            if boot_reg {
697                crate::serial_println!("[trace][heap] alloc slab lock acquired");
698            }
699            let ptr = slab.with_mut_and_token(|s, token| s.alloc_block(ci, token));
700            if ptr.is_null() {
701                return Err(record_heap_failure(
702                    layout,
703                    effective,
704                    KernelHeapBackend::Slab,
705                    KernelHeapAllocError::SlabRefillFailed {
706                        effective,
707                        class_size: SLAB_SIZES[ci],
708                    },
709                ));
710            }
711            ptr
712        }
713        KernelHeapBackend::Vmalloc => {
714            // --- vmalloc path (large allocation) ---
715            if boot_reg {
716                crate::serial_println!("[trace][heap] alloc vmalloc size={}", effective);
717            }
718
719            crate::sync::with_irqs_disabled(|token| {
720                crate::memory::allocate_kernel_virtual(effective, token).map_err(|error| {
721                    record_heap_failure(
722                        layout,
723                        effective,
724                        KernelHeapBackend::Vmalloc,
725                        KernelHeapAllocError::Vmalloc(error),
726                    )
727                })
728            })?
729        }
730    };
731
732    Ok(result)
733}
734
735#[global_allocator]
736static HEAP_ALLOCATOR: LockedHeap = LockedHeap;
737
738/// Compatibility facade over the current global kernel heap policy.
739///
740/// Callers that need an explicit heap allocation entry point, rather than
741/// relying on `Box`/`Vec`/`GlobalAlloc`, should use this helper. The selected
742/// backend remains the current heap policy:
743/// - small allocations -> slab
744/// - large allocations -> vmalloc
745#[inline]
746pub unsafe fn alloc_kernel_heap(layout: Layout) -> *mut u8 {
747    try_alloc_kernel_heap(layout).unwrap_or(ptr::null_mut())
748}
749
750/// Free memory previously returned by [`alloc_kernel_heap`].
751#[inline]
752pub unsafe fn dealloc_kernel_heap(ptr: *mut u8, layout: Layout) {
753    HEAP_ALLOCATOR.dealloc(ptr, layout);
754}
755
756fn log_common_oom_header(layout: Layout, effective: usize) {
757    let cpu = crate::arch::percpu::current_cpu_index();
758    let irq_enabled = crate::arch::interrupts_enabled();
759    let tid = crate::process::current_task_id()
760        .map(|t| t.as_u64())
761        .unwrap_or(0);
762    let task_name = crate::process::current_task_clone()
763        .map(|t| t.name)
764        .unwrap_or("<none>");
765
766    crate::serial_println!(
767        "[heap][oom] cpu={} irq={} tid={} task={} size={} align={} effective={}",
768        cpu,
769        irq_enabled,
770        tid,
771        task_name,
772        layout.size(),
773        layout.align(),
774        effective
775    );
776}
777
778fn log_buddy_snapshot() -> Option<(usize, usize, usize)> {
779    if let Some(guard) = crate::memory::buddy::get_allocator().try_lock() {
780        if let Some(alloc) = guard.as_ref() {
781            let (total_pages, allocated_pages) = alloc.page_totals();
782            let free_pages = total_pages.saturating_sub(allocated_pages);
783            let fail_counts = crate::memory::buddy::buddy_alloc_fail_counts_snapshot();
784
785            crate::serial_println!(
786                "[heap][oom] buddy: total={} alloc={} free={}",
787                total_pages,
788                allocated_pages,
789                free_pages
790            );
791
792            let mut fail_line = alloc::string::String::from("[heap][oom] buddy_fail_by_order:");
793            for (i, &count) in fail_counts.iter().enumerate() {
794                use core::fmt::Write;
795                let _ = write!(fail_line, " o{}={} ", i, count);
796            }
797            crate::serial_println!("{}", fail_line);
798            return Some((total_pages, allocated_pages, free_pages));
799        }
800        crate::serial_println!("[heap][oom] buddy: allocator uninitialized");
801        return None;
802    }
803
804    crate::serial_println!("[heap][oom] buddy: allocator locked");
805    None
806}
807
808fn log_heap_failure_policy(layout: Layout) {
809    match last_heap_failure_snapshot() {
810        Some(snapshot) => {
811            crate::serial_println!(
812                "[heap][oom] last_failure backend={:?} requested={} align={} effective={} error={:?}",
813                snapshot.backend,
814                snapshot.requested_size,
815                snapshot.align,
816                snapshot.effective_size,
817                snapshot.error
818            );
819            if snapshot.requested_size != layout.size() || snapshot.align != layout.align() {
820                crate::serial_println!(
821                    "[heap][oom] note=last_heap_failure does not exactly match current layout; using best-effort context"
822                );
823            }
824        }
825        None => crate::serial_println!("[heap][oom] last_heap_failure unavailable"),
826    }
827}
828
829/// Allocates error handler.
830#[alloc_error_handler]
831fn alloc_error_handler(layout: Layout) -> ! {
832    let effective = layout.size().max(layout.align());
833    let pages_needed = (effective.saturating_add(4095)) / 4096;
834    let order = if pages_needed == 0 {
835        0
836    } else {
837        pages_needed.next_power_of_two().trailing_zeros() as u8
838    };
839    log_common_oom_header(layout, effective);
840
841    if effective <= MAX_SLAB_SIZE {
842        crate::serial_println!(
843            "[heap][oom] backend=slab effective={} class_max={} refill_order=0",
844            effective,
845            MAX_SLAB_SIZE
846        );
847        log_heap_failure_policy(layout);
848        if let Some((total_pages, _, free_pages)) = log_buddy_snapshot() {
849            crate::serial_println!(
850                "[heap][oom] slab-refill pages={} buddy_order={}",
851                pages_needed,
852                order
853            );
854            if free_pages > (total_pages / 4) {
855                crate::serial_println!(
856                    "[heap][oom] diagnosis=slab order-0 refill failed despite remaining free pages \
857                     ({} free pages): allocator pressure, zone exhaustion, or transient allocator state",
858                    free_pages,
859                );
860            }
861        }
862    } else {
863        crate::serial_println!(
864            "[heap][oom] backend=vmalloc request_pages={} legacy_buddy_order_hint={}",
865            pages_needed,
866            order
867        );
868        log_heap_failure_policy(layout);
869        if let Some(snap) = last_heap_failure_snapshot() {
870            if let KernelHeapAllocError::AlignmentExceedsKernelPage { align } = snap.error {
871                crate::serial_println!(
872                    "[heap][oom] diagnosis=layout alignment {} B exceeds 4 KiB page alignment guaranteed by vmalloc heap path",
873                    align
874                );
875            }
876        }
877        match crate::memory::vmalloc::last_failure_snapshot() {
878            Some(snapshot) => {
879                crate::serial_println!(
880                    "[heap][oom] vmalloc_last_failure size={} pages={} error={:?}",
881                    snapshot.size,
882                    snapshot.pages,
883                    snapshot.error
884                );
885                match snapshot.error {
886                    crate::memory::vmalloc::VmallocError::SizeExceedsPolicy {
887                        requested,
888                        max_allowed,
889                    } => {
890                        crate::serial_println!(
891                            "[heap][oom] diagnosis=vmalloc policy limit exceeded requested={} max_allowed={}",
892                            requested,
893                            max_allowed
894                        );
895                    }
896                    crate::memory::vmalloc::VmallocError::VirtualRangeExhausted => {
897                        crate::serial_println!(
898                            "[heap][oom] diagnosis=kernel virtual allocation arena exhausted or fragmented"
899                        );
900                    }
901                    crate::memory::vmalloc::VmallocError::PhysicalMemoryExhausted => {
902                        crate::serial_println!(
903                            "[heap][oom] diagnosis=vmalloc could not acquire enough physical pages"
904                        );
905                    }
906                    crate::memory::vmalloc::VmallocError::MetadataAllocationFailed => {
907                        crate::serial_println!(
908                            "[heap][oom] diagnosis=vmalloc metadata allocation failed"
909                        );
910                    }
911                    crate::memory::vmalloc::VmallocError::KernelMapFailed => {
912                        crate::serial_println!(
913                            "[heap][oom] diagnosis=kernel page-table mapping failed during vmalloc"
914                        );
915                    }
916                    crate::memory::vmalloc::VmallocError::ZeroSize => {
917                        crate::serial_println!("[heap][oom] diagnosis=zero-sized vmalloc request");
918                    }
919                }
920            }
921            None => {
922                crate::serial_println!("[heap][oom] vmalloc_last_failure unavailable");
923            }
924        }
925        let _ = log_buddy_snapshot();
926    }
927    crate::serial_println!(
928        "[heap][oom] policy=fatal_global_alloc_path use try_alloc_kernel_heap()/allocate_kernel_virtual() on recoverable paths"
929    );
930    panic!("fatal kernel heap allocation failure: {:?}", layout)
931}
932
933/// Dump heap and buddy allocator diagnostics to the serial console.
934///
935/// Safe to call from the shell or debug tooling. Prints:
936/// - Total/allocated/free pages
937/// - Per-order buddy free list head counts
938/// - Buddy allocation failure counts by order (fragmentation indicator)
939/// - Slab free list head pointers
940pub fn dump_diagnostics() {
941    crate::serial_println!("[heap][diag] === Heap Diagnostics ===");
942
943    // Buddy allocator stats
944    if let Some(guard) = crate::memory::buddy::get_allocator().try_lock() {
945        if let Some(alloc) = guard.as_ref() {
946            let (total_pages, allocated_pages) = alloc.page_totals();
947            let mut zones =
948                [crate::memory::buddy::ZoneStats::empty(); crate::memory::zone::ZoneType::COUNT];
949            let zone_count = alloc.zone_snapshot(&mut zones);
950            crate::serial_println!(
951                "[heap][diag] buddy: total={} pages, allocated={} pages, free={} pages",
952                total_pages,
953                allocated_pages,
954                total_pages.saturating_sub(allocated_pages)
955            );
956
957            for info in zones.iter().take(zone_count) {
958                crate::serial_println!(
959                    "[heap][diag] zone={:?} state={:?} managed={} present={} reserved={} free={} cached={} cu/cm={}/{} avail={} segments={}/{} pageblocks=u{}/m{} u_free={} m_free={} watermarks={}/{}/{} reserve={} largest_order={:?}",
960                    info.zone_type,
961                    info.pressure(),
962                    info.managed_pages,
963                    info.present_pages,
964                    info.reserved_pages,
965                    info.free_pages,
966                    info.cached_pages,
967                    info.cached_unmovable_pages,
968                    info.cached_movable_pages,
969                    info.available_after_reserve_pages(),
970                    info.segment_count,
971                    info.segment_capacity,
972                    info.unmovable_pageblocks,
973                    info.movable_pageblocks,
974                    info.unmovable_free_pages,
975                    info.movable_free_pages,
976                    info.watermark_min,
977                    info.watermark_low,
978                    info.watermark_high,
979                    info.lowmem_reserve_pages,
980                    info.largest_free_order
981                );
982            }
983
984            // Per-zone free list heads by migratetype.
985            for zi in 0..zone_count {
986                let zone = alloc.get_zone(zi);
987                let info = zones[zi];
988                let mut line = alloc::string::String::from("[heap][diag] ");
989                use core::fmt::Write;
990                let _ = write!(line, "zone={:?} free_heads:", zone.zone_type);
991                for order in 0..=crate::memory::zone::MAX_ORDER {
992                    let unmovable = zone.free_list_count_for(
993                        order as u8,
994                        crate::memory::zone::Migratetype::Unmovable,
995                    );
996                    let movable = zone.free_list_count_for(
997                        order as u8,
998                        crate::memory::zone::Migratetype::Movable,
999                    );
1000                    if unmovable > 0 || movable > 0 {
1001                        let _ = write!(line, " o{}=u{}/m{} ", order, unmovable, movable);
1002                    }
1003                }
1004                crate::serial_println!("{}", line);
1005
1006                let mut frag = alloc::string::String::from("[heap][diag] ");
1007                let _ = write!(frag, "zone={:?} frag:", zone.zone_type);
1008                for order in 1..=crate::memory::zone::MAX_ORDER {
1009                    let score = zone.fragmentation_score(order as u8, info.cached_pages);
1010                    let _ = write!(frag, " o{}={}%", order, score);
1011                }
1012                crate::serial_println!("{}", frag);
1013            }
1014        }
1015    } else {
1016        crate::serial_println!("[heap][diag] buddy: allocator locked (retry later)");
1017    }
1018
1019    // Buddy failure counts
1020    let fail_counts = crate::memory::buddy::buddy_alloc_fail_counts_snapshot();
1021    let mut has_fails = false;
1022    for (i, &count) in fail_counts.iter().enumerate() {
1023        if count > 0 {
1024            has_fails = true;
1025        }
1026        crate::serial_println!("[heap][diag] buddy_fail[{}]: {}", i, count);
1027    }
1028    if has_fails {
1029        crate::serial_println!(
1030            "[heap][diag] => non-zero buddy_fail counts indicate fragmentation pressure"
1031        );
1032    }
1033
1034    // Slab stats
1035    {
1036        let alloc = SLAB_PAGES_ALLOCATED.load(AtomicOrdering::Relaxed);
1037        let reclaim = SLAB_PAGES_RECLAIMED.load(AtomicOrdering::Relaxed);
1038        crate::serial_println!(
1039            "[heap][diag] slab: pages_allocated={} pages_reclaimed={} pages_live={}",
1040            alloc,
1041            reclaim,
1042            alloc.saturating_sub(reclaim)
1043        );
1044    }
1045    if let Some(mut guard) = SLAB_ALLOC.try_lock() {
1046        // SAFETY: we hold the slab lock; raw pointer traversal is safe.
1047        guard.with_mut_and_token(|s, _| unsafe {
1048            for ci in 0..NUM_SLABS {
1049                let mut head = s.partial_pages[ci];
1050                if head.is_null() {
1051                    continue;
1052                }
1053                let mut page_count = 0usize;
1054                let mut free_blocks = 0u32;
1055                while !head.is_null() {
1056                    page_count += 1;
1057                    free_blocks = free_blocks.saturating_add((*head).free_count);
1058                    head = (*head).next_partial;
1059                }
1060                crate::serial_println!(
1061                    "[heap][diag] slab[{}]: partial_pages={} free_blocks={}",
1062                    SLAB_SIZES[ci],
1063                    page_count,
1064                    free_blocks
1065                );
1066            }
1067        });
1068    } else {
1069        crate::serial_println!("[heap][diag] slab: locked (retry later)");
1070    }
1071
1072    // Contiguous-physical allocation telemetry
1073    {
1074        let d = crate::memory::phys_contiguous_diag();
1075        crate::serial_println!(
1076            "[heap][diag] phys_contiguous: pages_allocated={} pages_freed={} pages_live={} alloc_failures={}",
1077            d.pages_allocated,
1078            d.pages_freed,
1079            d.pages_live,
1080            d.alloc_fail_count
1081        );
1082    }
1083
1084    if let Some(snapshot) = last_heap_failure_snapshot() {
1085        crate::serial_println!(
1086            "[heap][diag] last_heap_failure: backend={:?} requested={} align={} effective={} error={:?}",
1087            snapshot.backend,
1088            snapshot.requested_size,
1089            snapshot.align,
1090            snapshot.effective_size,
1091            snapshot.error
1092        );
1093    }
1094
1095    crate::memory::vmalloc::dump_diagnostics();
1096
1097    crate::serial_println!("[heap][diag] === End Diagnostics ===");
1098}