Skip to main content

strat9_kernel/memory/
buddy.rs

1// Buddy allocator implementation
2//
3// Refcount sentinel invariant (OSTD-style, fully enforced):
4//
5//   free-list frame => refcount == REFCOUNT_UNUSED  (u32::MAX)
6//   live frame      => refcount >= 1
7//
8// `mark_block_free()` stamps REFCOUNT_UNUSED on every free path.
9// `mark_block_allocated()` leaves refcount untouched (still REFCOUNT_UNUSED)
10// so that `FrameAllocOptions::allocate()` can perform a fail-fast
11// CAS(REFCOUNT_UNUSED => 1) that catches double-free / free-list corruption
12// immediately rather than silently aliasing memory.
13
14use crate::arch::xshim::PhysAddr;
15#[allow(unused_imports)]
16use crate::{
17    boot::entry::{MemoryKind, MemoryRegion},
18    memory::{
19        boot_alloc,
20        frame::{
21            frame_flags, get_meta, AllocError, FrameAllocator, PhysFrame, FRAME_META_LINK_NONE,
22        },
23        hhdm_offset, phys_to_virt,
24        zone::{
25            BuddyBitmap, Migratetype, Zone, ZoneSegment, ZoneType, MAX_ORDER, PAGEBLOCK_ORDER,
26            PAGEBLOCK_PAGES,
27        },
28    },
29    serial_println,
30    sync::{guardian::PreemptDisabled, IrqDisabledToken, SpinLock, SpinLockGuard},
31};
32use core::{
33    mem, ptr,
34    sync::atomic::{AtomicUsize, Ordering as AtomicOrdering},
35};
36
37const PAGE_SIZE: u64 = 4096;
38const DMA_MAX: u64 = 16 * 1024 * 1024;
39const NORMAL_MAX: u64 = 896 * 1024 * 1024;
40
41const LOCAL_CACHE_CAPACITY: usize = 256;
42const LOCAL_CACHE_REFILL_ORDER: u8 = 4;
43const LOCAL_CACHE_REFILL_FRAMES: usize = 1 << (LOCAL_CACHE_REFILL_ORDER as usize);
44const LOCAL_CACHE_FLUSH_BATCH: usize = 64;
45const LOCAL_CACHE_SLOTS: usize = Migratetype::COUNT * crate::arch::percpu::MAX_CPUS;
46const LOCAL_CACHED_ZONE_MIGRATETYPE_SLOTS: usize = Migratetype::COUNT * ZoneType::COUNT;
47/// Fragmentation score threshold for triggering compaction-assist.
48///
49/// When a higher-order allocation fails and the zone's fragmentation score
50/// exceeds this threshold, the allocator drains per-CPU caches before retrying.
51/// The value is expressed as a percentage (0-100). Lower values make compaction
52/// more aggressive; higher values make it more conservative.
53///
54/// Default: 35 (35% of free pages trapped below the requested order).
55static COMPACTION_FRAGMENTATION_THRESHOLD: AtomicUsize = AtomicUsize::new(35);
56const COMPACTION_SNAPSHOT_NONE: usize = usize::MAX;
57const UNMOVABLE_ZONE_ORDER: [usize; ZoneType::COUNT] = [
58    ZoneType::Normal as usize,
59    ZoneType::HighMem as usize,
60    ZoneType::DMA as usize,
61];
62const MOVABLE_ZONE_ORDER: [usize; ZoneType::COUNT] = [
63    ZoneType::HighMem as usize,
64    ZoneType::Normal as usize,
65    ZoneType::DMA as usize,
66];
67
68#[cfg(feature = "selftest")]
69macro_rules! buddy_dbg {
70    ($($arg:tt)*) => {
71        serial_println!($($arg)*);
72    };
73}
74
75#[cfg(not(feature = "selftest"))]
76macro_rules! buddy_dbg {
77    ($($arg:tt)*) => {};
78}
79
80pub struct BuddyAllocator {
81    zones: [Zone; ZoneType::COUNT],
82    /// Per-zone bitmap pool reserved from free memory: [start, end).
83    bitmap_pool: [(u64, u64); ZoneType::COUNT],
84}
85
86#[derive(Clone, Copy, Debug)]
87struct CompactionCandidate {
88    zone_idx: usize,
89    zone_type: ZoneType,
90    order: u8,
91    migratetype: Migratetype,
92    pressure: ZonePressure,
93    fragmentation_score: usize,
94    requested_pages: usize,
95    available_pages: usize,
96    usable_pages: usize,
97    cached_pages: usize,
98    pageblock_count: usize,
99    matching_pageblocks: usize,
100}
101
102impl BuddyAllocator {
103    /// Creates a new instance.
104    pub const fn new() -> Self {
105        BuddyAllocator {
106            zones: [
107                Zone::new(ZoneType::DMA),
108                Zone::new(ZoneType::Normal),
109                Zone::new(ZoneType::HighMem),
110            ],
111            bitmap_pool: [(0, 0); ZoneType::COUNT],
112        }
113    }
114
115    /// Performs the init operation.
116    pub fn init(&mut self, memory_regions: &[MemoryRegion]) {
117        crate::e9_mark!(b'b');
118        #[cfg(debug_assertions)]
119        debug_assert!(
120            hhdm_offset() != u64::MAX,
121            "HHDM offset sanity check failed unexpectedly"
122        );
123        crate::e9_mark!(b'1');
124
125        // Skip serial output during buddy init to avoid format_args function pointer issues.
126        crate::e9_mark!(b'2');
127
128        // Pass 1: compute per-zone address span (base + span_pages)
129        crate::e9_mark!(b'3');
130        self.pass_count(memory_regions);
131        crate::e9_mark!(b'4');
132
133        // Pass 2: reserve per-zone bitmap pools using an upper bound derived
134        // from the boot allocator's current free extents.
135        let mut candidates = [MemoryRegion {
136            base: 0,
137            size: 0,
138            kind: MemoryKind::Reserved,
139        }; boot_alloc::MAX_BOOT_ALLOC_REGIONS];
140        let candidate_len = boot_alloc::snapshot_free_regions(&mut candidates);
141        crate::e9_mark!(b'5');
142        self.pass_reserve_bitmap_pools(&candidates[..candidate_len]);
143        crate::e9_mark!(b'6');
144
145        ///////////
146        // Pass 3: reserve exact segment storage from the remaining accessible
147        // boot memory and then build the final segmented buddy layout from the boot
148        // allocator's remaining free ranges after bitmap and segment-storage
149        // reservations.
150        ///////////
151        // CRITICAL HERE : re-snapshot after pass_reserve_segment_storage
152        // because it consumes pages from the boot allocator.
153        // Using the stale snapshot would cause the buddy to build segments spanning pages that
154        // actually hold the ZoneSegment metadata, leading to silent corruption
155        // when those pages are later allocated and written by a live frame owner.
156        ///////////
157        let mut remaining = [MemoryRegion {
158            base: 0,
159            size: 0,
160            kind: MemoryKind::Reserved,
161        }; boot_alloc::MAX_BOOT_ALLOC_REGIONS];
162
163        let remaining_len = boot_alloc::snapshot_free_regions(&mut remaining);
164        crate::e9_mark!(b'7');
165        self.pass_reserve_segment_storage(&remaining[..remaining_len]);
166        crate::e9_mark!(b'8');
167
168        // Re-snapshot: segment-storage pages are now consumed from the boot
169        // allocator and must not appear in any buddy segment.
170        let remaining_len = boot_alloc::snapshot_free_regions(&mut remaining);
171
172        crate::e9_mark!(b'9');
173        self.pass_build_segments(&remaining[..remaining_len]);
174        crate::e9_mark!(b'a');
175        self.pass_finalize_zone_accounting();
176        crate::e9_mark!(b'b');
177        self.pass_setup_segment_bitmaps();
178        crate::e9_mark!(b'c');
179        self.pass_populate();
180        crate::e9_mark!(b'd');
181
182        // Seal the boot allocator: all its remaining free regions are now managed
183        // by buddy.  Any later boot_alloc::alloc_stack() call would otherwise
184        // double-allocate pages that buddy already tracks in its free lists.
185        boot_alloc::seal();
186        crate::e9_mark!(b'e');
187    }
188
189    /// Performs the pass count operation.
190    fn pass_count(&mut self, memory_regions: &[MemoryRegion]) {
191        let mut min_base = [u64::MAX; ZoneType::COUNT];
192        let mut max_end = [0u64; ZoneType::COUNT];
193        let mut present_pages = [0usize; ZoneType::COUNT];
194
195        for region in memory_regions {
196            for zi in 0..ZoneType::COUNT {
197                if let Some((start, end)) = Self::zone_intersection_aligned(region, zi) {
198                    present_pages[zi] =
199                        present_pages[zi].saturating_add(((end - start) / PAGE_SIZE) as usize);
200                    if start < min_base[zi] {
201                        min_base[zi] = start;
202                    }
203                    if end > max_end[zi] {
204                        max_end[zi] = end;
205                    }
206                }
207            }
208        }
209
210        for zi in 0..ZoneType::COUNT {
211            let zone = &mut self.zones[zi];
212            zone.base = PhysAddr::new(0);
213            zone.page_count = 0;
214            zone.present_pages = present_pages[zi];
215            zone.span_pages = 0;
216            zone.allocated = 0;
217            zone.reserved_pages = 0;
218            zone.lowmem_reserve_pages = 0;
219            zone.watermark_min = 0;
220            zone.watermark_low = 0;
221            zone.watermark_high = 0;
222            zone.clear_segments();
223
224            if min_base[zi] == u64::MAX || max_end[zi] <= min_base[zi] {
225                continue;
226            }
227
228            zone.base = PhysAddr::new(min_base[zi]);
229            zone.span_pages = ((max_end[zi] - min_base[zi]) / PAGE_SIZE) as usize;
230        }
231    }
232
233    /// Reserve per-zone segment tables sized to the actual fragmented layout.
234    fn pass_reserve_segment_storage(&mut self, memory_regions: &[MemoryRegion]) {
235        let mut segment_counts = [0usize; ZoneType::COUNT];
236
237        for region in memory_regions {
238            for (zi, count) in segment_counts.iter_mut().enumerate() {
239                if Self::zone_intersection_aligned(region, zi).is_some() {
240                    *count = count.saturating_add(1);
241                }
242            }
243        }
244
245        for (zi, &segment_count) in segment_counts.iter().enumerate() {
246            let zone = &mut self.zones[zi];
247            zone.clear_segments();
248
249            if segment_count == 0 {
250                continue;
251            }
252
253            let bytes = segment_count.saturating_mul(mem::size_of::<ZoneSegment>());
254            let storage_phys =
255                boot_alloc::alloc_bytes_accessible(bytes, mem::align_of::<ZoneSegment>())
256                    .unwrap_or_else(|| {
257                        panic!(
258                            "Buddy allocator: unable to reserve {} bytes for {:?} segment table",
259                            bytes, zone.zone_type
260                        )
261                    })
262                    .as_u64();
263            unsafe {
264                ptr::write_bytes(phys_to_virt(storage_phys) as *mut u8, 0, bytes);
265            }
266
267            zone.segment_capacity = segment_count;
268            zone.segments = phys_to_virt(storage_phys) as *mut ZoneSegment;
269        }
270    }
271
272    /// Reserve per-zone bitmap pools using a segmentation-safe upper bound.
273    fn pass_reserve_bitmap_pools(&mut self, memory_regions: &[MemoryRegion]) {
274        for zi in 0..ZoneType::COUNT {
275            crate::e9_mark!(b'P');
276            let managed_pages = memory_regions
277                .iter()
278                .filter_map(|region| Self::zone_intersection_aligned(region, zi))
279                .map(|(start, end)| ((end - start) / PAGE_SIZE) as usize)
280                .sum::<usize>();
281            let needed_bytes = Self::bitmap_bytes_upper_bound_for_pages(managed_pages);
282            let reserved_bytes = Self::align_up(needed_bytes as u64, PAGE_SIZE);
283
284            if reserved_bytes == 0 {
285                self.bitmap_pool[zi] = (0, 0);
286                crate::e9_mark!(b'Z');
287                continue;
288            }
289            crate::e9_mark!(b'A');
290
291            let pool_start = boot_alloc::alloc_bytes_accessible(needed_bytes, PAGE_SIZE as usize)
292                .unwrap_or_else(|| {
293                    panic!(
294                        "Buddy allocator: unable to reserve {} bytes for zone {:?} bitmaps",
295                        needed_bytes, self.zones[zi].zone_type
296                    )
297                })
298                .as_u64();
299            crate::e9_mark!(b'B');
300            let pool_end = pool_start.saturating_add(reserved_bytes);
301            self.bitmap_pool[zi] = (pool_start, pool_end);
302
303            // Zero stolen pages to initialize all bitmaps to 0.
304            // Use a byte loop to avoid memset function pointer issues.
305            let dst = phys_to_virt(pool_start) as *mut u8;
306            let count = (pool_end - pool_start) as usize;
307            crate::e9_mark!(b'C');
308            for i in 0..count {
309                unsafe {
310                    core::ptr::write_volatile(dst.add(i), 0);
311                }
312            }
313            crate::e9_mark!(b'D');
314        }
315    }
316
317    /// Finalise zone accounting once the managed segment set is known.
318    fn pass_finalize_zone_accounting(&mut self) {
319        for zone in &mut self.zones {
320            zone.reserved_pages = zone.present_pages.saturating_sub(zone.page_count);
321            zone.lowmem_reserve_pages =
322                Self::lowmem_reserve_target_pages(zone.zone_type, zone.page_count);
323            // Watermark parameters: (divisor, floor_pages, cap_pages)
324            // watermark_min: at least 16 pages, up to 2048, ~0.4% of zone
325            zone.watermark_min = Self::watermark_target_pages(zone.page_count, 256, 16, 2048);
326
327            // watermark_low/high: increment of at least 16 pages, up to 2048, ~0.2% of zone
328            let delta = Self::watermark_target_pages(zone.page_count, 512, 16, 2048);
329            zone.watermark_low = zone
330                .watermark_min
331                .saturating_add(delta)
332                .min(zone.page_count);
333            zone.watermark_high = zone
334                .watermark_low
335                .saturating_add(delta)
336                .min(zone.page_count);
337        }
338    }
339
340    /// Compute a bounded watermark target for a zone.
341    fn watermark_target_pages(
342        managed_pages: usize,
343        divisor: usize,
344        floor: usize,
345        cap: usize,
346    ) -> usize {
347        Self::bounded_zone_target(managed_pages, divisor, floor, cap, 8)
348    }
349
350    /// Compute a bounded low-memory reserve target.
351    ///
352    /// Parameters: (divisor, floor_pages, cap_pages, max_fraction_divisor)
353    /// - DMA: 12.5% of zone, min 16 pages, max 512 pages, at most 25% of zone
354    /// - Normal: ~1.6% of zone, min 64 pages, max 2048 pages, at most 12.5% of zone
355    /// - HighMem: no reserve (movable allocations prefer HighMem)
356    fn lowmem_reserve_target_pages(zone_type: ZoneType, managed_pages: usize) -> usize {
357        match zone_type {
358            ZoneType::DMA => Self::bounded_zone_target(managed_pages, 8, 16, 512, 4),
359            ZoneType::Normal => Self::bounded_zone_target(managed_pages, 64, 64, 2048, 8),
360            ZoneType::HighMem => 0,
361        }
362    }
363
364    /// Bound a policy target to something meaningful for the current zone size.
365    fn bounded_zone_target(
366        managed_pages: usize,
367        divisor: usize,
368        floor: usize,
369        cap: usize,
370        max_fraction_divisor: usize,
371    ) -> usize {
372        if managed_pages == 0 {
373            return 0;
374        }
375
376        let scaled = core::cmp::max(managed_pages / divisor, floor);
377        let capped = core::cmp::min(scaled, cap);
378        let max_for_zone = core::cmp::max(1, managed_pages / max_fraction_divisor);
379        core::cmp::min(capped, max_for_zone)
380    }
381
382    /// Build the final segmented physical layout from remaining boot allocator ranges.
383    fn pass_build_segments(&mut self, memory_regions: &[MemoryRegion]) {
384        for region in memory_regions {
385            for zi in 0..ZoneType::COUNT {
386                let Some((start, end)) = Self::zone_intersection_aligned(region, zi) else {
387                    continue;
388                };
389                let zone = &mut self.zones[zi];
390                if zone.segment_count >= zone.segment_capacity {
391                    panic!(
392                        "Buddy allocator: zone {:?} exceeded reserved segment capacity={} while processing phys=0x{:x}..0x{:x}",
393                        zone.zone_type,
394                        zone.segment_capacity,
395                        start,
396                        end,
397                    );
398                }
399
400                let slot = zone.segment_count;
401                zone.segments_mut()[slot] = ZoneSegment {
402                    base: PhysAddr::new(start),
403                    page_count: ((end - start) / PAGE_SIZE) as usize,
404                    free_lists: [[0; MAX_ORDER + 1]; Migratetype::COUNT],
405                    buddy_bitmaps: [BuddyBitmap::empty(); MAX_ORDER + 1],
406                    pageblock_tags: ptr::null_mut(),
407                    pageblock_count: 0,
408                    #[cfg(debug_assertions)]
409                    alloc_bitmap: BuddyBitmap::empty(),
410                };
411                zone.segment_count = slot + 1;
412                zone.page_count = zone
413                    .page_count
414                    .saturating_add(((end - start) / PAGE_SIZE) as usize);
415
416                buddy_dbg!(
417                    "  Zone {:?}: segment phys=0x{:x}..0x{:x} pages={}",
418                    zone.zone_type,
419                    start,
420                    end,
421                    ((end - start) / PAGE_SIZE) as usize,
422                );
423            }
424        }
425    }
426
427    /// Assign bitmap slices to each populated segment.
428    fn pass_setup_segment_bitmaps(&mut self) {
429        for zi in 0..ZoneType::COUNT {
430            let (pool_start, pool_end) = self.bitmap_pool[zi];
431            if pool_start == 0 || pool_end <= pool_start {
432                continue;
433            }
434
435            let zone = &mut self.zones[zi];
436            let default_pageblock_migratetype = Self::default_pageblock_migratetype(zone.zone_type);
437            let mut cursor = pool_start;
438            let segment_count = zone.segment_count;
439            for segment in zone.segments_mut().iter_mut().take(segment_count) {
440                let _exact_bitmap_bytes = Self::bitmap_bytes_for_span(segment.page_count);
441                for order in 0..=MAX_ORDER {
442                    let num_bits = Self::pairs_for_order(segment.page_count, order as u8);
443                    let num_bytes = Self::bits_to_bytes(num_bits) as u64;
444                    if num_bits == 0 {
445                        segment.buddy_bitmaps[order] = BuddyBitmap::empty();
446                        continue;
447                    }
448
449                    assert!(
450                        cursor + num_bytes <= pool_end,
451                        "buddy bitmap pool overflow: zone {:?} order {} needs {} bytes but only {} remaining",
452                        zone.zone_type, order, num_bytes, pool_end.saturating_sub(cursor),
453                    );
454                    segment.buddy_bitmaps[order] = BuddyBitmap {
455                        data: phys_to_virt(cursor) as *mut u8,
456                        num_bits,
457                    };
458                    cursor += num_bytes;
459                }
460
461                #[cfg(debug_assertions)]
462                {
463                    let num_bits = segment.page_count;
464                    let num_bytes = Self::bits_to_bytes(num_bits) as u64;
465                    if num_bits == 0 {
466                        segment.alloc_bitmap = BuddyBitmap::empty();
467                    } else {
468                        assert!(
469                            cursor + num_bytes <= pool_end,
470                            "buddy alloc bitmap pool overflow: zone {:?} needs {} bytes but only {} remaining",
471                            zone.zone_type, num_bytes, pool_end.saturating_sub(cursor),
472                        );
473                        segment.alloc_bitmap = BuddyBitmap {
474                            data: phys_to_virt(cursor) as *mut u8,
475                            num_bits,
476                        };
477                        cursor += num_bytes;
478                    }
479                }
480
481                let pageblock_count = segment.page_count.div_ceil(PAGEBLOCK_PAGES);
482                segment.pageblock_count = pageblock_count;
483                if pageblock_count == 0 {
484                    segment.pageblock_tags = ptr::null_mut();
485                } else {
486                    let num_bytes = pageblock_count as u64;
487                    assert!(
488                        cursor + num_bytes <= pool_end,
489                        "buddy pageblock tags pool overflow: zone {:?} needs {} bytes but only {} remaining",
490                        zone.zone_type, num_bytes, pool_end.saturating_sub(cursor),
491                    );
492                    segment.pageblock_tags = phys_to_virt(cursor) as *mut u8;
493                    unsafe {
494                        ptr::write_bytes(
495                            segment.pageblock_tags,
496                            default_pageblock_migratetype as u8,
497                            pageblock_count,
498                        );
499                    }
500                    cursor += num_bytes;
501                }
502            }
503
504            assert!(
505                cursor <= pool_end,
506                "buddy bitmap pool final check failed: zone {:?} cursor 0x{:x} exceeds pool_end 0x{:x}",
507                zone.zone_type, cursor, pool_end,
508            );
509        }
510    }
511
512    /// Seed each contiguous segment with greedy block insertion.
513    fn pass_populate(&mut self) {
514        for zi in 0..ZoneType::COUNT {
515            let zone_type = self.zones[zi].zone_type;
516            let segment_count = self.zones[zi].segment_count;
517            for si in 0..segment_count {
518                let (start, end) = {
519                    let segments = self.zones[zi].segments();
520                    let segment = &segments[si];
521                    (segment.base.as_u64(), segment.end_address())
522                };
523                let segment = &mut self.zones[zi].segments_mut()[si];
524                Self::seed_range_as_free(zone_type, segment, start, end);
525            }
526        }
527    }
528
529    /// Seeds a contiguous physical range `[start, end)` as free using greedy block insertion.
530    ///
531    /// Unlike the previous min/max span design, `segment` is guaranteed to be a
532    /// genuinely contiguous free extent. Greedy seeding therefore improves boot
533    /// time without ever making holes visible to the buddy topology.
534    fn seed_range_as_free(zone_type: ZoneType, segment: &mut ZoneSegment, start: u64, end: u64) {
535        if start >= end {
536            return;
537        }
538        let mut addr = start;
539
540        'seed: while addr < end {
541            if !segment.contains_address(PhysAddr::new(addr)) {
542                break;
543            }
544
545            if let Some(protected_end) = Self::protected_overlap_end(addr, addr + PAGE_SIZE) {
546                buddy_dbg!(
547                    "  Zone {:?}: skip protected range 0x{:x}..0x{:x}",
548                    zone_type,
549                    addr,
550                    protected_end
551                );
552                addr = core::cmp::min(protected_end, end);
553                continue;
554            }
555
556            let remaining_pages = ((end - addr) / PAGE_SIZE) as usize;
557            debug_assert!(remaining_pages != 0);
558            let mut order = ((remaining_pages.ilog2()) as u8).min(MAX_ORDER as u8);
559
560            while order > 0 {
561                let block_size = PAGE_SIZE << order;
562                if addr & (block_size - 1) == 0 {
563                    break;
564                }
565                order -= 1;
566            }
567
568            loop {
569                let block_size = PAGE_SIZE << order;
570                let block_end = addr.saturating_add(block_size);
571                if block_end > end {
572                    debug_assert!(order != 0);
573                    order -= 1;
574                    continue;
575                }
576
577                if Self::protected_overlap_end(addr, block_end).is_some() {
578                    if order == 0 {
579                        if let Some(skip_to) = Self::protected_overlap_end(addr, block_end) {
580                            buddy_dbg!("  Zone {:?}: skip protected page 0x{:x}", zone_type, addr);
581                            addr = core::cmp::min(skip_to, end);
582                            continue 'seed;
583                        }
584                    }
585                    order -= 1;
586                    continue;
587                }
588
589                let migratetype = Self::pageblock_migratetype(
590                    segment,
591                    addr,
592                    Self::default_pageblock_migratetype(zone_type),
593                );
594                Self::insert_free_block(segment, addr, order, migratetype);
595                addr = block_end;
596                continue 'seed;
597            }
598        }
599    }
600
601    /// Allocates from zone.
602    fn alloc_from_zone(
603        zone: &mut Zone,
604        zone_idx: usize,
605        order: u8,
606        migratetype: Migratetype,
607        honor_watermarks: bool,
608        token: &IrqDisabledToken,
609    ) -> Option<PhysFrame> {
610        if !Self::zone_allows_allocation(zone, zone_idx, order, honor_watermarks) {
611            return None;
612        }
613
614        for si in 0..zone.segment_count {
615            let frame_phys = {
616                let segment = &mut zone.segments_mut()[si];
617                Self::alloc_from_segment(segment, order, migratetype, token)
618            };
619            if let Some(frame_phys) = frame_phys {
620                zone.allocated += 1usize << order;
621                return PhysFrame::from_start_address(PhysAddr::new(frame_phys)).ok();
622            }
623        }
624        None
625    }
626
627    /// Allocate from one contiguous segment.
628    fn alloc_from_segment(
629        segment: &mut ZoneSegment,
630        order: u8,
631        requested_migratetype: Migratetype,
632        _token: &IrqDisabledToken,
633    ) -> Option<u64> {
634        for cur_order in order..=MAX_ORDER as u8 {
635            for donor_migratetype in requested_migratetype.fallback_order() {
636                let Some(frame_phys) = Self::free_list_pop(segment, cur_order, donor_migratetype)
637                else {
638                    continue;
639                };
640                debug_assert!(
641                    !crate::memory::frame::block_phys_has_poison_guard(frame_phys, cur_order),
642                    "buddy: poisoned block on free list (order {})",
643                    cur_order
644                );
645                let block_size = PAGE_SIZE << cur_order;
646                let block_end = frame_phys.saturating_add(block_size);
647                if Self::protected_overlap_end(frame_phys, block_end).is_some() {
648                    panic!(
649                        "Buddy allocator inconsistency: free block 0x{:x} order {} overlaps protected memory",
650                        frame_phys, cur_order
651                    );
652                }
653
654                let _ = Self::toggle_pair(segment, frame_phys, cur_order);
655
656                let mut split_order = cur_order;
657                while split_order > order {
658                    split_order -= 1;
659                    Self::retag_pageblock_range(
660                        segment,
661                        frame_phys,
662                        split_order,
663                        requested_migratetype,
664                    );
665                    let buddy_phys = frame_phys + ((1u64 << split_order) * PAGE_SIZE);
666                    let buddy_migratetype =
667                        Self::pageblock_migratetype(segment, buddy_phys, donor_migratetype);
668                    Self::mark_block_free(buddy_phys, split_order, buddy_migratetype);
669                    Self::free_list_push(segment, buddy_phys, split_order, buddy_migratetype);
670                    let _ = Self::toggle_pair(segment, frame_phys, split_order);
671                }
672                Self::retag_pageblock_range(segment, frame_phys, order, requested_migratetype);
673                Self::mark_block_allocated(frame_phys, order, requested_migratetype);
674
675                #[cfg(debug_assertions)]
676                Self::mark_allocated(segment, frame_phys, order, true);
677
678                return Some(frame_phys);
679            }
680        }
681        None
682    }
683
684    /// Find the segment containing a block at the given physical address.
685    ///
686    /// Uses binary search since segments are sorted by base address.
687    /// This is O(log n) instead of O(n) for the linear search.
688    #[inline]
689    fn find_segment_index(zone: &Zone, phys: u64, order: u8) -> Option<usize> {
690        let segments = zone.segments();
691        let count = zone.segment_count;
692        if count == 0 {
693            return None;
694        }
695
696        // Binary search: find the last segment whose base <= phys
697        let mut lo = 0usize;
698        let mut hi = count;
699        while lo < hi {
700            let mid = lo + (hi - lo) / 2;
701            if segments[mid].base.as_u64() <= phys {
702                lo = mid + 1;
703            } else {
704                hi = mid;
705            }
706        }
707
708        // lo is now the first segment with base > phys, so check lo - 1
709        if lo == 0 {
710            return None;
711        }
712        let idx = lo - 1;
713        if Self::segment_contains_block(&segments[idx], phys, order) {
714            Some(idx)
715        } else {
716            None
717        }
718    }
719
720    #[inline]
721    fn segment_contains_block(segment: &ZoneSegment, phys: u64, order: u8) -> bool {
722        if !segment.contains_address(PhysAddr::new(phys)) {
723            return false;
724        }
725        let block_end = phys.saturating_add(PAGE_SIZE << order);
726        block_end <= segment.end_address()
727    }
728
729    /// Releases to zone.
730    fn free_to_zone(zone: &mut Zone, frame: PhysFrame, order: u8, _token: &IrqDisabledToken) {
731        let frame_phys = frame.start_address.as_u64();
732        let block_size = PAGE_SIZE << order;
733        let block_end = frame_phys.saturating_add(block_size);
734        let migratetype = Self::block_migratetype(frame_phys);
735        let Some(segment_idx) = Self::find_segment_index(zone, frame_phys, order) else {
736            #[cfg(feature = "selftest")]
737            {
738                serial_println!(
739                    "[buddy] CRITICAL: frame 0x{:x} order {} not found in zone {:?} segments.",
740                    frame_phys,
741                    order,
742                    zone.zone_type,
743                );
744                serial_println!(
745                    "  segments={}/{} span_pages={} page_count={} allocated={}",
746                    zone.segment_count,
747                    zone.segment_capacity,
748                    zone.span_pages,
749                    zone.page_count,
750                    zone.allocated,
751                );
752                for si in 0..zone.segment_count {
753                    let seg = &zone.segments()[si];
754                    serial_println!(
755                        "    segment[{}]: base=0x{:x} pages={} end=0x{:x}",
756                        si,
757                        seg.base.as_u64(),
758                        seg.page_count,
759                        seg.end_address(),
760                    );
761                }
762            }
763
764            panic!(
765                "buddy free: frame 0x{:x} order {} does not belong to any segment in zone {:?}",
766                frame_phys, order, zone.zone_type,
767            );
768        };
769
770        debug_assert!(order <= MAX_ORDER as u8);
771        debug_assert!(frame.start_address.is_aligned(PAGE_SIZE << order));
772        debug_assert!(zone.contains_address(frame.start_address));
773
774        if Self::protected_overlap_end(frame_phys, block_end).is_some() {
775            serial_println!(
776                "[buddy] WARNING: free_to_zone: frame 0x{:x} order {} in zone {:?} overlaps protected memory 0x{:x}..0x{:x}",
777                frame_phys,
778                order,
779                zone.zone_type,
780                frame_phys,
781                block_end,
782            );
783            #[cfg(not(feature = "selftest"))]
784            panic!(
785                "buddy free: frame 0x{:x} order {} overlaps protected memory in zone {:?}",
786                frame_phys, order, zone.zone_type,
787            );
788        }
789
790        #[cfg(debug_assertions)]
791        {
792            let segment = &mut zone.segments_mut()[segment_idx];
793            Self::mark_allocated(segment, frame_phys, order, false);
794        }
795
796        {
797            let segment = &mut zone.segments_mut()[segment_idx];
798            if order as usize >= PAGEBLOCK_ORDER {
799                Self::retag_pageblock_range(segment, frame_phys, order, migratetype);
800            }
801            let free_migratetype = Self::pageblock_migratetype(segment, frame_phys, migratetype);
802            Self::mark_block_free(frame_phys, order, free_migratetype);
803            Self::insert_free_block(segment, frame_phys, order, free_migratetype);
804        }
805        zone.allocated = zone.allocated.saturating_sub(1usize << order);
806    }
807
808    /// Drops allocator accounting for a poisoned block without returning it to the free list.
809    ///
810    /// The block is **not** placed on any free list and its debug-bitmap entries
811    /// remain marked as "allocated" : because they genuinely are: the pages are
812    /// quarantined and inaccessible.  Clearing them would defeat the double-free
813    /// detector for any later attempt to free the same block.
814    fn quarantine_poisoned_block_in_zone(
815        zone: &mut Zone,
816        frame: PhysFrame,
817        order: u8,
818        _token: &IrqDisabledToken,
819    ) {
820        let frame_phys = frame.start_address.as_u64();
821        let block_size = PAGE_SIZE << order;
822        let block_end = frame_phys.saturating_add(block_size);
823        let Some(segment_idx) = Self::find_segment_index(zone, frame_phys, order) else {
824            panic!(
825                "buddy quarantine: frame 0x{:x} order {} does not belong to any segment in zone {:?}",
826                frame_phys,
827                order,
828                zone.zone_type,
829            );
830        };
831
832        debug_assert!(order <= MAX_ORDER as u8);
833        debug_assert!(frame.start_address.is_aligned(PAGE_SIZE << order));
834        debug_assert!(zone.contains_address(frame.start_address));
835        debug_assert!(Self::segment_contains_block(
836            &zone.segments()[segment_idx],
837            frame_phys,
838            order
839        ));
840
841        if Self::protected_overlap_end(frame_phys, block_end).is_some() {
842            return;
843        }
844
845        // Intentionally NO mark_allocated(false) here : pages stay "allocated"
846        // in the debug bitmap because they are quarantined, not freed.
847
848        zone.allocated = zone.allocated.saturating_sub(1usize << order);
849        POISON_QUARANTINE_PAGES.fetch_add(1usize << order, AtomicOrdering::Relaxed);
850    }
851
852    /// Linux-style parity-map coalescing insertion.
853    /// Returns after inserting the (potentially coalesced) block into the appropriate free list, without recursing further.
854    /// If the buddy bit is already set or we reach MAX_ORDER, the block is inserted as-is.
855    /// Otherwise, the buddy block is removed from its free list and coalesced with the current block, and the process repeats at the next order.
856    fn insert_free_block(
857        segment: &mut ZoneSegment,
858        frame_phys: u64,
859        initial_order: u8,
860        migratetype: Migratetype,
861    ) {
862        let mut current = frame_phys;
863        let mut order = initial_order;
864
865        loop {
866            let bit_is_set = Self::toggle_pair(segment, current, order);
867            if bit_is_set || order == MAX_ORDER as u8 {
868                Self::mark_block_free(current, order, migratetype);
869                Self::free_list_push(segment, current, order, migratetype);
870                break;
871            }
872
873            let Some(buddy) = Self::buddy_phys(segment, current, order) else {
874                Self::mark_block_free(current, order, migratetype);
875                Self::free_list_push(segment, current, order, migratetype);
876                break;
877            };
878
879            if !Self::can_merge_with_buddy(buddy, order, migratetype) {
880                Self::mark_block_free(current, order, migratetype);
881                Self::free_list_push(segment, current, order, migratetype);
882                break;
883            }
884
885            let removed = Self::free_list_remove(segment, buddy, order, migratetype);
886            if !removed {
887                panic!(
888                    "buddy inconsistency: free_list_remove failed for buddy 0x{:x} order {} migratetype {:?} during coalesce",
889                    buddy, order, migratetype,
890                );
891            }
892
893            current = core::cmp::min(current, buddy);
894            order += 1;
895        }
896    }
897
898    /// Performs the page index operation.
899    #[inline]
900    fn page_index(segment: &ZoneSegment, phys: u64) -> usize {
901        debug_assert!(segment.page_count > 0);
902        let base = segment.base.as_u64();
903        debug_assert!(phys >= base);
904        debug_assert!((phys - base).is_multiple_of(PAGE_SIZE));
905        ((phys - base) / PAGE_SIZE) as usize
906    }
907
908    /// Performs the pair index operation.
909    #[inline]
910    fn pair_index(segment: &ZoneSegment, phys: u64, order: u8) -> usize {
911        Self::page_index(segment, phys) >> (order as usize + 1)
912    }
913
914    /// Performs the toggle pair operation.
915    #[inline]
916    fn toggle_pair(segment: &mut ZoneSegment, phys: u64, order: u8) -> bool {
917        let bitmap = segment.buddy_bitmaps[order as usize];
918        if bitmap.is_empty() {
919            return true;
920        }
921        let idx = Self::pair_index(segment, phys, order);
922        debug_assert!(idx < bitmap.num_bits);
923        bitmap.toggle(idx)
924    }
925
926    /// Performs the buddy phys operation.
927    #[inline]
928    fn buddy_phys(segment: &ZoneSegment, phys: u64, order: u8) -> Option<u64> {
929        let base = segment.base.as_u64();
930        if phys < base {
931            return None;
932        }
933        let offset = phys - base;
934        let block_size = PAGE_SIZE << order;
935        let buddy_offset = offset ^ block_size;
936        let buddy_page = (buddy_offset / PAGE_SIZE) as usize;
937        if buddy_page >= segment.page_count {
938            return None;
939        }
940        Some(base + buddy_offset)
941    }
942
943    /// Performs the mark allocated operation.
944    #[cfg(debug_assertions)]
945    fn mark_allocated(segment: &mut ZoneSegment, frame_phys: u64, order: u8, allocated: bool) {
946        if segment.alloc_bitmap.is_empty() {
947            return;
948        }
949        let start = Self::page_index(segment, frame_phys);
950        let count = 1usize << order;
951        for i in 0..count {
952            let bit = start + i;
953            debug_assert!(bit < segment.alloc_bitmap.num_bits);
954            if allocated {
955                debug_assert!(
956                    !segment.alloc_bitmap.test(bit),
957                    "double allocation detected"
958                );
959                segment.alloc_bitmap.set(bit);
960            } else {
961                debug_assert!(segment.alloc_bitmap.test(bit), "double free detected");
962                segment.alloc_bitmap.clear(bit);
963            }
964        }
965    }
966
967    /// Insert a block at the head of the free list for the given order and migratetype.
968    ///
969    /// # Memory ordering
970    ///
971    /// All free-list link reads/writes use Acquire/Release ordering on the
972    /// MetaSlot atomic fields. This is correct under the current single-global-lock
973    /// design where all mutations are serialized. If the design ever moves to
974    /// per-zone locking, the ordering requirements must be re-evaluated:
975    /// concurrent push/pop on the same list would need CAS or a different
976    /// synchronization strategy.
977    fn free_list_push(segment: &mut ZoneSegment, phys: u64, order: u8, migratetype: Migratetype) {
978        debug_assert!(
979            !crate::memory::frame::block_phys_has_poison_guard(phys, order),
980            "buddy: refusing to push poisoned block to free list"
981        );
982        let head = segment.free_lists[migratetype.index()][order as usize];
983        Self::write_free_prev(phys, 0);
984        Self::write_free_next(phys, head);
985        if head != 0 {
986            Self::write_free_prev(head, phys);
987        }
988        segment.free_lists[migratetype.index()][order as usize] = phys;
989    }
990
991    /// Remove and return the head block from the free list.
992    ///
993    /// # Memory ordering
994    ///
995    /// See [`Self::free_list_push`] for ordering rationale. The Acquire/Release
996    /// pairs on MetaSlot links are sufficient under the single-global-lock model.
997    fn free_list_pop(
998        segment: &mut ZoneSegment,
999        order: u8,
1000        migratetype: Migratetype,
1001    ) -> Option<u64> {
1002        let head = segment.free_lists[migratetype.index()][order as usize];
1003        if head == 0 {
1004            return None;
1005        }
1006        let next = Self::read_free_next(head);
1007        segment.free_lists[migratetype.index()][order as usize] = next;
1008        if next != 0 {
1009            Self::write_free_prev(next, 0);
1010        }
1011        Self::write_free_next(head, 0);
1012        Self::write_free_prev(head, 0);
1013        Some(head)
1014    }
1015
1016    /// Unlink a specific block from the free list. Returns `true` on success.
1017    ///
1018    /// # Memory ordering
1019    ///
1020    /// See [`Self::free_list_push`] for ordering rationale. The function
1021    /// validates the head pointer before mutating it, which is safe under
1022    /// the single-global-lock model but would require CAS under concurrent access.
1023    fn free_list_remove(
1024        segment: &mut ZoneSegment,
1025        phys: u64,
1026        order: u8,
1027        migratetype: Migratetype,
1028    ) -> bool {
1029        let prev = Self::read_free_prev(phys);
1030        let next = Self::read_free_next(phys);
1031
1032        if prev == 0 {
1033            if segment.free_lists[migratetype.index()][order as usize] != phys {
1034                return false;
1035            }
1036            segment.free_lists[migratetype.index()][order as usize] = next;
1037        } else {
1038            Self::write_free_next(prev, next);
1039        }
1040
1041        if next != 0 {
1042            Self::write_free_prev(next, prev);
1043        }
1044
1045        Self::write_free_next(phys, 0);
1046        Self::write_free_prev(phys, 0);
1047        true
1048    }
1049
1050    /// Reads free next.
1051    #[inline]
1052    fn read_free_next(phys: u64) -> u64 {
1053        let next = get_meta(PhysAddr::new(phys)).next();
1054        if next == FRAME_META_LINK_NONE {
1055            0
1056        } else {
1057            next
1058        }
1059    }
1060
1061    /// Writes free next.
1062    #[inline]
1063    fn write_free_next(phys: u64, next: u64) {
1064        get_meta(PhysAddr::new(phys)).set_next(if next == 0 {
1065            FRAME_META_LINK_NONE
1066        } else {
1067            next
1068        });
1069    }
1070
1071    /// Reads free prev.
1072    #[inline]
1073    fn read_free_prev(phys: u64) -> u64 {
1074        let prev = get_meta(PhysAddr::new(phys)).prev();
1075        if prev == FRAME_META_LINK_NONE {
1076            0
1077        } else {
1078            prev
1079        }
1080    }
1081
1082    /// Writes free prev.
1083    #[inline]
1084    fn write_free_prev(phys: u64, prev: u64) {
1085        get_meta(PhysAddr::new(phys)).set_prev(if prev == 0 {
1086            FRAME_META_LINK_NONE
1087        } else {
1088            prev
1089        });
1090    }
1091
1092    /// Performs the zone index for addr operation.
1093    fn zone_index_for_addr(addr: u64) -> usize {
1094        if addr < DMA_MAX {
1095            ZoneType::DMA as usize
1096        } else if addr < NORMAL_MAX {
1097            ZoneType::Normal as usize
1098        } else {
1099            ZoneType::HighMem as usize
1100        }
1101    }
1102
1103    /// Performs the zone bounds operation.
1104    fn zone_bounds(zone_idx: usize) -> (u64, u64) {
1105        match zone_idx {
1106            x if x == ZoneType::DMA as usize => (0, DMA_MAX),
1107            x if x == ZoneType::Normal as usize => (DMA_MAX, NORMAL_MAX),
1108            _ => (NORMAL_MAX, u64::MAX),
1109        }
1110    }
1111
1112    /// Performs the zone intersection aligned operation.
1113    fn zone_intersection_aligned(region: &MemoryRegion, zone_idx: usize) -> Option<(u64, u64)> {
1114        if !matches!(region.kind, MemoryKind::Free | MemoryKind::Reclaim) {
1115            return None;
1116        }
1117
1118        let region_start = region.base;
1119        let region_end = region.base.saturating_add(region.size);
1120        let (zone_start, zone_end) = Self::zone_bounds(zone_idx);
1121
1122        let start = core::cmp::max(region_start, zone_start);
1123        let end = core::cmp::min(region_end, zone_end);
1124        if start >= end {
1125            return None;
1126        }
1127
1128        // Reserve physical address 0 as sentinel/not-usable.
1129        let start = Self::align_up(core::cmp::max(start, PAGE_SIZE), PAGE_SIZE);
1130        let end = Self::align_down(end, PAGE_SIZE);
1131        if start >= end {
1132            None
1133        } else {
1134            Some((start, end))
1135        }
1136    }
1137
1138    /// Performs the protected overlap end operation.
1139    fn protected_overlap_end(start: u64, end: u64) -> Option<u64> {
1140        for (base, size) in Self::protected_module_ranges().into_iter().flatten() {
1141            if size == 0 {
1142                continue;
1143            }
1144            let pstart = Self::align_down(base, PAGE_SIZE);
1145            let pend = Self::align_up(base.saturating_add(size), PAGE_SIZE);
1146            if end <= pstart || start >= pend {
1147                continue;
1148            }
1149            return Some(pend);
1150        }
1151        None
1152    }
1153
1154    /// Performs the protected module ranges operation.
1155    fn protected_module_ranges() -> [Option<(u64, u64)>; boot_alloc::MAX_PROTECTED_RANGES] {
1156        boot_alloc::protected_ranges_snapshot()
1157    }
1158
1159    /// Performs the pairs for order operation.
1160    #[inline]
1161    fn pairs_for_order(span_pages: usize, order: u8) -> usize {
1162        let pair_span = 1usize << (order as usize + 1);
1163        span_pages.div_ceil(pair_span)
1164    }
1165
1166    /// Performs the bits to bytes operation.
1167    #[inline]
1168    fn bits_to_bytes(bits: usize) -> usize {
1169        bits.div_ceil(8)
1170    }
1171
1172    /// Performs the bitmap bytes for span operation.
1173    fn bitmap_bytes_for_span(span_pages: usize) -> usize {
1174        let mut bytes = 0usize;
1175        for order in 0..=MAX_ORDER as u8 {
1176            bytes += Self::bits_to_bytes(Self::pairs_for_order(span_pages, order));
1177        }
1178        #[cfg(debug_assertions)]
1179        {
1180            bytes += Self::bits_to_bytes(span_pages);
1181        }
1182        bytes += Self::pageblock_tag_bytes_for_span(span_pages);
1183        bytes
1184    }
1185
1186    /// Upper bound for bitmap storage over any segmentation of `page_count` pages.
1187    ///
1188    /// For a single contiguous span of `s` pages, the buddy bitmap uses
1189    /// approximately `s` bits total across all orders (each order contributes
1190    /// `s / 2^(order+1)` pair bits, summing to ~`s`). We add a per-segment
1191    /// overhead to account for small segments where the bound is less tight.
1192    /// The factor of 2 provides safety margin for edge cases (segmentation,
1193    /// alignment, debug bitmaps).
1194    fn bitmap_bytes_upper_bound_for_pages(page_count: usize) -> usize {
1195        // Buddy bitmaps: ~1 bit per page across all orders (sum of s/2^(k+1) ≈ s)
1196        // Factor of 2 for safety margin and segmentation overhead
1197        #[allow(unused_mut)]
1198        let mut bits = page_count.saturating_mul(2);
1199        #[cfg(debug_assertions)]
1200        {
1201            // Debug alloc bitmap: 1 bit per page
1202            bits = bits.saturating_add(page_count);
1203        }
1204        Self::bits_to_bytes(bits)
1205            .saturating_add(Self::pageblock_tag_bytes_upper_bound_for_pages(page_count))
1206    }
1207
1208    /// Exact byte count required for pageblock migratetype tags over one contiguous span.
1209    #[inline]
1210    fn pageblock_tag_bytes_for_span(span_pages: usize) -> usize {
1211        span_pages.div_ceil(PAGEBLOCK_PAGES)
1212    }
1213
1214    /// Safe upper bound for pageblock-tag storage across any segmentation of `page_count` pages.
1215    #[inline]
1216    fn pageblock_tag_bytes_upper_bound_for_pages(page_count: usize) -> usize {
1217        page_count
1218    }
1219
1220    /// Performs the align up operation.
1221    #[inline]
1222    fn align_up(value: u64, align: u64) -> u64 {
1223        debug_assert!(align.is_power_of_two());
1224        (value + align - 1) & !(align - 1)
1225    }
1226
1227    /// Performs the align down operation.
1228    #[inline]
1229    fn align_down(value: u64, align: u64) -> u64 {
1230        debug_assert!(align.is_power_of_two());
1231        value & !(align - 1)
1232    }
1233
1234    /// Default pageblock migratetype assigned at bootstrap for one zone.
1235    #[inline]
1236    fn default_pageblock_migratetype(zone_type: ZoneType) -> Migratetype {
1237        match zone_type {
1238            ZoneType::HighMem => Migratetype::Movable,
1239            ZoneType::DMA | ZoneType::Normal => Migratetype::Unmovable,
1240        }
1241    }
1242
1243    /// Returns the pageblock index covering `phys` inside `segment`.
1244    #[inline]
1245    fn pageblock_index(segment: &ZoneSegment, phys: u64) -> usize {
1246        Self::page_index(segment, phys) / PAGEBLOCK_PAGES
1247    }
1248
1249    /// Decode one pageblock tag byte into a migratetype.
1250    #[inline]
1251    fn decode_pageblock_tag(tag: u8) -> Migratetype {
1252        match tag {
1253            x if x == Migratetype::Movable as u8 => Migratetype::Movable,
1254            _ => Migratetype::Unmovable,
1255        }
1256    }
1257
1258    /// Returns the current pageblock migratetype for a block start.
1259    #[inline]
1260    fn pageblock_migratetype(
1261        segment: &ZoneSegment,
1262        phys: u64,
1263        fallback: Migratetype,
1264    ) -> Migratetype {
1265        if segment.pageblock_count == 0 || segment.pageblock_tags.is_null() {
1266            return fallback;
1267        }
1268        let idx = Self::pageblock_index(segment, phys);
1269        debug_assert!(idx < segment.pageblock_count);
1270        unsafe { Self::decode_pageblock_tag(*segment.pageblock_tags.add(idx)) }
1271    }
1272
1273    /// Retag every pageblock overlapped by the buddy block `[phys, phys + 2^order * PAGE_SIZE)`.
1274    ///
1275    /// When a block is allocated or freed, its pageblocks are retagged to the
1276    /// new migratetype. This is intentional: it reinforces the grouping trend
1277    /// over time (movable allocations reinforce movable pageblocks, etc.).
1278    fn retag_pageblock_range(
1279        segment: &mut ZoneSegment,
1280        phys: u64,
1281        order: u8,
1282        migratetype: Migratetype,
1283    ) {
1284        if segment.pageblock_count == 0 || segment.pageblock_tags.is_null() {
1285            return;
1286        }
1287
1288        let start_page = Self::page_index(segment, phys);
1289        let end_page_exclusive = start_page.saturating_add(1usize << order);
1290        let start_idx = start_page / PAGEBLOCK_PAGES;
1291        let end_idx = end_page_exclusive.saturating_sub(1) / PAGEBLOCK_PAGES;
1292        assert!(
1293            end_idx < segment.pageblock_count,
1294            "buddy: pageblock retag out of bounds: end_idx={} >= pageblock_count={}",
1295            end_idx,
1296            segment.pageblock_count,
1297        );
1298
1299        for idx in start_idx..=end_idx {
1300            unsafe {
1301                *segment.pageblock_tags.add(idx) = migratetype as u8;
1302            }
1303        }
1304    }
1305
1306    /// Count pageblocks by migratetype for one zone.
1307    fn zone_pageblock_counts(zone: &Zone) -> [usize; Migratetype::COUNT] {
1308        let mut counts = [0usize; Migratetype::COUNT];
1309        for segment in zone.segments().iter().take(zone.segment_count) {
1310            if segment.pageblock_count == 0 || segment.pageblock_tags.is_null() {
1311                continue;
1312            }
1313            for idx in 0..segment.pageblock_count {
1314                let migratetype =
1315                    unsafe { Self::decode_pageblock_tag(*segment.pageblock_tags.add(idx)) };
1316                counts[migratetype.index()] = counts[migratetype.index()].saturating_add(1);
1317            }
1318        }
1319        counts
1320    }
1321
1322    fn zone_effective_free_pages(zone: &Zone, zone_idx: usize) -> usize {
1323        zone.available_pages()
1324            .saturating_add(LOCAL_CACHED_ZONE_FRAMES[zone_idx].load(AtomicOrdering::Relaxed))
1325    }
1326
1327    /// Returns whether the zone should be considered for the current request.
1328    fn zone_allows_allocation(
1329        zone: &Zone,
1330        zone_idx: usize,
1331        order: u8,
1332        honor_watermarks: bool,
1333    ) -> bool {
1334        if zone.page_count == 0 {
1335            return false;
1336        }
1337
1338        if !honor_watermarks {
1339            return true;
1340        }
1341
1342        let requested_pages = 1usize << order;
1343        let floor = zone.watermark_min.saturating_add(zone.lowmem_reserve_pages);
1344        Self::zone_effective_free_pages(zone, zone_idx) >= requested_pages.saturating_add(floor)
1345    }
1346
1347    /// Returns whether a buddy block is free and coalescible with `migratetype`.
1348    fn can_merge_with_buddy(phys: u64, order: u8, migratetype: Migratetype) -> bool {
1349        let meta = get_meta(PhysAddr::new(phys));
1350        let flags = meta.get_flags();
1351        flags & frame_flags::FREE != 0
1352            && meta.get_order() == order
1353            && Self::migratetype_from_flags(flags) == migratetype
1354            && !crate::memory::frame::block_phys_has_poison_guard(phys, order)
1355    }
1356
1357    /// Decode the block migratetype stored in frame metadata flags.
1358    fn block_migratetype(frame_phys: u64) -> Migratetype {
1359        Self::migratetype_from_flags(get_meta(PhysAddr::new(frame_phys)).get_flags())
1360    }
1361
1362    /// Decode a migratetype from frame flags.
1363    #[inline]
1364    fn migratetype_from_flags(flags: u32) -> Migratetype {
1365        if flags & frame_flags::MOVABLE != 0 {
1366            Migratetype::Movable
1367        } else {
1368            Migratetype::Unmovable
1369        }
1370    }
1371
1372    /// Encode the metadata flags for a free block of the given migratetype.
1373    #[inline]
1374    fn free_flags_for(migratetype: Migratetype) -> u32 {
1375        match migratetype {
1376            Migratetype::Unmovable => frame_flags::FREE,
1377            Migratetype::Movable => frame_flags::FREE | frame_flags::MOVABLE,
1378        }
1379    }
1380
1381    /// Encode the metadata flags for an allocated block of the given migratetype.
1382    #[inline]
1383    fn allocated_flags_for(migratetype: Migratetype) -> u32 {
1384        match migratetype {
1385            Migratetype::Unmovable => frame_flags::ALLOCATED,
1386            Migratetype::Movable => frame_flags::ALLOCATED | frame_flags::MOVABLE,
1387        }
1388    }
1389
1390    /// Try to allocate from the supplied zone order, first honoring reserves and then bypassing them.
1391    fn alloc_in_zone_order(
1392        &mut self,
1393        order: u8,
1394        migratetype: Migratetype,
1395        zone_order: &[usize],
1396        token: &IrqDisabledToken,
1397    ) -> Option<PhysFrame> {
1398        for honor_watermarks in [true, false] {
1399            for &zi in zone_order {
1400                if let Some(frame) = Self::alloc_from_zone(
1401                    &mut self.zones[zi],
1402                    zi,
1403                    order,
1404                    migratetype,
1405                    honor_watermarks,
1406                    token,
1407                ) {
1408                    return Some(frame);
1409                }
1410            }
1411        }
1412        None
1413    }
1414
1415    /// Returns the preferred zone scan order for one migratetype.
1416    ///
1417    /// Unmovable allocations still prefer `Normal` first because the current
1418    /// kernel hot-touches those pages directly. Movable allocations instead
1419    /// prefer `HighMem` first to preserve scarce low memory for pinned kernel
1420    /// structures and emergency paths.
1421    #[inline]
1422    fn preferred_zone_order(migratetype: Migratetype) -> &'static [usize; ZoneType::COUNT] {
1423        match migratetype {
1424            Migratetype::Unmovable => &UNMOVABLE_ZONE_ORDER,
1425            Migratetype::Movable => &MOVABLE_ZONE_ORDER,
1426        }
1427    }
1428
1429    #[inline]
1430    fn zone_pressure_for_free_pages(zone: &Zone, free_pages: usize) -> ZonePressure {
1431        let reserve_floor = zone.watermark_min.saturating_add(zone.lowmem_reserve_pages);
1432        let low_floor = zone.watermark_low.saturating_add(zone.lowmem_reserve_pages);
1433        let high_floor = zone
1434            .watermark_high
1435            .saturating_add(zone.lowmem_reserve_pages);
1436
1437        if free_pages <= reserve_floor {
1438            ZonePressure::Min
1439        } else if free_pages <= low_floor {
1440            ZonePressure::Low
1441        } else if free_pages <= high_floor {
1442            ZonePressure::High
1443        } else {
1444            ZonePressure::Healthy
1445        }
1446    }
1447
1448    fn compaction_candidate(
1449        &self,
1450        order: u8,
1451        migratetype: Migratetype,
1452        zone_order: &[usize],
1453    ) -> Option<CompactionCandidate> {
1454        if order == 0 {
1455            return None;
1456        }
1457
1458        let requested_pages = 1usize << order;
1459        let mut best: Option<CompactionCandidate> = None;
1460
1461        for &zone_idx in zone_order {
1462            let zone = &self.zones[zone_idx];
1463            if zone.page_count == 0 {
1464                continue;
1465            }
1466
1467            let cached_pages = LOCAL_CACHED_ZONE_FRAMES[zone_idx].load(AtomicOrdering::Relaxed);
1468            if cached_pages == 0 {
1469                continue;
1470            }
1471
1472            let effective_free = Self::zone_effective_free_pages(zone, zone_idx);
1473            let available_pages = effective_free
1474                .saturating_sub(zone.watermark_min.saturating_add(zone.lowmem_reserve_pages));
1475            if available_pages < requested_pages {
1476                continue;
1477            }
1478
1479            // Compute usable pages and fragmentation score in a single pass
1480            // to avoid redundant free list walks.
1481            let (usable_pages, fragmentation_score) =
1482                zone.usable_pages_and_fragmentation(order, cached_pages);
1483            if usable_pages >= requested_pages {
1484                continue;
1485            }
1486
1487            if fragmentation_score
1488                < COMPACTION_FRAGMENTATION_THRESHOLD.load(AtomicOrdering::Relaxed)
1489            {
1490                continue;
1491            }
1492
1493            let pageblocks = Self::zone_pageblock_counts(zone);
1494            let candidate = CompactionCandidate {
1495                zone_idx,
1496                zone_type: zone.zone_type,
1497                order,
1498                migratetype,
1499                pressure: Self::zone_pressure_for_free_pages(zone, effective_free),
1500                fragmentation_score,
1501                requested_pages,
1502                available_pages,
1503                usable_pages,
1504                cached_pages,
1505                pageblock_count: pageblocks[Migratetype::Unmovable.index()]
1506                    .saturating_add(pageblocks[Migratetype::Movable.index()]),
1507                matching_pageblocks: pageblocks[migratetype.index()],
1508            };
1509
1510            let replace = match best {
1511                None => true,
1512                Some(current) => {
1513                    candidate.fragmentation_score > current.fragmentation_score
1514                        || (candidate.fragmentation_score == current.fragmentation_score
1515                            && candidate.cached_pages > current.cached_pages)
1516                        || (candidate.fragmentation_score == current.fragmentation_score
1517                            && candidate.cached_pages == current.cached_pages
1518                            && candidate.matching_pageblocks > current.matching_pageblocks)
1519                }
1520            };
1521
1522            if replace {
1523                best = Some(candidate);
1524            }
1525        }
1526
1527        best
1528    }
1529
1530    #[inline]
1531    fn compaction_drain_budget(candidate: CompactionCandidate) -> usize {
1532        let pageblock_goal = if candidate.matching_pageblocks != 0 {
1533            PAGEBLOCK_PAGES
1534        } else {
1535            candidate.requested_pages
1536        };
1537        let target_pages = core::cmp::max(candidate.requested_pages, pageblock_goal)
1538            .saturating_mul(2)
1539            .max(LOCAL_CACHE_FLUSH_BATCH);
1540        core::cmp::min(target_pages, candidate.cached_pages)
1541    }
1542
1543    /// Allocate while the caller already owns the global allocator lock.
1544    fn alloc_locked_with_migratetype(
1545        &mut self,
1546        order: u8,
1547        migratetype: Migratetype,
1548        token: &IrqDisabledToken,
1549    ) -> Result<PhysFrame, AllocError> {
1550        if order > MAX_ORDER as u8 {
1551            return Err(AllocError::InvalidOrder);
1552        }
1553
1554        let cpu_idx = crate::arch::percpu::current_cpu_index();
1555        if ALLOC_IN_PROGRESS[cpu_idx].swap(true, core::sync::atomic::Ordering::Acquire) {
1556            panic!("Recursive allocation detected on CPU {}!", cpu_idx);
1557        }
1558
1559        let result = self
1560            .alloc_in_zone_order(
1561                order,
1562                migratetype,
1563                Self::preferred_zone_order(migratetype),
1564                token,
1565            )
1566            .ok_or_else(|| {
1567                crate::memory::buddy::record_buddy_alloc_fail(order);
1568                AllocError::OutOfMemory
1569            });
1570
1571        ALLOC_IN_PROGRESS[cpu_idx].store(false, core::sync::atomic::Ordering::Release);
1572        result
1573    }
1574
1575    /// Allocate from one explicit zone while the caller already owns the global allocator lock.
1576    fn alloc_zone_locked(
1577        &mut self,
1578        order: u8,
1579        zone: ZoneType,
1580        migratetype: Migratetype,
1581        token: &IrqDisabledToken,
1582    ) -> Result<PhysFrame, AllocError> {
1583        if order > MAX_ORDER as u8 {
1584            return Err(AllocError::InvalidOrder);
1585        }
1586
1587        let cpu_idx = crate::arch::percpu::current_cpu_index();
1588        if ALLOC_IN_PROGRESS[cpu_idx].swap(true, core::sync::atomic::Ordering::Acquire) {
1589            panic!("Recursive allocation detected on CPU {}!", cpu_idx);
1590        }
1591
1592        let zone_idx = zone as usize;
1593        let zone_order = [zone_idx];
1594        let result = self
1595            .alloc_in_zone_order(order, migratetype, &zone_order, token)
1596            .ok_or_else(|| {
1597                crate::memory::buddy::record_buddy_alloc_fail(order);
1598                AllocError::OutOfMemory
1599            });
1600
1601        ALLOC_IN_PROGRESS[cpu_idx].store(false, core::sync::atomic::Ordering::Release);
1602        result
1603    }
1604
1605    fn mark_block_allocated(frame_phys: u64, order: u8, migratetype: Migratetype) {
1606        let page_count = 1usize << order;
1607        for page_idx in 0..page_count {
1608            let phys = frame_phys + page_idx as u64 * PAGE_SIZE;
1609            let meta = get_meta(PhysAddr::new(phys));
1610            // Sentinel must still be intact at this point : if not, the frame
1611            // was never on the free list (double-alloc or metadata corruption).
1612            debug_assert_eq!(
1613                meta.get_refcount(),
1614                crate::memory::frame::REFCOUNT_UNUSED,
1615                "buddy: mark_block_allocated on frame {:#x} with unexpected refcount (corruption?)",
1616                phys,
1617            );
1618            meta.set_flags(Self::allocated_flags_for(migratetype));
1619            meta.set_order(order);
1620            // Leave refcount as REFCOUNT_UNUSED; FrameAllocOptions::allocate()
1621            // will perform CAS(REFCOUNT_UNUSED => 1) as the fail-fast handoff.
1622        }
1623    }
1624
1625    fn mark_block_free(frame_phys: u64, order: u8, migratetype: Migratetype) {
1626        Self::set_block_meta(
1627            frame_phys,
1628            order,
1629            Self::free_flags_for(migratetype),
1630            crate::memory::frame::REFCOUNT_UNUSED,
1631        );
1632    }
1633
1634    /// Stamp every 4 KiB [`MetaSlot`] in the buddy block (flags, order, free-list links, refcount).
1635    ///
1636    /// [`MetaSlot::reset_with_free_list_meta`] runs on **each** page, including non-head pages
1637    /// of a multi-page block: the whole block returns to the buddy as one unit, so vtable and
1638    /// guard bits are cleared (except poison preserved per-slot) on every constituent frame.
1639    fn set_block_meta(frame_phys: u64, order: u8, flags: u32, refcount: u32) {
1640        let page_count = 1usize << order;
1641        for page_idx in 0..page_count {
1642            let phys = frame_phys + page_idx as u64 * PAGE_SIZE;
1643            let meta = get_meta(PhysAddr::new(phys));
1644            meta.set_flags(flags);
1645            meta.set_order(order);
1646            meta.set_next(FRAME_META_LINK_NONE);
1647            meta.set_prev(FRAME_META_LINK_NONE);
1648            meta.set_refcount(refcount);
1649            meta.reset_with_free_list_meta();
1650        }
1651    }
1652}
1653
1654static BUDDY_ALLOCATOR: SpinLock<Option<BuddyAllocator>> = SpinLock::new(None);
1655
1656/// Per-order allocation failure counters.
1657///
1658/// `BUDDY_ALLOC_FAIL_COUNTS[order]` counts how many times a request for
1659/// `order` failed to find a free block at `order` or any higher order.
1660/// These are incremented in `alloc_from_zone` when the loop exhausts all
1661/// orders without finding a free block.
1662///
1663/// Read via `buddy_alloc_fail_counts_snapshot()` for diagnostics.
1664static BUDDY_ALLOC_FAIL_COUNTS: [core::sync::atomic::AtomicUsize;
1665    crate::memory::zone::MAX_ORDER + 1] =
1666    [const { core::sync::atomic::AtomicUsize::new(0) }; crate::memory::zone::MAX_ORDER + 1];
1667
1668static COMPACTION_ATTEMPTS: AtomicUsize = AtomicUsize::new(0);
1669static COMPACTION_SUCCESSES: AtomicUsize = AtomicUsize::new(0);
1670static COMPACTION_LAST_ORDER: AtomicUsize = AtomicUsize::new(COMPACTION_SNAPSHOT_NONE);
1671static COMPACTION_LAST_MIGRATETYPE: AtomicUsize = AtomicUsize::new(COMPACTION_SNAPSHOT_NONE);
1672static COMPACTION_LAST_ZONE: AtomicUsize = AtomicUsize::new(COMPACTION_SNAPSHOT_NONE);
1673static COMPACTION_LAST_PRESSURE: AtomicUsize = AtomicUsize::new(ZonePressure::SNAPSHOT_COUNT);
1674static COMPACTION_LAST_FRAGMENTATION: AtomicUsize = AtomicUsize::new(0);
1675static COMPACTION_LAST_REQUESTED_PAGES: AtomicUsize = AtomicUsize::new(0);
1676static COMPACTION_LAST_AVAILABLE_PAGES: AtomicUsize = AtomicUsize::new(0);
1677static COMPACTION_LAST_USABLE_PAGES: AtomicUsize = AtomicUsize::new(0);
1678static COMPACTION_LAST_CACHED_PAGES: AtomicUsize = AtomicUsize::new(0);
1679static COMPACTION_LAST_DRAINED_PAGES: AtomicUsize = AtomicUsize::new(0);
1680static COMPACTION_LAST_PAGEBLOCK_COUNT: AtomicUsize = AtomicUsize::new(0);
1681static COMPACTION_LAST_MATCHING_PAGEBLOCKS: AtomicUsize = AtomicUsize::new(0);
1682
1683/// Records a buddy allocation failure for the given order.
1684///
1685/// Called from `alloc_from_zone` when no free block is available at any
1686/// order >= `order`. Increments the per-order counter for diagnostics.
1687pub(crate) fn record_buddy_alloc_fail(order: u8) {
1688    let idx = order as usize;
1689    if idx <= crate::memory::zone::MAX_ORDER {
1690        BUDDY_ALLOC_FAIL_COUNTS[idx].fetch_add(1, core::sync::atomic::Ordering::Relaxed);
1691    }
1692}
1693
1694/// Returns the buddy allocation failure counts by order.
1695///
1696/// Use this for diagnostics : e.g., to determine whether a heap panic is
1697/// caused by genuine memory pressure or by high-order fragmentation.
1698pub fn buddy_alloc_fail_counts_snapshot() -> [usize; crate::memory::zone::MAX_ORDER + 1] {
1699    let mut out = [0usize; crate::memory::zone::MAX_ORDER + 1];
1700    for (i, counter) in BUDDY_ALLOC_FAIL_COUNTS.iter().enumerate() {
1701        out[i] = counter.load(core::sync::atomic::Ordering::Relaxed);
1702    }
1703    out
1704}
1705
1706fn snapshot_zone_type(value: usize) -> Option<ZoneType> {
1707    match value {
1708        x if x == ZoneType::DMA as usize => Some(ZoneType::DMA),
1709        x if x == ZoneType::Normal as usize => Some(ZoneType::Normal),
1710        x if x == ZoneType::HighMem as usize => Some(ZoneType::HighMem),
1711        _ => None,
1712    }
1713}
1714
1715fn snapshot_migratetype(value: usize) -> Option<Migratetype> {
1716    match value {
1717        x if x == Migratetype::Unmovable as usize => Some(Migratetype::Unmovable),
1718        x if x == Migratetype::Movable as usize => Some(Migratetype::Movable),
1719        _ => None,
1720    }
1721}
1722
1723fn record_compaction_attempt(candidate: CompactionCandidate, drained_pages: usize, success: bool) {
1724    COMPACTION_ATTEMPTS.fetch_add(1, AtomicOrdering::Relaxed);
1725    if success {
1726        COMPACTION_SUCCESSES.fetch_add(1, AtomicOrdering::Relaxed);
1727    }
1728
1729    COMPACTION_LAST_ORDER.store(candidate.order as usize, AtomicOrdering::Relaxed);
1730    COMPACTION_LAST_MIGRATETYPE.store(candidate.migratetype as usize, AtomicOrdering::Relaxed);
1731    COMPACTION_LAST_ZONE.store(candidate.zone_type as usize, AtomicOrdering::Relaxed);
1732    COMPACTION_LAST_PRESSURE.store(candidate.pressure.as_snapshot(), AtomicOrdering::Relaxed);
1733    COMPACTION_LAST_FRAGMENTATION.store(candidate.fragmentation_score, AtomicOrdering::Relaxed);
1734    COMPACTION_LAST_REQUESTED_PAGES.store(candidate.requested_pages, AtomicOrdering::Relaxed);
1735    COMPACTION_LAST_AVAILABLE_PAGES.store(candidate.available_pages, AtomicOrdering::Relaxed);
1736    COMPACTION_LAST_USABLE_PAGES.store(candidate.usable_pages, AtomicOrdering::Relaxed);
1737    COMPACTION_LAST_CACHED_PAGES.store(candidate.cached_pages, AtomicOrdering::Relaxed);
1738    COMPACTION_LAST_DRAINED_PAGES.store(drained_pages, AtomicOrdering::Relaxed);
1739    COMPACTION_LAST_PAGEBLOCK_COUNT.store(candidate.pageblock_count, AtomicOrdering::Relaxed);
1740    COMPACTION_LAST_MATCHING_PAGEBLOCKS
1741        .store(candidate.matching_pageblocks, AtomicOrdering::Relaxed);
1742}
1743
1744/// Snapshot compaction-assist telemetry without locking the allocator.
1745pub fn compaction_stats_snapshot() -> CompactionStats {
1746    let last_order = COMPACTION_LAST_ORDER.load(AtomicOrdering::Relaxed);
1747    let last_migratetype = COMPACTION_LAST_MIGRATETYPE.load(AtomicOrdering::Relaxed);
1748    let last_zone = COMPACTION_LAST_ZONE.load(AtomicOrdering::Relaxed);
1749    let last_pressure = COMPACTION_LAST_PRESSURE.load(AtomicOrdering::Relaxed);
1750
1751    CompactionStats {
1752        attempts: COMPACTION_ATTEMPTS.load(AtomicOrdering::Relaxed),
1753        successes: COMPACTION_SUCCESSES.load(AtomicOrdering::Relaxed),
1754        last_order: if last_order == COMPACTION_SNAPSHOT_NONE {
1755            None
1756        } else {
1757            Some(last_order as u8)
1758        },
1759        last_migratetype: snapshot_migratetype(last_migratetype),
1760        last_zone: snapshot_zone_type(last_zone),
1761        last_pressure: ZonePressure::from_snapshot(last_pressure),
1762        last_fragmentation_score: COMPACTION_LAST_FRAGMENTATION.load(AtomicOrdering::Relaxed),
1763        last_requested_pages: COMPACTION_LAST_REQUESTED_PAGES.load(AtomicOrdering::Relaxed),
1764        last_available_pages: COMPACTION_LAST_AVAILABLE_PAGES.load(AtomicOrdering::Relaxed),
1765        last_usable_pages: COMPACTION_LAST_USABLE_PAGES.load(AtomicOrdering::Relaxed),
1766        last_cached_pages: COMPACTION_LAST_CACHED_PAGES.load(AtomicOrdering::Relaxed),
1767        last_drained_pages: COMPACTION_LAST_DRAINED_PAGES.load(AtomicOrdering::Relaxed),
1768        last_pageblock_count: COMPACTION_LAST_PAGEBLOCK_COUNT.load(AtomicOrdering::Relaxed),
1769        last_matching_pageblocks: COMPACTION_LAST_MATCHING_PAGEBLOCKS.load(AtomicOrdering::Relaxed),
1770    }
1771}
1772
1773/// Pages permanently withheld from the buddy free lists due to [`meta_guard::POISONED`].
1774static POISON_QUARANTINE_PAGES: AtomicUsize = AtomicUsize::new(0);
1775
1776/// Snapshot of pages quarantined (not recycled) because frame metadata reported poison.
1777pub fn poison_quarantine_pages_snapshot() -> usize {
1778    POISON_QUARANTINE_PAGES.load(AtomicOrdering::Relaxed)
1779}
1780
1781/// Returns the global buddy lock address for deadlock tracing.
1782pub fn debug_buddy_lock_addr() -> usize {
1783    &BUDDY_ALLOCATOR as *const _ as usize
1784}
1785
1786/// Per-CPU flag to detect recursive allocations (deadlocks from logs/interrupts)
1787static ALLOC_IN_PROGRESS: [core::sync::atomic::AtomicBool; crate::arch::percpu::MAX_CPUS] =
1788    [const { core::sync::atomic::AtomicBool::new(false) }; crate::arch::percpu::MAX_CPUS];
1789
1790struct LocalFrameCache {
1791    len: usize,
1792    frames: [u64; LOCAL_CACHE_CAPACITY],
1793}
1794
1795impl LocalFrameCache {
1796    const fn new() -> Self {
1797        Self {
1798            len: 0,
1799            frames: [0; LOCAL_CACHE_CAPACITY],
1800        }
1801    }
1802
1803    fn clear(&mut self) {
1804        self.len = 0;
1805    }
1806
1807    fn pop(&mut self) -> Option<PhysFrame> {
1808        if self.len == 0 {
1809            return None;
1810        }
1811        self.len -= 1;
1812        Some(PhysFrame {
1813            start_address: PhysAddr::new(self.frames[self.len]),
1814        })
1815    }
1816
1817    fn push(&mut self, frame: PhysFrame) -> Result<(), PhysFrame> {
1818        if self.len >= LOCAL_CACHE_CAPACITY {
1819            return Err(frame);
1820        }
1821        self.frames[self.len] = frame.start_address.as_u64();
1822        self.len += 1;
1823        Ok(())
1824    }
1825
1826    fn pop_many(&mut self, out: &mut [u64]) -> usize {
1827        let count = core::cmp::min(self.len, out.len());
1828        for slot in out.iter_mut().take(count) {
1829            self.len -= 1;
1830            *slot = self.frames[self.len];
1831        }
1832        count
1833    }
1834
1835    fn pop_many_for_zone(&mut self, out: &mut [u64], zone_idx: usize) -> usize {
1836        let mut written = 0usize;
1837        let mut idx = 0usize;
1838
1839        while idx < self.len && written < out.len() {
1840            let phys = self.frames[idx];
1841            if zone_index_for_phys(phys) != zone_idx {
1842                idx += 1;
1843                continue;
1844            }
1845
1846            self.len -= 1;
1847            out[written] = phys;
1848            written += 1;
1849            self.frames[idx] = self.frames[self.len];
1850        }
1851
1852        written
1853    }
1854}
1855
1856/// Per-CPU frame caches protected by a `PreemptDisabled` guardian.
1857///
1858/// These caches are only accessed from `alloc_order0_cached` / `free_order0_cached`,
1859/// which are always called with IRQs already disabled by the caller (via
1860/// `IrqDisabledToken`).  Using `PreemptDisabled` instead of the default
1861/// `IrqDisabled` avoids redundant RFLAGS save/restore on every lock
1862/// acquisition while still preventing preemption-driven data races.
1863///
1864/// # Safety invariant
1865///
1866/// If any future code path acquires a `LOCAL_FRAME_CACHES` lock from an
1867/// interrupt handler or without IRQs disabled, this must be reverted to
1868/// `SpinLock<LocalFrameCache>` (default `IrqDisabled` guardian).
1869static LOCAL_FRAME_CACHES: [SpinLock<LocalFrameCache, PreemptDisabled>; LOCAL_CACHE_SLOTS] =
1870    [const { SpinLock::new(LocalFrameCache::new()) }; LOCAL_CACHE_SLOTS];
1871static LOCAL_CACHED_FRAMES: AtomicUsize = AtomicUsize::new(0);
1872static LOCAL_CACHED_ZONE_FRAMES: [AtomicUsize; ZoneType::COUNT] =
1873    [const { AtomicUsize::new(0) }; ZoneType::COUNT];
1874static LOCAL_CACHED_ZONE_MIGRATETYPE_FRAMES: [AtomicUsize; LOCAL_CACHED_ZONE_MIGRATETYPE_SLOTS] =
1875    [const { AtomicUsize::new(0) }; LOCAL_CACHED_ZONE_MIGRATETYPE_SLOTS];
1876
1877type GlobalGuard = SpinLockGuard<'static, Option<BuddyAllocator>>;
1878
1879struct OnDemandGlobalLock {
1880    guard: Option<GlobalGuard>,
1881}
1882
1883impl OnDemandGlobalLock {
1884    fn new() -> Self {
1885        Self { guard: None }
1886    }
1887
1888    fn unlock(&mut self) {
1889        self.guard = None;
1890    }
1891
1892    fn with_allocator<R>(
1893        &mut self,
1894        f: impl FnOnce(&mut BuddyAllocator, &IrqDisabledToken) -> R,
1895    ) -> Option<R> {
1896        let guard = self.guard.get_or_insert_with(|| BUDDY_ALLOCATOR.lock());
1897        guard.with_mut_and_token(|slot, token| slot.as_mut().map(|allocator| f(allocator, token)))
1898    }
1899
1900    fn alloc_with_migratetype(
1901        &mut self,
1902        order: u8,
1903        migratetype: Migratetype,
1904    ) -> Result<PhysFrame, AllocError> {
1905        self.with_allocator(|allocator, token| {
1906            allocator.alloc_locked_with_migratetype(order, migratetype, token)
1907        })
1908        .unwrap_or(Err(AllocError::OutOfMemory))
1909    }
1910
1911    fn free(&mut self, frame: PhysFrame, order: u8) {
1912        let _ = self.with_allocator(|allocator, token| allocator.free(frame, order, token));
1913    }
1914
1915    fn free_phys_batch(&mut self, phys_batch: &[u64], count: usize) {
1916        if count == 0 {
1917            return;
1918        }
1919        let _ = self.with_allocator(|allocator, token| {
1920            for phys in phys_batch.iter().take(count).copied() {
1921                allocator.free(
1922                    PhysFrame {
1923                        start_address: PhysAddr::new(phys),
1924                    },
1925                    0,
1926                    token,
1927                );
1928            }
1929        });
1930    }
1931}
1932
1933#[inline]
1934fn zone_index_for_phys(phys: u64) -> usize {
1935    if phys < DMA_MAX {
1936        ZoneType::DMA as usize
1937    } else if phys < NORMAL_MAX {
1938        ZoneType::Normal as usize
1939    } else {
1940        ZoneType::HighMem as usize
1941    }
1942}
1943
1944#[inline]
1945fn local_cache_slot(cpu_idx: usize, migratetype: Migratetype) -> usize {
1946    migratetype.index() * crate::arch::percpu::MAX_CPUS + cpu_idx
1947}
1948
1949#[inline]
1950fn local_cached_zone_migratetype_slot(zone_idx: usize, migratetype: Migratetype) -> usize {
1951    migratetype.index() * ZoneType::COUNT + zone_idx
1952}
1953
1954#[inline]
1955fn is_cacheable_phys_for(phys: u64, migratetype: Migratetype) -> bool {
1956    match migratetype {
1957        Migratetype::Unmovable => zone_index_for_phys(phys) == ZoneType::Normal as usize,
1958        Migratetype::Movable => zone_index_for_phys(phys) != ZoneType::DMA as usize,
1959    }
1960}
1961
1962#[inline]
1963fn local_cached_zone_migratetype_count(zone_idx: usize, migratetype: Migratetype) -> usize {
1964    LOCAL_CACHED_ZONE_MIGRATETYPE_FRAMES[local_cached_zone_migratetype_slot(zone_idx, migratetype)]
1965        .load(AtomicOrdering::Relaxed)
1966}
1967
1968#[inline]
1969fn local_cached_inc_phys(phys: u64, migratetype: Migratetype) {
1970    let zone_idx = zone_index_for_phys(phys);
1971    LOCAL_CACHED_FRAMES.fetch_add(1, AtomicOrdering::Relaxed);
1972    LOCAL_CACHED_ZONE_FRAMES[zone_idx].fetch_add(1, AtomicOrdering::Relaxed);
1973    LOCAL_CACHED_ZONE_MIGRATETYPE_FRAMES[local_cached_zone_migratetype_slot(zone_idx, migratetype)]
1974        .fetch_add(1, AtomicOrdering::Relaxed);
1975}
1976
1977#[inline]
1978fn local_cached_dec_phys(phys: u64, migratetype: Migratetype) {
1979    let prev_total = LOCAL_CACHED_FRAMES.fetch_sub(1, AtomicOrdering::Relaxed);
1980    debug_assert!(prev_total > 0);
1981    let zone = zone_index_for_phys(phys);
1982    let prev_zone = LOCAL_CACHED_ZONE_FRAMES[zone].fetch_sub(1, AtomicOrdering::Relaxed);
1983    debug_assert!(prev_zone > 0);
1984    let prev_zone_type = LOCAL_CACHED_ZONE_MIGRATETYPE_FRAMES
1985        [local_cached_zone_migratetype_slot(zone, migratetype)]
1986    .fetch_sub(1, AtomicOrdering::Relaxed);
1987    debug_assert!(prev_zone_type > 0);
1988}
1989
1990fn drain_local_caches_to_global(max_pages: usize, global: &mut OnDemandGlobalLock) -> usize {
1991    if max_pages == 0 {
1992        return 0;
1993    }
1994
1995    let mut drained = 0usize;
1996    let mut batch = [0u64; LOCAL_CACHE_FLUSH_BATCH];
1997    for migratetype in Migratetype::ALL {
1998        for cpu in 0..crate::arch::percpu::MAX_CPUS {
1999            if drained >= max_pages {
2000                break;
2001            }
2002            let target = core::cmp::min(batch.len(), max_pages.saturating_sub(drained));
2003            if target == 0 {
2004                break;
2005            }
2006
2007            let popped = {
2008                let mut cache = LOCAL_FRAME_CACHES[local_cache_slot(cpu, migratetype)].lock();
2009                cache.pop_many(&mut batch[..target])
2010            };
2011            if popped == 0 {
2012                continue;
2013            }
2014
2015            for phys in batch.iter().take(popped).copied() {
2016                local_cached_dec_phys(phys, migratetype);
2017            }
2018            global.free_phys_batch(&batch, popped);
2019
2020            // Keep lock acquisition on-demand during cross-CPU draining.
2021            global.unlock();
2022            drained += popped;
2023        }
2024    }
2025
2026    drained
2027}
2028
2029fn drain_local_caches_for_zone(
2030    max_pages: usize,
2031    zone_idx: usize,
2032    primary_migratetype: Migratetype,
2033    global: &mut OnDemandGlobalLock,
2034) -> usize {
2035    if max_pages == 0 {
2036        return 0;
2037    }
2038
2039    let mut drained = 0usize;
2040    let mut batch = [0u64; LOCAL_CACHE_FLUSH_BATCH];
2041
2042    for migratetype in primary_migratetype.fallback_order() {
2043        for cpu in 0..crate::arch::percpu::MAX_CPUS {
2044            if drained >= max_pages {
2045                return drained;
2046            }
2047
2048            let target = core::cmp::min(batch.len(), max_pages.saturating_sub(drained));
2049            if target == 0 {
2050                break;
2051            }
2052
2053            let popped = {
2054                let mut cache = LOCAL_FRAME_CACHES[local_cache_slot(cpu, migratetype)].lock();
2055                cache.pop_many_for_zone(&mut batch[..target], zone_idx)
2056            };
2057            if popped == 0 {
2058                continue;
2059            }
2060
2061            for phys in batch.iter().take(popped).copied() {
2062                local_cached_dec_phys(phys, migratetype);
2063            }
2064            global.free_phys_batch(&batch, popped);
2065            global.unlock();
2066            drained += popped;
2067        }
2068    }
2069
2070    if drained < max_pages {
2071        drained = drained.saturating_add(drain_local_caches_to_global(
2072            max_pages.saturating_sub(drained),
2073            global,
2074        ));
2075    }
2076
2077    drained
2078}
2079
2080/// Initializes buddy allocator.
2081pub fn init_buddy_allocator(memory_regions: &[MemoryRegion]) {
2082    crate::e9_mark!(b'B');
2083    for cache in &LOCAL_FRAME_CACHES {
2084        cache.lock().clear();
2085    }
2086    LOCAL_CACHED_FRAMES.store(0, AtomicOrdering::Relaxed);
2087    for zone_cached in &LOCAL_CACHED_ZONE_FRAMES {
2088        zone_cached.store(0, AtomicOrdering::Relaxed);
2089    }
2090    for zone_cached in &LOCAL_CACHED_ZONE_MIGRATETYPE_FRAMES {
2091        zone_cached.store(0, AtomicOrdering::Relaxed);
2092    }
2093    crate::e9_mark!(b'b');
2094
2095    {
2096        let mut guard = BUDDY_ALLOCATOR.lock();
2097        crate::e9_mark!(b'L');
2098        *guard = Some(BuddyAllocator::new());
2099        guard.with_mut_and_token(|slot, _token| {
2100            if let Some(allocator) = slot.as_mut() {
2101                allocator.init(memory_regions);
2102            }
2103        });
2104    }
2105    // Race/corruption diagnostic: register buddy lock for E9 LOCK-A/LOCK-R traces.
2106    crate::sync::debug_set_trace_buddy_addr(debug_buddy_lock_addr());
2107    crate::e9_mark!(b'F');
2108}
2109
2110/// Returns allocator.
2111pub fn get_allocator() -> &'static SpinLock<Option<BuddyAllocator>> {
2112    &BUDDY_ALLOCATOR
2113}
2114
2115fn refill_local_cache(
2116    cpu_idx: usize,
2117    global: &mut OnDemandGlobalLock,
2118    migratetype: Migratetype,
2119) -> Result<PhysFrame, AllocError> {
2120    // Critical path: refill in batches from the global allocator to amortize lock contention.
2121    let (base, order) = match global.alloc_with_migratetype(LOCAL_CACHE_REFILL_ORDER, migratetype) {
2122        Ok(frame) => (frame, LOCAL_CACHE_REFILL_ORDER),
2123        Err(AllocError::OutOfMemory) => (global.alloc_with_migratetype(0, migratetype)?, 0),
2124        Err(e) => return Err(e),
2125    };
2126    global.unlock();
2127
2128    let frame_count = 1usize << order;
2129    let mut overflow = [0u64; LOCAL_CACHE_REFILL_FRAMES];
2130    let mut overflow_len = 0usize;
2131    let mut ret = None;
2132
2133    {
2134        let mut cache = LOCAL_FRAME_CACHES[local_cache_slot(cpu_idx, migratetype)].lock();
2135        for idx in 0..frame_count {
2136            let phys = base.start_address.as_u64() + (idx as u64) * PAGE_SIZE;
2137            let frame = PhysFrame {
2138                start_address: PhysAddr::new(phys),
2139            };
2140            if !is_cacheable_phys_for(phys, migratetype) {
2141                overflow[overflow_len] = phys;
2142                overflow_len += 1;
2143                continue;
2144            }
2145            if ret.is_none() {
2146                // Re-publish the returned page as an allocated order-0 block.
2147                // The refcount must stay REFCOUNT_UNUSED so FrameAllocOptions
2148                // can still claim it via CAS(UNUSED -> 1).
2149                BuddyAllocator::mark_block_allocated(phys, 0, migratetype);
2150                ret = Some(frame);
2151                continue;
2152            }
2153            // Pages parked in the local cache are logically free and must
2154            // therefore carry the free-list sentinel invariant.
2155            BuddyAllocator::mark_block_free(phys, 0, migratetype);
2156            if cache.push(frame).is_ok() {
2157                local_cached_inc_phys(phys, migratetype);
2158            } else {
2159                overflow[overflow_len] = phys;
2160                overflow_len += 1;
2161            }
2162        }
2163    }
2164
2165    if overflow_len != 0 {
2166        global.free_phys_batch(&overflow, overflow_len);
2167    }
2168
2169    ret.ok_or(AllocError::OutOfMemory)
2170}
2171
2172fn steal_from_other_caches(cpu_idx: usize, migratetype: Migratetype) -> Option<PhysFrame> {
2173    let cpu_count = crate::arch::percpu::cpu_count()
2174        .max(1)
2175        .min(crate::arch::percpu::MAX_CPUS);
2176
2177    for step in 1..cpu_count {
2178        let peer = (cpu_idx + step) % cpu_count;
2179        let mut cache = LOCAL_FRAME_CACHES[local_cache_slot(peer, migratetype)].lock();
2180        if let Some(frame) = cache.pop() {
2181            BuddyAllocator::mark_block_allocated(frame.start_address.as_u64(), 0, migratetype);
2182            local_cached_dec_phys(frame.start_address.as_u64(), migratetype);
2183            return Some(frame);
2184        }
2185    }
2186    None
2187}
2188
2189fn alloc_order0_cached(migratetype: Migratetype) -> Result<PhysFrame, AllocError> {
2190    let cpu_idx = crate::arch::percpu::current_cpu_index();
2191
2192    {
2193        let mut cache = LOCAL_FRAME_CACHES[local_cache_slot(cpu_idx, migratetype)].lock();
2194        if let Some(frame) = cache.pop() {
2195            BuddyAllocator::mark_block_allocated(frame.start_address.as_u64(), 0, migratetype);
2196            local_cached_dec_phys(frame.start_address.as_u64(), migratetype);
2197            return Ok(frame);
2198        }
2199    }
2200
2201    let mut global = OnDemandGlobalLock::new();
2202
2203    if let Ok(frame) = refill_local_cache(cpu_idx, &mut global, migratetype) {
2204        return Ok(frame);
2205    }
2206    // Critical lock-order rule: never hold global while probing local caches.
2207    global.unlock();
2208
2209    if let Some(frame) = steal_from_other_caches(cpu_idx, migratetype) {
2210        return Ok(frame);
2211    }
2212
2213    global.alloc_with_migratetype(0, migratetype)
2214}
2215
2216fn free_order0_cached(frame: PhysFrame, migratetype: Migratetype) {
2217    // NOTE: O(2^order) MetaSlot scan : acceptable here because order is always 0
2218    // (single-page check) on this hot path.
2219    //
2220    // # Safety: poison guard check vs cache push
2221    //
2222    // The poison guard check happens before the frame is pushed into the local
2223    // cache. This is safe because:
2224    //   1. The frame is exclusively owned by the caller (no other CPU can modify it)
2225    //   2. IRQs are disabled during this function (via IrqDisabledToken)
2226    // Therefore no concurrent poisoning is possible between the check and the push.
2227    if crate::memory::frame::block_phys_has_poison_guard(frame.start_address.as_u64(), 0) {
2228        let mut global = OnDemandGlobalLock::new();
2229        global.free(frame, 0);
2230        return;
2231    }
2232
2233    if !is_cacheable_phys_for(frame.start_address.as_u64(), migratetype) {
2234        let mut global = OnDemandGlobalLock::new();
2235        global.free(frame, 0);
2236        return;
2237    }
2238
2239    let cpu_idx = crate::arch::percpu::current_cpu_index();
2240    let mut spill = [0u64; LOCAL_CACHE_FLUSH_BATCH];
2241
2242    let spill_len = {
2243        let mut cache = LOCAL_FRAME_CACHES[local_cache_slot(cpu_idx, migratetype)].lock();
2244        if cache.push(frame).is_ok() {
2245            // Mark free only on the success path: the incoming frame transitions
2246            // from "caller-allocated" to "cache sentinel" (REFCOUNT_UNUSED).
2247            BuddyAllocator::mark_block_free(frame.start_address.as_u64(), 0, migratetype);
2248            local_cached_inc_phys(frame.start_address.as_u64(), migratetype);
2249            return;
2250        }
2251
2252        // Cache full: pop existing frames to spill to buddy, then retry the push.
2253        let mut spill_len = cache.pop_many(&mut spill);
2254        for phys in spill.iter().take(spill_len).copied() {
2255            local_cached_dec_phys(phys, migratetype);
2256        }
2257
2258        if cache.push(frame).is_ok() {
2259            BuddyAllocator::mark_block_free(frame.start_address.as_u64(), 0, migratetype);
2260            local_cached_inc_phys(frame.start_address.as_u64(), migratetype);
2261        } else {
2262            // Still full after spilling : the incoming frame joins the spill batch.
2263            // It will be marked free by free_phys_batch => free_to_zone.
2264            spill[spill_len] = frame.start_address.as_u64();
2265            spill_len += 1;
2266        }
2267        spill_len
2268    };
2269
2270    if spill_len != 0 {
2271        let mut global = OnDemandGlobalLock::new();
2272        global.free_phys_batch(&spill, spill_len);
2273    }
2274}
2275
2276/// Allocate frames with per-CPU caching on order-0 requests.
2277///
2278/// `_token` is a compile-time proof that interrupts are disabled on the calling CPU,
2279/// preventing re-entrant allocation through an interrupt handler on the same lock.
2280pub fn alloc(_token: &IrqDisabledToken, order: u8) -> Result<PhysFrame, AllocError> {
2281    alloc_migratetype(_token, order, Migratetype::Unmovable)
2282}
2283
2284/// Allocate frames with an explicit migratetype preference.
2285///
2286/// Order-0 allocations use a per-CPU cache partitioned by migratetype so the
2287/// fast path preserves the caller's mobility class.
2288pub fn alloc_migratetype(
2289    _token: &IrqDisabledToken,
2290    order: u8,
2291    migratetype: Migratetype,
2292) -> Result<PhysFrame, AllocError> {
2293    if crate::silo::debug_boot_reg_active() {
2294        crate::serial_println!(
2295            "[trace][buddy] alloc enter order={} migratetype={:?} buddy_lock={:#x}",
2296            order,
2297            migratetype,
2298            &BUDDY_ALLOCATOR as *const _ as usize
2299        );
2300    }
2301    if order == 0 {
2302        alloc_order0_cached(migratetype)
2303    } else {
2304        let mut global = OnDemandGlobalLock::new();
2305        match global.alloc_with_migratetype(order, migratetype) {
2306            Ok(frame) => Ok(frame),
2307            Err(AllocError::OutOfMemory) => {
2308                let candidate = global
2309                    .with_allocator(|allocator, _token| {
2310                        allocator.compaction_candidate(
2311                            order,
2312                            migratetype,
2313                            BuddyAllocator::preferred_zone_order(migratetype),
2314                        )
2315                    })
2316                    .flatten();
2317
2318                if let Some(candidate) = candidate {
2319                    let budget = BuddyAllocator::compaction_drain_budget(candidate);
2320                    global.unlock();
2321                    let drained = drain_local_caches_for_zone(
2322                        budget,
2323                        candidate.zone_idx,
2324                        migratetype,
2325                        &mut global,
2326                    );
2327                    let retry = global.alloc_with_migratetype(order, migratetype);
2328                    record_compaction_attempt(candidate, drained, retry.is_ok());
2329                    if retry.is_ok() || drained != 0 {
2330                        return retry;
2331                    }
2332                } else {
2333                    global.unlock();
2334                }
2335
2336                let _ = drain_local_caches_to_global(usize::MAX, &mut global);
2337                global.alloc_with_migratetype(order, migratetype)
2338            }
2339            Err(e) => Err(e),
2340        }
2341    }
2342}
2343
2344/// Free frames with per-CPU caching on order-0 requests.
2345///
2346/// `_token` is a compile-time proof that interrupts are disabled on the calling CPU.
2347pub fn free(_token: &IrqDisabledToken, frame: PhysFrame, order: u8) {
2348    let migratetype = BuddyAllocator::block_migratetype(frame.start_address.as_u64());
2349    if order == 0 {
2350        free_order0_cached(frame, migratetype);
2351    } else {
2352        let mut global = OnDemandGlobalLock::new();
2353        global.free(frame, order);
2354    }
2355}
2356
2357impl FrameAllocator for BuddyAllocator {
2358    /// Performs the alloc operation.
2359    fn alloc(&mut self, order: u8, token: &IrqDisabledToken) -> Result<PhysFrame, AllocError> {
2360        self.alloc_locked_with_migratetype(order, Migratetype::Unmovable, token)
2361    }
2362
2363    /// Performs the free operation.
2364    fn free(&mut self, frame: PhysFrame, order: u8, token: &IrqDisabledToken) {
2365        let cpu_idx = crate::arch::percpu::current_cpu_index();
2366        if ALLOC_IN_PROGRESS[cpu_idx].swap(true, core::sync::atomic::Ordering::Acquire) {
2367            panic!("Recursive deallocation detected on CPU {}!", cpu_idx);
2368        }
2369
2370        let frame_phys = frame.start_address.as_u64();
2371        let mut zi = Self::zone_index_for_addr(frame_phys);
2372
2373        // Verify the address-selected zone actually contains this frame in its
2374        // segment geometry. If not, search all zones to find the correct one.
2375        // This handles edge cases where boot-allocator consumption of DMA-region
2376        // pages (frame metadata, bitmap pools) leaves physical addresses below
2377        // DMA_MAX that are outside any DMA segment.
2378        if Self::find_segment_index(&self.zones[zi], frame_phys, order).is_none() {
2379            let mut found = false;
2380            for candidate_zi in 0..ZoneType::COUNT {
2381                if candidate_zi == zi {
2382                    continue;
2383                }
2384                if Self::find_segment_index(&self.zones[candidate_zi], frame_phys, order).is_some()
2385                {
2386                    #[cfg(feature = "selftest")]
2387                    serial_println!(
2388                        "[buddy] WARN: frame 0x{:x} order {} belongs to zone[{}] {:?}, not zone[{}] {:?}; forwarding.",
2389                        frame_phys, order,
2390                        candidate_zi, self.zones[candidate_zi].zone_type,
2391                        zi, self.zones[zi].zone_type,
2392                    );
2393                    zi = candidate_zi;
2394                    found = true;
2395                    break;
2396                }
2397            }
2398            if !found {
2399                #[cfg(feature = "selftest")]
2400                {
2401                    serial_println!(
2402                        "[buddy] free: frame 0x{:x} order {} not in any zone segment; checking all zones...",
2403                        frame_phys, order,
2404                    );
2405                    for zzi in 0..ZoneType::COUNT {
2406                        let z = &self.zones[zzi];
2407                        serial_println!(
2408                            "  zone[{}] {:?}: segments={} page_count={}",
2409                            zzi,
2410                            z.zone_type,
2411                            z.segment_count,
2412                            z.page_count,
2413                        );
2414                        for si in 0..z.segment_count {
2415                            let seg = &z.segments()[si];
2416                            serial_println!(
2417                                "    segment[{}]: base=0x{:x} pages={} end=0x{:x}",
2418                                si,
2419                                seg.base.as_u64(),
2420                                seg.page_count,
2421                                seg.end_address(),
2422                            );
2423                        }
2424                    }
2425                    serial_println!(
2426                        "[buddy] CRITICAL: frame 0x{:x} order {} belongs to no zone segment!",
2427                        frame_phys,
2428                        order,
2429                    );
2430                }
2431            }
2432        }
2433
2434        let zone = &mut self.zones[zi];
2435        // NOTE: O(2^order) MetaSlot scan. Acceptable for large-order frees
2436        // (kernel stacks, vmalloc) which are rare; order-0 path is handled
2437        // separately in free_order0_cached with a single-page check.
2438        if crate::memory::frame::block_phys_has_poison_guard(frame_phys, order) {
2439            Self::quarantine_poisoned_block_in_zone(zone, frame, order, token);
2440        } else {
2441            Self::free_to_zone(zone, frame, order, token);
2442        }
2443
2444        ALLOC_IN_PROGRESS[cpu_idx].store(false, core::sync::atomic::Ordering::Release);
2445    }
2446}
2447
2448impl BuddyAllocator {
2449    /// Allocate explicitly from one zone (e.g. DMA-only callers).
2450    pub fn alloc_zone(
2451        &mut self,
2452        order: u8,
2453        zone: ZoneType,
2454        token: &IrqDisabledToken,
2455    ) -> Result<PhysFrame, AllocError> {
2456        self.alloc_zone_locked(order, zone, Migratetype::Unmovable, token)
2457    }
2458
2459    /// Allocate explicitly from one zone with a migratetype hint.
2460    ///
2461    /// This keeps the target zone fixed but still selects the preferred
2462    /// free-list class and fallback donor order from `migratetype`.
2463    pub fn alloc_zone_migratetype(
2464        &mut self,
2465        order: u8,
2466        zone: ZoneType,
2467        migratetype: Migratetype,
2468        token: &IrqDisabledToken,
2469    ) -> Result<PhysFrame, AllocError> {
2470        self.alloc_zone_locked(order, zone, migratetype, token)
2471    }
2472}
2473
2474/// Derived pressure state for a zone snapshot.
2475///
2476/// Thresholds are evaluated against the zone's effective free pages, including
2477/// pages parked in order-0 per-CPU caches.
2478#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2479pub enum ZonePressure {
2480    /// Free pages are above the high watermark.
2481    Healthy,
2482    /// Free pages dropped below the high watermark.
2483    High,
2484    /// Free pages dropped below the low watermark.
2485    Low,
2486    /// Free pages reached the minimum watermark plus reserve floor.
2487    Min,
2488}
2489
2490impl ZonePressure {
2491    const SNAPSHOT_COUNT: usize = 4;
2492
2493    #[inline]
2494    const fn as_snapshot(self) -> usize {
2495        match self {
2496            Self::Healthy => 0,
2497            Self::High => 1,
2498            Self::Low => 2,
2499            Self::Min => 3,
2500        }
2501    }
2502
2503    #[inline]
2504    const fn from_snapshot(value: usize) -> Option<Self> {
2505        match value {
2506            0 => Some(Self::Healthy),
2507            1 => Some(Self::High),
2508            2 => Some(Self::Low),
2509            3 => Some(Self::Min),
2510            _ => None,
2511        }
2512    }
2513}
2514
2515/// Snapshot statistics for a single memory zone.
2516///
2517/// The struct is plain data on purpose so low-level diagnostics and crash paths
2518/// can snapshot it onto the stack without heap allocation.
2519#[derive(Debug, Clone, Copy)]
2520pub struct ZoneStats {
2521    /// Zone classification.
2522    pub zone_type: ZoneType,
2523    /// Lowest physical address covered by the zone span.
2524    pub base: u64,
2525    /// Pages currently managed by buddy in this zone.
2526    pub managed_pages: usize,
2527    /// Pages reported as usable by the firmware map before reservations.
2528    pub present_pages: usize,
2529    /// Outer span in pages, including holes.
2530    pub spanned_pages: usize,
2531    /// Pages removed from management during bootstrap.
2532    pub reserved_pages: usize,
2533    /// Pages allocated to live callers.
2534    pub allocated_pages: usize,
2535    /// Order-0 pages currently parked in per-CPU caches.
2536    pub cached_pages: usize,
2537    /// Cached pages parked in unmovable per-CPU caches.
2538    pub cached_unmovable_pages: usize,
2539    /// Cached pages parked in movable per-CPU caches.
2540    pub cached_movable_pages: usize,
2541    /// Effective free pages, including cached pages.
2542    pub free_pages: usize,
2543    /// Free pages tracked in movable free lists.
2544    pub movable_free_pages: usize,
2545    /// Free pages tracked in unmovable free lists.
2546    pub unmovable_free_pages: usize,
2547    /// Number of populated contiguous segments.
2548    pub segment_count: usize,
2549    /// Reserved segment-table capacity.
2550    pub segment_capacity: usize,
2551    /// Total number of pageblocks tracked across all segments.
2552    pub pageblock_count: usize,
2553    /// Pageblocks currently tagged unmovable.
2554    pub unmovable_pageblocks: usize,
2555    /// Pageblocks currently tagged movable.
2556    pub movable_pageblocks: usize,
2557    /// Minimum watermark.
2558    pub watermark_min: usize,
2559    /// Low watermark.
2560    pub watermark_low: usize,
2561    /// High watermark.
2562    pub watermark_high: usize,
2563    /// Low-memory reserve kept for lower-priority paths.
2564    pub lowmem_reserve_pages: usize,
2565    /// Largest currently available free order.
2566    pub largest_free_order: Option<u8>,
2567}
2568
2569impl ZoneStats {
2570    /// Empty snapshot entry for stack-allocated arrays.
2571    pub const fn empty() -> Self {
2572        Self {
2573            zone_type: ZoneType::DMA,
2574            base: 0,
2575            managed_pages: 0,
2576            present_pages: 0,
2577            spanned_pages: 0,
2578            reserved_pages: 0,
2579            allocated_pages: 0,
2580            cached_pages: 0,
2581            cached_unmovable_pages: 0,
2582            cached_movable_pages: 0,
2583            free_pages: 0,
2584            movable_free_pages: 0,
2585            unmovable_free_pages: 0,
2586            segment_count: 0,
2587            segment_capacity: 0,
2588            pageblock_count: 0,
2589            unmovable_pageblocks: 0,
2590            movable_pageblocks: 0,
2591            watermark_min: 0,
2592            watermark_low: 0,
2593            watermark_high: 0,
2594            lowmem_reserve_pages: 0,
2595            largest_free_order: None,
2596        }
2597    }
2598
2599    /// Returns the number of hole pages inside the zone span.
2600    #[inline]
2601    pub fn hole_pages(&self) -> usize {
2602        self.spanned_pages.saturating_sub(self.managed_pages)
2603    }
2604
2605    /// Returns the effective reserve floor enforced by policy.
2606    #[inline]
2607    pub fn reserve_floor_pages(&self) -> usize {
2608        self.watermark_min.saturating_add(self.lowmem_reserve_pages)
2609    }
2610
2611    /// Returns the free pages remaining after the reserve floor is discounted.
2612    #[inline]
2613    pub fn available_after_reserve_pages(&self) -> usize {
2614        self.free_pages.saturating_sub(self.reserve_floor_pages())
2615    }
2616
2617    /// Returns the derived pressure state from the current zone watermarks.
2618    pub fn pressure(&self) -> ZonePressure {
2619        let reserve_floor = self.reserve_floor_pages();
2620        let low_floor = self.watermark_low.saturating_add(self.lowmem_reserve_pages);
2621        let high_floor = self
2622            .watermark_high
2623            .saturating_add(self.lowmem_reserve_pages);
2624
2625        if self.free_pages <= reserve_floor {
2626            ZonePressure::Min
2627        } else if self.free_pages <= low_floor {
2628            ZonePressure::Low
2629        } else if self.free_pages <= high_floor {
2630            ZonePressure::High
2631        } else {
2632            ZonePressure::Healthy
2633        }
2634    }
2635}
2636
2637/// Snapshot of the last fragmentation-driven compaction assist attempt.
2638///
2639/// The fields are intentionally plain data so crash dumps and shell commands
2640/// can read them without locking or heap allocation.
2641#[derive(Debug, Clone, Copy)]
2642pub struct CompactionStats {
2643    /// Number of targeted compaction assists attempted after an allocation miss.
2644    pub attempts: usize,
2645    /// Number of attempts that yielded a successful retry.
2646    pub successes: usize,
2647    /// Last requested buddy order that triggered a targeted drain.
2648    pub last_order: Option<u8>,
2649    /// Mobility class of the last assisted allocation.
2650    pub last_migratetype: Option<Migratetype>,
2651    /// Zone selected as the preferred compaction target.
2652    pub last_zone: Option<ZoneType>,
2653    /// Pressure state observed on that zone before draining caches.
2654    pub last_pressure: Option<ZonePressure>,
2655    /// Fragmentation score that justified the assist path.
2656    pub last_fragmentation_score: usize,
2657    /// Pages requested by the original allocation.
2658    pub last_requested_pages: usize,
2659    /// Effective free pages left above reserves in the chosen zone.
2660    pub last_available_pages: usize,
2661    /// Free pages already available at or above the requested order.
2662    pub last_usable_pages: usize,
2663    /// Order-0 pages parked in local caches for the chosen zone.
2664    pub last_cached_pages: usize,
2665    /// Pages actually drained from local caches during the last attempt.
2666    pub last_drained_pages: usize,
2667    /// Total pageblocks tracked in the selected zone.
2668    pub last_pageblock_count: usize,
2669    /// Pageblocks already tagged with the requested migratetype.
2670    pub last_matching_pageblocks: usize,
2671}
2672
2673impl CompactionStats {
2674    /// Empty snapshot used before any assisted drain happened.
2675    pub const fn empty() -> Self {
2676        Self {
2677            attempts: 0,
2678            successes: 0,
2679            last_order: None,
2680            last_migratetype: None,
2681            last_zone: None,
2682            last_pressure: None,
2683            last_fragmentation_score: 0,
2684            last_requested_pages: 0,
2685            last_available_pages: 0,
2686            last_usable_pages: 0,
2687            last_cached_pages: 0,
2688            last_drained_pages: 0,
2689            last_pageblock_count: 0,
2690            last_matching_pageblocks: 0,
2691        }
2692    }
2693}
2694
2695impl BuddyAllocator {
2696    /// Fast totals without heap allocation (safe in low-level paths).
2697    pub fn page_totals(&self) -> (usize, usize) {
2698        let mut total_pages = 0usize;
2699        let mut allocated_pages = 0usize;
2700        for zone in &self.zones {
2701            total_pages = total_pages.saturating_add(zone.page_count);
2702            allocated_pages = allocated_pages.saturating_add(zone.allocated);
2703        }
2704        let cached_pages = LOCAL_CACHED_FRAMES.load(AtomicOrdering::Relaxed);
2705        allocated_pages = allocated_pages.saturating_sub(cached_pages);
2706        (total_pages, allocated_pages)
2707    }
2708
2709    /// Get a reference to a zone by index.
2710    pub fn get_zone(&self, idx: usize) -> &Zone {
2711        &self.zones[idx]
2712    }
2713
2714    /// Snapshot zones without heap allocation.
2715    /// Returns the number of entries written to `out`.
2716    pub fn zone_snapshot(&self, out: &mut [ZoneStats]) -> usize {
2717        let n = core::cmp::min(out.len(), self.zones.len());
2718        for (i, zone) in self.zones.iter().take(n).enumerate() {
2719            let cached_unmovable = local_cached_zone_migratetype_count(i, Migratetype::Unmovable);
2720            let cached_movable = local_cached_zone_migratetype_count(i, Migratetype::Movable);
2721            let cached = cached_unmovable.saturating_add(cached_movable);
2722            let pageblocks = Self::zone_pageblock_counts(zone);
2723            let mut free_by_type = zone.free_pages_by_migratetype();
2724            free_by_type[Migratetype::Unmovable.index()] =
2725                free_by_type[Migratetype::Unmovable.index()].saturating_add(cached_unmovable);
2726            free_by_type[Migratetype::Movable.index()] =
2727                free_by_type[Migratetype::Movable.index()].saturating_add(cached_movable);
2728            out[i] = ZoneStats {
2729                zone_type: zone.zone_type,
2730                base: zone.base.as_u64(),
2731                managed_pages: zone.page_count,
2732                present_pages: zone.present_pages,
2733                spanned_pages: zone.span_pages,
2734                reserved_pages: zone.reserved_pages,
2735                allocated_pages: zone.allocated.saturating_sub(cached),
2736                cached_pages: cached,
2737                cached_unmovable_pages: cached_unmovable,
2738                cached_movable_pages: cached_movable,
2739                free_pages: Self::zone_effective_free_pages(zone, i),
2740                movable_free_pages: free_by_type[Migratetype::Movable.index()],
2741                unmovable_free_pages: free_by_type[Migratetype::Unmovable.index()],
2742                segment_count: zone.segment_count,
2743                segment_capacity: zone.segment_capacity,
2744                pageblock_count: pageblocks[Migratetype::Unmovable.index()]
2745                    .saturating_add(pageblocks[Migratetype::Movable.index()]),
2746                unmovable_pageblocks: pageblocks[Migratetype::Unmovable.index()],
2747                movable_pageblocks: pageblocks[Migratetype::Movable.index()],
2748                watermark_min: zone.watermark_min,
2749                watermark_low: zone.watermark_low,
2750                watermark_high: zone.watermark_high,
2751                lowmem_reserve_pages: zone.lowmem_reserve_pages,
2752                largest_free_order: zone.largest_free_order(),
2753            };
2754        }
2755        n
2756    }
2757}
2758
2759/// Set the compaction fragmentation threshold (percentage, 0-100).
2760///
2761/// Lower values make compaction more aggressive; higher values make it more
2762/// conservative. The default is 35.
2763pub fn set_compaction_threshold(threshold: usize) {
2764    COMPACTION_FRAGMENTATION_THRESHOLD.store(threshold.min(100), AtomicOrdering::Relaxed);
2765}
2766
2767/// Get the current compaction fragmentation threshold.
2768pub fn compaction_threshold() -> usize {
2769    COMPACTION_FRAGMENTATION_THRESHOLD.load(AtomicOrdering::Relaxed)
2770}