Skip to main content

strat9_kernel/arch/x86_64/
cpuid.rs

1//! CPU feature detection via CPUID instruction.
2//!
3//! Provides a `CpuInfo` struct populated at boot time with vendor, model,
4//! feature flags, and XSAVE geometry. All subsequent queries go through
5//! `host()` which returns the cached result.
6
7use crate::sync::SpinLock;
8use alloc::string::String;
9use bitflags::bitflags;
10use core::sync::atomic::{AtomicBool, AtomicU64, Ordering};
11
12bitflags! {
13    /// Logical internal bitmap.
14    ///
15    /// Bit positions do NOT match raw CPUID register positions: flags coming
16    /// from different leaves/sub-registers are reallocated into one `u64`
17    /// space (SMEP/SMAP are deliberately relocated high to avoid collisions).
18    /// Only compare through the named constants : never raw bit positions.
19    #[derive(Debug, Clone, Copy, PartialEq, Eq)]
20    pub struct CpuFeatures: u64 {
21        //  Leaf 0x01 ECX
22        const SSE3      = 1 << 0;
23        const SSSE3     = 1 << 1;
24        const FMA       = 1 << 2;
25        const SSE4_1    = 1 << 3;
26        const SSE4_2    = 1 << 4;
27        const POPCNT    = 1 << 5;
28        const AES_NI    = 1 << 6;
29        const XSAVE     = 1 << 7;
30        const OSXSAVE   = 1 << 8;
31        const AVX       = 1 << 12;
32        const F16C      = 1 << 9;
33        const VMX       = 1 << 10;
34        const X2APIC    = 1 << 11;
35        //  Leaf 0x01 EDX
36        const FPU       = 1 << 16;
37        const TSC       = 1 << 17;
38        const APIC      = 1 << 18;
39        const SSE       = 1 << 19;
40        const SSE2      = 1 << 20;
41        const FXSR      = 1 << 21;
42        //  Leaf 0x07 ECX
43        const SMEP      = 1 << 57; // CPUID(7).EBX bit 7 (relocated to avoid collision)
44        const SMAP      = 1 << 58; // CPUID(7).EBX bit 20 (relocated to avoid collision)
45        //  Leaf 0x07 EBX
46        const AVX2      = 1 << 32;
47        const AVX512F   = 1 << 33;
48        const AVX512BW  = 1 << 34;
49        const AVX512VL  = 1 << 35;
50        const SHA       = 1 << 36;
51        //  Leaf 0x80000001 EDX
52        const NX        = 1 << 48;
53        const PAGES_1G  = 1 << 49;
54        const RDTSCP    = 1 << 50;
55        const LONG_MODE = 1 << 51;
56        //  Leaf 0x80000001 ECX
57        const SVM       = 1 << 56;
58    }
59}
60
61/// XCR0 component bits.
62pub const XCR0_X87: u64 = 1 << 0;
63pub const XCR0_SSE: u64 = 1 << 1;
64pub const XCR0_AVX: u64 = 1 << 2;
65pub const XCR0_OPMASK: u64 = 1 << 5;
66pub const XCR0_ZMM_HI256: u64 = 1 << 6;
67pub const XCR0_HI16_ZMM: u64 = 1 << 7;
68
69#[derive(Debug, Clone, Copy, PartialEq, Eq)]
70pub enum CpuVendor {
71    Intel,
72    Amd,
73    Unknown,
74}
75
76/// Cached CPU identification and feature information.
77#[derive(Debug, Clone)]
78pub struct CpuInfo {
79    pub vendor: CpuVendor,
80    /// Raw CPUID(0) vendor string (EBX:EDX:ECX, 12 bytes, not NUL-terminated).
81    /// Preserves vendors classified as [`CpuVendor::Unknown`] by
82    /// [`Self::vendor_string`].
83    pub vendor_id: [u8; 12],
84    pub features: CpuFeatures,
85    /// Bitmap of extended state components the CPU supports
86    /// (CPUID.(0D,0):EDX:EAX). Capability bitmap, not a maximum value.
87    pub supported_xcr0: u64,
88    /// XSAVE area size (bytes) for the components enabled in XCR0 at
89    /// detect() time (CPUID.(0D,0):EBX). Depends on the OS-enabled state —
90    /// use [`xsave_size_for_xcr0`] for arbitrary masks.
91    pub xsave_size_current: usize,
92    /// XSAVE area size (bytes) required if every supported component were
93    /// enabled (CPUID.(0D,0):ECX). Upper bound for synthetic XCR0 masks.
94    pub xsave_size_max: usize,
95    pub family: u16,
96    pub model: u8,
97    pub stepping: u8,
98    pub model_name: [u8; 48],
99    model_name_len: usize,
100}
101
102impl CpuInfo {
103    /// Return the model name as a `&str`.
104    pub fn model_name_str(&self) -> &str {
105        let bytes = &self.model_name[..self.model_name_len];
106        core::str::from_utf8(bytes).unwrap_or("Unknown")
107    }
108
109    /// Return a vendor id string (e.g. "GenuineIntel").
110    pub fn vendor_string(&self) -> &'static str {
111        match self.vendor {
112            CpuVendor::Intel => "GenuineIntel",
113            CpuVendor::Amd => "AuthenticAMD",
114            CpuVendor::Unknown => "Unknown",
115        }
116    }
117}
118
119static HOST_CPU: SpinLock<Option<CpuInfo>> = SpinLock::new(None);
120static INITIALIZED: AtomicBool = AtomicBool::new(false);
121
122/// Lock-free cache of the host's default XCR0 mask, written once during
123/// `init()`.  Used by `normalized_xcr0()` in the context-switch hot path
124/// to avoid acquiring the `HOST_CPU` spinlock with interrupts disabled.
125pub(crate) static HOST_DEFAULT_XCR0_CACHE: AtomicU64 = AtomicU64::new(0);
126
127/// Cached extended-state save profile, frozen once at `init()`.
128///
129/// Precomputed so context-switch / task-creation code never walks CPUID
130/// leaf 0x0D sub-leaves at runtime (see `xsave_size_for_xcr0`, which is
131/// correct but too slow for hot paths).
132#[derive(Debug, Clone, Copy)]
133pub struct XsaveProfile {
134    /// XCR0 mask the kernel programs (x87|SSE|AVX|AVX-512 as available).
135    pub xcr0_mask: u64,
136    /// Save-area size in bytes for that mask, rounded UP to a 64-byte
137    /// multiple so allocators can over-allocate safely.
138    pub area_size: usize,
139    /// Alignment required by the save instruction: XSAVE/XRSTOR fault (#GP)
140    /// on non-64-byte operands; FXSAVE/FXRSTOR only require 16 bytes, but we
141    /// standardize on 64 everywhere (over-alignment is harmless).
142    pub align: usize,
143}
144
145static PROFILE_XCR0: AtomicU64 = AtomicU64::new(0);
146static PROFILE_AREA_SIZE: core::sync::atomic::AtomicUsize =
147    core::sync::atomic::AtomicUsize::new(512);
148
149/// Return the frozen XSAVE profile. Before `init()` (or without XSAVE),
150/// returns the conservative FXSAVE baseline (mask=x87|SSE, 512 bytes).
151pub fn boot_xsave_profile() -> XsaveProfile {
152    let mask = PROFILE_XCR0.load(Ordering::Acquire);
153    let size = PROFILE_AREA_SIZE.load(Ordering::Acquire);
154    XsaveProfile {
155        xcr0_mask: mask,
156        area_size: size,
157        align: 64,
158    }
159}
160
161/// Detect and cache CPU information. Must be called once at BSP boot.
162pub fn init() {
163    crate::e9_mark!(b'X');
164    let info = detect();
165    crate::e9_mark!(b'Y');
166    crate::serial_println!(
167        "[CPUID] {} {} (family={} model={} stepping={})",
168        info.vendor_string(),
169        info.model_name_str(),
170        info.family,
171        info.model,
172        info.stepping,
173    );
174    crate::serial_println!(
175        "[CPUID] features={:?}, supported_xcr0={:#x}, xsave_cur={} xsave_max={}",
176        info.features,
177        info.supported_xcr0,
178        info.xsave_size_current,
179        info.xsave_size_max,
180    );
181    // Compute the default XCR0 *before* publishing `info`, so no reader can
182    // observe INITIALIZED=true with a stale/zero XCR0 cache (the previous
183    // order relied on the locked fallback to paper over the window).
184    let default_xcr0 = default_xcr0_for(&info);
185    // Freeze the boot-time save profile: area size for the kernel's chosen
186    // mask, rounded up to the 64-byte instruction-alignment granularity.
187    //
188    // NOTE: compute the size from the *local* `info`, not via the global
189    // cache (`xsave_size_for_xcr0`/`host_uses_xsave`). Those consult
190    // `INITIALIZED`, which is only published below : so during `init()` they
191    // always fall back to 512 bytes even when AVX/AVX-512 is present, leaving
192    // the XSAVE area too small and corrupting state on the first AVX context
193    // switch.
194    let profile_size = xsave_size_for_info(&info, default_xcr0).div_ceil(64) * 64;
195    PROFILE_XCR0.store(default_xcr0, Ordering::Release);
196    PROFILE_AREA_SIZE.store(profile_size, Ordering::Release);
197
198    crate::e9_mark!(b'J');
199    *HOST_CPU.lock() = Some(info);
200    crate::e9_mark!(b'K');
201    HOST_DEFAULT_XCR0_CACHE.store(default_xcr0, Ordering::Release);
202    INITIALIZED.store(true, Ordering::Release);
203    crate::e9_mark!(b'i');
204}
205
206/// Return a clone of the cached host CPU info. Panics if `init()` not called.
207pub fn host() -> CpuInfo {
208    HOST_CPU
209        .lock()
210        .clone()
211        .expect("cpuid::init() not called yet")
212}
213
214/// Whether XSAVE is supported by the host.
215pub fn host_uses_xsave() -> bool {
216    INITIALIZED.load(Ordering::Acquire)
217        && HOST_CPU
218            .lock()
219            .as_ref()
220            .map_or(false, |h| h.features.contains(CpuFeatures::XSAVE))
221}
222
223/// Detect CPU features by interrogating CPUID leaves.
224fn detect() -> CpuInfo {
225    let cpuid = super::cpuid;
226
227    crate::e9_mark!(b'D');
228    //  Vendor (leaf 0): keep the raw 12-byte id, then classify.
229    let (max_leaf, ebx0, ecx0, edx0) = cpuid(0, 0);
230    crate::e9_mark!(b'E');
231    crate::e9_mark!(b'a');
232    let mut vendor_id = [0u8; 12];
233    crate::e9_mark!(b'b');
234    let b = ebx0.to_le_bytes();
235    vendor_id[0] = b[0];
236    vendor_id[1] = b[1];
237    vendor_id[2] = b[2];
238    vendor_id[3] = b[3];
239    crate::e9_mark!(b'c');
240    let d = edx0.to_le_bytes();
241    vendor_id[4] = d[0];
242    vendor_id[5] = d[1];
243    vendor_id[6] = d[2];
244    vendor_id[7] = d[3];
245    crate::e9_mark!(b'd');
246    let c = ecx0.to_le_bytes();
247    vendor_id[8] = c[0];
248    vendor_id[9] = c[1];
249    vendor_id[10] = c[2];
250    vendor_id[11] = c[3];
251    crate::e9_mark!(b'e');
252    let vendor = match (ebx0, edx0, ecx0) {
253        (0x756E_6547, 0x4965_6E69, 0x6C65_746E) => CpuVendor::Intel,
254        (0x6874_7541, 0x6974_6E65, 0x444D_4163) => CpuVendor::Amd,
255        _ => CpuVendor::Unknown,
256    };
257    crate::e9_mark!(b'f');
258
259    let mut features = CpuFeatures::empty();
260
261    //  Leaf 0x01: main feature bits
262    crate::e9_mark!(b'1');
263    let (eax1, _ebx1, ecx1, edx1) = if max_leaf >= 1 {
264        cpuid(1, 0)
265    } else {
266        (0, 0, 0, 0)
267    };
268    crate::e9_mark!(b'2');
269    crate::e9_mark!(b'd');
270    crate::e9_mark!(b'f');
271    crate::e9_mark!(b'e');
272
273    let stepping = (eax1 & 0xF) as u8;
274    let base_family = (eax1 >> 8) & 0xF;
275    let base_model = (eax1 >> 4) & 0xF;
276    let ext_model = (eax1 >> 16) & 0xF;
277    let ext_family = (eax1 >> 20) & 0xFF;
278    let mut family_full: u16 = base_family as u16;
279    let mut model: u8 = base_model as u8;
280    if base_family == 6 || base_family == 15 {
281        model |= (ext_model << 4) as u8;
282    }
283    if base_family == 15 {
284        family_full += ext_family as u16;
285    }
286    let family = family_full;
287
288    if ecx1 & (1 << 0) != 0 {
289        features |= CpuFeatures::SSE3;
290    }
291    if ecx1 & (1 << 9) != 0 {
292        features |= CpuFeatures::SSSE3;
293    }
294    if ecx1 & (1 << 12) != 0 {
295        features |= CpuFeatures::FMA;
296    }
297    if ecx1 & (1 << 19) != 0 {
298        features |= CpuFeatures::SSE4_1;
299    }
300    if ecx1 & (1 << 20) != 0 {
301        features |= CpuFeatures::SSE4_2;
302    }
303    if ecx1 & (1 << 23) != 0 {
304        features |= CpuFeatures::POPCNT;
305    }
306    if ecx1 & (1 << 25) != 0 {
307        features |= CpuFeatures::AES_NI;
308    }
309    if ecx1 & (1 << 26) != 0 {
310        features |= CpuFeatures::XSAVE;
311    }
312    if ecx1 & (1 << 27) != 0 {
313        features |= CpuFeatures::OSXSAVE;
314    }
315    if ecx1 & (1 << 28) != 0 {
316        features |= CpuFeatures::AVX;
317    }
318    if ecx1 & (1 << 21) != 0 {
319        features |= CpuFeatures::X2APIC;
320    }
321    if ecx1 & (1 << 29) != 0 {
322        features |= CpuFeatures::F16C;
323    }
324    if ecx1 & (1 << 5) != 0 {
325        features |= CpuFeatures::VMX;
326    }
327
328    if edx1 & (1 << 0) != 0 {
329        features |= CpuFeatures::FPU;
330    }
331    if edx1 & (1 << 4) != 0 {
332        features |= CpuFeatures::TSC;
333    }
334    if edx1 & (1 << 9) != 0 {
335        features |= CpuFeatures::APIC;
336    }
337    if edx1 & (1 << 24) != 0 {
338        features |= CpuFeatures::FXSR;
339    }
340    if edx1 & (1 << 25) != 0 {
341        features |= CpuFeatures::SSE;
342    }
343    if edx1 & (1 << 26) != 0 {
344        features |= CpuFeatures::SSE2;
345    }
346
347    //  Leaf 0x07: extended features
348    crate::e9_mark!(b'3');
349    if max_leaf >= 7 {
350        let (_eax7, ebx7, _ecx7, _edx7) = cpuid(7, 0);
351        crate::e9_mark!(b'5');
352        if ebx7 & (1 << 5) != 0 {
353            features |= CpuFeatures::AVX2;
354        }
355        if ebx7 & (1 << 16) != 0 {
356            features |= CpuFeatures::AVX512F;
357        }
358        if ebx7 & (1 << 29) != 0 {
359            features |= CpuFeatures::SHA;
360        }
361        if ebx7 & (1 << 30) != 0 {
362            features |= CpuFeatures::AVX512BW;
363        }
364        if ebx7 & (1 << 31) != 0 {
365            features |= CpuFeatures::AVX512VL;
366        }
367        // SMEP (EBX bit 7): Supervisor Mode Execution Prevention
368        if ebx7 & (1 << 7) != 0 {
369            features |= CpuFeatures::SMEP;
370        }
371        // SMAP (EBX bit 20): Supervisor Mode Access Prevention
372        if ebx7 & (1 << 20) != 0 {
373            features |= CpuFeatures::SMAP;
374        }
375    }
376
377    //  Leaf 0x0D: XSAVE geometry
378    //
379    // CPUID.(0D,0):EDX:EAX = bitmap of components supported by the CPU
380    // (XCR0 candidates). EBX = size of the save area for the components
381    // currently enabled in XCR0 (OS-chosen, not a constant). ECX = size if
382    // every supported component were enabled : the correct upper bound for
383    // synthetic masks. Supervisor states (managed via IA32_XSS, not XCR0)
384    // are filtered out so `supported_xcr0` stays a true XCR0 bitmap.
385    crate::e9_mark!(b'F');
386    let mut supported_xcr0 = XCR0_X87 | XCR0_SSE;
387    let mut xsave_size_current = 512usize;
388    let mut xsave_size_max = 512usize;
389    crate::e9_mark!(b'G');
390    let xsave_check = features.contains(CpuFeatures::XSAVE);
391    crate::e9_mark!(b'g');
392    let leaf0d_check = max_leaf >= 0x0D;
393    crate::e9_mark!(b'h');
394    if xsave_check && leaf0d_check {
395        crate::e9_mark!(b'H');
396        let (eax_d, ebx_d, ecx_d, edx_d) = cpuid(0x0D, 0);
397        crate::e9_mark!(b'I');
398        let raw = ((edx_d as u64) << 32) | eax_d as u64;
399        supported_xcr0 = raw;
400        xsave_size_current = ebx_d as usize;
401        xsave_size_max = ecx_d as usize;
402    } else {
403        crate::e9_mark!(b'H');
404    }
405
406    //  Leaf 0x80000001: extended features (AMD-V, NX, 1G pages)
407    crate::e9_mark!(b'J');
408    let (max_ext, _, _, _) = cpuid(0x8000_0000, 0);
409    crate::e9_mark!(b'K');
410    if max_ext >= 0x8000_0001 {
411        let (_eax_e, _ebx_e, ecx_e, edx_e) = cpuid(0x8000_0001, 0);
412        if edx_e & (1 << 20) != 0 {
413            features |= CpuFeatures::NX;
414        }
415        if edx_e & (1 << 26) != 0 {
416            features |= CpuFeatures::PAGES_1G;
417        }
418        if edx_e & (1 << 27) != 0 {
419            features |= CpuFeatures::RDTSCP;
420        }
421        if edx_e & (1 << 29) != 0 {
422            features |= CpuFeatures::LONG_MODE;
423        }
424        if ecx_e & (1 << 2) != 0 {
425            features |= CpuFeatures::SVM;
426        }
427    }
428    crate::e9_mark!(b'L');
429
430    //  Leaves 0x80000002-0x80000004: brand string
431    let mut model_name = [0u8; 48];
432    let mut model_name_len = 0usize;
433    if max_ext >= 0x8000_0004 {
434        for (i, leaf) in (0x8000_0002u32..=0x8000_0004).enumerate() {
435            let (a, b, c, d) = cpuid(leaf, 0);
436            let offset = i * 16;
437            let ab = a.to_le_bytes();
438            model_name[offset] = ab[0];
439            model_name[offset + 1] = ab[1];
440            model_name[offset + 2] = ab[2];
441            model_name[offset + 3] = ab[3];
442            let bb = b.to_le_bytes();
443            model_name[offset + 4] = bb[0];
444            model_name[offset + 5] = bb[1];
445            model_name[offset + 6] = bb[2];
446            model_name[offset + 7] = bb[3];
447            let cb = c.to_le_bytes();
448            model_name[offset + 8] = cb[0];
449            model_name[offset + 9] = cb[1];
450            model_name[offset + 10] = cb[2];
451            model_name[offset + 11] = cb[3];
452            let db = d.to_le_bytes();
453            model_name[offset + 12] = db[0];
454            model_name[offset + 13] = db[1];
455            model_name[offset + 14] = db[2];
456            model_name[offset + 15] = db[3];
457        }
458        model_name_len = model_name
459            .iter()
460            .rposition(|&b| b != 0 && b != b' ')
461            .map_or(0, |p| p + 1);
462    }
463
464    CpuInfo {
465        vendor,
466        vendor_id,
467        features,
468        supported_xcr0,
469        xsave_size_current,
470        xsave_size_max,
471        family,
472        model,
473        stepping,
474        model_name,
475        model_name_len,
476    }
477}
478
479/// Pure helper: the XCR0 mask this kernel would enable for `info`
480/// (x87 + SSE always, plus AVX / AVX-512 states when the CPU announces
481/// them AND the hardware supports the required XCR0 bits).
482/// Shared by `init()` (pre-lock computation) and `host()`.
483fn default_xcr0_for(info: &CpuInfo) -> u64 {
484    const SSE_BASE: u64 = XCR0_X87 | XCR0_SSE;
485    const AVX_STATE: u64 = SSE_BASE | XCR0_AVX;
486    const AVX512_STATE: u64 = AVX_STATE | XCR0_OPMASK | XCR0_ZMM_HI256 | XCR0_HI16_ZMM;
487
488    if !info.features.contains(CpuFeatures::XSAVE) {
489        return SSE_BASE;
490    }
491
492    let available = info.supported_xcr0;
493    let mut wanted = SSE_BASE;
494
495    if info.features.contains(CpuFeatures::AVX) && (available & AVX_STATE) == AVX_STATE {
496        wanted = AVX_STATE;
497    }
498
499    if info.features.contains(CpuFeatures::AVX512F) && (available & AVX512_STATE) == AVX512_STATE {
500        wanted = AVX512_STATE;
501    }
502
503    wanted
504}
505
506impl CpuInfo {
507    /// Whether AVX may actually be executed: the hardware announces it AND
508    /// the kernel enabled the required states (CR4.OSXSAVE via XSAVE +
509    /// XCR0 bits 0|1|2). Never gate code paths on `features.contains(AVX)`
510    /// alone : that only reflects CPUID, not what the OS programmed.
511    pub fn avx_usable(&self) -> bool {
512        const REQUIRED: u64 = XCR0_X87 | XCR0_SSE | XCR0_AVX;
513        self.features
514            .contains(CpuFeatures::AVX | CpuFeatures::XSAVE | CpuFeatures::OSXSAVE)
515            && (self.supported_xcr0 & REQUIRED) == REQUIRED
516            && crate::arch::x86_64::cpuid_osxsave_enabled()
517            && host_default_xcr0() & XCR0_AVX != 0
518    }
519
520    /// Whether AVX-512 may actually be executed: hardware support plus all
521    /// required XCR0 states enabled (opmask, ZMM_Hi256, HI16_ZMM on top of
522    /// x87/SSE/AVX). See `avx_usable`.
523    pub fn avx512_usable(&self) -> bool {
524        const REQUIRED: u64 =
525            XCR0_X87 | XCR0_SSE | XCR0_AVX | XCR0_OPMASK | XCR0_ZMM_HI256 | XCR0_HI16_ZMM;
526        self.features.contains(
527            CpuFeatures::AVX512F | CpuFeatures::AVX | CpuFeatures::XSAVE | CpuFeatures::OSXSAVE,
528        ) && (self.supported_xcr0 & REQUIRED) == REQUIRED
529            && crate::arch::x86_64::cpuid_osxsave_enabled()
530            && host_default_xcr0() & REQUIRED == REQUIRED
531    }
532}
533
534/// Compute the XCR0 mask for a given set of allowed features,
535/// clamped to what the host actually supports.
536///
537/// NOTE: currently the host default mask is returned regardless of
538/// `features` (the kernel enables x87+SSE+AVX(+512) as one global profile).
539/// Kept as an API seam for per-feature XCR0 profiles.
540pub fn xcr0_for_features(_features: CpuFeatures) -> u64 {
541    let h = host();
542    default_xcr0_for(&h)
543}
544
545/// Compute the XSAVE area size needed for a given XCR0 mask directly from a
546/// `CpuInfo`, without consulting the global cache (`host_uses_xsave`/`host`).
547/// This is what `init()` must use, because the cache is not yet published when
548/// `init()` runs : otherwise AVX/AVX-512 would be given a 512-byte area.
549fn xsave_size_for_info(info: &CpuInfo, xcr0: u64) -> usize {
550    if !info.features.contains(CpuFeatures::XSAVE) {
551        return 512;
552    }
553    // Clamp to what the CPU actually supports; ignore unknown bits.
554    let xcr0 = xcr0 & info.supported_xcr0;
555    if xcr0 == info.supported_xcr0 {
556        // Fast path: requesting all components : use the ECX upper-bound.
557        return info.xsave_size_max.max(576);
558    }
559    let mut size = 576usize; // legacy area (512) + xsave header (64)
560    for comp in 2..64 {
561        if xcr0 & (1u64 << comp) == 0 {
562            continue;
563        }
564        let (eax, ebx, ecx, _edx) = super::cpuid(0x0D, comp);
565        // ECX bit 0: component managed via XCR0 (0) or IA32_XSS (1).
566        // Skip supervisor states : not saved by a user-space XCR0 mask.
567        if ecx & 1 != 0 {
568            continue;
569        }
570        let comp_size = eax as usize;
571        let comp_offset = ebx as usize;
572        if comp_size != 0 {
573            size = size.max(comp_offset.saturating_add(comp_size));
574        }
575    }
576    size.min(info.xsave_size_max).max(576)
577}
578
579/// Compute the XSAVE area size needed for a given XCR0 mask.
580/// Falls back to 512 (FXSAVE) if XSAVE is not supported.
581pub fn xsave_size_for_xcr0(xcr0: u64) -> usize {
582    if !host_uses_xsave() {
583        return 512;
584    }
585    let h = host();
586    if xcr0 == h.supported_xcr0 {
587        return h.xsave_size_max;
588    }
589
590    // Enumerate each enabled XCR0 component via CPUID leaf 0xD sub-leaves.
591    // Sub-leaf n returns offset (EBX) and size (EAX) for component n.
592    // The total save area is max(offset + size) across all enabled components.
593    // NOTE: this walks CPUID per component : do not call from context-switch
594    // or task-creation hot paths; precompute profiles instead.
595    // The ceiling is xsave_size_max (CPUID.0D.0:ECX), valid for any mask;
596    // EBX would only be valid for the XCR0 currently programmed.
597    let mut size = 576usize; // legacy area (512) + xsave header (64)
598    for comp in 2..64 {
599        if xcr0 & (1u64 << comp) == 0 {
600            continue;
601        }
602        let (eax, ebx, ecx, _edx) = super::cpuid(0x0D, comp);
603        // ECX bit 0: component managed via XCR0 (0) or IA32_XSS (1).
604        // Skip supervisor states : they are not saved by XSAVE/XRSTOR
605        // with a user-space XCR0 mask.
606        if ecx & 1 != 0 {
607            continue;
608        }
609        let comp_size = eax as usize;
610        let comp_offset = ebx as usize;
611        if comp_size != 0 {
612            size = size.max(comp_offset.saturating_add(comp_size));
613        }
614    }
615    size.min(h.xsave_size_max).max(576)
616}
617
618/// Return the host's default XCR0 mask from the lock-free cache, falling back
619/// to the locked query before `init()` is complete.
620#[inline]
621pub fn host_default_xcr0_fast() -> u64 {
622    let cached = HOST_DEFAULT_XCR0_CACHE.load(Ordering::Acquire);
623    if cached != 0 {
624        return cached;
625    }
626    host_default_xcr0()
627}
628
629/// Return the host's default XCR0 mask (all supported features).
630/// Safe to call before `init()` : returns `XCR0_X87 | XCR0_SSE` if not yet initialized.
631///
632/// Try to read the TSC frequency from CPUID leaf 0x15 (Time Stamp Counter and
633/// Core Crystal Clock Information).  Returns kHz, or None if not available.
634///
635/// This is the preferred method on Intel/AMD CPUs that support invariant TSC.
636/// Linux uses this as its primary TSC calibration source.
637pub fn tsc_frequency_khz() -> Option<u64> {
638    let max_leaf = super::cpuid(0, 0).0;
639    if max_leaf < 0x15 {
640        //  TSC frequency from CPUID leaf 0x15
641        return None;
642    }
643    let (eax, ebx, ecx, _edx) = super::cpuid(0x15, 0);
644    let core_crystal_hz = ecx as u64;
645
646    // If the core crystal clock is known, compute TSC frequency.
647    if core_crystal_hz != 0 {
648        let denom = eax as u64;
649        let num = ebx as u64;
650        if denom != 0 {
651            // TSC freq = core_crystal_hz * num / denom / 1_000
652            // Use checked arithmetic to avoid silent overflow.
653            return core_crystal_hz
654                .checked_mul(num)?
655                .checked_div(denom)?
656                .checked_div(1_000);
657        }
658    }
659    None
660}
661
662pub fn host_default_xcr0() -> u64 {
663    if INITIALIZED.load(Ordering::Acquire) {
664        HOST_CPU
665            .lock()
666            .as_ref()
667            .map_or(XCR0_X87 | XCR0_SSE, default_xcr0_for)
668    } else {
669        XCR0_X87 | XCR0_SSE
670    }
671}
672
673/// Build a Linux-style `flags` string from CPU features.
674pub fn features_to_flags_string(f: CpuFeatures) -> String {
675    let mut flags = String::new();
676    let table: &[(CpuFeatures, &str)] = &[
677        (CpuFeatures::FPU, "fpu"),
678        (CpuFeatures::TSC, "tsc"),
679        (CpuFeatures::APIC, "apic"),
680        (CpuFeatures::FXSR, "fxsr"),
681        (CpuFeatures::SSE, "sse"),
682        (CpuFeatures::SSE2, "sse2"),
683        (CpuFeatures::SSE3, "sse3"),
684        (CpuFeatures::SSSE3, "ssse3"),
685        (CpuFeatures::SSE4_1, "sse4_1"),
686        (CpuFeatures::SSE4_2, "sse4_2"),
687        (CpuFeatures::POPCNT, "popcnt"),
688        (CpuFeatures::AES_NI, "aes"),
689        (CpuFeatures::XSAVE, "xsave"),
690        (CpuFeatures::OSXSAVE, "osxsave"),
691        (CpuFeatures::AVX, "avx"),
692        (CpuFeatures::F16C, "f16c"),
693        (CpuFeatures::FMA, "fma"),
694        (CpuFeatures::AVX2, "avx2"),
695        (CpuFeatures::AVX512F, "avx512f"),
696        (CpuFeatures::AVX512BW, "avx512bw"),
697        (CpuFeatures::AVX512VL, "avx512vl"),
698        (CpuFeatures::SHA, "sha_ni"),
699        (CpuFeatures::X2APIC, "x2apic"),
700        (CpuFeatures::NX, "nx"),
701        (CpuFeatures::PAGES_1G, "pdpe1gb"),
702        (CpuFeatures::RDTSCP, "rdtscp"),
703        (CpuFeatures::LONG_MODE, "lm"),
704        (CpuFeatures::VMX, "vmx"),
705        (CpuFeatures::SVM, "svm"),
706        (CpuFeatures::SMEP, "smep"),
707        (CpuFeatures::SMAP, "smap"),
708    ];
709    for &(feat, name) in table {
710        if f.contains(feat) {
711            if !flags.is_empty() {
712                flags.push(' ');
713            }
714            flags.push_str(name);
715        }
716    }
717    flags
718}