Skip to main content

strat9_kernel/process/
elf.rs

1//! ELF64 loader for Strat9-OS.
2//!
3//! Parses ELF64 headers and loads PT_LOAD segments into a user address space,
4//! then creates a kernel task that trampolines into Ring 3 via IRETQ.
5//!
6//! Supports :
7//!   - ET_EXEC
8//!   - ET_DYN (PIE/static-PIE)
9//!   - ELF64 little-endian x86_64 binaries.
10//!
11//!
12//! Does not support (or need future fix) :
13//!
14//!   - ~~Fix TODO : allocation heap during ELF loading~~
15//!     PARTIALLY DONE (#67) : program-header vectors and boot-stack argv
16//!     pointers now use `try_reserve_exact`-based fallible collection and
17//!     fail the load with an error instead of aborting. Full rollback of
18//!     already-mapped segments still requires a fallible GlobalAlloc.
19//!
20//!   - ~~Fix TODO : find_free_vma_range : fallback hardcoded 0x1000_0000~~
21//!     DONE : the fallback was removed; a failed search now fails the load
22//!     with "No virtual range for ET_DYN image".
23//!
24//! Security:
25//!   - User stack has a guard page (user_stack_base() - 4096) that is intentionally
26//!     left unmapped.  Stack underflows hit it and page-fault.
27//!
28use crate::{
29    arch::xshim::{Size4KiB, VirtAddr},
30    x86_crate_shim::structures::paging::{Mapper, Page},
31};
32use alloc::{sync::Arc, vec::Vec};
33
34use crate::{
35    capability::Capability,
36    memory::address_space::{AddressSpace, VmaFlags, VmaPageSize, VmaType},
37    process::{
38        task::{CpuContext, KernelStack, ResumeKind, SyncUnsafeCell, Task},
39        TaskId, TaskPriority, TaskState,
40    },
41};
42
43macro_rules! elf_trace {
44    ($($arg:tt)*) => {
45        #[cfg(debug_assertions)] {
46            crate::e9_println!($($arg)*);
47            crate::serial_println!($($arg)*);
48        }
49    };
50}
51
52// ---------------------------------------------------------------------------
53// ELF64 constants (relocation & dynamic tags : not covered by xmas-elf)
54// ---------------------------------------------------------------------------
55
56const ET_EXEC: u16 = 2;
57const ET_DYN: u16 = 3;
58const PT_LOAD: u32 = 1;
59const PT_DYNAMIC: u32 = 2;
60const PT_INTERP: u32 = 3;
61const PT_TLS: u32 = 7;
62const PT_GNU_STACK: u32 = 0x6474_e551;
63const PT_GNU_RELRO: u32 = 0x6474_e552;
64const PF_X: u32 = 1;
65const PF_W: u32 = 2;
66const PF_R: u32 = 4;
67const DT_NULL: i64 = 0;
68const DT_RELA: i64 = 7;
69const DT_RELASZ: i64 = 8;
70const DT_RELAENT: i64 = 9;
71const DT_STRTAB: i64 = 5;
72const DT_SYMTAB: i64 = 6;
73const DT_SYMENT: i64 = 11;
74const DT_JMPREL: i64 = 23;
75const DT_PLTRELSZ: i64 = 2;
76const DT_PLTREL: i64 = 20;
77const DT_RELACOUNT: i64 = 0x6fff_fff9;
78const DT_RELR: i64 = 36;
79const DT_RELRSZ: i64 = 35;
80const DT_RELRENT: i64 = 37;
81const R_X86_64_RELATIVE: u32 = 8;
82const R_X86_64_64: u32 = 1;
83const R_X86_64_COPY: u32 = 5;
84const R_X86_64_GLOB_DAT: u32 = 6;
85const R_X86_64_JUMP_SLOT: u32 = 7;
86const R_X86_64_TPOFF64: u32 = 18;
87const R_X86_64_DTPMOD64: u32 = 16;
88const R_X86_64_DTPOFF64: u32 = 17;
89const R_X86_64_IRELATIVE: u32 = 37;
90
91/// Maximum virtual address we accept for user-space mappings.
92///
93/// Strat9 userspace components are statically linked (no-pie) at ET_EXEC
94/// 0xFFFFFFFF80000000 : the higher-half window. Each user AddressSpace owns
95/// a private copy of the PML4[511] PDP (see address_space::new_user), with
96/// the kernel-image slot removed, so processes can use the full canonical
97/// higher-half range without touching kernel pages. The guard therefore only
98/// rejects non-canonical addresses.
99pub const USER_ADDR_MAX: u64 = crate::memory::userslice::USER_SPACE_END;
100
101/// Number of 4 KiB pages for the user stack (16 pages = 64 KiB).
102///
103/// This is the *default*: the loader accepts a per-process stack size via
104/// [`load_and_run_elf_with_stack`] (issue #64).
105pub const USER_STACK_PAGES: usize = 16;
106/// Lower bound on a per-process user stack (4 pages = 16 KiB): the boot
107/// stack layout alone (argv/envp/auxv + guard margins) needs at least one
108/// page, and tiny stacks would fault immediately.
109pub const USER_STACK_MIN_PAGES: usize = 4;
110/// Upper bound on a per-process user stack (2048 pages = 8 MiB), to keep a
111/// buggy or hostile caller from exhausting the user address space / frames.
112pub const USER_STACK_MAX_PAGES: usize = 2048;
113/// Standard user-mode RFLAGS: IF=1, reserved bit 1 set.
114const USER_RFLAGS: u64 = 0x202;
115
116/// Get the randomized user stack base address.
117fn user_stack_base() -> u64 {
118    crate::kaslr::stack_base()
119}
120
121/// Get the randomized user stack top address.
122fn user_stack_top() -> u64 {
123    crate::kaslr::stack_top()
124}
125
126/// Get the guard page address below the user stack.
127fn user_stack_guard() -> u64 {
128    crate::kaslr::stack_guard()
129}
130
131/// Get the randomized PIE base address for ELF loading.
132fn pie_base() -> u64 {
133    crate::kaslr::pie_base()
134}
135
136/// Result of loading an ELF image into an address space.
137#[derive(Debug, Clone, Copy)]
138pub struct LoadedElfInfo {
139    pub runtime_entry: u64,
140    pub program_entry: u64,
141    pub phdr_vaddr: u64,
142    pub phent: u16,
143    pub phnum: u16,
144    pub interp_base: Option<u64>,
145    pub tls_vaddr: u64,
146    pub tls_filesz: u64,
147    pub tls_memsz: u64,
148    pub tls_align: u64,
149    pub stack_exec: bool,
150}
151
152// ---------------------------------------------------------------------------
153// ELF64 structures for kernel-internal use
154// ---------------------------------------------------------------------------
155
156/// Parsed ELF64 file header (copy-friendly, no borrows).
157#[derive(Debug, Clone, Copy)]
158struct Elf64Header {
159    e_type: u16,
160    e_entry: u64,
161    e_phoff: u64,
162    e_phentsize: u16,
163    e_phnum: u16,
164}
165
166/// Parsed ELF64 program header (copy-friendly, packed for raw byte reading).
167#[repr(C, packed)]
168#[derive(Debug, Clone, Copy)]
169struct Elf64Phdr {
170    p_type: u32,
171    p_flags: u32,
172    p_offset: u64,
173    p_vaddr: u64,
174    p_paddr: u64,
175    p_filesz: u64,
176    p_memsz: u64,
177    p_align: u64,
178}
179
180#[repr(C, packed)]
181#[derive(Debug, Clone, Copy)]
182struct Elf64Dyn {
183    d_tag: i64,
184    d_val: u64,
185}
186
187#[repr(C, packed)]
188#[derive(Debug, Clone, Copy)]
189struct Elf64Rela {
190    r_offset: u64,
191    r_info: u64,
192    r_addend: i64,
193}
194
195#[repr(C, packed)]
196#[derive(Debug, Clone, Copy)]
197struct Elf64Sym {
198    st_name: u32,
199    st_info: u8,
200    st_other: u8,
201    st_shndx: u16,
202    st_value: u64,
203    st_size: u64,
204}
205
206// ---------------------------------------------------------------------------
207// Parsing (uses xmas-elf for header validation)
208// ---------------------------------------------------------------------------
209
210/// Parse and validate the ELF64 file header from raw bytes.
211///
212/// Uses `xmas-elf` for magic/class/machine/version validation, then copies
213/// the fields we need into a local `Copy` struct.
214fn parse_header(data: &[u8]) -> Result<Elf64Header, &'static str> {
215    let elf = xmas_elf::ElfFile::new(data).map_err(|e| {
216        log::error!("[elf] xmas_elf::ElfFile::new failed: {:?}", e);
217        "Invalid ELF header"
218    })?;
219
220    let hdr = elf.header.pt2;
221
222    // Reject non-x86_64 binaries early.
223    let machine = hdr.machine().as_machine();
224    if machine != xmas_elf::header::Machine::X86_64 {
225        log::error!(
226            "[elf] Rejecting binary: machine={:?} (expected X86_64)",
227            machine
228        );
229        return Err("Not an x86_64 ELF binary");
230    }
231
232    // Type: executable or shared object (PIE/static PIE)
233    let e_type = hdr.type_().0;
234    if e_type != ET_EXEC && e_type != ET_DYN {
235        log::error!(
236            "[elf] Rejecting binary: e_type={} (expected ET_EXEC={} or ET_DYN={})",
237            e_type,
238            ET_EXEC,
239            ET_DYN
240        );
241        return Err("Unsupported ELF type (expected ET_EXEC or ET_DYN)");
242    }
243
244    let e_entry = hdr.entry_point();
245    // Entry point must be canonical user space (for ET_DYN this is relative and
246    // validated again after relocation). ET_EXEC with e_entry=0 is handled later.
247    if e_entry >= USER_ADDR_MAX {
248        return Err("Entry point outside user address range");
249    }
250
251    let e_phentsize = hdr.ph_entry_size();
252    let e_phoff = hdr.ph_offset();
253    let e_phnum = hdr.ph_count();
254
255    // Sanity check program headers.
256    // Compare against our packed Elf64Phdr (56 bytes = standard ELF64), not
257    // xmas_elf::ProgramHeader which may have padding due to #[repr(C)].
258    if e_phentsize as usize != core::mem::size_of::<Elf64Phdr>() {
259        log::error!(
260            "[elf] Rejecting binary: e_phentsize={} expected={}",
261            e_phentsize,
262            core::mem::size_of::<Elf64Phdr>()
263        );
264        return Err("Unexpected phentsize");
265    }
266
267    let ph_end = (e_phoff as usize)
268        .checked_add((e_phnum as usize) * (e_phentsize as usize))
269        .ok_or("Program header table overflows")?;
270    if ph_end > data.len() {
271        return Err("Program headers extend past file");
272    }
273
274    Ok(Elf64Header {
275        e_type,
276        e_entry,
277        e_phoff,
278        e_phentsize,
279        e_phnum,
280    })
281}
282
283/// Iterate over program headers in the ELF.
284fn program_headers<'a>(
285    data: &'a [u8],
286    header: &Elf64Header,
287) -> impl Iterator<Item = Elf64Phdr> + 'a {
288    let phoff = header.e_phoff as usize;
289    let phsize = header.e_phentsize as usize;
290    let phnum = header.e_phnum as usize;
291
292    (0..phnum).map(move |i| {
293        let offset = phoff + i * phsize;
294        // SAFETY: parse_header already validated that all program headers fit
295        // within `data`, and Elf64Phdr is packed (align 1).
296        unsafe { core::ptr::read_unaligned(data.as_ptr().add(offset) as *const Elf64Phdr) }
297    })
298}
299
300/// Collects an exact-size-hint iterator into a `Vec`, failing cleanly on
301/// allocation failure instead of aborting (issue #67).
302///
303/// The exact reservation means the `push` loop below can never reallocate:
304/// either the whole vector is built, or we return `Err` having allocated
305/// nothing (beyond the failed reservation itself).
306fn try_collect_exact<T, I>(iter: I) -> Result<Vec<T>, &'static str>
307where
308    I: IntoIterator<Item = T>,
309{
310    let iter = iter.into_iter();
311    let (lower, Some(upper)) = iter.size_hint() else {
312        return Err("ELF: iterator has inexact size hint");
313    };
314    if lower != upper {
315        return Err("ELF: iterator has inexact size hint");
316    }
317    let mut v = Vec::new();
318    v.try_reserve_exact(lower)
319        .map_err(|_| "ELF: out of memory while collecting program headers")?;
320    for item in iter {
321        v.push(item);
322    }
323    Ok(v)
324}
325
326/// Maximum PT_INTERP path length we accept.
327///
328/// Real interpreters (ld-linux.so, musl libc loader, …) are well under
329/// 4 KiB; a path longer than a single page is almost certainly corrupt or
330/// hostile.  We cap before any further parsing so a pathological `.interp`
331/// section cannot waste cycles on a doomed allocation / UTF-8 check.
332const MAX_INTERP_PATH_LEN: usize = 4096;
333
334/// Parses interp path.
335///
336/// Cheap sanity checks (`p_filesz` bounds, max length, in-file range) run
337/// first so we reject absurd or hostile `.interp` sections without paying
338/// the full scan / UTF-8 validation cost.
339fn parse_interp_path<'a>(
340    elf_data: &'a [u8],
341    phdrs: &[Elf64Phdr],
342) -> Result<Option<&'a str>, &'static str> {
343    let Some(interp) = phdrs.iter().find(|ph| ph.p_type == PT_INTERP) else {
344        return Ok(None);
345    };
346    if interp.p_filesz == 0 {
347        return Err("PT_INTERP has empty path");
348    }
349    // Reject absurdly large .interp sections before any further work.
350    if (interp.p_filesz as usize) > MAX_INTERP_PATH_LEN {
351        return Err("PT_INTERP path exceeds MAX_INTERP_PATH_LEN");
352    }
353    let start = interp.p_offset as usize;
354    let end = start
355        .checked_add(interp.p_filesz as usize)
356        .ok_or("PT_INTERP range overflow")?;
357    if end > elf_data.len() {
358        return Err("PT_INTERP extends past file");
359    }
360    let raw = &elf_data[start..end];
361    let nul = raw
362        .iter()
363        .position(|&b| b == 0)
364        .ok_or("PT_INTERP path is not NUL terminated")?;
365    let s = core::str::from_utf8(&raw[..nul]).map_err(|_| "PT_INTERP path is not UTF-8")?;
366    if s.is_empty() {
367        return Err("PT_INTERP path is empty");
368    }
369    Ok(Some(s))
370}
371
372/// Performs the find relocated phdr vaddr operation.
373fn find_relocated_phdr_vaddr(
374    header: &Elf64Header,
375    phdrs: &[Elf64Phdr],
376    load_bias: u64,
377) -> Result<u64, &'static str> {
378    let phoff = header.e_phoff;
379    for ph in phdrs {
380        if ph.p_type != PT_LOAD || ph.p_filesz == 0 {
381            continue;
382        }
383        let file_start = ph.p_offset;
384        let file_end = ph
385            .p_offset
386            .checked_add(ph.p_filesz)
387            .ok_or("PHDR location overflow")?;
388        if phoff >= file_start && phoff < file_end {
389            let delta = phoff - file_start;
390            let vaddr = ph
391                .p_vaddr
392                .checked_add(delta)
393                .and_then(|v| v.checked_add(load_bias))
394                .ok_or("Relocated PHDR address overflow")?;
395            if vaddr >= USER_ADDR_MAX {
396                return Err("Relocated PHDR outside user address space");
397            }
398            return Ok(vaddr);
399        }
400    }
401    // Static (-no-pie) binaries place the program header array before the
402    // first PT_LOAD (classic layout: e_phoff=64, LOADs start at 0x1000), so
403    // it is legitimately not covered by any segment. AT_PHDR is optional per
404    // the ABI: report 0 and let push_auxv skip it rather than failing the
405    // whole load for a statically-linked image.
406    Ok(0)
407}
408
409/// Reads elf from vfs.
410fn read_elf_from_vfs(path: &str) -> Result<Vec<u8>, &'static str> {
411    const MAX_ELF_SIZE: usize = 64 * 1024 * 1024;
412    let resolved_path =
413        crate::vfs::resolve_and_check_path_for_current_task(path, true, false, true)
414            .map_err(|_| "PT_INTERP execute denied")?;
415    let fd = crate::vfs::open(&resolved_path, crate::vfs::OpenFlags::READ)
416        .map_err(|_| "PT_INTERP open failed")?;
417    let mut out = Vec::new();
418    let mut buf = [0u8; 4096];
419
420    // Read the first chunk to validate ELF magic before loading the whole file.
421    let n = match crate::vfs::read(fd, &mut buf) {
422        Ok(0) => {
423            let _ = crate::vfs::close(fd);
424            return Err("PT_INTERP file is empty");
425        }
426        Ok(n) => n,
427        Err(_) => {
428            let _ = crate::vfs::close(fd);
429            return Err("PT_INTERP read failed");
430        }
431    };
432    if n < 4 || buf[..4] != [0x7F, b'E', b'L', b'F'] {
433        let _ = crate::vfs::close(fd);
434        return Err("PT_INTERP file is not an ELF");
435    }
436    out.extend_from_slice(&buf[..n]);
437
438    // Continue reading the rest of the file.
439    loop {
440        let n = match crate::vfs::read(fd, &mut buf) {
441            Ok(0) => break,
442            Ok(n) => n,
443            Err(_) => {
444                let _ = crate::vfs::close(fd);
445                return Err("PT_INTERP read failed");
446            }
447        };
448        if out.len().saturating_add(n) > MAX_ELF_SIZE {
449            let _ = crate::vfs::close(fd);
450            return Err("PT_INTERP file too large");
451        }
452        out.extend_from_slice(&buf[..n]);
453    }
454    let _ = crate::vfs::close(fd);
455    Ok(out)
456}
457
458/// Compute total mapped bounds for all PT_LOAD segments.
459///
460/// Validates each PT_LOAD individually (alignment, size, address range) and
461/// additionally enforces, in the spirit of the FreeBSD post-CVE-2018-6924
462/// hardening and the Linux loader, that :
463///   - PT_LOAD `p_vaddr` values appear in strictly non-decreasing order
464///     across the program header table;
465///   - PT_LOAD segments do not overlap in virtual memory after page-alignment
466///     (overlapping segments would alias user pages and produce undefined
467///     behaviour when one is later relaxed via RELRO/mprotect).
468///
469/// Returns the page-aligned `(min_vaddr, max_vaddr)` of the image.
470fn compute_load_bounds(phdrs: &[Elf64Phdr]) -> Result<(u64, u64), &'static str> {
471    let mut min_vaddr = u64::MAX;
472    let mut max_vaddr = 0u64;
473    let mut saw_load = false;
474    // Tracks the end of the last page-aligned segment to detect both
475    // disorder and overlap.  Initialised to 0 (page 0 is never a valid
476    // user-space segment start in our loader), so the first LOAD is always
477    // accepted.
478    let mut last_seg_end_page: u64 = 0;
479
480    for phdr in phdrs {
481        if phdr.p_type != PT_LOAD {
482            continue;
483        }
484        if phdr.p_memsz == 0 {
485            continue;
486        }
487        saw_load = true;
488
489        if phdr.p_memsz < phdr.p_filesz {
490            return Err("PT_LOAD memsz < filesz");
491        }
492
493        // ELF requires p_vaddr % page == p_offset % page for PT_LOAD.
494        if ((phdr.p_vaddr ^ phdr.p_offset) & 0xFFF) != 0 {
495            return Err("PT_LOAD alignment mismatch (vaddr/offset)");
496        }
497
498        let seg_end = phdr
499            .p_vaddr
500            .checked_add(phdr.p_memsz)
501            .ok_or("PT_LOAD vaddr+memsz overflow")?;
502        if seg_end > USER_ADDR_MAX {
503            return Err("PT_LOAD exceeds user address space");
504        }
505
506        let seg_start_page = phdr.p_vaddr & !0xFFF;
507        let seg_end_page = (seg_end + 0xFFF) & !0xFFF;
508
509        // Strictly non-decreasing p_vaddr: reject disorder that would defeat
510        // the linear PT_LOAD scan performed by the loader and the dynamic
511        // linker.  Linux's `load_elf_binary` makes the same assumption.
512        // The first LOAD is allowed to be at any address (last_seg_end_page
513        // starts at 0, a placeholder that no real user segment can reach
514        // because the loader never maps page 0).
515        if last_seg_end_page != 0 {
516            if seg_start_page < last_seg_end_page {
517                return Err("PT_LOAD segments overlap or are out of order");
518            }
519        }
520
521        last_seg_end_page = seg_end_page;
522        min_vaddr = min_vaddr.min(seg_start_page);
523        max_vaddr = max_vaddr.max(seg_end_page);
524    }
525
526    if !saw_load {
527        return Err("ELF has no PT_LOAD segments");
528    }
529    Ok((min_vaddr, max_vaddr))
530}
531
532/// Compute load bias and relocated entry for ET_EXEC / ET_DYN.
533fn compute_load_bias_and_entry(
534    user_as: &AddressSpace,
535    header: &Elf64Header,
536    phdrs: &[Elf64Phdr],
537) -> Result<(u64, u64), &'static str> {
538    let (min_vaddr, max_vaddr) = compute_load_bounds(phdrs)?;
539    let span = max_vaddr
540        .checked_sub(min_vaddr)
541        .ok_or("Invalid PT_LOAD bounds")?;
542
543    let load_bias = if header.e_type == ET_EXEC {
544        0
545    } else {
546        let n_pages = (span as usize).div_ceil(4096);
547        // No hardcoded fallback: if the randomized PIE base cannot host the
548        // image (address space full), fail the load loudly instead of
549        // silently colliding with existing mappings (issue #68).
550        let load_base = user_as
551            .find_free_vma_range(pie_base(), n_pages, VmaPageSize::Small)
552            .ok_or("No virtual range for ET_DYN image")?;
553        load_base
554            .checked_sub(min_vaddr)
555            .ok_or("ET_DYN load bias underflow")?
556    };
557
558    let relocated_end = max_vaddr
559        .checked_add(load_bias)
560        .ok_or("Relocated PT_LOAD range overflow")?;
561    if relocated_end > USER_ADDR_MAX {
562        return Err("Relocated PT_LOAD range exceeds user space");
563    }
564
565    if header.e_type == ET_EXEC && header.e_entry == 0 {
566        return Err("ET_EXEC has null entry point");
567    }
568
569    let entry_raw = header.e_entry;
570
571    let relocated_entry = entry_raw
572        .checked_add(load_bias)
573        .ok_or("Relocated entry overflow")?;
574    if relocated_entry == 0 || relocated_entry >= USER_ADDR_MAX {
575        return Err("Relocated entry outside user space");
576    }
577
578    Ok((load_bias, relocated_entry))
579}
580
581/// Performs the apply segment permissions operation.
582///
583/// # SMP / TLB invariants (DO NOT BREAK)
584///
585/// During ELF loading, the caller (the loader) is the *sole* user of the
586/// target `AddressSpace`: the address space was just created with
587/// [`AddressSpace::new_user`] and is not yet attached to any task.  Because
588/// of this, the function only performs a **local** TLB invalidation on the
589/// current CPU when CR3 matches the address space.
590///
591/// ## Hard constraint
592///
593/// **This function MUST NOT be reused as a generic `mprotect` after the
594/// image has started executing.**  Once a user task has been scheduled,
595/// the address space may be active on another CPU and a local-only flush
596/// would let stale writable mappings survive on remote CPUs, breaking
597/// RELRO guarantees and creating an exploitable window.  A future
598/// `mprotect` implementation must use the cross-CPU TLB shootdown path
599/// (`tlb::shootdown_range`) instead of this helper.
600fn apply_segment_permissions(
601    user_as: &AddressSpace,
602    page_start: u64,
603    page_count: usize,
604    flags: VmaFlags,
605) -> Result<(), &'static str> {
606    use crate::x86_crate_shim::registers::control::Cr3;
607
608    // Defensive: refuse to relax permissions via the loader path once the
609    // address space has been installed on any CPU.  The check is only
610    // meaningful on architectures with a remote-CPU tracking field on
611    // AddressSpace; if such a field is added later, gate the assertion on
612    // its presence.
613    #[cfg(all(debug_assertions, feature = "elf_loader_assert_remote_active"))]
614    {
615        if user_as.is_active_on_remote_cpu() {
616            return Err(
617                "apply_segment_permissions called on an address space already active on another CPU",
618            );
619        }
620    }
621
622    let pte_flags = flags.to_page_flags();
623    // SAFETY: loader owns this AddressSpace during image construction.
624    let mut mapper = unsafe { user_as.mapper() };
625    for i in 0..page_count {
626        let vaddr = page_start
627            .checked_add((i as u64) * 4096)
628            .ok_or("Permission update address overflow")?;
629        let page = Page::<Size4KiB>::from_start_address(VirtAddr::new(vaddr))
630            .map_err(|_| "Invalid page while updating segment flags")?;
631        // SAFETY: the page is already mapped by map_region for this segment.
632        let _ = unsafe {
633            mapper
634                .update_flags(page, pte_flags)
635                .map_err(|_| "Failed to update segment page flags")?
636        };
637        // We ignore flush here and do a targeted flush decision below.
638    }
639
640    // During ELF loading we update a freshly-created user address space that is
641    // not active on other CPUs.  Cross-CPU shootdowns here only add boot-time
642    // latency and can timeout while APs are not yet servicing IPIs.
643    // If this address space is currently active on this CPU, local invalidation
644    // is enough for the loader path.
645    let (current_cr3, _) = Cr3::read();
646    if current_cr3.start_address() == user_as.cr3() {
647        let end = page_start + (page_count as u64) * 4096;
648        crate::arch::tlb::local_range(VirtAddr::new(page_start), VirtAddr::new(end));
649    }
650
651    Ok(())
652}
653
654/// Reads user mapped bytes.
655///
656/// Uses a single-slot page translation cache to avoid a page-table walk
657/// on every successive byte within the same page.  This is the same
658/// optimisation already applied to [`load_segment`] and dramatically
659/// speeds up binaries with thousands of RELA / RELR relocations.
660fn read_user_mapped_bytes(
661    user_as: &AddressSpace,
662    mut vaddr: u64,
663    out: &mut [u8],
664) -> Result<(), &'static str> {
665    let end = vaddr
666        .checked_add(out.len() as u64)
667        .ok_or("Read range overflow")?;
668    if end > USER_ADDR_MAX {
669        return Err("Read range outside user space");
670    }
671    let mut copied = 0usize;
672    let mut cached_page_vaddr: u64 = u64::MAX;
673    let mut cached_hhdm: usize = 0;
674    // SMAP: temporarily disable supervisor-mode access prevention while
675    // reading from user-space pages through the HHDM.
676    crate::arch::stac();
677    while copied < out.len() {
678        let page_vaddr = vaddr & !0xFFF;
679        let page_off = (vaddr & 0xFFF) as usize;
680        let chunk = core::cmp::min(out.len() - copied, 4096 - page_off);
681
682        if page_vaddr != cached_page_vaddr {
683            let phys = user_as
684                .translate(VirtAddr::new(vaddr))
685                .ok_or("Failed to translate mapped user bytes")?;
686            let paddr = phys.as_u64();
687            if paddr == 0 {
688                crate::arch::clac();
689                return Err("Translated physical address is null");
690            }
691            let hhdm_ptr = crate::memory::phys_to_virt(paddr) as *const u8;
692            if hhdm_ptr.is_null() {
693                crate::arch::clac();
694                return Err("HHDM-mapped source is null");
695            }
696            cached_page_vaddr = page_vaddr;
697            cached_hhdm = hhdm_ptr as usize;
698        }
699
700        let src = cached_hhdm as *const u8;
701        // SAFETY: src points to mapped physical memory via HHDM.
702        // The address was just validated non-null, and the translate()
703        // call guarantees the virtual address is backed by a valid frame.
704        unsafe {
705            core::ptr::copy_nonoverlapping(src.add(page_off), out.as_mut_ptr().add(copied), chunk)
706        };
707        copied += chunk;
708        vaddr = vaddr
709            .checked_add(chunk as u64)
710            .ok_or("Virtual address overflow while reading mapped bytes")?;
711    }
712    crate::arch::clac();
713    Ok(())
714}
715
716/// Writes user mapped bytes.
717///
718/// Mirror of [`read_user_mapped_bytes`]: a single-slot page cache collapses
719/// thousands of RELA / RELR writes that touch the same page into a single
720/// page-table walk.
721fn write_user_mapped_bytes(
722    user_as: &AddressSpace,
723    mut vaddr: u64,
724    src: &[u8],
725) -> Result<(), &'static str> {
726    let end = vaddr
727        .checked_add(src.len() as u64)
728        .ok_or("Write range overflow")?;
729    if end > USER_ADDR_MAX {
730        return Err("Write range outside user space");
731    }
732    let mut written = 0usize;
733    let mut cached_page_vaddr: u64 = u64::MAX;
734    let mut cached_hhdm: usize = 0;
735    // SMAP: temporarily disable supervisor-mode access prevention while
736    // writing to user-space pages through the HHDM.
737    crate::arch::stac();
738    while written < src.len() {
739        let page_vaddr = vaddr & !0xFFF;
740        let page_off = (vaddr & 0xFFF) as usize;
741        let chunk = core::cmp::min(src.len() - written, 4096 - page_off);
742
743        if page_vaddr != cached_page_vaddr {
744            let phys = user_as
745                .translate(VirtAddr::new(vaddr))
746                .ok_or("Failed to translate relocation target")?;
747            let paddr = phys.as_u64();
748            if paddr == 0 {
749                crate::arch::clac();
750                return Err("Translated physical address is null");
751            }
752            let hhdm_ptr = crate::memory::phys_to_virt(paddr) as *mut u8;
753            if hhdm_ptr.is_null() {
754                crate::arch::clac();
755                return Err("HHDM-mapped destination is null");
756            }
757            cached_page_vaddr = page_vaddr;
758            cached_hhdm = hhdm_ptr as usize;
759        }
760
761        let dst = cached_hhdm as *mut u8;
762        // SAFETY: destination points to mapped user frame through HHDM.
763        // The address was just validated non-null, and the translate()
764        // call guarantees the virtual address is backed by a valid frame.
765        unsafe {
766            core::ptr::copy_nonoverlapping(src.as_ptr().add(written), dst.add(page_off), chunk)
767        };
768        written += chunk;
769        vaddr = vaddr
770            .checked_add(chunk as u64)
771            .ok_or("Virtual address overflow while writing mapped bytes")?;
772    }
773    crate::arch::clac();
774    Ok(())
775}
776
777/// Reads user u64.
778fn read_user_u64(user_as: &AddressSpace, vaddr: u64) -> Result<u64, &'static str> {
779    let mut raw = [0u8; 8];
780    read_user_mapped_bytes(user_as, vaddr, &mut raw)?;
781    Ok(u64::from_le_bytes(raw))
782}
783
784/// Writes user u64.
785fn write_user_u64(user_as: &AddressSpace, vaddr: u64, value: u64) -> Result<(), &'static str> {
786    write_user_mapped_bytes(user_as, vaddr, &value.to_le_bytes())
787}
788
789/// Calls a user-space IFUNC resolver function and returns its result.
790///
791/// The resolver is located at `resolver_vaddr` in the user address space.
792/// All RELATIVE relocations for this binary must have been applied first so
793/// that the resolver's own calls/addresses are correct.
794///
795/// # Security note (audit 2026-09-07)
796///
797/// IFUNC resolvers execute as ordinary user-space functions, but this helper
798/// calls them from Ring 0 via HHDM.  A malicious or corrupted resolver can
799/// read/write kernel memory and escalate privileges.  This is acceptable for
800/// a single-address-space kernel that loads only trusted binaries, but must
801/// NOT be used if untrusted ELF images are ever loaded.
802///
803/// To make accidental misuse hard, the helper enforces, at runtime, that
804/// the resolver lives in a PT_LOAD marked as `non-writable & executable`
805/// (resolvers must be `.text`, never `.data`).  A hostile binary that
806/// plants an IFUNC resolver in a writable page will fail this check.
807///
808/// Future hardening:
809///   - compile the resolver under a sandbox (no `syscall`, no `iret`);
810///   - require an opt-in build flag (`features = "ifunc_resolver"`) so the
811///     unsafe code path is *absent* by default in production kernels.
812fn call_ifunc_resolver(user_as: &AddressSpace, resolver_vaddr: u64) -> Result<u64, &'static str> {
813    if resolver_vaddr >= USER_ADDR_MAX {
814        return Err("IFUNC resolver address outside user space");
815    }
816    let phys = user_as
817        .translate(VirtAddr::new(resolver_vaddr))
818        .ok_or("IFUNC resolver page not mapped")?;
819    let hhdm_ptr = crate::memory::phys_to_virt(phys.as_u64());
820
821    // Runtime safety check: a resolver must live in an executable,
822    // non-writable PT_LOAD.  We re-derive the page flags from the VMA
823    // rather than walking the ELF again so this stays O(1).
824    //
825    // The check is gated on a feature flag because AddressSpace does not
826    // yet expose a public `vma_containing` API.  Until it does, the
827    // check is opt-in via the kernel feature `ifunc_resolver_vma_check`
828    // so production builds do not silently rely on an unexported
829    // helper.
830    #[cfg(all(debug_assertions, feature = "ifunc_resolver_vma_check"))]
831    {
832        let page_vaddr = resolver_vaddr & !0xFFF;
833        if let Some(vma) = user_as.vma_containing(page_vaddr) {
834            if vma.flags.writable || !vma.flags.executable {
835                return Err("IFUNC resolver page is not (.text, non-writable)");
836            }
837        } else {
838            return Err("IFUNC resolver page has no VMA");
839        }
840    }
841
842    log::warn!(
843        "[elf] IFUNC resolver at {:#x} executing in Ring 0 : security risk if binary is untrusted",
844        resolver_vaddr
845    );
846    // SAFETY: hhdm_ptr points to a user page containing executable code.
847    // The resolver is a simple function that returns a u64; it must not
848    // access kernel state.  All RELATIVE relocations for this binary have
849    // already been applied, so the resolver's own target addresses are valid.
850    //
851    // WARNING: this executes user code in Ring 0.  Safe only for trusted binaries.
852    let resolver: extern "C" fn() -> u64 = unsafe { core::mem::transmute(hhdm_ptr as *const ()) };
853    Ok(resolver())
854}
855
856/// Performs the apply relr relocations operation.
857fn apply_relr_relocations(
858    user_as: &AddressSpace,
859    load_bias: u64,
860    relr_base: u64,
861    relr_size: usize,
862    relr_ent: usize,
863) -> Result<usize, &'static str> {
864    if relr_size == 0 {
865        return Ok(0);
866    }
867    if relr_ent != core::mem::size_of::<u64>() {
868        return Err("Unsupported DT_RELRENT size");
869    }
870    if relr_size % relr_ent != 0 {
871        return Err("DT_RELR table size is not aligned");
872    }
873
874    let count = relr_size / relr_ent;
875    let mut applied = 0usize;
876    let mut where_addr = 0u64;
877
878    for i in 0..count {
879        let entry_addr = relr_base
880            .checked_add((i * relr_ent) as u64)
881            .ok_or("DT_RELR walk overflow")?;
882        let entry = read_user_u64(user_as, entry_addr)?;
883
884        if (entry & 1) == 0 {
885            where_addr = load_bias
886                .checked_add(entry)
887                .ok_or("DT_RELR absolute relocation overflow")?;
888            if where_addr >= USER_ADDR_MAX {
889                return Err("DT_RELR target outside user space");
890            }
891            let cur = read_user_u64(user_as, where_addr)?;
892            write_user_u64(
893                user_as,
894                where_addr,
895                cur.checked_add(load_bias)
896                    .ok_or("DT_RELR relocated value overflow")?,
897            )?;
898            where_addr = where_addr
899                .checked_add(8)
900                .ok_or("DT_RELR where pointer overflow")?;
901            applied += 1;
902        } else {
903            if where_addr == 0 {
904                return Err("DT_RELR bitmap entry before initial address entry");
905            }
906            let mut bitmap = entry >> 1;
907            for bit in 0..63u64 {
908                if (bitmap & 1) != 0 {
909                    let slot = where_addr
910                        .checked_add(bit * 8)
911                        .ok_or("DT_RELR bitmap target overflow")?;
912                    if slot >= USER_ADDR_MAX {
913                        return Err("DT_RELR bitmap target outside user space");
914                    }
915                    let cur = read_user_u64(user_as, slot)?;
916                    write_user_u64(
917                        user_as,
918                        slot,
919                        cur.checked_add(load_bias)
920                            .ok_or("DT_RELR bitmap relocated value overflow")?,
921                    )?;
922                    applied += 1;
923                }
924                bitmap >>= 1;
925                if bitmap == 0 {
926                    break;
927                }
928            }
929            where_addr = where_addr
930                .checked_add(64 * 8)
931                .ok_or("DT_RELR where advance overflow")?;
932        }
933    }
934    Ok(applied)
935}
936
937/// Performs the apply dynamic relocations operation.
938fn apply_dynamic_relocations(
939    user_as: &AddressSpace,
940    phdrs: &[Elf64Phdr],
941    elf_type: u16,
942    load_bias: u64,
943) -> Result<(), &'static str> {
944    if elf_type != ET_DYN {
945        return Ok(());
946    }
947
948    let dynamic = phdrs.iter().find(|ph| ph.p_type == PT_DYNAMIC);
949    let Some(dynamic_ph) = dynamic else {
950        return Ok(());
951    };
952    if dynamic_ph.p_filesz == 0 {
953        return Ok(());
954    }
955
956    let dyn_addr = dynamic_ph
957        .p_vaddr
958        .checked_add(load_bias)
959        .ok_or("PT_DYNAMIC relocated address overflow")?;
960    let dyn_file_size = dynamic_ph.p_filesz as usize;
961    let dyn_count = dyn_file_size / core::mem::size_of::<Elf64Dyn>();
962    // Read the entire .dynamic section at once to avoid O(n) page-table walks.
963    let mut dyn_buf = alloc::vec![0u8; dyn_file_size];
964    read_user_mapped_bytes(user_as, dyn_addr, &mut dyn_buf)?;
965    let dyn_slice: &[Elf64Dyn] =
966        unsafe { core::slice::from_raw_parts(dyn_buf.as_ptr() as *const Elf64Dyn, dyn_count) };
967
968    let mut rela_addr: Option<u64> = None;
969    let mut rela_size: usize = 0;
970    let mut rela_ent: usize = core::mem::size_of::<Elf64Rela>();
971    let mut jmprel_addr: Option<u64> = None;
972    let mut jmprel_size: usize = 0;
973    let mut pltrel_kind: Option<u64> = None;
974    let mut symtab_addr: Option<u64> = None;
975    let mut sym_ent: usize = core::mem::size_of::<Elf64Sym>();
976    // Relocated DT_STRTAB: stored now so future symbol-by-name PLT/GOT
977    // lazy-binding resolution can map Elf64Sym::st_name offsets to strings
978    // (issue #66). Not consumed yet beyond tracing.
979    let mut strtab_addr: Option<u64> = None;
980    let mut rela_count_hint: Option<usize> = None;
981    let mut relr_addr: Option<u64> = None;
982    let mut relr_size: usize = 0;
983    let mut relr_ent: usize = 0;
984
985    for i in 0..dyn_count {
986        let dyn_entry = &dyn_slice[i];
987
988        match dyn_entry.d_tag {
989            DT_NULL => break,
990            DT_RELA => {
991                rela_addr = Some(
992                    dyn_entry
993                        .d_val
994                        .checked_add(load_bias)
995                        .ok_or("DT_RELA relocated address overflow")?,
996                )
997            }
998            DT_RELASZ => rela_size = dyn_entry.d_val as usize,
999            DT_RELAENT => rela_ent = dyn_entry.d_val as usize,
1000            DT_RELACOUNT => rela_count_hint = Some(dyn_entry.d_val as usize),
1001            DT_JMPREL => {
1002                jmprel_addr = Some(
1003                    dyn_entry
1004                        .d_val
1005                        .checked_add(load_bias)
1006                        .ok_or("DT_JMPREL relocated address overflow")?,
1007                )
1008            }
1009            DT_PLTRELSZ => jmprel_size = dyn_entry.d_val as usize,
1010            DT_PLTREL => pltrel_kind = Some(dyn_entry.d_val),
1011            DT_SYMTAB => {
1012                symtab_addr = Some(
1013                    dyn_entry
1014                        .d_val
1015                        .checked_add(load_bias)
1016                        .ok_or("DT_SYMTAB relocated address overflow")?,
1017                )
1018            }
1019            DT_SYMENT => sym_ent = dyn_entry.d_val as usize,
1020            DT_STRTAB => {
1021                strtab_addr = Some(
1022                    dyn_entry
1023                        .d_val
1024                        .checked_add(load_bias)
1025                        .ok_or("DT_STRTAB relocated address overflow")?,
1026                );
1027            }
1028            DT_RELR => {
1029                relr_addr = Some(
1030                    dyn_entry
1031                        .d_val
1032                        .checked_add(load_bias)
1033                        .ok_or("DT_RELR relocated address overflow")?,
1034                )
1035            }
1036            DT_RELRSZ => relr_size = dyn_entry.d_val as usize,
1037            DT_RELRENT => relr_ent = dyn_entry.d_val as usize,
1038            _ => {}
1039        }
1040    }
1041
1042    let mut relr_applied = 0usize;
1043    if let Some(relr_base) = relr_addr {
1044        relr_applied = apply_relr_relocations(user_as, load_bias, relr_base, relr_size, relr_ent)?;
1045    } else if relr_size != 0 || relr_ent != 0 {
1046        return Err("DT_RELR metadata present without DT_RELR base");
1047    }
1048    if rela_ent != core::mem::size_of::<Elf64Rela>() {
1049        return Err("Unsupported DT_RELAENT size");
1050    }
1051    if sym_ent != core::mem::size_of::<Elf64Sym>() {
1052        return Err("Unsupported DT_SYMENT size");
1053    }
1054    if pltrel_kind.is_some() && pltrel_kind != Some(DT_RELA as u64) {
1055        return Err("Only DT_PLTREL=DT_RELA is supported");
1056    }
1057
1058    elf_trace!(
1059        "[elf] dynamic: symtab={:?} strtab={:?}",
1060        symtab_addr,
1061        strtab_addr
1062    );
1063
1064    let read_sym_entry = |sym_idx: u32| -> Result<Elf64Sym, &'static str> {
1065        let symtab = symtab_addr.ok_or("Missing DT_SYMTAB for symbol relocations")?;
1066        let sym_addr = symtab
1067            .checked_add((sym_idx as u64) * (sym_ent as u64))
1068            .ok_or("Symbol table address overflow")?;
1069        let mut raw = [0u8; core::mem::size_of::<Elf64Sym>()];
1070        read_user_mapped_bytes(user_as, sym_addr, &mut raw)?;
1071        Ok(unsafe { core::ptr::read_unaligned(raw.as_ptr() as *const Elf64Sym) })
1072    };
1073
1074    let resolve_sym =
1075        |sym_idx: u32, with_bias: bool, check_def: bool| -> Result<u64, &'static str> {
1076            if sym_idx == 0 {
1077                return Ok(0);
1078            }
1079            let sym = read_sym_entry(sym_idx)?;
1080            if check_def && sym.st_shndx == 0 {
1081                return Err("Undefined symbol relocation not supported");
1082            }
1083            if with_bias {
1084                sym.st_value
1085                    .checked_add(load_bias)
1086                    .ok_or("Symbol value relocation overflow")
1087            } else {
1088                Ok(sym.st_value)
1089            }
1090        };
1091
1092    let resolve_size = |sym_idx: u32| -> Result<u64, &'static str> {
1093        if sym_idx == 0 {
1094            return Ok(0);
1095        }
1096        let sym = read_sym_entry(sym_idx)?;
1097        Ok(sym.st_size)
1098    };
1099
1100    // Variant II TLS: tp = tls_base + aligned_memsz.  We need the aligned
1101    // memsz for TPOFF64/DTPOFF64 calculations.
1102    let tls_aligned_memsz: i128 = phdrs
1103        .iter()
1104        .find(|ph| ph.p_type == PT_TLS)
1105        .map(|tls| {
1106            let memsz = tls.p_memsz;
1107            let align = tls.p_align.max(1);
1108            let aligned = (memsz + align - 1) & !(align - 1);
1109            aligned as i128
1110        })
1111        .unwrap_or(0);
1112
1113    let apply_rela_table = |table_base: u64,
1114                            table_size: usize,
1115                            count_hint: Option<usize>|
1116     -> Result<usize, &'static str> {
1117        if table_size == 0 {
1118            return Ok(0);
1119        }
1120        // Use the table size as the authoritative entry count.  DT_RELACOUNT is
1121        // a *hint* from the linker that may undercount; honouring it with min()
1122        // silently drops valid relocations.  We only validate the hint as a
1123        // sanity bound (if provided).
1124        let count = table_size / rela_ent;
1125        if let Some(hint) = count_hint {
1126            if hint > count {
1127                return Err("DT_RELACOUNT exceeds actual RELA table size");
1128            }
1129        }
1130        let mut applied = 0usize;
1131        for i in 0..count {
1132            let rela_addr_i = table_base
1133                .checked_add((i * rela_ent) as u64)
1134                .ok_or("Rela table overflow")?;
1135            let mut raw = [0u8; core::mem::size_of::<Elf64Rela>()];
1136            read_user_mapped_bytes(user_as, rela_addr_i, &mut raw)?;
1137            // SAFETY: raw has exact size of Elf64Rela.
1138            let rela = unsafe { core::ptr::read_unaligned(raw.as_ptr() as *const Elf64Rela) };
1139
1140            let r_type = (rela.r_info & 0xffff_ffff) as u32;
1141            let r_sym = (rela.r_info >> 32) as u32;
1142            let target = rela
1143                .r_offset
1144                .checked_add(load_bias)
1145                .ok_or("Relocation target overflow")?;
1146            if target >= USER_ADDR_MAX {
1147                return Err("Relocation target outside user space");
1148            }
1149
1150            let value = match r_type {
1151                R_X86_64_RELATIVE => {
1152                    if r_sym != 0 {
1153                        return Err("R_X86_64_RELATIVE with non-zero symbol");
1154                    }
1155                    (load_bias as i128)
1156                        .checked_add(rela.r_addend as i128)
1157                        .ok_or("Relocation value overflow")?
1158                }
1159                R_X86_64_GLOB_DAT | R_X86_64_JUMP_SLOT | R_X86_64_64 => {
1160                    let sym_val = resolve_sym(r_sym, true, true)? as i128;
1161                    sym_val
1162                        .checked_add(rela.r_addend as i128)
1163                        .ok_or("Relocation value overflow")?
1164                }
1165                R_X86_64_COPY => {
1166                    let sym_val = resolve_sym(r_sym, true, true)?;
1167                    if sym_val == 0 {
1168                        continue;
1169                    }
1170                    let sym_sz = resolve_size(r_sym)?;
1171                    if sym_sz == 0 {
1172                        log::warn!("[elf] R_X86_64_COPY with zero st_size for symbol {}", r_sym);
1173                    }
1174                    if sym_sz > 0 && sym_val < USER_ADDR_MAX {
1175                        let mut tmp = [0u8; 256];
1176                        let mut off = 0u64;
1177                        while off < sym_sz {
1178                            let chunk = core::cmp::min(256, (sym_sz - off) as usize);
1179                            let src = sym_val.checked_add(off).ok_or("COPY source overflow")?;
1180                            let dst = target.checked_add(off).ok_or("COPY target overflow")?;
1181                            read_user_mapped_bytes(user_as, src, &mut tmp[..chunk])?;
1182                            write_user_mapped_bytes(user_as, dst, &tmp[..chunk])?;
1183                            off += chunk as u64;
1184                        }
1185                    }
1186                    applied += 1;
1187                    continue;
1188                }
1189                R_X86_64_TPOFF64 => {
1190                    let sym_val = if r_sym != 0 {
1191                        resolve_sym(r_sym, false, false)? as i128
1192                    } else {
1193                        0i128
1194                    };
1195                    // Variant II: tp = tls_base + aligned_memsz, so offset from tp is
1196                    // (sym.st_value - aligned_memsz + r_addend).  Use i128 to avoid
1197                    // underflow when sym_val < aligned_memsz.
1198                    sym_val
1199                        .checked_sub(tls_aligned_memsz)
1200                        .and_then(|v| v.checked_add(rela.r_addend as i128))
1201                        .ok_or("TPOFF64 value overflow")?
1202                }
1203                R_X86_64_DTPMOD64 => {
1204                    // For single-binary loading (no dynamic linker), the module ID is always 1.
1205                    1i128
1206                }
1207                R_X86_64_DTPOFF64 => {
1208                    // DTV-relative offset: same as TPOFF64 for single-binary loading.
1209                    let sym_val = if r_sym != 0 {
1210                        resolve_sym(r_sym, false, false)? as i128
1211                    } else {
1212                        0i128
1213                    };
1214                    sym_val
1215                        .checked_sub(tls_aligned_memsz)
1216                        .and_then(|v| v.checked_add(rela.r_addend as i128))
1217                        .ok_or("DTPOFF64 value overflow")?
1218                }
1219                R_X86_64_IRELATIVE => {
1220                    // IRELATIVE: the target is a resolver function that must be
1221                    // *called* to obtain the final value.  All RELATIVE
1222                    // relocations for this binary have already been applied, so
1223                    // the resolver's own addresses are correct.
1224                    let resolver_vaddr = (load_bias as i128)
1225                        .checked_add(rela.r_addend as i128)
1226                        .ok_or("IRELATIVE resolver address overflow")?;
1227                    if resolver_vaddr < 0 || resolver_vaddr as u64 >= USER_ADDR_MAX {
1228                        return Err("IRELATIVE resolver outside user space");
1229                    }
1230                    let resolved = call_ifunc_resolver(user_as, resolver_vaddr as u64)?;
1231                    resolved as i128
1232                }
1233                _ => {
1234                    log::warn!("[elf] Unsupported relocation type {}", r_type);
1235                    continue;
1236                }
1237            };
1238            if value < 0 || value > u64::MAX as i128 {
1239                return Err("Relocation value out of range");
1240            }
1241            let val_u64 = value as u64;
1242            #[cfg(debug_assertions)]
1243            if applied < 5 {
1244                let r_addend_copy = rela.r_addend;
1245                let mut before = [0u8; 8];
1246                let _ = read_user_mapped_bytes(user_as, target, &mut before);
1247                let before_val = u64::from_le_bytes(before);
1248                log::trace!(
1249                    "[reloc] [{i}] r_type={} target={:#x} r_addend={:#x} value={:#x} before={:#x}",
1250                    r_type,
1251                    target,
1252                    r_addend_copy,
1253                    val_u64,
1254                    before_val
1255                );
1256            }
1257            write_user_mapped_bytes(user_as, target, &val_u64.to_le_bytes())?;
1258            #[cfg(debug_assertions)]
1259            if applied < 5 {
1260                let mut after = [0u8; 8];
1261                let _ = read_user_mapped_bytes(user_as, target, &mut after);
1262                let after_val = u64::from_le_bytes(after);
1263                log::trace!(
1264                    "[reloc] [{i}] after_write={:#x} (expected={:#x})",
1265                    after_val,
1266                    val_u64
1267                );
1268            }
1269            #[cfg(debug_assertions)]
1270            if val_u64 >= 0xffff_8000_0000_0000 {
1271                let r_addend_copy = rela.r_addend;
1272                log::trace!(
1273                    "[reloc-KERNEL-ADDR] [{i}] r_type={} target={:#x} r_addend={:#x} val={:#x} bias={:#x}",
1274                    r_type, target, r_addend_copy, val_u64, load_bias
1275                );
1276            }
1277            applied += 1;
1278        }
1279        Ok(applied)
1280    };
1281
1282    let mut total_applied = 0usize;
1283    #[cfg(debug_assertions)]
1284    log::trace!(
1285        "[reloc] apply_dynamic_relocations: bias={:#x} rela_addr={:?} rela_size={} rela_count={:?}",
1286        load_bias,
1287        rela_addr,
1288        rela_size,
1289        rela_count_hint
1290    );
1291    if let Some(rela_base) = rela_addr {
1292        let _ = total_applied += apply_rela_table(rela_base, rela_size, rela_count_hint)?;
1293    }
1294    if let Some(jmprel_base) = jmprel_addr {
1295        let _ = total_applied += apply_rela_table(jmprel_base, jmprel_size, None)?;
1296    }
1297
1298    #[cfg(debug_assertions)]
1299    if total_applied > 0 {
1300        log::trace!(
1301            "[reloc] applied {} RELA relocations (bias={:#x})",
1302            total_applied,
1303            load_bias
1304        );
1305    }
1306    if relr_applied > 0 {
1307        log::debug!("[elf] Applied {} RELR relocations", relr_applied);
1308    }
1309    Ok(())
1310}
1311
1312// ---------------------------------------------------------------------------
1313// Loading
1314// ---------------------------------------------------------------------------
1315
1316/// Convert ELF p_flags to VmaFlags.
1317fn elf_flags_to_vma(p_flags: u32) -> VmaFlags {
1318    VmaFlags {
1319        readable: p_flags & PF_R != 0,
1320        writable: p_flags & PF_W != 0,
1321        executable: p_flags & PF_X != 0,
1322        user_accessible: true,
1323    }
1324}
1325
1326/// Load a single PT_LOAD segment into the given address space.
1327///
1328/// Allocates physical frames, maps them with appropriate permissions, and
1329/// copies file data into the mapping. BSS (memsz > filesz) is already
1330/// zero-filled because `map_region` zeroes newly allocated frames.
1331fn load_segment(
1332    user_as: &AddressSpace,
1333    elf_data: &[u8],
1334    phdr: &Elf64Phdr,
1335    load_bias: u64,
1336) -> Result<(), &'static str> {
1337    let vaddr = phdr
1338        .p_vaddr
1339        .checked_add(load_bias)
1340        .ok_or("PT_LOAD relocated vaddr overflow")?;
1341    let memsz = phdr.p_memsz;
1342    let filesz = phdr.p_filesz;
1343    let offset = phdr.p_offset;
1344
1345    // Validate addresses are in user space
1346    if vaddr >= USER_ADDR_MAX {
1347        return Err("PT_LOAD vaddr outside user space");
1348    }
1349    let end = vaddr
1350        .checked_add(memsz)
1351        .ok_or("PT_LOAD vaddr+memsz overflows")?;
1352    if end > USER_ADDR_MAX {
1353        return Err("PT_LOAD segment extends past user space");
1354    }
1355
1356    // Validate file region
1357    let file_end = (offset as usize)
1358        .checked_add(filesz as usize)
1359        .ok_or("PT_LOAD offset+filesz overflows")?;
1360    if file_end > elf_data.len() {
1361        return Err("PT_LOAD file data extends past ELF");
1362    }
1363
1364    // Calculate page-aligned mapping
1365    let page_start = vaddr & !0xFFF;
1366    let page_end = (end + 0xFFF) & !0xFFF;
1367    let page_count = ((page_end - page_start) / 4096) as usize;
1368
1369    // Map writable during copy, then restore final ELF flags.
1370    let actual_flags = elf_flags_to_vma(phdr.p_flags);
1371    let load_flags = VmaFlags {
1372        readable: true,
1373        writable: true, // Need write access to copy data in
1374        executable: actual_flags.executable,
1375        user_accessible: true,
1376    };
1377
1378    let vma_type = if actual_flags.executable {
1379        VmaType::Code
1380    } else {
1381        VmaType::Anonymous
1382    };
1383    log::debug!(
1384        "[elf] map PT_LOAD: start={:#x} pages={} filesz={:#x}",
1385        page_start,
1386        page_count,
1387        filesz
1388    );
1389    user_as.map_region(
1390        page_start,
1391        page_count,
1392        load_flags,
1393        vma_type,
1394        VmaPageSize::Small,
1395    )?;
1396
1397    // Copy file data into the mapped pages.
1398    // Batch-translate all pages at once to avoid page-table walks per-chunk.
1399    if filesz > 0 {
1400        let src = &elf_data[offset as usize..file_end];
1401        let mut copied = 0usize;
1402
1403        // Collect physical addresses for all pages in the range.
1404        let n_vaddrs = ((page_end - page_start) / 4096) as usize;
1405        let mut phys_pages = alloc::vec::Vec::with_capacity(n_vaddrs);
1406        for i in 0..n_vaddrs {
1407            let vaddr = page_start + (i as u64) * 4096;
1408            let phys = user_as
1409                .translate(VirtAddr::new(vaddr))
1410                .ok_or("Failed to translate user page after mapping")?;
1411            phys_pages.push(phys);
1412        }
1413
1414        while copied < src.len() {
1415            let dst_vaddr = vaddr + copied as u64;
1416            let page_idx = ((dst_vaddr - page_start) / 4096) as usize;
1417            let page_offset = (dst_vaddr & 0xFFF) as usize;
1418            let chunk = core::cmp::min(src.len() - copied, 4096 - page_offset);
1419
1420            let phys = phys_pages[page_idx];
1421            let hhdm_ptr = crate::memory::phys_to_virt(phys.as_u64()) as *mut u8;
1422            // SAFETY: hhdm_ptr points to a freshly mapped, zeroed frame via HHDM.
1423            unsafe {
1424                core::ptr::copy_nonoverlapping(
1425                    src.as_ptr().add(copied),
1426                    hhdm_ptr.add(page_offset),
1427                    chunk,
1428                );
1429            }
1430            copied += chunk;
1431        }
1432    }
1433
1434    // Tighten PTE permissions after copy.
1435    apply_segment_permissions(user_as, page_start, page_count, actual_flags)?;
1436
1437    log::debug!(
1438        "  PT_LOAD: {:#x}..{:#x} ({} pages, file {:#x}+{:#x}, flags {:?})",
1439        page_start,
1440        page_end,
1441        page_count,
1442        offset,
1443        filesz,
1444        actual_flags,
1445    );
1446
1447    Ok(())
1448}
1449
1450// ---------------------------------------------------------------------------
1451// Task creation with IRETQ trampoline
1452// ---------------------------------------------------------------------------
1453
1454/// Parameters for the Ring 3 trampoline, stored in a static so the
1455/// Trampoline that switches to user address space and does IRETQ to Ring 3.
1456///
1457/// Parameters (entry point, stack top, arg0, address space) are read from the
1458/// *current task* so that each ELF task carries its own copy.  This makes the
1459/// trampoline safe under SMP: two tasks can run their trampolines concurrently
1460/// on different CPUs without any shared mutable state.
1461extern "C" fn elf_ring3_trampoline() -> ! {
1462    use crate::arch::gdt;
1463    use core::sync::atomic::Ordering;
1464
1465    elf_trace!("[trace][elf] ring3_trampoline before current_task");
1466    let Some(task) = crate::process::scheduler::current_task_clone_spin_debug("ring3_trampoline")
1467    else {
1468        log::error!("[elf] ring3_trampoline: no current task, aborting");
1469        loop {
1470            crate::x86_crate_shim::instructions::hlt();
1471        }
1472    };
1473    elf_trace!(
1474        "[trace][elf] ring3_trampoline enter tid={} name={}",
1475        task.id.as_u64(),
1476        task.name
1477    );
1478    task.set_resume_kind(crate::process::task::ResumeKind::IretFrame);
1479
1480    let user_rip = task.trampoline_entry.load(Ordering::Acquire);
1481    let user_rsp = task.trampoline_stack_top.load(Ordering::Acquire);
1482    let user_arg0 = task.trampoline_arg0.load(Ordering::Acquire);
1483    elf_trace!(
1484        "[trace][elf] ring3_trampoline args tid={} rip={:#x} rsp={:#x} arg0={:#x}",
1485        task.id.as_u64(),
1486        user_rip,
1487        user_rsp,
1488        user_arg0
1489    );
1490
1491    // Probe: read GOT entries via HHDM before switching to user AS.
1492    // This is the last kernel-owned moment before user execution begins.
1493    // If values here are wrong, the bug is in load/relocation, not in
1494    // something that happens after this point.
1495    #[cfg(debug_assertions)]
1496    {
1497        // SAFETY: Kernel still holds the boot/kernel CR3. HHDM is valid.
1498        unsafe {
1499            let as_ref = task.process.address_space_arc();
1500            let task_name: &str = &task.name;
1501            for test_off in [0x12920u64, 0x12928u64, 0x12930u64] {
1502                let vaddr = 0x100000000u64.wrapping_add(test_off);
1503                if let Some(phys) = as_ref.translate(VirtAddr::new(vaddr)) {
1504                    let ptr = crate::memory::phys_to_virt(phys.as_u64()) as *const u64;
1505                    let val = core::ptr::read_unaligned(ptr);
1506                    elf_trace!(
1507                        "[trampoline-got] tid={} name={} GOT[{:#x}]=phys={:#x} val={:#x}",
1508                        task.id.as_u64(),
1509                        task_name,
1510                        vaddr,
1511                        phys.as_u64(),
1512                        val
1513                    );
1514                } else {
1515                    elf_trace!(
1516                        "[trampoline-got] tid={} name={} GOT[{:#x}]=<not mapped>",
1517                        task.id.as_u64(),
1518                        task_name,
1519                        vaddr
1520                    );
1521                }
1522            }
1523        }
1524    }
1525
1526    // Switch to the user address space stored in the task.
1527    // SAFETY: The address space was set up during task creation and is valid.
1528    unsafe {
1529        let as_ref = task.process.address_space_arc();
1530        as_ref.switch_to();
1531    }
1532    elf_trace!(
1533        "[trace][elf] ring3_trampoline switch_to done tid={}",
1534        task.id.as_u64()
1535    );
1536
1537    let user_cs = gdt::user_code_selector().0 as u64;
1538    let user_ss = gdt::user_data_selector().0 as u64;
1539    let user_rflags: u64 = USER_RFLAGS;
1540    elf_trace!(
1541        "[trace][elf] ring3_trampoline iret tid={} cs={:#x} ss={:#x} rflags={:#x}",
1542        task.id.as_u64(),
1543        user_cs,
1544        user_ss,
1545        user_rflags
1546    );
1547
1548    // ----- Pre-iret LAPIC timer diagnostic -----
1549    // Verify that the APIC timer is actually running on this CPU before we
1550    // enter Ring 3 (if it is not, no timer tick = no heartbeat = silent hang).
1551    unsafe {
1552        let lvt = crate::arch::apic::read_reg(crate::arch::apic::REG_LVT_TIMER);
1553        let init_cnt = crate::arch::apic::read_reg(crate::arch::apic::REG_TIMER_INIT);
1554        let _cur_cnt = crate::arch::apic::read_reg(crate::arch::apic::REG_TIMER_CURRENT);
1555        let _rflags_now: u64;
1556        core::arch::asm!("pushfq; pop {}", out(reg) _rflags_now, options(nostack));
1557        elf_trace!(
1558            "[trace][elf] pre-iret LAPIC: LVT={:#x} init={} cur={} IF={}",
1559            lvt,
1560            init_cnt,
1561            _cur_cnt,
1562            (_rflags_now >> 9) & 1
1563        );
1564        if lvt & (1 << 16) != 0 {
1565            elf_trace!(
1566                "[trace][elf] WARNING: LAPIC timer is MASKED (bit 16 set) : no ticks will fire!"
1567            );
1568        }
1569        if init_cnt == 0 {
1570            elf_trace!("[trace][elf] WARNING: LAPIC timer init_count=0 : timer not started!");
1571        }
1572    }
1573
1574    crate::arch::ring3_diag::validate_ring3_state(
1575        user_rip,
1576        user_rsp,
1577        user_cs as u16,
1578        user_ss as u16,
1579    );
1580
1581    elf_trace!(
1582        "[elf] PRE-IRETQ tid={} rip={:#x} rsp={:#x} rflags={:#x}",
1583        task.id.as_u64(),
1584        user_rip,
1585        user_rsp,
1586        user_rflags
1587    );
1588
1589    // E9 probe: validate_ring3_state passed, entering asm block.
1590    // If '0' is visible but not '1', the compiler inserted code between
1591    // the two that crashed (unlikely, but this rules it out).
1592    elf_trace!(
1593        "E9[0] pre-asm rip={:#x} rsp={:#x} cs={:#x} ss={:#x}",
1594        user_rip,
1595        user_rsp,
1596        user_cs,
1597        user_ss,
1598    );
1599
1600    // SAFETY: Valid user mappings have been set up. IRETQ switches to Ring 3.
1601    //
1602    // Interrupts must be masked in the final kernel instructions before
1603    // `swapgs ; iretq`. Otherwise a timer IRQ can land after `swapgs` but
1604    // before `iretq`, with `CS=0x8` and `GS=user`, and the first `gs:[..]`
1605    // access in the handler faults in the swapgs->iretq window.
1606
1607    // Each `out 0xe9, al` writes an ASCII character to QEMU's E9 port.
1608    // Debug builds include probes 1-4 for diagnosing IRETQ failures;
1609    // release builds omit them to avoid port I/O overhead.
1610    #[cfg(debug_assertions)]
1611    unsafe {
1612        core::arch::asm!(
1613            // Close the IRQ window before touching GS. `iretq` restores IF=1
1614            // from the user RFLAGS frame, so user mode still starts with
1615            // interrupts enabled.
1616            "cli",
1617
1618            //  Probe 1: entering the asm block
1619            "push rax",
1620            "mov al, 0x31",     // '1'
1621            "out 0xe9, al",
1622            "pop rax",
1623
1624            //  Build the iretq frame
1625            // Order required by IRETQ (popped in reverse order):
1626            //   [RSP+32] SS
1627            //   [RSP+24] user RSP
1628            //   [RSP+16] RFLAGS
1629            //   [RSP+8]  CS
1630            //   [RSP+0]  RIP  <--- RSP here after the 5 pushes
1631            "push {ss}",
1632            "push {rsp_val}",
1633            "push {rflags}",
1634            "push {cs}",
1635            "push {rip}",
1636
1637            //  Probe 2: frame iretq complete
1638            "push rax",
1639            "mov al, 0x32",     // '2'
1640            "out 0xe9, al",
1641            "pop rax",
1642
1643            //  Pre-fault the user code page
1644            // Touch the first byte at user_rip to trigger a demand page fault
1645            // while GS is still the kernel per-CPU block. Without this, the
1646            // iretq instruction itself can fault in the SWAPGS->Ring3 window,
1647            // producing a SWAPGS-WINDOW page fault (CS=Ring0 but GS=user).
1648            "mov rax, {rip}",
1649            "movzx rax, byte ptr [rax]",
1650
1651            //  Load arg0 into RDI
1652            "mov rdi, {arg0}",
1653
1654            //  Probe 3: RDI loaded, just before SWAPGS
1655            "push rax",
1656            "mov al, 0x33",     // '3'
1657            "out 0xe9, al",
1658            "pop rax",
1659
1660            //  SWAPGS: GS.base kernel <-> GS.base user
1661            "swapgs",
1662
1663            //  Probe 4: SWAPGS succeeded, IRETQ imminent
1664            "push rax",
1665            "mov al, 0x34",     // '4'
1666            "out 0xe9, al",
1667            "pop rax",
1668
1669            //  IRETQ: point of no return
1670            "iretq",
1671
1672            ss      = in(reg) user_ss,
1673            rsp_val = in(reg) user_rsp,
1674            rflags  = in(reg) user_rflags,
1675            cs      = in(reg) user_cs,
1676            rip     = in(reg) user_rip,
1677            arg0    = in(reg) user_arg0,
1678            options(noreturn),
1679        );
1680    }
1681
1682    #[cfg(not(debug_assertions))]
1683    unsafe {
1684        core::arch::asm!(
1685            "cli",
1686
1687            //  Build the iretq frame
1688            "push {ss}",
1689            "push {rsp_val}",
1690            "push {rflags}",
1691            "push {cs}",
1692            "push {rip}",
1693
1694            //  Pre-fault the user code page
1695            "mov rax, {rip}",
1696            "movzx rax, byte ptr [rax]",
1697
1698            //  Load arg0 into RDI
1699            "mov rdi, {arg0}",
1700
1701            //  SWAPGS: GS.base kernel <-> GS.base user
1702            "swapgs",
1703
1704            //  IRETQ: point of no return
1705            "iretq",
1706
1707            ss      = in(reg) user_ss,
1708            rsp_val = in(reg) user_rsp,
1709            rflags  = in(reg) user_rflags,
1710            cs      = in(reg) user_cs,
1711            rip     = in(reg) user_rip,
1712            arg0    = in(reg) user_arg0,
1713            options(noreturn),
1714        );
1715    }
1716}
1717
1718// ---------------------------------------------------------------------------
1719// Public API
1720// ---------------------------------------------------------------------------
1721/// Load an ELF64 binary and schedule it as a Ring 3 user task.
1722///
1723/// # Arguments
1724/// * `elf_data` : raw ELF file bytes (must remain valid until load completes).
1725/// * `name` : name for the task (debugging purposes).
1726///
1727/// # Returns
1728/// `Ok(())` on success, `Err` with a static error message on failure.
1729pub fn load_and_run_elf(elf_data: &[u8], name: &'static str) -> Result<TaskId, &'static str> {
1730    load_and_run_elf_with_caps(elf_data, name, &[])
1731}
1732
1733/// Load an ELF64 binary with command-line arguments and schedule it as a Ring 3 task.
1734///
1735/// `extra_args` maps to `argv[1..]`; `argv[0]` is always `name`.
1736pub fn load_and_run_elf_with_args(
1737    elf_data: &[u8],
1738    name: &'static str,
1739    extra_args: &[&str],
1740) -> Result<TaskId, &'static str> {
1741    let task = load_elf_task_inner(elf_data, name, extra_args, &[], USER_STACK_PAGES)?;
1742    let task_id = task.id;
1743    crate::process::add_task(task);
1744    Ok(task_id)
1745}
1746
1747/// Load an ELF64 binary with an explicit per-process user stack size.
1748///
1749/// `stack_pages` is expressed in 4 KiB pages and clamped to
1750/// [`USER_STACK_MIN_PAGES`]..=[`USER_STACK_MAX_PAGES`] (issue #64).
1751pub fn load_and_run_elf_with_stack(
1752    elf_data: &[u8],
1753    name: &'static str,
1754    extra_args: &[&str],
1755    seed_caps: &[Capability],
1756    stack_pages: usize,
1757) -> Result<TaskId, &'static str> {
1758    let task = load_elf_task_inner(elf_data, name, extra_args, seed_caps, stack_pages)?;
1759    let task_id = task.id;
1760    crate::process::add_task(task);
1761    Ok(task_id)
1762}
1763
1764/// Thin public wrapper that keeps the existing API stable.
1765pub fn load_elf_task_with_caps(
1766    elf_data: &[u8],
1767    name: &'static str,
1768    seed_caps: &[Capability],
1769) -> Result<Arc<Task>, &'static str> {
1770    load_elf_task_inner(elf_data, name, &[], seed_caps, USER_STACK_PAGES)
1771}
1772
1773/// Performs the load and run elf with caps operation.
1774pub fn load_and_run_elf_with_caps(
1775    elf_data: &[u8],
1776    name: &'static str,
1777    seed_caps: &[Capability],
1778) -> Result<TaskId, &'static str> {
1779    log::trace!(
1780        "[trace][elf] load_and_run_elf enter name={} size={}",
1781        name,
1782        elf_data.len()
1783    );
1784    let task = load_elf_task_inner(elf_data, name, &[], seed_caps, USER_STACK_PAGES)?;
1785    let task_id = task.id;
1786    let runtime_entry = task
1787        .trampoline_entry
1788        .load(core::sync::atomic::Ordering::Acquire);
1789    let boot_stack_top = task
1790        .trampoline_stack_top
1791        .load(core::sync::atomic::Ordering::Acquire);
1792    log::trace!(
1793        "[trace][elf] load_and_run_elf add_task begin tid={} entry={:#x}",
1794        task_id.as_u64(),
1795        runtime_entry
1796    );
1797    crate::process::add_task(task);
1798    log::trace!(
1799        "[trace][elf] load_and_run_elf add_task done tid={}",
1800        task_id.as_u64()
1801    );
1802
1803    log::info!(
1804        "[elf] Task '{}' created: entry={:#x}, stack_top={:#x}",
1805        name,
1806        runtime_entry,
1807        boot_stack_top,
1808    );
1809
1810    Ok(task_id)
1811}
1812
1813const AT_PHDR: u64 = 3;
1814const AT_PHENT: u64 = 4;
1815const AT_PHNUM: u64 = 5;
1816const AT_PAGESZ: u64 = 6;
1817const AT_BASE: u64 = 7;
1818const AT_ENTRY: u64 = 9;
1819const AT_RANDOM: u64 = 25;
1820
1821fn generate_aux_random_seed() -> [u8; 16] {
1822    let mut seed = [0u8; 16];
1823    crate::e9_mark!(b'1');
1824    crate::entropy::fill_random(&mut seed);
1825    crate::e9_mark!(b'2');
1826    seed
1827}
1828
1829/// Performs the push auxv operation.
1830fn push_auxv(user_as: &AddressSpace, sp: &mut u64, tag: u64, val: u64) -> Result<(), &'static str> {
1831    *sp -= 8;
1832    write_user_u64(user_as, *sp, val)?;
1833    *sp -= 8;
1834    write_user_u64(user_as, *sp, tag)?;
1835    Ok(())
1836}
1837
1838/// Performs the setup boot user stack operation.
1839/// Sets up the initial user-space stack for a freshly loaded ELF task.
1840///
1841/// Stack layout (low addr at bottom = first word read by `_start`):
1842/// ```
1843/// [sp+0]             argc
1844/// [sp+8]             argv[0] ptr  (program name)
1845/// [sp+8*(2..=argc)]  argv[1..] ptrs  (extra_args)
1846/// [sp+8*(argc+1)]    NULL  (argv terminator)
1847/// [sp+8*(argc+2)]    NULL  (envp terminator)
1848/// ...                auxv pairs
1849/// ```
1850fn setup_boot_user_stack(
1851    user_as: &AddressSpace,
1852    name: &str,
1853    extra_args: &[&str],
1854    phdr_vaddr: u64,
1855    phent: u16,
1856    phnum: u16,
1857    program_entry: u64,
1858    interp_base: Option<u64>,
1859    stack_base: u64,
1860    stack_top: u64,
1861) -> Result<u64, &'static str> {
1862    let mut sp = stack_top;
1863
1864    // Write argv[0] = program name (null-terminated)
1865    let name_nul_len = (name.len() + 1) as u64;
1866    sp -= name_nul_len;
1867    if sp < stack_base {
1868        return Err("User stack overflow during boot stack setup");
1869    }
1870    let argv0_ptr = sp;
1871    write_user_mapped_bytes(user_as, sp, name.as_bytes())?;
1872    write_user_mapped_bytes(user_as, sp + name.len() as u64, &[0])?;
1873
1874    // Write extra arg strings and record their user-space pointers
1875    let mut extra_ptrs: alloc::vec::Vec<u64> = alloc::vec::Vec::new();
1876    // Fallible pre-allocation: argv can be large; fail the load instead of
1877    // aborting on OOM (issue #67). Exact capacity => pushes never realloc.
1878    extra_ptrs
1879        .try_reserve_exact(extra_args.len())
1880        .map_err(|_| "User stack overflow during boot stack setup")?;
1881    for &arg in extra_args.iter() {
1882        let arg_nul_len = (arg.len() + 1) as u64;
1883        sp -= arg_nul_len;
1884        if sp < stack_base {
1885            return Err("User stack overflow during boot stack setup");
1886        }
1887        extra_ptrs.push(sp);
1888        write_user_mapped_bytes(user_as, sp, arg.as_bytes())?;
1889        write_user_mapped_bytes(user_as, sp + arg.len() as u64, &[0])?;
1890    }
1891
1892    sp &= !0xF;
1893    sp -= 16;
1894    if sp < stack_base {
1895        return Err("User stack overflow during boot stack setup");
1896    }
1897    let random_ptr = sp;
1898    let random_seed = generate_aux_random_seed();
1899    write_user_mapped_bytes(user_as, sp, &random_seed)?;
1900    let auxv_pairs = if interp_base.is_some() { 8u64 } else { 7u64 };
1901    // argc(1) + argv[0..=N](1+N) + argv_NULL(1) + envp_NULL(1) + auxv(pairs*2)
1902    let stack_words = 4u64 + extra_args.len() as u64 + auxv_pairs * 2;
1903    let align_pad = (0u64.wrapping_sub(stack_words * 8)) & 0xF;
1904    sp -= align_pad;
1905
1906    // Auxv (written high-to-low since push_auxv decrements sp)
1907    push_auxv(user_as, &mut sp, 0, 0)?; // AT_NULL
1908    push_auxv(user_as, &mut sp, AT_RANDOM, random_ptr)?;
1909    push_auxv(user_as, &mut sp, AT_ENTRY, program_entry)?;
1910    if let Some(base) = interp_base {
1911        push_auxv(user_as, &mut sp, AT_BASE, base)?;
1912    }
1913    push_auxv(user_as, &mut sp, AT_PAGESZ, 4096)?;
1914    push_auxv(user_as, &mut sp, AT_PHNUM, phnum as u64)?;
1915    push_auxv(user_as, &mut sp, AT_PHENT, phent as u64)?;
1916    // AT_PHDR is optional: 0 means "not mapped" (static binaries with the
1917    // program header array outside any PT_LOAD).
1918    if phdr_vaddr != 0 {
1919        push_auxv(user_as, &mut sp, AT_PHDR, phdr_vaddr)?;
1920    }
1921
1922    // envp NULL terminator
1923    sp -= 8;
1924    write_user_u64(user_as, sp, 0)?;
1925
1926    // argv NULL terminator
1927    sp -= 8;
1928    write_user_u64(user_as, sp, 0)?;
1929
1930    // Extra argv pointers in reverse (last arg highest in stack, first arg lowest)
1931    for &ptr in extra_ptrs.iter().rev() {
1932        sp -= 8;
1933        write_user_u64(user_as, sp, ptr)?;
1934    }
1935
1936    // argv[0] = program name
1937    sp -= 8;
1938    write_user_u64(user_as, sp, argv0_ptr)?;
1939
1940    // argc = 1 (name) + extra_args
1941    sp -= 8;
1942    write_user_u64(user_as, sp, 1u64 + extra_args.len() as u64)?;
1943
1944    debug_assert_eq!(sp & 0xF, 0);
1945    Ok(sp)
1946}
1947
1948/// Internal ELF task builder used by all public loading APIs.
1949/// `extra_args` are written to the user stack as argv[1..] after the program name.
1950fn load_elf_task_inner(
1951    elf_data: &[u8],
1952    name: &'static str,
1953    extra_args: &[&str],
1954    seed_caps: &[Capability],
1955    stack_pages: usize,
1956) -> Result<Arc<Task>, &'static str> {
1957    if !(USER_STACK_MIN_PAGES..=USER_STACK_MAX_PAGES).contains(&stack_pages) {
1958        return Err("User stack size out of range");
1959    }
1960    log::trace!(
1961        "[trace][elf] load_elf_task enter name={} size={}",
1962        name,
1963        elf_data.len()
1964    );
1965    log::info!("[elf] Loading ELF '{}'...", name);
1966
1967    // Step 1: Parse and validate ELF header
1968    log::trace!("[trace][elf] load_elf_task parse_header begin");
1969    let header = match parse_header(elf_data) {
1970        Ok(h) => h,
1971        Err(e) => {
1972            log::error!("[elf] parse_header FAILED for '{}': {}", name, e);
1973            return Err(e);
1974        }
1975    };
1976    log::trace!(
1977        "[trace][elf] load_elf_task parse_header ok type={}",
1978        if header.e_type == ET_DYN {
1979            "ET_DYN"
1980        } else {
1981            "ET_EXEC"
1982        }
1983    );
1984    // Step 2: Create user address space
1985    log::trace!("[trace][elf] load_elf_task user_as begin");
1986    let user_as = Arc::new(AddressSpace::new_user()?);
1987    log::trace!("[trace][elf] load_elf_task user_as done");
1988
1989    let phdrs: Vec<Elf64Phdr> = try_collect_exact(program_headers(elf_data, &header))?;
1990    let interp_path = parse_interp_path(elf_data, &phdrs)?;
1991    let (load_bias, entry) = compute_load_bias_and_entry(&user_as, &header, &phdrs)?;
1992    let phdr_vaddr = find_relocated_phdr_vaddr(&header, &phdrs, load_bias)?;
1993
1994    let phnum = header.e_phnum;
1995    log::trace!(
1996        "[trace][elf] load_elf_task layout entry={:#x} bias={:#x} phdrs={}",
1997        entry,
1998        load_bias,
1999        phnum
2000    );
2001    log::info!(
2002        "[elf] ELF '{}': type={}, entry={:#x}, bias={:#x}, {} program headers",
2003        name,
2004        if header.e_type == ET_DYN {
2005            "ET_DYN"
2006        } else {
2007            "ET_EXEC"
2008        },
2009        entry,
2010        load_bias,
2011        phnum,
2012    );
2013
2014    // Step 3: Load all PT_LOAD segments
2015    let mut load_count = 0u32;
2016    for phdr in phdrs.iter() {
2017        if phdr.p_type == PT_LOAD && phdr.p_memsz != 0 {
2018            load_segment(&user_as, elf_data, phdr, load_bias)?;
2019            load_count += 1;
2020        }
2021    }
2022    if interp_path.is_none() {
2023        apply_dynamic_relocations(&user_as, &phdrs, header.e_type, load_bias)?;
2024    }
2025
2026    // PT_GNU_RELRO: mark the RELRO range read-only after relocations.
2027    if let Some(relro) = phdrs.iter().find(|ph| ph.p_type == PT_GNU_RELRO) {
2028        if relro.p_memsz > 0 {
2029            let relro_start = relro.p_vaddr.wrapping_add(load_bias) & !0xFFF;
2030            // A partial trailing page may contain writable data outside RELRO.
2031            // Protect only through the last complete page, never round up.
2032            let relro_end = (relro.p_vaddr.wrapping_add(load_bias) + relro.p_memsz) & !0xFFF;
2033            if relro_end > relro_start && relro_end <= USER_ADDR_MAX {
2034                let ro_flags = VmaFlags {
2035                    readable: true,
2036                    writable: false,
2037                    executable: false,
2038                    user_accessible: true,
2039                };
2040                let relro_pages = ((relro_end - relro_start) / 4096) as usize;
2041                apply_segment_permissions(&user_as, relro_start, relro_pages, ro_flags)?;
2042                log::debug!(
2043                    "[elf] PT_GNU_RELRO: {:#x}..{:#x} made read-only",
2044                    relro_start,
2045                    relro_end
2046                );
2047            }
2048        }
2049    }
2050
2051    log::trace!(
2052        "[trace][elf] load_elf_task segments_done count={} has_interp={}",
2053        load_count,
2054        interp_path.is_some()
2055    );
2056    log::info!("[elf] Loaded {} PT_LOAD segment(s)", load_count);
2057
2058    let mut runtime_entry = entry;
2059    let mut interp_base: Option<u64> = None;
2060    if let Some(path) = interp_path {
2061        let interp_data = read_elf_from_vfs(path)?;
2062        let interp_header = parse_header(&interp_data)?;
2063        let interp_phdrs: Vec<Elf64Phdr> =
2064            try_collect_exact(program_headers(&interp_data, &interp_header))?;
2065        if parse_interp_path(&interp_data, &interp_phdrs)?.is_some() {
2066            return Err("Nested PT_INTERP is not supported");
2067        }
2068        let (interp_bias, interp_entry) =
2069            compute_load_bias_and_entry(&user_as, &interp_header, &interp_phdrs)?;
2070        let (interp_min_vaddr, _) = compute_load_bounds(&interp_phdrs)?;
2071        let mut interp_load_count = 0u32;
2072        for phdr in interp_phdrs.iter() {
2073            if phdr.p_type == PT_LOAD && phdr.p_memsz != 0 {
2074                load_segment(&user_as, &interp_data, phdr, interp_bias)?;
2075                interp_load_count += 1;
2076            }
2077        }
2078        apply_dynamic_relocations(&user_as, &interp_phdrs, interp_header.e_type, interp_bias)?;
2079        runtime_entry = interp_entry;
2080        interp_base = Some(interp_min_vaddr.saturating_add(interp_bias));
2081        log::info!(
2082            "[elf] PT_INTERP '{}' loaded: {} PT_LOAD, entry={:#x}",
2083            path,
2084            interp_load_count,
2085            runtime_entry
2086        );
2087    }
2088
2089    // TLS setup (Variant II: data at negative offsets from FS:0)
2090    let mut user_fs_base_val = 0u64;
2091    if let Some(tls) = phdrs.iter().find(|p| p.p_type == PT_TLS) {
2092        let tls_memsz = tls.p_memsz;
2093        let tls_filesz = tls.p_filesz;
2094        let tls_align = core::cmp::max(tls.p_align, 8).next_power_of_two();
2095        let aligned_memsz = (tls_memsz + tls_align - 1) & !(tls_align - 1);
2096        let total_size = aligned_memsz + 8;
2097        let n_tls_pages = ((total_size + 4095) / 4096) as usize;
2098        let tls_flags = VmaFlags {
2099            readable: true,
2100            writable: true,
2101            executable: false,
2102            user_accessible: true,
2103        };
2104        let tls_base = user_as
2105            .find_free_vma_range(0x7FFF_E000_0000, n_tls_pages, VmaPageSize::Small)
2106            .ok_or("No space for TLS block")?;
2107        user_as.map_region(
2108            tls_base,
2109            n_tls_pages,
2110            tls_flags,
2111            VmaType::Anonymous,
2112            VmaPageSize::Small,
2113        )?;
2114        if tls_filesz > 0 {
2115            let src_off = tls.p_offset as usize;
2116            let src_end = src_off
2117                .checked_add(tls_filesz as usize)
2118                .ok_or("PT_TLS offset+filesz overflows")?;
2119            if src_end > elf_data.len() {
2120                return Err("PT_TLS file data extends past ELF");
2121            }
2122            write_user_mapped_bytes(&user_as, tls_base, &elf_data[src_off..src_end])?;
2123        }
2124        let tp = tls_base + aligned_memsz;
2125        write_user_u64(&user_as, tp, tp)?;
2126        user_fs_base_val = tp;
2127    }
2128
2129    // Step 4: Map user stack
2130    // Per-process ASLR (issue #62): on top of the boot-time KASLR offset, each
2131    // image draws its own page-aligned jitter so two processes never share
2132    // identical stack addresses. The stack size is configurable per process
2133    // (issue #64). A single guard page below the stack is left unmapped :
2134    // underflow faults instead of silently corrupting neighbours.
2135    let stack_base = crate::kaslr::stack_base_with_jitter(crate::kaslr::draw_stack_jitter());
2136    let stack_top = crate::kaslr::stack_top_for(stack_base, stack_pages);
2137    // PT_GNU_STACK with PF_X means the stack should be executable (legacy ABI).
2138    // Without PT_GNU_STACK or without PF_X, the stack is NX (modern default).
2139    //
2140    // Linux semantics: when several PT_GNU_STACK entries are present (rare,
2141    // but happens with hand-crafted or malicious binaries), only the *last*
2142    // one counts.  Using `.any(...)` would honour whichever PT_GNU_STACK
2143    // appears first and silently let an executable stack slip in even if
2144    // a later entry resets PF_X to 0.  Iterate from the back to match the
2145    // documented Linux behaviour.
2146    let stack_exec = phdrs
2147        .iter()
2148        .rev()
2149        .find_map(|ph| {
2150            if ph.p_type == PT_GNU_STACK {
2151                Some((ph.p_flags & PF_X) != 0)
2152            } else {
2153                None
2154            }
2155        })
2156        .unwrap_or(false);
2157    let stack_flags = VmaFlags {
2158        readable: true,
2159        writable: true,
2160        executable: stack_exec,
2161        user_accessible: true,
2162    };
2163    user_as.map_region(
2164        stack_base,
2165        stack_pages,
2166        stack_flags,
2167        VmaType::Stack,
2168        VmaPageSize::Small,
2169    )?;
2170    // Guard page: a single unmapped page below the stack.  Any stack
2171    // underflow (push past the bottom) hits this and faults.  The page
2172    // is intentionally left unmapped : no VMA, no PTE.
2173    log::debug!(
2174        "[elf] User stack: {:#x}..{:#x} ({} pages), guard at {:#x}",
2175        stack_base,
2176        stack_top,
2177        stack_pages,
2178        user_stack_guard(),
2179    );
2180
2181    // Stack canary (issue #63): a random per-process value occupies the
2182    // top word of the stack; all boot data lives strictly below it. The
2183    // kernel re-checks it at task exit (Task::verify_user_stack_canary).
2184    let mut canary_bytes = [0u8; 8];
2185    crate::e9_mark!(b'3');
2186    crate::entropy::fill_random(&mut canary_bytes);
2187    crate::e9_mark!(b'4');
2188    let stack_canary = u64::from_le_bytes(canary_bytes) | 1; // never 0
2189    crate::e9_mark!(b'5');
2190    write_user_u64(&user_as, stack_top - 8, stack_canary)?;
2191    crate::e9_mark!(b'6');
2192
2193    let boot_sp = setup_boot_user_stack(
2194        &user_as,
2195        name,
2196        extra_args,
2197        phdr_vaddr,
2198        header.e_phentsize,
2199        header.e_phnum,
2200        entry,
2201        interp_base,
2202        stack_base,
2203        stack_top - 8, // data must stay below the canary slot
2204    )?;
2205
2206    // Step 5: Create kernel task : trampoline params are stored inside the task
2207    // itself so that concurrent SMP execution of multiple trampolines is safe.
2208    log::trace!(
2209        "[trace][elf] load_elf_task kstack_begin size={}",
2210        Task::DEFAULT_STACK_SIZE
2211    );
2212    let kernel_stack = KernelStack::allocate(Task::DEFAULT_STACK_SIZE)?;
2213    log::trace!(
2214        "[trace][elf] load_elf_task kstack_done virt={:#x} top={:#x}",
2215        kernel_stack.virt_base.as_u64(),
2216        kernel_stack.virt_base.as_u64() + kernel_stack.size as u64
2217    );
2218    let context = CpuContext::new(elf_ring3_trampoline as *const () as u64, &kernel_stack);
2219    let (pid, tid, tgid) = Task::allocate_process_ids();
2220    let fpu_state = crate::process::task::ExtendedState::new();
2221    let xcr0_mask = fpu_state.xcr0_mask;
2222
2223    let task = Arc::new(Task {
2224        id: TaskId::new(),
2225        pid,
2226        tid,
2227        tgid,
2228        pgid: core::sync::atomic::AtomicU32::new(pid),
2229        sid: core::sync::atomic::AtomicU32::new(pid),
2230        uid: core::sync::atomic::AtomicU32::new(0),
2231        euid: core::sync::atomic::AtomicU32::new(0),
2232        gid: core::sync::atomic::AtomicU32::new(0),
2233        egid: core::sync::atomic::AtomicU32::new(0),
2234        state: core::sync::atomic::AtomicU8::new(TaskState::Ready as u8),
2235        priority: TaskPriority::Normal,
2236        context: SyncUnsafeCell::new(context),
2237        resume_kind: SyncUnsafeCell::new(ResumeKind::RetFrame),
2238        interrupt_rsp: core::sync::atomic::AtomicU64::new(0),
2239        kernel_stack,
2240        user_stack: Some(crate::process::task::UserStack {
2241            virt_base: x86_64::VirtAddr::new(stack_base),
2242            size: stack_pages * 4096,
2243        }),
2244        stack_canary: core::sync::atomic::AtomicU64::new(stack_canary),
2245        stack_canary_addr: core::sync::atomic::AtomicU64::new(stack_top - 8),
2246        kernel_stack_user: SyncUnsafeCell::new(None),
2247        name,
2248        process: Arc::new(crate::process::process::Process::new(pid, user_as)),
2249        pending_signals: super::signal::SignalSet::new(),
2250        blocked_signals: super::signal::SignalSet::new(),
2251        irq_signal_delivery_blocked: core::sync::atomic::AtomicBool::new(false),
2252        signal_stack: SyncUnsafeCell::new(None),
2253        itimers: super::timer::ITimers::new(),
2254        wake_pending: core::sync::atomic::AtomicBool::new(false),
2255        wake_deadline_ns: core::sync::atomic::AtomicU64::new(0),
2256        trampoline_entry: core::sync::atomic::AtomicU64::new(runtime_entry),
2257        trampoline_stack_top: core::sync::atomic::AtomicU64::new(boot_sp),
2258        trampoline_arg0: core::sync::atomic::AtomicU64::new(0),
2259        ticks: core::sync::atomic::AtomicU64::new(0),
2260        sched_policy: crate::process::task::SyncUnsafeCell::new(Task::default_sched_policy(
2261            TaskPriority::Normal,
2262        )),
2263        home_cpu: core::sync::atomic::AtomicUsize::new(usize::MAX),
2264        last_cpu: core::sync::atomic::AtomicUsize::new(usize::MAX),
2265        affinity_mask: core::sync::atomic::AtomicU64::new(0),
2266        vruntime: core::sync::atomic::AtomicU64::new(0),
2267        fair_rq_generation: core::sync::atomic::AtomicU64::new(0),
2268        fair_on_rq: core::sync::atomic::AtomicBool::new(false),
2269        clear_child_tid: core::sync::atomic::AtomicU64::new(0),
2270        robust_list_head: core::sync::atomic::AtomicU64::new(0),
2271        robust_list_len: core::sync::atomic::AtomicUsize::new(0),
2272        user_fs_base: core::sync::atomic::AtomicU64::new(user_fs_base_val),
2273        fpu_state: crate::process::task::SyncUnsafeCell::new(fpu_state),
2274        xcr0_mask: core::sync::atomic::AtomicU64::new(xcr0_mask),
2275        rt_link: intrusive_collections::LinkedListLink::new(),
2276        rt_budget_remaining: core::sync::atomic::AtomicU64::new(0),
2277        rt_budget_period_start: core::sync::atomic::AtomicU64::new(0),
2278        rt_degraded: core::sync::atomic::AtomicBool::new(false),
2279        fair_wait_ticks: core::sync::atomic::AtomicU64::new(0),
2280    });
2281
2282    log::trace!(
2283        "[trace][elf] load_elf_task task_built tid={} pid={} entry={:#x} sp={:#x}",
2284        task.id.as_u64(),
2285        task.pid,
2286        runtime_entry,
2287        boot_sp
2288    );
2289    // Seed capabilities into the new task (before scheduling).
2290    let mut bootstrap_handle: Option<u64> = None;
2291    if !seed_caps.is_empty() {
2292        let caps = unsafe { &mut *task.process.capabilities.get() };
2293        for cap in seed_caps {
2294            let id = caps.insert(cap.clone());
2295            if bootstrap_handle.is_none()
2296                && cap.resource_type == crate::capability::ResourceType::Volume
2297            {
2298                bootstrap_handle = Some(id.as_u64());
2299            }
2300        }
2301    }
2302
2303    // Setup stdin/stdout/stderr (fd 0/1/2) pointing to /dev/console
2304    // SAFETY: task is not yet scheduled, exclusive access to fd_table
2305    {
2306        let fd_table = unsafe { &mut *task.process.fd_table.get() };
2307        crate::vfs::console_scheme::setup_stdio(fd_table);
2308    }
2309
2310    if let Some(h) = bootstrap_handle {
2311        // Program entry will see this in its first argument register (RDI).
2312        task.trampoline_arg0
2313            .store(h, core::sync::atomic::Ordering::Release);
2314    }
2315
2316    task.seed_interrupt_frame(crate::syscall::SyscallFrame {
2317        r15: 0,
2318        r14: 0,
2319        r13: 0,
2320        r12: 0,
2321        rbp: 0,
2322        rbx: 0,
2323        r11: USER_RFLAGS,
2324        r10: 0,
2325        r9: 0,
2326        r8: 0,
2327        rsi: 0,
2328        rdi: task
2329            .trampoline_arg0
2330            .load(core::sync::atomic::Ordering::Acquire),
2331        rdx: 0,
2332        rcx: runtime_entry,
2333        rax: 0,
2334        iret_rip: runtime_entry,
2335        iret_cs: crate::arch::gdt::user_code_selector().0 as u64,
2336        iret_rflags: USER_RFLAGS,
2337        iret_rsp: boot_sp,
2338        iret_ss: crate::arch::gdt::user_data_selector().0 as u64,
2339    });
2340
2341    {
2342        let arc_data_ptr = alloc::sync::Arc::as_ptr(&task) as usize;
2343        let fpu_ptr = task.fpu_state.get() as usize;
2344        if let Some(cur) = crate::process::scheduler::current_task_clone() {
2345            let cur_data_ptr = alloc::sync::Arc::as_ptr(&cur) as usize;
2346            let cur_strong = alloc::sync::Arc::strong_count(&cur);
2347            log::info!(
2348                "[elf] Task '{}' prepared: entry={:#x}, stack_top={:#x} \
2349                 new_arc={:#x} new_fpu={:#x} cur_arc={:#x} cur_strong={}",
2350                name,
2351                runtime_entry,
2352                boot_sp,
2353                arc_data_ptr,
2354                fpu_ptr,
2355                cur_data_ptr,
2356                cur_strong,
2357            );
2358        } else {
2359            log::info!(
2360                "[elf] Task '{}' prepared: entry={:#x}, stack_top={:#x} \
2361                 new_arc={:#x} new_fpu={:#x} (no current task)",
2362                name,
2363                runtime_entry,
2364                boot_sp,
2365                arc_data_ptr,
2366                fpu_ptr,
2367            );
2368        }
2369    }
2370
2371    Ok(task)
2372}
2373
2374/// Load an ELF binary into the provided address space.
2375/// Returns the entry point address.
2376///
2377/// # Duplication note (audit 2026-09-07)
2378///
2379/// This function duplicates the parsing / PT_LOAD / RELRO / PT_INTERP /
2380/// TLS-extraction steps of [`load_elf_task_inner`].  Asterinas avoids the
2381/// duplication by separating `load_elf_to_vmar` (image-loading sink)
2382/// from `do_execve` (process setup).  In Strat9-OS the two paths have
2383/// diverged over time : this function does *not* allocate the TLS block
2384/// itself (the caller in [`crate::syscall::exec`] does), and it does *not*
2385/// build a Task struct.
2386///
2387/// A future refactor should extract the common `parse → load_segments →
2388/// RELRO → interp-load → tls-extract` sequence into a single private
2389/// helper parameterised by an `ImageSink` trait (Task-bound vs bare AS),
2390/// mirroring Asterinas.  Until then, every loader-side fix must be applied
2391/// to **both** functions.
2392pub fn load_elf_image(
2393    elf_data: &[u8],
2394    user_as: &AddressSpace,
2395) -> Result<LoadedElfInfo, &'static str> {
2396    let header = match parse_header(elf_data) {
2397        Ok(h) => h,
2398        Err(e) => {
2399            log::error!("[elf] load_elf_image parse_header FAILED: {}", e);
2400            return Err(e);
2401        }
2402    };
2403    let phdrs: Vec<Elf64Phdr> = try_collect_exact(program_headers(elf_data, &header))?;
2404    let interp_path = parse_interp_path(elf_data, &phdrs)?;
2405    let (load_bias, entry) = compute_load_bias_and_entry(user_as, &header, &phdrs)?;
2406    let phdr_vaddr = find_relocated_phdr_vaddr(&header, &phdrs, load_bias)?;
2407
2408    for phdr in phdrs.iter() {
2409        if phdr.p_type == PT_LOAD && phdr.p_memsz != 0 {
2410            load_segment(user_as, elf_data, phdr, load_bias)?;
2411        }
2412    }
2413    if interp_path.is_none() {
2414        apply_dynamic_relocations(user_as, &phdrs, header.e_type, load_bias)?;
2415    }
2416
2417    // PT_GNU_RELRO: mark the RELRO range read-only after relocations.
2418    if let Some(relro) = phdrs.iter().find(|ph| ph.p_type == PT_GNU_RELRO) {
2419        if relro.p_memsz > 0 {
2420            let relro_start = relro.p_vaddr.wrapping_add(load_bias) & !0xFFF;
2421            // A partial trailing page may contain writable data outside RELRO.
2422            // Protect only through the last complete page, never round up.
2423            let relro_end = (relro.p_vaddr.wrapping_add(load_bias) + relro.p_memsz) & !0xFFF;
2424            if relro_end > relro_start && relro_end <= USER_ADDR_MAX {
2425                let ro_flags = VmaFlags {
2426                    readable: true,
2427                    writable: false,
2428                    executable: false,
2429                    user_accessible: true,
2430                };
2431                let relro_pages = ((relro_end - relro_start) / 4096) as usize;
2432                apply_segment_permissions(user_as, relro_start, relro_pages, ro_flags)?;
2433                log::debug!(
2434                    "[elf] PT_GNU_RELRO: {:#x}..{:#x} made read-only",
2435                    relro_start,
2436                    relro_end
2437                );
2438            }
2439        }
2440    }
2441
2442    let (tls_vaddr, tls_filesz, tls_memsz, tls_align) =
2443        if let Some(tls) = phdrs.iter().find(|ph| ph.p_type == PT_TLS) {
2444            let align = core::cmp::max(tls.p_align, 1).next_power_of_two();
2445            (
2446                tls.p_vaddr.saturating_add(load_bias),
2447                tls.p_filesz,
2448                tls.p_memsz,
2449                align,
2450            )
2451        } else {
2452            (0, 0, 0, 1)
2453        };
2454
2455    let mut runtime_entry = entry;
2456    let mut interp_base = None;
2457    if let Some(path) = interp_path {
2458        let interp_data = read_elf_from_vfs(path)?;
2459        let interp_header = parse_header(&interp_data)?;
2460        let interp_phdrs: Vec<Elf64Phdr> =
2461            try_collect_exact(program_headers(&interp_data, &interp_header))?;
2462        if parse_interp_path(&interp_data, &interp_phdrs)?.is_some() {
2463            return Err("Nested PT_INTERP is not supported");
2464        }
2465        let (interp_bias, interp_entry) =
2466            compute_load_bias_and_entry(user_as, &interp_header, &interp_phdrs)?;
2467        let (interp_min_vaddr, _) = compute_load_bounds(&interp_phdrs)?;
2468        for phdr in interp_phdrs.iter() {
2469            if phdr.p_type == PT_LOAD && phdr.p_memsz != 0 {
2470                load_segment(user_as, &interp_data, phdr, interp_bias)?;
2471            }
2472        }
2473        apply_dynamic_relocations(user_as, &interp_phdrs, interp_header.e_type, interp_bias)?;
2474        runtime_entry = interp_entry;
2475        interp_base = Some(interp_min_vaddr.saturating_add(interp_bias));
2476    }
2477
2478    // PT_GNU_STACK: Linux semantics dictate that only the *last* PT_GNU_STACK
2479    // entry counts (see `load_elf_task_inner` for the rationale).  Walk the
2480    // phdrs in reverse to find the last PT_GNU_STACK.
2481    let stack_exec = phdrs
2482        .iter()
2483        .rev()
2484        .find_map(|ph| {
2485            if ph.p_type == PT_GNU_STACK {
2486                Some((ph.p_flags & PF_X) != 0)
2487            } else {
2488                None
2489            }
2490        })
2491        .unwrap_or(false);
2492
2493    Ok(LoadedElfInfo {
2494        runtime_entry,
2495        program_entry: entry,
2496        phdr_vaddr,
2497        phent: header.e_phentsize,
2498        phnum: header.e_phnum,
2499        interp_base,
2500        tls_vaddr,
2501        tls_filesz,
2502        tls_memsz,
2503        tls_align,
2504        stack_exec,
2505    })
2506}
2507
2508/// Reads user mapped bytes pub.
2509pub fn read_user_mapped_bytes_pub(
2510    user_as: &AddressSpace,
2511    vaddr: u64,
2512    out: &mut [u8],
2513) -> Result<(), &'static str> {
2514    read_user_mapped_bytes(user_as, vaddr, out)
2515}
2516
2517/// Writes user mapped bytes pub.
2518pub fn write_user_mapped_bytes_pub(
2519    user_as: &AddressSpace,
2520    vaddr: u64,
2521    src: &[u8],
2522) -> Result<(), &'static str> {
2523    write_user_mapped_bytes(user_as, vaddr, src)
2524}
2525
2526/// Writes user u64 pub.
2527pub fn write_user_u64_pub(
2528    user_as: &AddressSpace,
2529    vaddr: u64,
2530    value: u64,
2531) -> Result<(), &'static str> {
2532    write_user_u64(user_as, vaddr, value)
2533}