Skip to main content

strat9_kernel/process/scheduler/
runtime_ops.rs

1use super::*;
2
3static FINISH_INTERRUPT_TRACE_BUDGET: core::sync::atomic::AtomicU32 =
4    core::sync::atomic::AtomicU32::new(32);
5
6/// Initialize the scheduler
7pub fn init_scheduler() {
8    // Build scheduler state only for CPUs that are actually online. Using the
9    // registered per-CPU count here can strand runnable tasks on AP slots that
10    // never reached the scheduler gate.
11    let cpu_count = crate::arch::smp::cpu_count().max(1);
12    crate::serial_println!(
13        "[trace][sched] init_scheduler enter cpu_count={}",
14        cpu_count
15    );
16    // Build the scheduler outside the global scheduler lock to avoid
17    // lock-order inversions (`GLOBAL_SCHED_STATE -> allocator`) during task/stack
18    // allocation in `GlobalSchedState::new`.
19    let new_sched = GlobalSchedState::new();
20    // Initialize per-CPU scheduler state for each active CPU.
21    for i in 0..cpu_count {
22        let cpu_sched = super::core_impl::create_cpu_scheduler(i);
23        *LOCAL_SCHEDULERS[i].lock() = Some(cpu_sched);
24    }
25    crate::serial_println!("[trace][sched] init_scheduler new() done");
26
27    // Race/corruption diagnostic: register scheduler lock for E9 LOCK-A/LOCK-R traces.
28    crate::sync::debug_set_trace_lock_addr(debug_scheduler_lock_addr());
29
30    let mut scheduler = GLOBAL_SCHED_STATE.lock();
31    *scheduler = Some(new_sched);
32    drop(scheduler); // Release the lock
33
34    // Only initialize legacy PIT if APIC timer is not active
35    if !timer::is_apic_timer_active() {
36        timer::init_pit(100); // 100Hz = 10ms interval for quantum
37        log::info!("Scheduler: using legacy PIT timer (100Hz)");
38    } else {
39        log::info!("Scheduler: using APIC timer (100Hz)");
40    }
41    crate::serial_println!("[trace][sched] init_scheduler exit");
42}
43
44/// Add a task to the scheduler
45pub fn add_task(task: Arc<Task>) {
46    let tid = task.id;
47    crate::serial_force_println!(
48        "[trace][sched] add_task enter tid={} name={}",
49        tid.as_u64(),
50        task.name
51    );
52    crate::serial_force_println!(
53        "[trace][sched] lock addrs sched={:#x} slab={:#x} buddy={:#x}",
54        crate::process::scheduler::debug_scheduler_lock_addr(),
55        crate::memory::heap::debug_slab_lock_addr(),
56        crate::memory::buddy::debug_buddy_lock_addr()
57    );
58    let mut spins = 0usize;
59    let mut scheduler = loop {
60        if let Some(guard) = GLOBAL_SCHED_STATE.try_lock() {
61            break guard;
62        }
63        spins = spins.saturating_add(1);
64        if spins == 2_000_000 {
65            crate::serial_force_println!(
66                "[trace][sched] add_task waiting lock tid={} owner_cpu={}",
67                tid.as_u64(),
68                GLOBAL_SCHED_STATE.owner_cpu()
69            );
70            spins = 0;
71        }
72        core::hint::spin_loop();
73    };
74    crate::serial_force_println!("[trace][sched] add_task lock acquired tid={}", tid.as_u64());
75    let ipi_to_cpu = if let Some(ref mut sched) = *scheduler {
76        crate::serial_force_println!(
77            "[trace][sched] add_task scheduler present tid={}",
78            tid.as_u64()
79        );
80        let ipi = sched.add_task(task);
81        crate::serial_force_println!("[trace][sched] add_task done tid={}", tid.as_u64());
82        ipi
83    } else {
84        crate::serial_force_println!(
85            "[trace][sched] add_task scheduler missing tid={}",
86            tid.as_u64()
87        );
88        None
89    };
90    drop(scheduler);
91    if let Some(ci) = ipi_to_cpu {
92        send_resched_ipi_to_cpu(ci);
93    }
94}
95
96/// Add a task and register a parent/child relation.
97pub fn add_task_with_parent(task: Arc<Task>, parent: TaskId) {
98    let ipi_to_cpu = {
99        let mut scheduler = GLOBAL_SCHED_STATE.lock();
100        if let Some(ref mut sched) = *scheduler {
101            sched.add_task_with_parent(task, parent)
102        } else {
103            None
104        }
105    };
106    if let Some(ci) = ipi_to_cpu {
107        send_resched_ipi_to_cpu(ci);
108    }
109}
110
111/// Start the scheduler (called from kernel_main)
112///
113/// Picks the first task and starts running it. Never returns.
114pub fn schedule() -> ! {
115    let cpu_index = current_cpu_index();
116    schedule_on_cpu(cpu_index)
117}
118
119/// Performs the schedule on cpu operation.
120pub fn schedule_on_cpu(cpu_index: usize) -> ! {
121    crate::e9_println!("BD-ENTER cpu={}", cpu_index);
122    // TEMP DEBUG raw pulse: BD-ENTER via raw marks too (formatted E9 may be silent).
123    unsafe {
124        core::arch::asm!("out 0xe9, al", in("al") b'B', options(nomem, nostack));
125        core::arch::asm!("out 0xe9, al", in("al") b'D', options(nomem, nostack));
126    }
127    // Disable interrupts for the entire critical section.
128    //
129    // On the BSP, IF may be 1 (interrupts were enabled in Phase 9).
130    // Without CLI, a timer interrupt between `pick_next_task` (which sets
131    // `current_task`) and `restore_first_task` would let `maybe_preempt()`
132    // call `switch_context` on the *init stack*, corrupting the task's
133    // `saved_rsp` and creating an infinite loop.
134    //
135    // APs already arrive here with IF=0 (from the trampoline), but the
136    // explicit CLI makes the contract clear for all callers.
137    //
138    // Interrupts are re-enabled by the RFLAGS seed (0x202, IF=1) stored in
139    // each task's bootstrap interrupt frame, so no explicit sti() is needed.
140    crate::arch::cli();
141
142    // APs may arrive here before the BSP has called init_scheduler().
143    // Spin-wait (releasing the lock each iteration) until the scheduler
144    // is initialized, then pick the first task.
145    let mut wait_iters: u64 = 0;
146    let first_task = loop {
147        let scheduler = GLOBAL_SCHED_STATE.lock();
148        if let Some(ref _sched) = *scheduler {
149            if wait_iters > 0 {
150                crate::e9_println!("BD first_task cpu={} waited={}", cpu_index, wait_iters);
151            }
152            drop(scheduler);
153            // Pick first task via LOCAL per-CPU state.
154            let idx = if cpu_index < active_cpu_count() {
155                cpu_index
156            } else {
157                0
158            };
159            unsafe {
160                core::arch::asm!("out 0xe9, al", in("al") b'M', options(nomem, nostack));
161            }
162            let mut local = LOCAL_SCHEDULERS[idx].lock();
163            unsafe {
164                core::arch::asm!("out 0xe9, al", in("al") b'N', options(nomem, nostack));
165            }
166            if let Some(ref mut cpu) = *local {
167                break super::core_impl::pick_next_task_local(cpu, idx);
168            }
169            // LOCAL not ready yet : drop and spin.
170            drop(local);
171            wait_iters = wait_iters.saturating_add(1);
172            if wait_iters == 1 || (wait_iters % 1_000_000 == 0 && wait_iters > 0) {
173                crate::e9_println!("BD-WAIT-LOCAL cpu={} iters={}", cpu_index, wait_iters);
174            }
175            core::hint::spin_loop();
176            continue;
177        }
178        // Drop lock before spinning so the BSP can initialize the scheduler.
179        drop(scheduler);
180        wait_iters = wait_iters.saturating_add(1);
181        if wait_iters == 1 || (wait_iters % 1_000_000 == 0 && wait_iters > 0) {
182            crate::e9_println!("BD-WAIT cpu={} iters={}", cpu_index, wait_iters);
183        }
184        core::hint::spin_loop();
185    }; // Lock is released here before jumping to first task
186    unsafe {
187        core::arch::asm!("out 0xe9, al", in("al") b'R', options(nomem, nostack));
188    }
189    super::task_ops::flush_deferred_silo_cleanups();
190    unsafe {
191        core::arch::asm!("out 0xe9, al", in("al") b'r', options(nomem, nostack));
192    }
193
194    // NOTE: serial_force_println! (formatted format_args!) hangs in this kernel build
195    // (known vtable issue). Raw E9 marks only in the scheduler hot path.
196
197    // Set TSS.rsp0 and SYSCALL kernel RSP for the first task
198    {
199        let stack_top =
200            first_task.kernel_stack.virt_base.as_u64() + first_task.kernel_stack.size as u64;
201        crate::arch::tss::set_kernel_stack(x86_64::VirtAddr::new(stack_top));
202        crate::arch::syscall::set_kernel_rsp(stack_top);
203        unsafe {
204            core::arch::asm!("out 0xe9, al", in("al") b'S', options(nomem, nostack));
205        }
206        crate::serial_force_println!(
207            "[trace][sched] schedule_on_cpu stacks set cpu={} rsp0={:#x}",
208            cpu_index,
209            stack_top
210        );
211    }
212
213    // Switch to the first task's address space (no-op for kernel tasks)
214    // SAFETY: The first task's address space is valid (kernel AS at boot).
215    unsafe {
216        core::arch::asm!("out 0xe9, al", in("al") b'U', options(nomem, nostack));
217    }
218    if let Err(e) = validate_task_context(&first_task) {
219        panic!(
220            "scheduler: invalid first task '{}' (id={:?}): {}",
221            first_task.name, first_task.id, e
222        );
223    }
224    crate::serial_force_println!(
225        "[trace][sched] schedule_on_cpu first_task ctx valid cpu={} tid={}",
226        cpu_index,
227        first_task.id.as_u64()
228    );
229    unsafe {
230        first_task.process.address_space_arc().switch_to();
231    }
232    unsafe {
233        core::arch::asm!("out 0xe9, al", in("al") b'K', options(nomem, nostack));
234    }
235
236    // Jump to the first task (never returns)
237    // SAFETY: The context was set up by CpuContext::new with a valid stack frame.
238    // Interrupts are disabled; the trampoline's `sti` re-enables them.
239
240    unsafe {
241        core::arch::asm!("out 0xe9, al", in("al") b'L', options(nomem, nostack));
242        // Pass the stack frame pointer (saved_rsp points TO the frame, not the context struct)
243        let frame_ptr = (*first_task.context.get()).saved_rsp as *const u64;
244        crate::process::task::do_restore_first_task(
245            frame_ptr,
246            first_task.fpu_state.get() as *const u8,
247            first_task
248                .xcr0_mask
249                .load(core::sync::atomic::Ordering::Relaxed),
250        );
251    }
252}
253
254/// Called immediately after a context switch completes (in the new task's context).
255/// This safely re-queues the previously running task now that its state is fully saved.
256///
257/// Mirrors Redox switch_finish_hook: minimal work, no serial (avoids lock contention).
258pub fn finish_switch() {
259    let _perf = super::perf_counters::PerfScope::new(
260        &super::perf_counters::CTX_SWITCH_TSC,
261        &super::perf_counters::CTX_SWITCH_COUNT,
262    );
263    let cpu_index = current_cpu_index();
264    let mut task_to_drop = None;
265    {
266        // Lock order: LOCAL only (rank 4).
267        let mut spins = 0usize;
268        let mut guard = loop {
269            if let Some(g) = LOCAL_SCHEDULERS[cpu_index].try_lock_no_irqsave() {
270                break g;
271            }
272            spins = spins.saturating_add(1);
273            if spins % 1_000_000 == 0 {
274                unsafe { core::arch::asm!("mov al, 'W'; out 0xe9, al", out("al") _) };
275            }
276            core::hint::spin_loop();
277        };
278        if let Some(ref mut cpu) = *guard {
279            if let Some(ref task) = cpu.current_task {
280                unsafe { task.process.address_space_arc().switch_to() };
281            }
282            task_to_drop = super::core_impl::drain_post_switch_local(cpu, true);
283        }
284    }
285
286    core::sync::atomic::fence(core::sync::atomic::Ordering::SeqCst);
287
288    // Process deferred work raised by the timer interrupt.
289    // This is a safe point: we're on the new task's kernel stack with no
290    // scheduler locks held. Deferred work items (interval timers, wake
291    // deadlines, per-task accounting) acquire locks internally via try_lock.
292    super::deferred_work::process_deferred_work();
293
294    // Debug invariant check after every context switch completes.
295    // Uses try_lock to avoid blocking; validates that task containers
296    // are consistent (no task in two places, no orphaned zombies, etc.).
297    super::validate_scheduler_invariants();
298
299    super::task_ops::flush_deferred_silo_cleanups();
300    drop(task_to_drop);
301}
302
303/// Finalize a preemption-driven switch once the raw timer stub has already
304/// moved onto the next task's kernel stack.
305///
306/// This mirrors Redox's `switch_finish_hook` and Maestro's `switch_finish`:
307/// requeue/drop of the old task happens only after the architectural switch is
308/// complete, so another CPU cannot steal the old task while its FPU/stack state
309/// is still in flight.
310pub fn finish_interrupt_switch() {
311    let _perf = super::perf_counters::PerfScope::new(
312        &super::perf_counters::CTX_SWITCH_TSC,
313        &super::perf_counters::CTX_SWITCH_COUNT,
314    );
315    let cpu_index = current_cpu_index();
316    let should_trace = FINISH_INTERRUPT_TRACE_BUDGET
317        .try_update(
318            core::sync::atomic::Ordering::AcqRel,
319            core::sync::atomic::Ordering::Relaxed,
320            |budget| budget.checked_sub(1),
321        )
322        .is_ok();
323    let entry_rsp0 = crate::arch::tss::kernel_stack_for(cpu_index)
324        .map(|addr| addr.as_u64())
325        .unwrap_or(0);
326    if should_trace {
327        crate::e9_println!("[ifs-enter] cpu={} rsp0={:#x}", cpu_index, entry_rsp0);
328    }
329
330    // Spin until GLOBAL_SCHED_STATE is available (released by maybe_preempt_from_interrupt
331    // before returning to this assembly stub).  We must not block with IRQs enabled
332    // because we are still inside the timer interrupt handler.  try_lock_no_irqsave
333    // requires IRQs already disabled, which is guaranteed here.
334    //
335    // The spin is bounded: if the lock is not released within MAX_IFS_SPINS
336    // iterations something is fundamentally broken (holder deadlocked, or
337    // lock corruption).  Panic so we get a stack trace instead of a silent
338    // hang.
339    // This is around few seconds on recent CPU.
340
341    // Use LOCAL lock : no spinning on GLOBAL_SCHED_STATE. The LOCAL lock for this CPU
342    // is released quickly by maybe_preempt_from_interrupt before we get here.
343    const MAX_IFS_SPINS: usize = 50_000_000;
344    let mut task_to_drop = None;
345    let mut spins = 0usize;
346    loop {
347        if let Some(mut guard) = LOCAL_SCHEDULERS[cpu_index].try_lock_no_irqsave() {
348            if let Some(ref mut cpu) = *guard {
349                // REQUEUE OLD TASK FIRST (while current AS is still active/stable)
350                task_to_drop = super::core_impl::drain_post_switch_local(cpu, false);
351
352                // NOW SWITCH TO NEW ADDRESS SPACE
353                if let Some(ref task) = cpu.current_task {
354                    let task_stack_top =
355                        task.kernel_stack.virt_base.as_u64() + task.kernel_stack.size as u64;
356                    if should_trace {
357                        crate::e9_println!(
358                            "[ifs-task] cpu={} tid={} rsp0={:#x} expected={:#x}",
359                            cpu_index,
360                            task.id.as_u64(),
361                            entry_rsp0,
362                            task_stack_top
363                        );
364                    }
365                    if should_trace && entry_rsp0 != 0 && entry_rsp0 != task_stack_top {
366                        crate::e9_println!(
367                            "[ifs-rsp0-mismatch] cpu={} tid={} rsp0={:#x} expected={:#x}",
368                            cpu_index,
369                            task.id.as_u64(),
370                            entry_rsp0,
371                            task_stack_top
372                        );
373                    }
374                    unsafe { task.process.address_space_arc().switch_to() };
375                }
376            }
377            break;
378        }
379        spins = spins.saturating_add(1);
380        if spins >= MAX_IFS_SPINS {
381            crate::e9_println!(
382                "[BUG] finish_interrupt_switch: LOCAL lock not released after {} spins, cpu={}",
383                spins,
384                cpu_index
385            );
386            panic!(
387                "finish_interrupt_switch: LOCAL lock stuck after {} spins on cpu {}",
388                spins, cpu_index
389            );
390        }
391        core::hint::spin_loop();
392    }
393
394    // Process deferred work raised by the timer interrupt.
395    // Safe point: we're on the new task's kernel stack with no scheduler
396    // locks held. This handles interval timers, wake deadlines, and
397    // per-task accounting that were deferred from the hardirq handler.
398    super::deferred_work::process_deferred_work();
399
400    let _ = task_to_drop;
401}
402
403/// Yield the current task to allow other tasks to run (cooperative).
404///
405/// Disables interrupts around the scheduler lock to prevent deadlock
406/// with the timer handler's `maybe_preempt()`.
407///
408/// Returns immediately (no-op) if preemption is disabled on this CPU.
409pub fn yield_task() {
410    // Respect the preemption guard: if a `PreemptGuard` is held, do nothing.
411    if !percpu::is_preemptible() {
412        return;
413    }
414    let _perf = super::perf_counters::PerfScope::new(
415        &super::perf_counters::SCHED_YIELD_TSC,
416        &super::perf_counters::SCHED_YIELD_COUNT,
417    );
418
419    let saved_flags = save_flags_and_cli();
420    let cpu_index = current_cpu_index();
421
422    // Lock order: LOCAL only (rank 4).
423    let switch_target = {
424        lockdep_acquire(LockRank::Local, Some(cpu_index));
425        let mut local = LOCAL_SCHEDULERS[cpu_index].lock();
426        let target = if let Some(ref mut cpu) = *local {
427            super::core_impl::yield_cpu_local(cpu, cpu_index)
428        } else {
429            None
430        };
431        lockdep_release(LockRank::Local);
432        target
433    }; // Lock released here, before the actual context switch
434
435    if let Some(ref target) = switch_target {
436        // SAFETY: no scheduler locks held, interrupts disabled.
437        unsafe {
438            crate::process::task::do_switch_context(target);
439        }
440        // finish_switch() processes deferred work internally.
441        finish_switch();
442    }
443
444    restore_flags(saved_flags);
445}
446
447/// Force a context switch away from the current task, unconditionally.
448///
449/// Unlike [`yield_task`], this function **ignores the preemption guard**.
450/// It must only be called from [`super::task_ops::exit_current_task`] after:
451///
452/// 1. The task has been marked [`TaskState::Dead`].
453/// 2. All scheduler locks have been released.
454/// 3. No spinlock-guarded per-CPU data is being accessed by this task.
455///
456/// At that point the preempt_count is irrelevant : the task will never run
457/// again, so bypassing the guard is both safe and necessary to prevent the
458/// dead task from spinning in a `hlt()` loop.
459pub fn yield_dead_task() {
460    let saved_flags = save_flags_and_cli();
461    let cpu_index = current_cpu_index();
462
463    let switch_target = {
464        let mut local = LOCAL_SCHEDULERS[cpu_index].lock();
465        if let Some(ref mut cpu) = *local {
466            super::core_impl::yield_cpu_local(cpu, cpu_index)
467        } else {
468            None
469        }
470    }; // Lock released before the context switch.
471
472    if let Some(ref target) = switch_target {
473        // SAFETY: Pointers are valid (Arc<Task> contexts kept alive by the
474        // scheduler).  Interrupts are disabled via save_flags_and_cli().
475        unsafe {
476            crate::process::task::do_switch_context(target);
477        }
478        finish_switch();
479    }
480
481    restore_flags(saved_flags);
482}
483
484#[inline]
485fn interrupt_frame_fits(task: &Arc<Task>, rsp: u64) -> bool {
486    let stack_base = task.kernel_stack.virt_base.as_u64();
487    let stack_top = stack_base + task.kernel_stack.size as u64;
488    let frame_size = core::mem::size_of::<crate::syscall::SyscallFrame>() as u64;
489    rsp >= stack_base && rsp.saturating_add(frame_size) <= stack_top
490}
491
492/// Called from the timer interrupt handler (or a resched IPI) to potentially
493/// preempt the current task.
494///
495/// This is safe to call from interrupt context because:
496/// 1. IF is already cleared by the CPU when entering the interrupt.
497/// 2. We use `try_lock()` - if the scheduler is already locked
498///    (e.g., `yield_task()` is in progress), we simply skip preemption
499///    for this tick.
500/// 3. We honour the `PreemptGuard`: if preemption is disabled, we return.
501pub fn maybe_preempt() {
502    let _perf = super::perf_counters::PerfScope::new(
503        &super::perf_counters::SCHED_PREEMPT_TSC,
504        &super::perf_counters::SCHED_PREEMPT_COUNT,
505    );
506    let cpu_index = current_cpu_index();
507    if cpu_is_valid(cpu_index) {
508        RESCHED_IPI_PENDING[cpu_index].store(false, Ordering::Release);
509    }
510
511    // Honour the preemption guard - never preempt a section that asked for it.
512    if !percpu::is_preemptible() {
513        return;
514    }
515
516    // Lock order: LOCAL only (rank 4). No GLOBAL, BLOCKED, or IDENTITY.
517    let switch_target = {
518        lockdep_acquire(LockRank::Local, Some(cpu_index));
519        let mut guard = match LOCAL_SCHEDULERS[cpu_index].try_lock_no_irqsave() {
520            Some(g) => g,
521            None => {
522                lockdep_release(LockRank::Local);
523                note_try_lock_fail_on_cpu(cpu_index);
524                return;
525            }
526        };
527        let cpu = match guard.as_mut() {
528            Some(c) => c,
529            None => {
530                lockdep_release(LockRank::Local);
531                return;
532            }
533        };
534        if take_force_resched_hint(cpu_index) {
535            cpu.need_resched = true;
536        }
537        if cpu.current_task.is_none() || !cpu.need_resched {
538            lockdep_release(LockRank::Local);
539            return;
540        }
541        if let Some(current) = cpu.current_task.as_ref() {
542            sched_trace(format_args!(
543                "cpu={} preempt request task={} rt_delta={}",
544                cpu_index,
545                current.id.as_u64(),
546                cpu.current_runtime.period_delta_ticks
547            ));
548        }
549        cpu.need_resched = false;
550        let target = super::core_impl::yield_cpu_local(cpu, cpu_index);
551        lockdep_release(LockRank::Local);
552        target
553    }; // LOCAL lock released here
554
555    if let Some(ref target) = switch_target {
556        if cpu_is_valid(cpu_index) {
557            if !FIRST_PREEMPT_LOGGED[cpu_index].swap(true, Ordering::Relaxed) {
558                let _preempt_n = CPU_PREEMPT_COUNT[cpu_index].load(Ordering::Relaxed);
559            }
560            CPU_PREEMPT_COUNT[cpu_index].fetch_add(1, Ordering::Relaxed);
561        }
562        // SAFETY: no scheduler locks held, no allocation, no IPI.
563        unsafe {
564            crate::process::task::do_switch_context(target);
565        }
566        finish_switch();
567    }
568}
569
570/// Interrupt-aware preemption path.
571///
572/// Unlike the legacy `ret`-based scheduler path, the full interrupted user
573/// context is already materialized as a `SyscallFrame` on the current kernel
574/// stack. This lets us save the outgoing task immediately, select the next
575/// runnable task under the scheduler lock, and return an `iretq`-compatible
576/// frame pointer for the raw timer stub.
577pub fn maybe_preempt_from_interrupt(
578    cpu_index: usize,
579    current_frame: &mut crate::syscall::SyscallFrame,
580) -> Option<crate::arch::idt::InterruptReturnDecision> {
581    if cpu_is_valid(cpu_index) {
582        RESCHED_IPI_PENDING[cpu_index].store(false, Ordering::Release);
583    }
584
585    if !percpu::is_preemptible() {
586        return None;
587    }
588
589    let current_frame_rsp = current_frame as *mut crate::syscall::SyscallFrame as u64;
590    let mut _task_to_drop: Option<Arc<Task>> = None;
591
592    // Lock order: LOCAL only (rank 4). No GLOBAL, BLOCKED, or IDENTITY.
593    lockdep_acquire(LockRank::Local, Some(cpu_index));
594    let decision = {
595        let mut guard = match LOCAL_SCHEDULERS[cpu_index].try_lock_no_irqsave() {
596            Some(g) => g,
597            None => {
598                lockdep_release(LockRank::Local);
599                note_try_lock_fail_on_cpu(cpu_index);
600                return None;
601            }
602        };
603        let cpu = match guard.as_mut() {
604            Some(c) => c,
605            None => {
606                lockdep_release(LockRank::Local);
607                return None;
608            }
609        };
610
611        if take_force_resched_hint(cpu_index) {
612            cpu.need_resched = true;
613        }
614        if cpu.current_task.is_none() || !cpu.need_resched {
615            return None;
616        }
617
618        let current = match cpu.current_task.as_ref() {
619            Some(t) => t.clone(),
620            None => {
621                unsafe { core::arch::asm!("mov al, 'X'; out 0xe9, al", out("al") _) };
622                return None;
623            }
624        };
625        current.set_resume_kind(crate::process::task::ResumeKind::IretFrame);
626        current.set_interrupt_rsp(current_frame_rsp);
627
628        let next = super::core_impl::pick_next_task_local(cpu, cpu_index);
629
630        if Arc::ptr_eq(&current, &next) {
631            cpu.need_resched = false;
632            _task_to_drop = cpu.task_to_drop.take();
633            // No context switch: return current task's FPU area for save/restore.
634            let current_fpu = current.fpu_state.get() as *mut u8;
635            Some(crate::arch::idt::InterruptReturnDecision {
636                next_rsp: 0,
637                old_fpu: current_fpu,
638                new_fpu: current_fpu,
639            })
640        } else {
641            let mut next_rsp = next.interrupt_rsp();
642            if next.resume_kind() == crate::process::task::ResumeKind::RetFrame {
643                // TEMP: do NOT seed synthetic frames from the IRQ path yet.
644                // Resuming a synthetic frame from the raw timer stub derails the
645                // resumed task (observed: it re-runs boot_alloc init/validate code
646                // in a tight IRQs-off loop, starving the timer). Leave first-launch
647                // tasks to the legacy ret-based scheduler path; only tasks with a
648                // REAL iret frame (previously preempted from Ring 3) switch here.
649                unsafe { core::arch::asm!("mov al, 'S'; out 0xe9, al", out("al") _) };
650                cpu.need_resched = false;
651                _task_to_drop = cpu.task_to_drop.take();
652                if let Some(prev) = cpu.task_to_requeue.take() {
653                    prev.set_state(TaskState::Running);
654                    cpu.current_task = Some(prev);
655                } else {
656                    current.set_state(TaskState::Running);
657                    cpu.current_task = Some(current.clone());
658                }
659                next.set_state(TaskState::Ready);
660                let class = cpu.class_table.class_for_task(&next);
661                cpu.class_rqs.enqueue(class, next);
662                let current_fpu = current.fpu_state.get() as *mut u8;
663                return Some(crate::arch::idt::InterruptReturnDecision {
664                    next_rsp: 0,
665                    old_fpu: current_fpu,
666                    new_fpu: current_fpu,
667                });
668            }
669            let fits = interrupt_frame_fits(&next, next_rsp);
670            if next_rsp == 0 || !fits {
671                unsafe { core::arch::asm!("mov al, 'A'; out 0xe9, al", out("al") _) };
672                let is_idle_fallback = Arc::ptr_eq(&next, &cpu.idle_task);
673                _task_to_drop = cpu.task_to_drop.take();
674
675                if let Some(prev) = cpu.task_to_requeue.take() {
676                    prev.set_state(TaskState::Running);
677                    cpu.current_task = Some(prev);
678                } else {
679                    current.set_state(TaskState::Running);
680                    cpu.current_task = Some(current.clone());
681                }
682
683                if !is_idle_fallback {
684                    next.set_state(TaskState::Ready);
685                    let class = cpu.class_table.class_for_task(&next);
686                    cpu.class_rqs.enqueue(class, next);
687                }
688                // Abort switch: return current task's FPU area for save/restore.
689                let current_fpu = current.fpu_state.get() as *mut u8;
690                return Some(crate::arch::idt::InterruptReturnDecision {
691                    next_rsp: 0,
692                    old_fpu: current_fpu,
693                    new_fpu: current_fpu,
694                });
695            } else {
696                next.set_resume_kind(crate::process::task::ResumeKind::IretFrame);
697                cpu.need_resched = false;
698                // Do NOT drain `task_to_requeue` / `task_to_drop` here.
699                // The raw timer stub still has to save the outgoing FPU state and
700                // pivot onto the next task's stack. Defer that finalization to
701                // `finish_interrupt_switch()` on the new stack.
702
703                let stack_top =
704                    next.kernel_stack.virt_base.as_u64() + next.kernel_stack.size as u64;
705
706                crate::arch::tss::set_kernel_stack(x86_64::VirtAddr::new(stack_top));
707                crate::arch::syscall::set_kernel_rsp(stack_top);
708
709                let old_fpu = current.fpu_state.get() as *mut u8;
710                let new_fpu = next.fpu_state.get() as *const u8;
711
712                // TEMP DEBUG: pulse the picked task id + stack top.
713                unsafe {
714                    let hex = b"0123456789abcdef";
715                    core::arch::asm!("out 0xe9, al", in("al") b'@', options(nomem, nostack));
716                    let tid = next.id.as_u64();
717                    for sh in [28usize, 24, 20, 16, 12, 8, 4, 0] {
718                        let nib = hex[((tid >> sh) & 0xF) as usize];
719                        core::arch::asm!("out 0xe9, al", in("al") nib, options(nomem, nostack));
720                    }
721                    core::arch::asm!("out 0xe9, al", in("al") b'\n', options(nomem, nostack));
722                }
723                Some(crate::arch::idt::InterruptReturnDecision {
724                    next_rsp,
725                    old_fpu,
726                    new_fpu,
727                })
728            }
729        }
730    };
731    lockdep_release(LockRank::Local);
732    // LOCAL lock released here
733
734    if decision.is_some() && cpu_is_valid(cpu_index) {
735        CPU_PREEMPT_COUNT[cpu_index].fetch_add(1, Ordering::Relaxed);
736    }
737
738    decision
739}
740
741/// Enable or disable verbose scheduler tracing.
742pub fn set_verbose(enabled: bool) {
743    SCHED_VERBOSE.store(enabled, Ordering::Relaxed);
744    log::info!(
745        "[sched][trace] verbose={}",
746        if enabled { "on" } else { "off" }
747    );
748}
749
750/// Return current verbose tracing state.
751pub fn verbose_enabled() -> bool {
752    SCHED_VERBOSE.load(Ordering::Relaxed)
753}
754
755/// Return the scheduler class-table currently in use.
756pub fn class_table() -> crate::process::sched::SchedClassTable {
757    let saved_flags = save_flags_and_cli();
758    let out = {
759        let scheduler = GLOBAL_SCHED_STATE.lock();
760        if let Some(ref sched) = *scheduler {
761            sched.class_table
762        } else {
763            crate::process::sched::SchedClassTable::default()
764        }
765    };
766    restore_flags(saved_flags);
767    out
768}
769
770/// Configure scheduler class pick/steal order at runtime.
771///
772/// ## Lock order
773///
774/// GLOBAL (rank 1) → LOCAL[cpu] (rank 4), sequentially for each CPU.
775/// No BLOCKED or IDENTITY locks needed.  IPIs are sent after all locks
776/// are released.
777///
778/// ## Contention note
779///
780/// GLOBAL is held while iterating all LOCALs.  This blocks cold-path
781/// operations (fork, exit, wake) on other CPUs for the duration.
782/// Acceptable since class table reconfiguration is infrequent.  Hot-path
783/// `try_lock` on GLOBAL (e.g., `steal_task_local`) will skip rather than
784/// block.
785pub fn configure_class_table(table: crate::process::sched::SchedClassTable) -> bool {
786    if !table.validate() {
787        return false;
788    }
789    let saved_flags = save_flags_and_cli();
790    let mut ipi_targets = [false; crate::arch::percpu::MAX_CPUS];
791    let my_cpu = current_cpu_index();
792
793    // Lock order: GLOBAL (rank 1) → LOCAL[cpu] (rank 4), sequentially.
794    lockdep_acquire(LockRank::Global, None);
795    let applied = {
796        let mut scheduler = GLOBAL_SCHED_STATE.lock();
797        if let Some(ref mut sched) = *scheduler {
798            let prev = sched.class_table;
799            sched.class_table = table;
800            let n = active_cpu_count();
801            for cpu_idx in 0..n {
802                lockdep_acquire(LockRank::Local, Some(cpu_idx));
803                if let Some(ref mut local_cpu) = *LOCAL_SCHEDULERS[cpu_idx].lock() {
804                    local_cpu.class_table = table;
805                    local_cpu.need_resched = true;
806                }
807                lockdep_release(LockRank::Local);
808                if cpu_idx != my_cpu && cpu_is_valid(cpu_idx) {
809                    ipi_targets[cpu_idx] = true;
810                }
811            }
812            if prev.policy_map() != sched.class_table.policy_map() {
813                sched.migrate_ready_tasks_for_new_class_table();
814            }
815            true
816        } else {
817            false
818        }
819    }; // GLOBAL released here
820    lockdep_release(LockRank::Global);
821    restore_flags(saved_flags);
822    // IPIs sent after all locks released — never under a scheduler lock.
823    for (cpu, send) in ipi_targets.iter().copied().enumerate() {
824        if send {
825            send_resched_ipi_to_cpu(cpu);
826        }
827    }
828    applied
829}
830
831/// Dump per-cpu scheduler queues for tracing/debug.
832pub fn log_state(label: &str) {
833    let saved_flags = save_flags_and_cli();
834    let scheduler = GLOBAL_SCHED_STATE.lock();
835    if let Some(ref sched) = *scheduler {
836        let pick = sched.class_table.pick_order();
837        let steal = sched.class_table.steal_order();
838        log::info!(
839            "[sched][state] label={} class_table.pick=[{},{},{}] class_table.steal=[{},{}]",
840            label,
841            pick[0].as_str(),
842            pick[1].as_str(),
843            pick[2].as_str(),
844            steal[0].as_str(),
845            steal[1].as_str()
846        );
847        let n = active_cpu_count();
848        for cpu_id in 0..n {
849            use crate::process::sched::SchedClassRq;
850            let local_guard = LOCAL_SCHEDULERS[cpu_id].lock();
851            if let Some(ref cpu) = *local_guard {
852                let current = cpu
853                    .current_task
854                    .as_ref()
855                    .map(|t| t.id.as_u64())
856                    .unwrap_or(u64::MAX);
857                let blocked_len = super::BLOCKED_TASKS.lock().len();
858                log::info!(
859                    "[sched][state] label={} cpu={} current={} rq_rt={} rq_fair={} rq_idle={} blocked={} need_resched={}",
860                    label,
861                    cpu_id,
862                    current,
863                    cpu.class_rqs.real_time.len(),
864                    cpu.class_rqs.fair.len(),
865                    cpu.class_rqs.idle.len(),
866                    blocked_len,
867                    cpu.need_resched
868                );
869            }
870        }
871    }
872    drop(scheduler);
873
874    // Deferred work metrics (lock-free, read after GLOBAL_SCHED_STATE is released).
875    let dw = super::deferred_work::metrics_snapshot();
876    let n = active_cpu_count();
877    for cpu_id in 0..n {
878        log::info!(
879            "[sched][state] label={} cpu={} dwork_raised={} dwork_processed={}",
880            label,
881            cpu_id,
882            dw.raised[cpu_id],
883            dw.processed[cpu_id],
884        );
885    }
886
887    restore_flags(saved_flags);
888}
889
890/// Structured scheduler state snapshot for shell/top/debug tooling.
891pub fn state_snapshot() -> SchedulerStateSnapshot {
892    let dw = super::deferred_work::metrics_snapshot();
893    let mut out = SchedulerStateSnapshot {
894        initialized: false,
895        boot_phase: 0,
896        cpu_count: 0,
897        pick_order: [
898            crate::process::sched::SchedClassId::RealTime,
899            crate::process::sched::SchedClassId::Fair,
900            crate::process::sched::SchedClassId::Idle,
901        ],
902        steal_order: [
903            crate::process::sched::SchedClassId::Fair,
904            crate::process::sched::SchedClassId::RealTime,
905        ],
906        blocked_tasks: 0,
907        current_task: [u64::MAX; crate::arch::percpu::MAX_CPUS],
908        rq_rt: [0; crate::arch::percpu::MAX_CPUS],
909        rq_fair: [0; crate::arch::percpu::MAX_CPUS],
910        rq_idle: [0; crate::arch::percpu::MAX_CPUS],
911        need_resched: [false; crate::arch::percpu::MAX_CPUS],
912        deferred_work_raised: dw.raised,
913        deferred_work_processed: dw.processed,
914    };
915
916    let saved_flags = save_flags_and_cli();
917    {
918        let scheduler = GLOBAL_SCHED_STATE.lock();
919        if let Some(ref sched) = *scheduler {
920            use crate::process::sched::SchedClassRq;
921            let cpu_count = active_cpu_count().min(crate::arch::percpu::MAX_CPUS);
922            out.initialized = true;
923            out.boot_phase = if cpu_count > 0 { 2 } else { 1 };
924            out.cpu_count = cpu_count;
925            out.pick_order = *sched.class_table.pick_order();
926            out.steal_order = *sched.class_table.steal_order();
927            out.blocked_tasks = super::BLOCKED_TASKS.lock().len();
928            for i in 0..cpu_count {
929                let local_guard = LOCAL_SCHEDULERS[i].lock();
930                if let Some(ref cpu) = *local_guard {
931                    out.current_task[i] = cpu
932                        .current_task
933                        .as_ref()
934                        .map(|t| t.id.as_u64())
935                        .unwrap_or(u64::MAX);
936                    out.rq_rt[i] = cpu.class_rqs.real_time.len();
937                    out.rq_fair[i] = cpu.class_rqs.fair.len();
938                    out.rq_idle[i] = cpu.class_rqs.idle.len();
939                    out.need_resched[i] = cpu.need_resched;
940                }
941            }
942        }
943    }
944    restore_flags(saved_flags);
945    out
946}
947
948/// The main function for the idle task
949pub(super) extern "C" fn idle_task_main() -> ! {
950    let cpu = crate::arch::percpu::current_cpu_index();
951    crate::serial_force_println!("[trace][sched] idle_task_main start cpu={}", cpu);
952    loop {
953        // Process any deferred work before halting. This ensures that
954        // timer ticks raised during interrupt context get processed even
955        // when the CPU has no other runnable tasks.
956        super::deferred_work::process_deferred_work();
957
958        // Be explicit on SMP: never rely on inherited IF state.
959        // If IF=0, HLT can deadlock that CPU forever.
960        crate::arch::sti();
961
962        // Halt until next interrupt (saves power, timer will wake us)
963        crate::arch::hlt();
964    }
965}