Skip to main content

strat9_kernel/process/scheduler/
deferred_work.rs

1//! Deferred work queue — softirq-equivalent for Strat9-OS.
2//!
3//! Moves heavy tick processing out of the hardirq handler into safe
4//! execution contexts (post-switch, idle loop, syscall return).
5//!
6//! # Design
7//!
8//! Each CPU has a lock-free `AtomicU32` bitmask of pending work items.
9//! The hardirq timer handler calls `raise_deferred_work()` which sets bits
10//! atomically — no locks, no allocation, O(1) cost.
11//!
12//! Processing happens at safe points where IRQs may be re-enabled and
13//! scheduler locks can be acquired without deadlock:
14//!
15//! - `finish_switch()` — after cooperative context switch completes
16//! - `finish_interrupt_switch()` — after interrupt-driven switch completes
17//! - `idle_task_main()` — idle loop before HLT
18//! - `maybe_preempt()` — at preemption entry (for work raised during IRQ)
19//!
20//! # What stays in hardirq
21//!
22//! - TICK_COUNT, CPU_TOTAL_TICKS, CPU_LOCAL_TICKS increments (lock-free)
23//! - FORCE_RESCHED_HINT setting (lock-free)
24//! - speaker_tick, n3_watchdog_tick (minimal)
25//! - EOI (must be immediate)
26//!
27//! # What moves to deferred
28//!
29//! - `tick_all_timers()` — interval timer processing
30//! - `check_wake_deadlines()` — deadline-based wakeups
31//! - Per-task Fair class accounting (runnable_len, tick_update_wait)
32
33use core::sync::atomic::{AtomicBool, AtomicU32, Ordering};
34
35/// Work item types that can be deferred from hardirq context.
36#[derive(Debug, Clone, Copy, PartialEq, Eq)]
37#[repr(u32)]
38pub enum DeferredWork {
39    /// Process interval timers (ITIMER_REAL, ITIMER_VIRTUAL, ITIMER_PROF).
40    IntervalTimers = 1 << 0,
41    /// Check wake deadlines and wake expired tasks.
42    WakeDeadlines = 1 << 1,
43    /// Per-task Fair class accounting (runnable_len, tick_update_wait).
44    PerTaskAccounting = 1 << 2,
45    /// NIC watchdog / IRQ-less N2 service (safe outside hardirq).
46    NicPoll = 1 << 3,
47}
48
49impl DeferredWork {
50    /// All work items combined.
51    const ALL: u32 = Self::IntervalTimers as u32
52        | Self::WakeDeadlines as u32
53        | Self::PerTaskAccounting as u32
54        | Self::NicPoll as u32;
55}
56
57/// Per-CPU deferred work state.
58struct DeferredWorkCpu {
59    /// Pending work bitmask (set by raise, cleared by process).
60    pending: AtomicU32,
61    /// Whether processing is currently in progress (prevents re-entrance).
62    processing: AtomicBool,
63    /// Total work items raised (for metrics).
64    raised_count: AtomicU32,
65    /// Total work items processed (for metrics).
66    processed_count: AtomicU32,
67}
68
69impl DeferredWorkCpu {
70    const fn new() -> Self {
71        Self {
72            pending: AtomicU32::new(0),
73            processing: AtomicBool::new(false),
74            raised_count: AtomicU32::new(0),
75            processed_count: AtomicU32::new(0),
76        }
77    }
78}
79
80/// Per-CPU deferred work queues.
81static DEFERRED_WORK: [DeferredWorkCpu; crate::arch::percpu::MAX_CPUS] =
82    [const { DeferredWorkCpu::new() }; crate::arch::percpu::MAX_CPUS];
83
84// ---------------------------------------------------------------------------
85// Raise (called from hardirq context)
86// ---------------------------------------------------------------------------
87
88/// Raise deferred work items for the current CPU.
89///
90/// Called from the timer interrupt handler to signal that heavy processing
91/// should happen at the next safe point. This is lock-free and O(1).
92///
93/// # Safety
94///
95/// Safe to call from any context (including hardirq with interrupts disabled).
96pub fn raise_deferred_work(work: DeferredWork) {
97    let cpu = crate::arch::percpu::current_cpu_index();
98    if cpu >= crate::arch::percpu::MAX_CPUS {
99        return;
100    }
101    DEFERRED_WORK[cpu]
102        .pending
103        .fetch_or(work as u32, Ordering::Release);
104    DEFERRED_WORK[cpu]
105        .raised_count
106        .fetch_add(1, Ordering::Relaxed);
107}
108
109/// Raise all tick-related deferred work items for the current CPU.
110///
111/// Convenience function called from `timer_tick()` to defer all heavy
112/// processing that was previously done inline in hardirq.
113pub fn raise_tick_deferred_work() {
114    raise_deferred_work(DeferredWork::IntervalTimers);
115    raise_deferred_work(DeferredWork::WakeDeadlines);
116    raise_deferred_work(DeferredWork::PerTaskAccounting);
117    raise_deferred_work(DeferredWork::NicPoll);
118}
119
120// ---------------------------------------------------------------------------
121// Process (called from safe context)
122// ---------------------------------------------------------------------------
123
124/// Process deferred work items for the current CPU.
125///
126/// Called from safe contexts: finish_switch, finish_interrupt_switch,
127/// idle loop entry, etc. Drains the pending bitmask and executes each
128/// work item with interrupts potentially re-enabled.
129///
130/// Returns `true` if any work was processed.
131pub fn process_deferred_work() -> bool {
132    let cpu = crate::arch::percpu::current_cpu_index();
133    if cpu >= crate::arch::percpu::MAX_CPUS {
134        return false;
135    }
136
137    let work_cpu = &DEFERRED_WORK[cpu];
138
139    // Fast path: nothing pending.
140    let pending = work_cpu.pending.swap(0, Ordering::AcqRel);
141    if pending == 0 {
142        return false;
143    }
144
145    // Only measure actual work processing, not the fast-path no-op.
146    let _perf = super::perf_counters::PerfScope::new(
147        &super::perf_counters::DEFERRED_WORK_TSC,
148        &super::perf_counters::DEFERRED_WORK_COUNT,
149    );
150
151    // Prevent re-entrance (e.g., if a work item triggers a reschedule
152    // that re-enters this path).
153    if work_cpu.processing.swap(true, Ordering::AcqRel) {
154        // Re-raise the items we couldn't process.
155        work_cpu.pending.fetch_or(pending, Ordering::Release);
156        work_cpu.processing.store(false, Ordering::Release);
157        return false;
158    }
159
160    work_cpu.processed_count.fetch_add(1, Ordering::Relaxed);
161
162    // Process each pending work item.
163    // These acquire scheduler locks internally via try_lock, so they
164    // are safe to call from most kernel contexts.
165    if pending & DeferredWork::IntervalTimers as u32 != 0 {
166        let tick = super::ticks();
167        let current_time_ns = tick * crate::arch::timer::NS_PER_TICK;
168        crate::process::timer::tick_all_timers(current_time_ns);
169    }
170
171    if pending & DeferredWork::WakeDeadlines as u32 != 0 {
172        let tick = super::ticks();
173        let current_time_ns = tick * crate::arch::timer::NS_PER_TICK;
174        super::timer_ops::check_wake_deadlines(current_time_ns);
175    }
176
177    if pending & DeferredWork::PerTaskAccounting as u32 != 0 {
178        process_per_task_accounting();
179    }
180
181    if pending & DeferredWork::NicPoll as u32 != 0 {
182        // Safe here: deferred work runs outside the hardirq swapgs window.
183        crate::hardware::nic::poll_all();
184    }
185
186    work_cpu.processing.store(false, Ordering::Release);
187    true
188}
189
190/// Process per-task Fair class accounting for the current CPU.
191///
192/// This is the deferred version of the per-task block that was previously
193/// inline in `timer_tick()`. Acquires LOCAL_SCHEDULERS[cpu] via try_lock.
194fn process_per_task_accounting() {
195    use super::{CPU_FAIR_RUNTIME_TICKS, CPU_IDLE_TICKS, CPU_RT_RUNTIME_TICKS, LOCAL_SCHEDULERS};
196    use crate::process::sched::SchedClassId;
197
198    let cpu_idx = crate::arch::percpu::current_cpu_index();
199    if !super::cpu_is_valid(cpu_idx) {
200        return;
201    }
202
203    const TICK_LOCK_RETRIES: usize = 3;
204    for attempt in 0..TICK_LOCK_RETRIES {
205        if let Some(mut guard) = LOCAL_SCHEDULERS[cpu_idx].try_lock_no_irqsave() {
206            if let Some(ref mut cpu) = *guard {
207                let should_resched = if let Some(ref current_task) = cpu.current_task {
208                    let class = cpu.class_table.class_for_task(current_task);
209                    match class {
210                        SchedClassId::RealTime => {
211                            CPU_RT_RUNTIME_TICKS[cpu_idx].fetch_add(1, Ordering::Relaxed);
212                        }
213                        SchedClassId::Fair => {
214                            CPU_FAIR_RUNTIME_TICKS[cpu_idx].fetch_add(1, Ordering::Relaxed);
215                        }
216                        SchedClassId::Idle => {
217                            CPU_IDLE_TICKS[cpu_idx].fetch_add(1, Ordering::Relaxed);
218                        }
219                    }
220                    current_task.ticks.fetch_add(1, Ordering::Relaxed);
221                    cpu.current_runtime.update();
222                    cpu.class_rqs.update_current(
223                        &cpu.current_runtime,
224                        current_task,
225                        false,
226                        &cpu.class_table,
227                    )
228                } else {
229                    false
230                };
231                // Increment Fair starvation counters for all queued tasks.
232                cpu.class_rqs.tick_update_wait();
233                if should_resched {
234                    cpu.need_resched = true;
235                }
236            }
237            return;
238        }
239        if attempt == TICK_LOCK_RETRIES - 1 {
240            super::note_try_lock_fail_on_cpu(cpu_idx);
241        }
242    }
243}
244
245// ---------------------------------------------------------------------------
246// Metrics
247// ---------------------------------------------------------------------------
248
249/// Snapshot of deferred work metrics across all CPUs.
250pub struct DeferredWorkMetrics {
251    pub cpu_count: usize,
252    pub raised: [u32; crate::arch::percpu::MAX_CPUS],
253    pub processed: [u32; crate::arch::percpu::MAX_CPUS],
254    pub pending: [u32; crate::arch::percpu::MAX_CPUS],
255}
256
257/// Take a snapshot of deferred work metrics.
258pub fn metrics_snapshot() -> DeferredWorkMetrics {
259    let n = super::active_cpu_count();
260    let mut raised = [0u32; crate::arch::percpu::MAX_CPUS];
261    let mut processed = [0u32; crate::arch::percpu::MAX_CPUS];
262    let mut pending = [0u32; crate::arch::percpu::MAX_CPUS];
263    for i in 0..n {
264        raised[i] = DEFERRED_WORK[i].raised_count.load(Ordering::Relaxed);
265        processed[i] = DEFERRED_WORK[i].processed_count.load(Ordering::Relaxed);
266        pending[i] = DEFERRED_WORK[i].pending.load(Ordering::Relaxed);
267    }
268    DeferredWorkMetrics {
269        cpu_count: n,
270        raised,
271        processed,
272        pending,
273    }
274}
275
276/// Reset deferred work metrics.
277pub fn reset_metrics() {
278    let n = super::active_cpu_count();
279    for i in 0..n {
280        DEFERRED_WORK[i].raised_count.store(0, Ordering::Relaxed);
281        DEFERRED_WORK[i].processed_count.store(0, Ordering::Relaxed);
282    }
283}
284
285/// Check if there is pending deferred work on the current CPU.
286#[inline]
287pub fn has_pending() -> bool {
288    let cpu = crate::arch::percpu::current_cpu_index();
289    if cpu >= crate::arch::percpu::MAX_CPUS {
290        return false;
291    }
292    DEFERRED_WORK[cpu].pending.load(Ordering::Acquire) != 0
293}