Skip to main content

strat9_kernel/process/scheduler/
task_ops.rs

1use super::{active_cpu_count, runtime_ops::finish_switch, *};
2use crate::{memory::UserSliceWrite, sync::FixedQueue};
3
4const PENDING_SILO_CLEANUPS_CAPACITY: usize = 256;
5
6static PENDING_SILO_CLEANUPS: SpinLock<FixedQueue<TaskId, PENDING_SILO_CLEANUPS_CAPACITY>> =
7    SpinLock::new(FixedQueue::new());
8
9/// Mark the current task as Dead and yield to the scheduler.
10///
11/// Called by SYS_PROC_EXIT. The task will not be re-queued because
12/// `pick_next_task()` only re-queues tasks in `Running` state.
13/// This function does not return.
14pub fn exit_current_task(exit_code: i32) -> ! {
15    // -- clear_child_tid (POSIX pthread join) --
16    // Must happen BEFORE we drop the address space - write 0 to the TID pointer
17    // and do a futex_wake so any waiting pthread_join() can proceed.
18    if let Some(task) = current_task_clone() {
19        let tidptr = task
20            .clear_child_tid
21            .load(core::sync::atomic::Ordering::Relaxed);
22        if tidptr != 0 {
23            let zero = 0u32.to_ne_bytes();
24            // POSIX clear_child_tid targets a userspace u32; keep the existing
25            // alignment check for futex semantics, but validate the mapping via
26            // UserSliceWrite instead of dereferencing the raw userspace pointer.
27            if (tidptr & 3) == 0 {
28                if let Ok(user) = UserSliceWrite::new(tidptr, zero.len()) {
29                    user.copy_from(&zero);
30                }
31                // Futex wake: wake all threads waiting on this address (e.g. pthread_join).
32                let _ = crate::syscall::futex::sys_futex_wake(tidptr, u32::MAX);
33            }
34        }
35
36        // -- robust_list (clean up held mutexes) --
37        // Walk the robust list and mark any held futexes as FUTEX_OWNER_DIED,
38        // then wake waiters. Prevents deadlocks when a thread dies holding a mutex.
39        crate::syscall::robust_list::cleanup_robust_list(&task);
40
41        // -- user-stack canary check (issue #63) --
42        // The address space is still alive here (it dies with the Process
43        // Arc later), so the top-of-stack word can still be read.
44        task.verify_user_stack_canary();
45
46        // -- kernel-owned user stack (/thread/create) --
47        // Reclaim the user stack the kernel allocated for this thread. Done
48        // while still the running task so the address space is current on
49        // this CPU; idempotent with the cleanup_task_resources path.
50        crate::process::thread_ops::reclaim_kernel_user_stack(&task);
51    }
52
53    let cpu_index = current_cpu_index();
54    let mut parent_to_signal: Option<TaskId> = None;
55    let mut ipi_to_cpu: Option<usize> = None;
56    {
57        let saved_flags = save_flags_and_cli();
58        // Lock order: GLOBAL (rank 1) -> IDENTITY write (rank 2).
59        lockdep_acquire(LockRank::Global, None);
60        let mut scheduler = GLOBAL_SCHED_STATE.lock();
61        let current = {
62            lockdep_acquire(LockRank::Local, Some(cpu_index));
63            let local = LOCAL_SCHEDULERS[cpu_index].lock();
64            let task = local.as_ref().and_then(|cpu| cpu.current_task.clone());
65            lockdep_release(LockRank::Local);
66            drop(local);
67            task
68        };
69        if let Some(ref mut sched) = *scheduler {
70            if let Some(current) = current {
71                let current_id = current.id;
72                let current_pid = current.pid;
73                let parent = {
74                    lockdep_acquire(LockRank::IdentityR, None);
75                    let identity = SCHED_IDENTITY.read();
76                    let p = identity.parent_of.get(&current_id).copied();
77                    lockdep_release(LockRank::IdentityR);
78                    drop(identity);
79                    p
80                };
81                let _ = sched.clear_task_wake_deadline_locked(current_id);
82                current.set_state(TaskState::Dead);
83                sched.task_cpu.remove(&current_id);
84                {
85                    lockdep_acquire(LockRank::IdentityW, None);
86                    let mut identity = SCHED_IDENTITY.write();
87                    GlobalSchedState::unregister_identity_locked(
88                        &mut identity,
89                        current_id,
90                        current_pid,
91                        current.tid,
92                    );
93                    identity.parent_of.remove(&current_id);
94                    lockdep_release(LockRank::IdentityW);
95                    drop(identity);
96                }
97
98                {
99                    lockdep_acquire(LockRank::IdentityW, None);
100                    let mut identity = SCHED_IDENTITY.write();
101                    ipi_to_cpu = reparent_children(sched, &mut identity, current_id);
102                    lockdep_release(LockRank::IdentityW);
103                    drop(identity);
104                }
105
106                if parent.is_some() {
107                    sched.zombies.insert(current_id, (exit_code, current_pid));
108                }
109                if let Some(parent_id) = parent {
110                    let (_, ipi_wake) = sched.wake_task_locked(parent_id);
111                    if ipi_to_cpu.is_none() {
112                        ipi_to_cpu = ipi_wake;
113                    }
114                    parent_to_signal = Some(parent_id);
115                }
116            }
117        }
118        lockdep_release(LockRank::Global);
119        drop(scheduler);
120        restore_flags(saved_flags);
121    }
122    if let Some(ci) = ipi_to_cpu {
123        send_resched_ipi_to_cpu(ci);
124    }
125
126    if let Some(parent_id) = parent_to_signal {
127        // Must happen outside scheduler lock to avoid lock recursion.
128        let _ =
129            crate::process::signal::send_signal(parent_id, crate::process::signal::Signal::SIGCHLD);
130    }
131
132    // Yield to pick the next task. Since we're Dead, we won't come back.
133    // Use yield_dead_task() which bypasses the PreemptGuard check : the task
134    // is already marked Dead and will never run again, so the guard is irrelevant.
135    // Using yield_task() here would silently return if a PreemptGuard is active,
136    // leaving the dead task spinning in the hlt() loop below.
137    yield_dead_task();
138
139    // Safety net - should never reach here
140    loop {
141        crate::arch::hlt();
142    }
143}
144
145/// Get the current task's ID (if any task is running).
146pub fn current_task_id() -> Option<TaskId> {
147    let saved_flags = save_flags_and_cli();
148    let cpu_index = current_cpu_index();
149    let id = LOCAL_SCHEDULERS[cpu_index]
150        .lock()
151        .as_ref()
152        .and_then(|cpu| cpu.current_task.as_ref().map(|t| t.id));
153    restore_flags(saved_flags);
154    id
155}
156
157/// Get the current task's ID without blocking (safe for exceptions).
158pub fn current_task_id_try() -> Option<TaskId> {
159    let saved_flags = save_flags_and_cli();
160    let cpu_index = current_cpu_index();
161    let id = LOCAL_SCHEDULERS[cpu_index]
162        .try_lock_no_irqsave()
163        .and_then(|guard| {
164            guard
165                .as_ref()
166                .and_then(|cpu| cpu.current_task.as_ref().map(|t| t.id))
167        });
168    restore_flags(saved_flags);
169    id
170}
171
172/// Get the current process ID (POSIX pid).
173pub fn current_pid() -> Option<Pid> {
174    current_task_clone().map(|t| t.pid)
175}
176
177/// Get the current thread ID (POSIX tid).
178pub fn current_tid() -> Option<Tid> {
179    current_task_clone().map(|t| t.tid)
180}
181
182/// Get the current process group id.
183pub fn current_pgid() -> Option<Pid> {
184    current_task_clone().map(|t| t.pgid.load(Ordering::Relaxed))
185}
186
187/// Get the current session id.
188pub fn current_sid() -> Option<Pid> {
189    current_task_clone().map(|t| t.sid.load(Ordering::Relaxed))
190}
191
192/// Get the current task (cloned Arc), if any.
193#[track_caller]
194pub fn current_task_clone() -> Option<Arc<Task>> {
195    let saved_flags = save_flags_and_cli();
196    let cpu_index = current_cpu_index();
197    let caller = core::panic::Location::caller();
198    let task = LOCAL_SCHEDULERS[cpu_index].lock().as_ref().and_then(|cpu| {
199        let arc = cpu.current_task.as_ref()?;
200        let strong = Arc::strong_count(arc);
201        // Heuristic only: strong_count can move concurrently, so this is a
202        // diagnostic signal for suspicious scheduler state, not a formal
203        // corruption proof. Keep the warning but do not mutate scheduler state.
204        if strong == 0 || strong > (isize::MAX as usize) / 2 {
205            let ptr = Arc::as_ptr(arc) as *const u8;
206            crate::serial_println!(
207                "[sched] suspicious Arc refcount (heuristic): cpu={} strong={:#x} ptr={:p} caller={}:{}",
208                cpu_index,
209                strong,
210                ptr,
211                caller.file(),
212                caller.line(),
213            );
214        }
215        Some(arc.clone())
216    });
217    restore_flags(saved_flags);
218    task
219}
220
221/// Best-effort, non-blocking variant of [`current_task_clone`].
222///
223/// Returns `None` when the scheduler lock is contended.
224/// Useful in cleanup paths where blocking on `GLOBAL_SCHED_STATE.lock()` could deadlock.
225#[track_caller]
226pub fn current_task_clone_try() -> Option<Arc<Task>> {
227    let saved_flags = save_flags_and_cli();
228    let cpu_index = current_cpu_index();
229    let caller = core::panic::Location::caller();
230    let task = LOCAL_SCHEDULERS[cpu_index]
231        .try_lock_no_irqsave()
232        .and_then(|guard| {
233            guard.as_ref().and_then(|cpu| {
234                let arc = cpu.current_task.as_ref()?;
235                let strong = Arc::strong_count(arc);
236                // Heuristic only: strong_count can move concurrently, so this is a
237                // diagnostic signal for suspicious scheduler state, not a formal
238                // corruption proof.
239                if strong == 0 || strong > (isize::MAX as usize) / 2 {
240                    let ptr = Arc::as_ptr(arc) as *const u8;
241                    crate::serial_println!(
242                        "[sched] suspicious Arc refcount (heuristic): cpu={} strong={:#x} ptr={:p} caller={}:{}",
243                        cpu_index,
244                        strong,
245                        ptr,
246                        caller.file(),
247                        caller.line(),
248                    );
249                }
250                Some(arc.clone())
251            })
252        });
253    restore_flags(saved_flags);
254    task
255}
256
257/// Debug-only blocking variant used to diagnose early ring3 entry stalls.
258///
259/// Spins with `try_lock()` so we can emit progress logs instead of blocking
260/// silently on `GLOBAL_SCHED_STATE.lock()`.
261pub fn current_task_clone_spin_debug(trace_label: &str) -> Option<Arc<Task>> {
262    let saved_flags = save_flags_and_cli();
263    let cpu_index = current_cpu_index();
264    let mut spins = 0usize;
265    let result = loop {
266        if let Some(guard) = LOCAL_SCHEDULERS[cpu_index].try_lock_no_irqsave() {
267            break guard.as_ref().and_then(|cpu| {
268                if cpu.current_task.is_none() {
269                    debug_trace_putc(b'N');
270                    return None;
271                }
272                let arc = cpu.current_task.as_ref().unwrap();
273                let strong = Arc::strong_count(arc);
274                // Racy, pifometric diagnostic only: strong_count can move
275                // concurrently, so this is a heuristic for suspicious
276                // scheduler state, not a formal corruption proof.
277                if strong == 0 || strong > (isize::MAX as usize) / 2 {
278                    let ptr = Arc::as_ptr(arc) as *const u8;
279                    crate::serial_force_println!(
280                        "[trace][sched] {} suspicious current_task heuristic cpu={} strong={:#x} ptr={:p}",
281                        trace_label,
282                        cpu_index,
283                        strong,
284                        ptr,
285                    );
286                }
287                Some(arc.clone())
288            });
289        }
290
291        spins = spins.saturating_add(1);
292        if spins == 2_000_000 {
293            crate::serial_force_println!(
294                "[trace][sched] {} waiting current_task cpu={} owner_cpu={}",
295                trace_label,
296                cpu_index,
297                GLOBAL_SCHED_STATE.owner_cpu()
298            );
299            spins = 0;
300        }
301        core::hint::spin_loop();
302    };
303    restore_flags(saved_flags);
304    result
305}
306
307/// Resolve a POSIX pid to internal TaskId.
308pub fn get_task_id_by_pid(pid: Pid) -> Option<TaskId> {
309    SCHED_IDENTITY.read().pid_to_task.get(&pid).copied()
310}
311
312/// Resolve a POSIX pid to the corresponding task.
313pub fn get_task_by_pid(pid: Pid) -> Option<Arc<Task>> {
314    let tid = get_task_id_by_pid(pid)?;
315    get_task_by_id(tid)
316}
317
318/// Resolve a direct child of `parent` by POSIX pid.
319///
320/// Unlike the global pid index, this remains valid after the child has called
321/// exit and before it is reaped, because the task object stays in `all_tasks`
322/// until waitpid consumes the zombie.
323///
324/// Lock order: `GLOBAL_SCHED_STATE` before `SCHED_IDENTITY` (see module docs).
325pub fn get_child_task_id_by_pid(parent: TaskId, pid: Pid) -> Option<TaskId> {
326    let saved_flags = save_flags_and_cli();
327    let out = {
328        let scheduler = GLOBAL_SCHED_STATE.lock();
329        if let Some(ref sched) = *scheduler {
330            let children = {
331                let identity = SCHED_IDENTITY.read();
332                identity
333                    .children_of
334                    .get(&parent)
335                    .cloned()
336                    .unwrap_or_default()
337            };
338            if children.is_empty() {
339                None
340            } else {
341                children.iter().copied().find(|child_id| {
342                    sched
343                        .all_tasks
344                        .get(child_id)
345                        .map(|task| task.pid == pid)
346                        .unwrap_or(false)
347                })
348            }
349        } else {
350            None
351        }
352    };
353    restore_flags(saved_flags);
354    out
355}
356
357/// Resolve a POSIX tid to the corresponding internal task id.
358pub fn get_task_id_by_tid(tid: Tid) -> Option<TaskId> {
359    let identity = SCHED_IDENTITY.read();
360    identity
361        .tid_to_task
362        .get(&tid)
363        .copied()
364        .or_else(|| identity.pid_to_task.get(&(tid as Pid)).copied())
365}
366
367/// Resolve a direct child of `parent` by POSIX tid.
368///
369/// This remains valid for dead-but-not-yet-reaped threads because it scans the
370/// caller's child set and the retained task object instead of relying on the
371/// global tid index removed during exit.
372///
373/// Lock order: `GLOBAL_SCHED_STATE` before `SCHED_IDENTITY` (see module docs).
374pub fn get_child_task_id_by_tid(parent: TaskId, tid: Tid) -> Option<TaskId> {
375    let saved_flags = save_flags_and_cli();
376    let out = {
377        let scheduler = GLOBAL_SCHED_STATE.lock();
378        if let Some(ref sched) = *scheduler {
379            let children = {
380                let identity = SCHED_IDENTITY.read();
381                identity
382                    .children_of
383                    .get(&parent)
384                    .cloned()
385                    .unwrap_or_default()
386            };
387            if children.is_empty() {
388                None
389            } else {
390                children.iter().copied().find(|child_id| {
391                    sched
392                        .all_tasks
393                        .get(child_id)
394                        .map(|task| task.tid == tid)
395                        .unwrap_or(false)
396                })
397            }
398        } else {
399            None
400        }
401    };
402    restore_flags(saved_flags);
403    out
404}
405
406/// Resolve a PID to the current process group id.
407pub fn get_pgid_by_pid(pid: Pid) -> Option<Pid> {
408    SCHED_IDENTITY.read().pid_to_pgid.get(&pid).copied()
409}
410
411/// Resolve a PID to the current session id.
412pub fn get_sid_by_pid(pid: Pid) -> Option<Pid> {
413    SCHED_IDENTITY.read().pid_to_sid.get(&pid).copied()
414}
415
416/// Collect task IDs that currently belong to process group `pgid`.
417pub fn get_task_ids_in_pgid(pgid: Pid) -> alloc::vec::Vec<TaskId> {
418    use alloc::vec::Vec;
419    SCHED_IDENTITY
420        .read()
421        .pgid_members
422        .get(&pgid)
423        .cloned()
424        .unwrap_or_else(Vec::new)
425}
426
427/// Collect task IDs that currently belong to thread group `tgid`.
428pub fn get_task_ids_in_tgid(tgid: Pid) -> alloc::vec::Vec<TaskId> {
429    use alloc::vec::Vec;
430    let saved_flags = save_flags_and_cli();
431    let out = {
432        let scheduler = GLOBAL_SCHED_STATE.lock();
433        if let Some(ref sched) = *scheduler {
434            sched
435                .all_tasks
436                .values()
437                .filter(|task| task.tgid == tgid)
438                .map(|task| task.id)
439                .collect::<Vec<_>>()
440        } else {
441            Vec::new()
442        }
443    };
444    restore_flags(saved_flags);
445    out
446}
447
448/// Set process group id for `target_pid` (or current if `None`).
449pub fn set_process_group(
450    requester: TaskId,
451    target_pid: Option<Pid>,
452    new_pgid: Option<Pid>,
453) -> Result<Pid, crate::syscall::error::SyscallError> {
454    use crate::syscall::error::SyscallError;
455
456    let saved_flags = save_flags_and_cli();
457    let result = (|| -> Result<Pid, SyscallError> {
458        let scheduler = GLOBAL_SCHED_STATE.lock();
459        let sched = scheduler.as_ref().ok_or(SyscallError::Fault)?;
460
461        let requester_task = sched
462            .all_tasks
463            .get(&requester)
464            .cloned()
465            .ok_or(SyscallError::Fault)?;
466        let requester_sid = requester_task.sid.load(Ordering::Relaxed);
467
468        let target_id = match target_pid {
469            None => requester,
470            Some(pid) => SCHED_IDENTITY
471                .read()
472                .pid_to_task
473                .get(&pid)
474                .copied()
475                .ok_or(SyscallError::NotFound)?,
476        };
477
478        if target_id != requester {
479            let is_child = SCHED_IDENTITY
480                .read()
481                .children_of
482                .get(&requester)
483                .map(|children| children.iter().any(|child| *child == target_id))
484                .unwrap_or(false);
485            if !is_child {
486                return Err(SyscallError::PermissionDenied);
487            }
488        }
489
490        let target_task = sched
491            .all_tasks
492            .get(&target_id)
493            .cloned()
494            .ok_or(SyscallError::NotFound)?;
495        let target_pid_value = target_task.pid;
496        let target_sid = target_task.sid.load(Ordering::Relaxed);
497
498        if target_sid != requester_sid {
499            return Err(SyscallError::PermissionDenied);
500        }
501
502        if target_pid_value == target_sid {
503            return Err(SyscallError::PermissionDenied);
504        }
505
506        let desired_pgid = new_pgid.unwrap_or(target_pid_value);
507        if desired_pgid != target_pid_value {
508            let group_leader_tid = SCHED_IDENTITY
509                .read()
510                .pid_to_task
511                .get(&desired_pgid)
512                .copied()
513                .ok_or(SyscallError::NotFound)?;
514            let group_leader = sched
515                .all_tasks
516                .get(&group_leader_tid)
517                .ok_or(SyscallError::NotFound)?;
518            if group_leader.sid.load(Ordering::Relaxed) != target_sid {
519                return Err(SyscallError::PermissionDenied);
520            }
521        }
522
523        // Mutate identity maps under SCHED_IDENTITY lock while still holding GLOBAL_SCHED_STATE.
524        let old_pgid = target_task.pgid.load(Ordering::Relaxed);
525        let actual_new_pgid = new_pgid.unwrap_or(target_task.pid);
526        target_task.pgid.store(actual_new_pgid, Ordering::Relaxed);
527        {
528            let mut identity = SCHED_IDENTITY.write();
529            GlobalSchedState::member_remove(&mut identity.pgid_members, old_pgid, target_id);
530            GlobalSchedState::member_add(&mut identity.pgid_members, actual_new_pgid, target_id);
531            identity
532                .pid_to_pgid
533                .insert(target_task.pid, actual_new_pgid);
534        }
535        Ok(actual_new_pgid)
536    })();
537    restore_flags(saved_flags);
538    result
539}
540
541/// Create a new session for the calling task.
542pub fn create_session(requester: TaskId) -> Result<Pid, crate::syscall::error::SyscallError> {
543    use crate::syscall::error::SyscallError;
544
545    let saved_flags = save_flags_and_cli();
546    let result = (|| -> Result<Pid, SyscallError> {
547        let scheduler = GLOBAL_SCHED_STATE.lock();
548        let sched = scheduler.as_ref().ok_or(SyscallError::Fault)?;
549        let requester_task = sched
550            .all_tasks
551            .get(&requester)
552            .cloned()
553            .ok_or(SyscallError::Fault)?;
554
555        let pid = requester_task.pid;
556        if requester_task.pgid.load(Ordering::Relaxed) == pid {
557            return Err(SyscallError::PermissionDenied);
558        }
559
560        let old_sid = requester_task.sid.load(Ordering::Relaxed);
561        let old_pgid = requester_task.pgid.load(Ordering::Relaxed);
562        requester_task.sid.store(pid, Ordering::Relaxed);
563        requester_task.pgid.store(pid, Ordering::Relaxed);
564        {
565            let mut identity = SCHED_IDENTITY.write();
566            GlobalSchedState::member_remove(&mut identity.sid_members, old_sid, requester);
567            GlobalSchedState::member_remove(&mut identity.pgid_members, old_pgid, requester);
568            GlobalSchedState::member_add(&mut identity.sid_members, pid, requester);
569            GlobalSchedState::member_add(&mut identity.pgid_members, pid, requester);
570            identity.pid_to_sid.insert(pid, pid);
571            identity.pid_to_pgid.insert(pid, pid);
572        }
573        Ok(pid)
574    })();
575    restore_flags(saved_flags);
576    result
577}
578
579/// Get a task by its TaskId (if still registered).
580pub fn get_task_by_id(id: TaskId) -> Option<Arc<Task>> {
581    let saved_flags = save_flags_and_cli();
582    let task = {
583        let scheduler = GLOBAL_SCHED_STATE.lock();
584        if let Some(ref sched) = *scheduler {
585            sched.all_tasks.get(&id).cloned()
586        } else {
587            None
588        }
589    };
590    restore_flags(saved_flags);
591    task
592}
593
594/// Update a task scheduling policy and requeue if needed.
595pub fn set_task_sched_policy(id: TaskId, policy: crate::process::sched::SchedPolicy) -> bool {
596    let saved_flags = save_flags_and_cli();
597    let mut ipi_to_cpu: Option<usize> = None;
598    let updated = {
599        let mut scheduler = GLOBAL_SCHED_STATE.lock();
600        if let Some(ref mut sched) = *scheduler {
601            let cpu_index = sched.task_cpu.get(&id).copied().unwrap_or(0);
602            let task = match sched.all_tasks.get(&id).cloned() {
603                Some(t) => t,
604                None => return false,
605            };
606            task.set_sched_policy(policy);
607            let class = sched.class_table.class_for_task(&task);
608
609            if let Some(ref mut local_cpu) = *LOCAL_SCHEDULERS[cpu_index].lock() {
610                // If task is queued in ready classes, migrate it to the new class.
611                if local_cpu.class_rqs.remove(id) {
612                    local_cpu.class_rqs.enqueue(class, task.clone());
613                }
614                local_cpu.need_resched = true;
615            }
616            if cpu_index != current_cpu_index() {
617                ipi_to_cpu = Some(cpu_index);
618            }
619            sched_trace(format_args!(
620                "set_policy task={} cpu={} policy={:?}",
621                id.as_u64(),
622                cpu_index,
623                policy
624            ));
625            true
626        } else {
627            false
628        }
629    };
630    if let Some(ci) = ipi_to_cpu {
631        send_resched_ipi_to_cpu(ci);
632    }
633    restore_flags(saved_flags);
634    updated
635}
636
637/// Get parent task ID for a child task.
638pub fn get_parent_id(child: TaskId) -> Option<TaskId> {
639    SCHED_IDENTITY.read().parent_of.get(&child).copied()
640}
641
642/// Get parent process ID for a child task.
643pub fn get_parent_pid(child: TaskId) -> Option<Pid> {
644    let parent_tid = get_parent_id(child)?;
645    let parent = get_task_by_id(parent_tid)?;
646    Some(parent.pid)
647}
648
649/// Try to reap a zombie child.
650///
651/// `target=None` means "any child".
652pub fn try_wait_child(parent: TaskId, target: Option<TaskId>) -> WaitChildResult {
653    let saved_flags = save_flags_and_cli();
654    let result = {
655        let mut scheduler = GLOBAL_SCHED_STATE.lock();
656        if let Some(ref mut sched) = *scheduler {
657            sched.try_reap_child_locked(parent, target)
658        } else {
659            WaitChildResult::NoChildren
660        }
661    };
662    restore_flags(saved_flags);
663    result
664}
665
666/// Block the current task and yield to the scheduler.
667///
668/// The current task is moved from Running to Blocked state and placed
669/// in the `blocked_tasks` map. It will not be re-scheduled until
670/// `wake_task(id)` is called.
671///
672/// ## Lock design
673///
674/// This function acquires **only** the `BLOCKED_TASKS` lock + the current
675/// CPU's `LOCAL_SCHEDULERS[cpu]` lock. It does **not** touch
676/// `GLOBAL_SCHED_STATE`, avoiding contention with cold-path operations
677/// (fork, exit, kill).
678///
679/// ## Lost-wakeup prevention
680///
681/// Before actually blocking, this function checks the task's `wake_pending`
682/// flag. If a concurrent `wake_task()` fired between the moment the task
683/// added itself to a `WaitQueue` and this call, the flag will be set and
684/// the function returns immediately without blocking.
685///
686/// Must NOT be called with interrupts disabled or while holding the
687/// scheduler lock (this function acquires both).
688pub fn block_current_task() {
689    let saved_flags = save_flags_and_cli();
690    let cpu_index = current_cpu_index();
691
692    let switch_target = {
693        // Lock order: BLOCKED (rank 3) -> LOCAL (rank 4).
694        lockdep_acquire(LockRank::Blocked, None);
695        let mut blocked = super::BLOCKED_TASKS.lock();
696        lockdep_acquire(LockRank::Local, Some(cpu_index));
697        let mut local = LOCAL_SCHEDULERS[cpu_index].lock();
698        let out = if let Some(ref mut cpu) = *local {
699            if let Some(ref current) = cpu.current_task {
700                if current
701                    .wake_pending
702                    .swap(false, core::sync::atomic::Ordering::AcqRel)
703                {
704                    // Pending wakeup consumed - do not block.
705                    None
706                } else {
707                    current.set_state(TaskState::Blocked);
708                    // Record home CPU so wake_task can route without GLOBAL.
709                    current
710                        .home_cpu
711                        .store(cpu_index, core::sync::atomic::Ordering::Relaxed);
712                    blocked.insert(current.id, current.clone());
713                    super::core_impl::yield_cpu_local(cpu, cpu_index)
714                }
715            } else {
716                None
717            }
718        } else {
719            None
720        };
721        lockdep_release(LockRank::Local);
722        drop(local);
723        lockdep_release(LockRank::Blocked);
724        drop(blocked);
725        out
726    }; // Locks released
727
728    if let Some(ref target) = switch_target {
729        // SAFETY: no scheduler locks held, no allocation, no IPI.
730        unsafe {
731            crate::process::task::do_switch_context(target);
732        }
733        finish_switch();
734    }
735
736    restore_flags(saved_flags);
737}
738
739/// Wake a blocked task by its ID.
740///
741/// Moves the task from `blocked_tasks` to the ready queue and sets its
742/// state to Ready. Returns `true` if the task was found and woken.
743///
744/// ## Lock design
745///
746/// The primary path (task found in `BLOCKED_TASKS`) acquires **only** the
747/// `BLOCKED_TASKS` lock + the target CPU's `LOCAL_SCHEDULERS[cpu]` lock.
748/// It does **not** touch `GLOBAL_SCHED_STATE`, avoiding contention with
749/// cold-path operations (fork, exit, kill).
750///
751/// ## Lost-wakeup prevention
752///
753/// If the task is not yet in `blocked_tasks` (it is still transitioning
754/// from Ready -> Blocked inside `block_current_task()`), this function sets
755/// the task's `wake_pending` flag so that `block_current_task()` will see
756/// the pending wakeup and return immediately without actually blocking.
757pub fn wake_task(id: TaskId) -> bool {
758    let saved_flags = save_flags_and_cli();
759
760    // --- Primary path: task is in BLOCKED_TASKS ---
761    // Lock order: BLOCKED (rank 3) -> LOCAL (rank 4).
762    let mut ipi_cpu: Option<usize> = None;
763    let mut woken = false;
764
765    {
766        lockdep_acquire(LockRank::Blocked, None);
767        let mut blocked = super::BLOCKED_TASKS.lock();
768        if let Some(task) = blocked.remove(&id) {
769            task.set_state(TaskState::Ready);
770
771            // --- CPU placement: prefer last_cpu, then home_cpu ---
772            let last = task.last_cpu.load(core::sync::atomic::Ordering::Relaxed);
773            let home = task.home_cpu.load(core::sync::atomic::Ordering::Relaxed);
774            let n = crate::arch::smp::cpu_count()
775                .max(1)
776                .min(crate::arch::percpu::MAX_CPUS);
777
778            let cpu_index = if last < n {
779                let last_ok = {
780                    lockdep_acquire(LockRank::Local, Some(last));
781                    let ok = LOCAL_SCHEDULERS[last]
782                        .lock()
783                        .as_ref()
784                        .map(|c| c.class_rqs.runnable_len() <= 2)
785                        .unwrap_or(false);
786                    lockdep_release(LockRank::Local);
787                    ok
788                };
789                if last_ok {
790                    last
791                } else if home < n {
792                    home
793                } else {
794                    0
795                }
796            } else if home < n {
797                home
798            } else {
799                0
800            };
801
802            let class = {
803                use crate::process::sched::SchedClassId;
804                match task.sched_policy() {
805                    crate::process::sched::SchedPolicy::RealTimeRR { .. }
806                    | crate::process::sched::SchedPolicy::RealTimeFifo { .. } => {
807                        SchedClassId::RealTime
808                    }
809                    crate::process::sched::SchedPolicy::Fair(_) => SchedClassId::Fair,
810                    crate::process::sched::SchedPolicy::Idle => SchedClassId::Idle,
811                }
812            };
813
814            lockdep_acquire(LockRank::Local, Some(cpu_index));
815            if let Some(ref mut local_cpu) = *LOCAL_SCHEDULERS[cpu_index].lock() {
816                local_cpu.class_rqs.enqueue(class, task.clone());
817                local_cpu.need_resched = true;
818            }
819            lockdep_release(LockRank::Local);
820
821            ipi_cpu = if cpu_index != current_cpu_index() {
822                Some(cpu_index)
823            } else {
824                None
825            };
826            woken = true;
827        }
828        lockdep_release(LockRank::Blocked);
829        drop(blocked);
830    } // BLOCKED_TASKS lock released
831
832    if woken {
833        if let Some(ci) = ipi_cpu {
834            send_resched_ipi_to_cpu(ci);
835        }
836        restore_flags(saved_flags);
837        return true;
838    }
839
840    // === Fallback path: task not yet in BLOCKED_TASKS =================================
841    // Lock order: GLOBAL (rank 1) -> BLOCKED (rank 3) -> LOCAL (rank 4).
842    {
843        lockdep_acquire(LockRank::Global, None);
844        let mut scheduler = GLOBAL_SCHED_STATE.lock();
845        if let Some(ref mut sched) = *scheduler {
846            let (fallback_woken, fallback_ipi) = sched.wake_task_locked(id);
847            woken = fallback_woken;
848            if ipi_cpu.is_none() {
849                ipi_cpu = fallback_ipi;
850            }
851        }
852        lockdep_release(LockRank::Global);
853    }
854
855    if let Some(ci) = ipi_cpu {
856        send_resched_ipi_to_cpu(ci);
857    }
858    restore_flags(saved_flags);
859    woken
860}
861
862/// Sets task wake deadline.
863pub fn set_task_wake_deadline(id: TaskId, deadline_ns: u64) -> bool {
864    let saved_flags = save_flags_and_cli();
865    let out = {
866        let mut scheduler = GLOBAL_SCHED_STATE.lock();
867        if let Some(ref mut sched) = *scheduler {
868            sched.set_task_wake_deadline_locked(id, deadline_ns)
869        } else {
870            false
871        }
872    };
873    restore_flags(saved_flags);
874    out
875}
876
877/// Performs the clear task wake deadline operation.
878pub fn clear_task_wake_deadline(id: TaskId) -> bool {
879    set_task_wake_deadline(id, 0)
880}
881
882/// Suspend a task by ID (best-effort).
883///
884/// Moves the task to the blocked map and marks it Blocked.
885/// - If the task is the *current* task on *this* CPU, a context switch is
886///   performed immediately.
887/// - If the task is the *current* task on *another* CPU, an IPI is sent to
888///   trigger preemption on that CPU. The task will not be re-queued at the
889///   next tick because its state is Blocked.
890///
891/// ## Lock order
892///
893/// Canonical: BLOCKED (rank 3) → LOCAL (rank 4).
894/// First path (current task): re-acquires LOCAL only for yield_cpu_local,
895/// after BLOCKED has been released.
896/// Ready-queue path: acquires LOCAL to remove, releases it, then acquires
897/// BLOCKED to insert — this is safe because LOCAL is fully released before
898/// BLOCKED is acquired (no nesting).
899pub fn suspend_task(id: TaskId) -> bool {
900    let saved_flags = save_flags_and_cli();
901
902    let mut switch_target: Option<SwitchTarget> = None;
903    let mut suspended = false;
904    let mut ipi_to_cpu: Option<usize> = None;
905
906    let my_cpu = current_cpu_index();
907    let n = active_cpu_count();
908
909    // Path 1: Check if the task is the current task on any CPU.
910    // Lock order: LOCAL (read-only probe) → release → BLOCKED → LOCAL (yield).
911    // The first LOCAL is acquired and released before BLOCKED; no nesting.
912    for ci in 0..n {
913        lockdep_acquire(LockRank::Local, Some(ci));
914        let task_id_on_cpu = LOCAL_SCHEDULERS[ci]
915            .lock()
916            .as_ref()
917            .and_then(|cpu| cpu.current_task.as_ref().map(|t| (t.id, t.clone())));
918        lockdep_release(LockRank::Local);
919        if let Some((tid, current)) = task_id_on_cpu {
920            if tid == id {
921                current.set_state(TaskState::Blocked);
922                current
923                    .home_cpu
924                    .store(ci, core::sync::atomic::Ordering::Relaxed);
925                // Insert into BLOCKED_TASKS (rank 3).
926                lockdep_acquire(LockRank::Blocked, None);
927                super::BLOCKED_TASKS
928                    .lock()
929                    .insert(current.id, current.clone());
930                lockdep_release(LockRank::Blocked);
931                suspended = true;
932                if ci == my_cpu {
933                    // Re-acquire LOCAL (rank 4) to yield. Safe because IRQs
934                    // are disabled and BLOCKED is released.
935                    lockdep_acquire(LockRank::Local, Some(ci));
936                    let mut local = LOCAL_SCHEDULERS[ci].lock();
937                    if let Some(ref mut cpu) = *local {
938                        switch_target = super::core_impl::yield_cpu_local(cpu, ci);
939                    }
940                    lockdep_release(LockRank::Local);
941                } else {
942                    ipi_to_cpu = Some(ci);
943                }
944                break;
945            }
946        }
947    }
948
949    // Path 2: Remove from ready queues (task was not running anywhere).
950    // Lock order: LOCAL (remove from queue) → release → BLOCKED (insert).
951    // No nesting: LOCAL is fully released before BLOCKED is acquired.
952    if !suspended {
953        for ci in 0..n {
954            lockdep_acquire(LockRank::Local, Some(ci));
955            let removed = {
956                let mut local = LOCAL_SCHEDULERS[ci].lock();
957                if let Some(ref mut cpu) = *local {
958                    cpu.class_rqs.remove(id)
959                } else {
960                    false
961                }
962            };
963            lockdep_release(LockRank::Local);
964            if removed {
965                if let Some(task) = get_task_by_id(id) {
966                    task.set_state(TaskState::Blocked);
967                    task.home_cpu
968                        .store(ci, core::sync::atomic::Ordering::Relaxed);
969                    lockdep_acquire(LockRank::Blocked, None);
970                    super::BLOCKED_TASKS.lock().insert(task.id, task.clone());
971                    lockdep_release(LockRank::Blocked);
972                }
973                suspended = true;
974                break;
975            }
976        }
977    }
978
979    // Already blocked.
980    if !suspended {
981        lockdep_acquire(LockRank::Blocked, None);
982        if super::BLOCKED_TASKS.lock().contains_key(&id) {
983            suspended = true;
984        }
985        lockdep_release(LockRank::Blocked);
986    }
987
988    if let Some(ref target) = switch_target {
989        unsafe {
990            crate::process::task::do_switch_context(target);
991        }
992        finish_switch();
993    }
994
995    if let Some(ci) = ipi_to_cpu {
996        send_resched_ipi_to_cpu(ci);
997    }
998
999    restore_flags(saved_flags);
1000    suspended
1001}
1002
1003/// Resume a previously suspended task by ID.
1004///
1005/// Moves the task from blocked to ready queue and marks it Ready.
1006///
1007/// ## Lock order
1008///
1009/// BLOCKED (rank 3) → release → LOCAL (rank 4).  No nesting:
1010/// BLOCKED is fully released before LOCAL is acquired.
1011pub fn resume_task(id: TaskId) -> bool {
1012    let saved_flags = save_flags_and_cli();
1013    let mut ipi_to_cpu: Option<usize> = None;
1014
1015    let mut task_to_enqueue: Option<Arc<Task>> = None;
1016    {
1017        // Lock order: BLOCKED (rank 3).
1018        lockdep_acquire(LockRank::Blocked, None);
1019        let mut blocked = super::BLOCKED_TASKS.lock();
1020        if let Some(task) = blocked.remove(&id) {
1021            task.set_state(TaskState::Ready);
1022
1023            // --- CPU placement: prefer last_cpu, then home_cpu ---
1024            let last = task.last_cpu.load(core::sync::atomic::Ordering::Relaxed);
1025            let home = task.home_cpu.load(core::sync::atomic::Ordering::Relaxed);
1026            let n = crate::arch::smp::cpu_count()
1027                .max(1)
1028                .min(crate::arch::percpu::MAX_CPUS);
1029            let cpu_index = if last < n {
1030                let last_ok = LOCAL_SCHEDULERS[last]
1031                    .lock()
1032                    .as_ref()
1033                    .map(|c| c.class_rqs.runnable_len() <= 2)
1034                    .unwrap_or(false);
1035                if last_ok {
1036                    last
1037                } else if home < n {
1038                    home
1039                } else {
1040                    0
1041                }
1042            } else if home < n {
1043                home
1044            } else {
1045                0
1046            };
1047
1048            let class = {
1049                use crate::process::sched::SchedClassId;
1050                match task.sched_policy() {
1051                    crate::process::sched::SchedPolicy::RealTimeRR { .. }
1052                    | crate::process::sched::SchedPolicy::RealTimeFifo { .. } => {
1053                        SchedClassId::RealTime
1054                    }
1055                    crate::process::sched::SchedPolicy::Fair(_) => SchedClassId::Fair,
1056                    crate::process::sched::SchedPolicy::Idle => SchedClassId::Idle,
1057                }
1058            };
1059
1060            // Release BLOCKED before acquiring LOCAL.
1061            lockdep_release(LockRank::Blocked);
1062            drop(blocked);
1063
1064            // Lock order: LOCAL (rank 4).
1065            lockdep_acquire(LockRank::Local, Some(cpu_index));
1066            if let Some(ref mut local_cpu) = *LOCAL_SCHEDULERS[cpu_index].lock() {
1067                local_cpu.class_rqs.enqueue(class, task.clone());
1068                local_cpu.need_resched = true;
1069            }
1070            lockdep_release(LockRank::Local);
1071
1072            if cpu_index != current_cpu_index() {
1073                ipi_to_cpu = Some(cpu_index);
1074            }
1075            task_to_enqueue = Some(task);
1076        } else {
1077            lockdep_release(LockRank::Blocked);
1078            drop(blocked);
1079        }
1080    }
1081
1082    if let Some(ci) = ipi_to_cpu {
1083        send_resched_ipi_to_cpu(ci);
1084    }
1085    restore_flags(saved_flags);
1086    task_to_enqueue.is_some()
1087}
1088
1089/// Kill a task by ID (best-effort).
1090///
1091/// - Ready / blocked tasks are removed and marked Dead immediately.
1092/// - If the task is the *current* task on *this* CPU, a context switch is
1093///   performed immediately.
1094/// - If the task is the *current* task on *another* CPU, an IPI triggers
1095///   preemption on that CPU; the task will not be re-queued because its
1096///   state is Dead.
1097///
1098/// Returns `true` if the task was found and killed.
1099pub fn kill_task(id: TaskId) -> bool {
1100    let pid = crate::process::get_task_by_id(id)
1101        .map(|t| t.pid)
1102        .unwrap_or(0);
1103    crate::audit::log(
1104        crate::audit::AuditCategory::Process,
1105        pid,
1106        crate::silo::task_silo_id(id).unwrap_or(0),
1107        alloc::format!("kill_task tid={}", id.as_u64()),
1108    );
1109    let saved_flags = save_flags_and_cli();
1110
1111    let mut switch_target: Option<SwitchTarget> = None;
1112    let mut killed = false;
1113    let mut ipi_to_cpu: Option<usize> = None;
1114    let mut parent_to_signal: Option<TaskId> = None;
1115
1116    {
1117        lockdep_acquire(LockRank::Global, None);
1118        let mut scheduler = GLOBAL_SCHED_STATE.lock();
1119        if let Some(ref mut sched) = *scheduler {
1120            const FORCED_KILL_EXIT_CODE: i32 = 1;
1121            let my_cpu = current_cpu_index();
1122
1123            // Check if the task is the current task on any CPU.
1124            let n = active_cpu_count();
1125            let mut running_hit: Option<(usize, Arc<Task>)> = None;
1126            for ci in 0..n {
1127                lockdep_acquire(LockRank::Local, Some(ci));
1128                let hit = LOCAL_SCHEDULERS[ci].lock().as_ref().and_then(|cpu| {
1129                    cpu.current_task
1130                        .as_ref()
1131                        .map(|t| (t.id, t.get_state(), t.clone()))
1132                });
1133                lockdep_release(LockRank::Local);
1134                if let Some((tid, state, current)) = hit {
1135                    if tid == id {
1136                        if state != TaskState::Dead {
1137                            running_hit = Some((ci, current));
1138                        }
1139                        break;
1140                    }
1141                }
1142            }
1143            if let Some((ci, current)) = running_hit {
1144                let task_pid = current.pid;
1145                let _ = sched.clear_task_wake_deadline_locked(id);
1146                current.set_state(TaskState::Dead);
1147                sched.task_cpu.remove(&id);
1148                // Single identity write: unregister + reparent + remove parent.
1149                {
1150                    lockdep_acquire(LockRank::IdentityW, None);
1151                    let mut identity = SCHED_IDENTITY.write();
1152                    GlobalSchedState::unregister_identity_locked(
1153                        &mut identity,
1154                        id,
1155                        task_pid,
1156                        current.tid,
1157                    );
1158                    identity.parent_of.remove(&current.id);
1159                    lockdep_release(LockRank::IdentityW);
1160                    drop(identity);
1161                }
1162                // reparent_children needs a separate identity write since it
1163                // also needs sched.zombies check — but we can merge if we
1164                // pass a pre-built identity. For now keep separate.
1165                let (parent, ipi_death) = {
1166                    lockdep_acquire(LockRank::IdentityW, None);
1167                    let mut identity = SCHED_IDENTITY.write();
1168                    let result = reparent_children(sched, &mut identity, id);
1169                    // re-check parent after reparent may have removed it
1170                    let parent = identity.parent_of.get(&id).copied();
1171                    lockdep_release(LockRank::IdentityW);
1172                    drop(identity);
1173                    if let Some(parent_id) = parent {
1174                        sched.zombies.insert(id, (FORCED_KILL_EXIT_CODE, task_pid));
1175                        let (_, ipi_wake) = sched.wake_task_locked(parent_id);
1176                        (Some(parent_id), result.or(ipi_wake))
1177                    } else {
1178                        // Parent was not in identity, but might still need zombie entry
1179                        // if the task had a parent. Use reparent result as signal.
1180                        (None, result)
1181                    }
1182                };
1183                parent_to_signal = parent;
1184                killed = true;
1185                if ci == my_cpu {
1186                    lockdep_acquire(LockRank::Local, Some(ci));
1187                    let mut local = LOCAL_SCHEDULERS[ci].lock();
1188                    if let Some(ref mut cpu) = *local {
1189                        switch_target = super::core_impl::yield_cpu_local(cpu, ci);
1190                    }
1191                    lockdep_release(LockRank::Local);
1192                    drop(local);
1193                } else {
1194                    ipi_to_cpu = Some(ci);
1195                }
1196                if ipi_to_cpu.is_none() {
1197                    ipi_to_cpu = ipi_death;
1198                }
1199            }
1200
1201            // Remove from ready queues.
1202            if !killed {
1203                let mut removed_from_ready = false;
1204                for ci in 0..n {
1205                    lockdep_acquire(LockRank::Local, Some(ci));
1206                    let removed = {
1207                        let mut local = LOCAL_SCHEDULERS[ci].lock();
1208                        if let Some(ref mut cpu) = *local {
1209                            cpu.class_rqs.remove(id)
1210                        } else {
1211                            false
1212                        }
1213                    };
1214                    lockdep_release(LockRank::Local);
1215                    if removed {
1216                        removed_from_ready = true;
1217                        break;
1218                    }
1219                }
1220                if removed_from_ready {
1221                    let _ = sched.clear_task_wake_deadline_locked(id);
1222                    if let Some(task) = sched.remove_all_task_locked(id) {
1223                        let task_pid = task.pid;
1224                        task.set_state(TaskState::Dead);
1225                        cleanup_task_resources(&task);
1226                        sched.task_cpu.remove(&id);
1227                        {
1228                            lockdep_acquire(LockRank::IdentityW, None);
1229                            let mut identity = SCHED_IDENTITY.write();
1230                            GlobalSchedState::unregister_identity_locked(
1231                                &mut identity,
1232                                id,
1233                                task_pid,
1234                                task.tid,
1235                            );
1236                            lockdep_release(LockRank::IdentityW);
1237                            drop(identity);
1238                        }
1239                        let (parent, ipi_death) =
1240                            finalize_forced_death(sched, id, FORCED_KILL_EXIT_CODE, task_pid);
1241                        parent_to_signal = parent;
1242                        if ipi_to_cpu.is_none() {
1243                            ipi_to_cpu = ipi_death;
1244                        }
1245                    }
1246                    killed = true;
1247                }
1248            }
1249
1250            // Remove from blocked map.
1251            if !killed {
1252                lockdep_acquire(LockRank::Blocked, None);
1253                if let Some(task) = super::BLOCKED_TASKS.lock().remove(&id) {
1254                    lockdep_release(LockRank::Blocked);
1255                    let task_pid = task.pid;
1256                    let _ = sched.clear_task_wake_deadline_locked(id);
1257                    task.set_state(TaskState::Dead);
1258                    cleanup_task_resources(&task);
1259                    let _ = sched.remove_all_task_locked(id);
1260                    sched.task_cpu.remove(&id);
1261                    {
1262                        lockdep_acquire(LockRank::IdentityW, None);
1263                        let mut identity = SCHED_IDENTITY.write();
1264                        GlobalSchedState::unregister_identity_locked(
1265                            &mut identity,
1266                            id,
1267                            task_pid,
1268                            task.tid,
1269                        );
1270                        lockdep_release(LockRank::IdentityW);
1271                        drop(identity);
1272                    }
1273                    let (parent, ipi_death) =
1274                        finalize_forced_death(sched, id, FORCED_KILL_EXIT_CODE, task_pid);
1275                    parent_to_signal = parent;
1276                    if ipi_to_cpu.is_none() {
1277                        ipi_to_cpu = ipi_death;
1278                    }
1279                    killed = true;
1280                } else {
1281                    lockdep_release(LockRank::Blocked);
1282                }
1283            }
1284        }
1285        lockdep_release(LockRank::Global);
1286    } // scheduler lock released before IPI and context switch
1287
1288    if let Some(ref target) = switch_target {
1289        unsafe {
1290            crate::process::task::do_switch_context(target);
1291        }
1292        finish_switch();
1293    }
1294
1295    if let Some(ci) = ipi_to_cpu {
1296        send_resched_ipi_to_cpu(ci);
1297    }
1298
1299    if let Some(parent_id) = parent_to_signal {
1300        // Must happen outside scheduler lock to avoid lock recursion.
1301        let _ =
1302            crate::process::signal::send_signal(parent_id, crate::process::signal::Signal::SIGCHLD);
1303    }
1304
1305    restore_flags(saved_flags);
1306    killed
1307}
1308
1309/// Performs the finalize forced death operation.
1310///
1311/// Lock order: caller holds GLOBAL_SCHED_STATE (rank 1).
1312/// This function acquires SCHED_IDENTITY write (rank 2) exactly once.
1313fn finalize_forced_death(
1314    sched: &mut GlobalSchedState,
1315    task_id: TaskId,
1316    exit_code: i32,
1317    task_pid: Pid,
1318) -> (Option<TaskId>, Option<usize>) {
1319    // Single SCHED_IDENTITY.write() for all identity mutations.
1320    lockdep_acquire(LockRank::IdentityW, None);
1321    let mut identity = SCHED_IDENTITY.write();
1322    let ipi_reparent = reparent_children(sched, &mut identity, task_id);
1323    let parent = identity.parent_of.remove(&task_id);
1324    lockdep_release(LockRank::IdentityW);
1325    drop(identity);
1326
1327    if let Some(parent_id) = parent {
1328        sched.zombies.insert(task_id, (exit_code, task_pid));
1329        let (_, ipi_wake) = sched.wake_task_locked(parent_id);
1330        (Some(parent_id), ipi_reparent.or(ipi_wake))
1331    } else {
1332        (None, ipi_reparent)
1333    }
1334}
1335
1336/// Performs the reparent children operation.
1337/// Reparent children of a dying task to PID 1 (init), or drop parent links
1338/// if PID 1 is not available.
1339///
1340/// # Orphan policy
1341///
1342/// - **PID 1 exists:** Orphans are reparented to init, which is responsible
1343///   for reaping them. This matches standard Unix semantics.
1344/// - **PID 1 does not exist:** Parent links are dropped entirely. Orphans
1345///   become parentless : they continue running but cannot be `wait()`-ed on.
1346///   This avoids the nondeterministic fallback of adopting an arbitrary task
1347///   (which might be short-lived or unsuitable for reaping).
1348/// - **PID 1 is the dying task:** Same as above : parent links are dropped.
1349fn reparent_children(
1350    sched: &mut GlobalSchedState,
1351    identity: &mut SchedIdentity,
1352    dying: TaskId,
1353) -> Option<usize> {
1354    let children = match identity.children_of.remove(&dying) {
1355        Some(c) => c,
1356        None => return None,
1357    };
1358
1359    // Preferred reaper: PID 1 (standard init process).
1360    let init_id = identity.pid_to_task.get(&1).copied();
1361
1362    let Some(init_id) = init_id else {
1363        // No PID 1: drop parent links. Orphans become parentless.
1364        for child in &children {
1365            identity.parent_of.remove(child);
1366        }
1367        return None;
1368    };
1369    if init_id == dying {
1370        // PID 1 is dying : cannot reparent to self. Drop parent links.
1371        for child in &children {
1372            identity.parent_of.remove(child);
1373        }
1374        return None;
1375    }
1376
1377    let mut has_zombie = false;
1378    let init_children = identity.children_of.entry(init_id).or_default();
1379    for child in children {
1380        if !has_zombie && sched.zombies.contains_key(&child) {
1381            has_zombie = true;
1382        }
1383        identity.parent_of.insert(child, init_id);
1384        init_children.push(child);
1385    }
1386    if has_zombie {
1387        let (_, ipi) = sched.wake_task_locked(init_id);
1388        ipi
1389    } else {
1390        None
1391    }
1392}
1393
1394/// Performs the cleanup task resources operation.
1395///
1396/// Called when a task exits or is killed to release ports, capabilities,
1397/// and user address space mappings.
1398///
1399/// # Safety
1400/// Must be called with the scheduler lock held and the task no longer
1401/// accessible from any global map (all_tasks, current_task, etc.).
1402fn queue_silo_cleanup(task_id: TaskId) {
1403    let mut guard = PENDING_SILO_CLEANUPS.lock();
1404    guard
1405        .push_back(task_id)
1406        .unwrap_or_else(|_| panic!("pending silo cleanup queue overflow"));
1407}
1408
1409pub fn flush_deferred_silo_cleanups() {
1410    let mut guard = match PENDING_SILO_CLEANUPS.try_lock() {
1411        Some(g) => g,
1412        None => return, // Lock held by preempted task or other CPU, skip safely
1413    };
1414    if guard.is_empty() {
1415        return;
1416    }
1417    let mut drained = FixedQueue::<TaskId, PENDING_SILO_CLEANUPS_CAPACITY>::new();
1418    core::mem::swap(&mut *guard, &mut drained);
1419    drop(guard);
1420    while let Some(task_id) = drained.pop_front() {
1421        crate::silo::on_task_terminated(task_id);
1422    }
1423}
1424
1425pub(crate) fn cleanup_task_resources(task: &Arc<Task>) {
1426    crate::ipc::port::cleanup_ports_for_task(task.id);
1427    queue_silo_cleanup(task.id);
1428
1429    // Kernel-owned user stack (/thread/create): reclaim if not already done
1430    // by exit_current_task. take()-based, so this is a cheap no-op otherwise.
1431    crate::process::thread_ops::reclaim_kernel_user_stack(task);
1432
1433    // SAFETY: strong_count is racy (a concurrent get_task_by_id may temporarily
1434    // hold an extra Arc ref). Worst case: cleanup is deferred until the last ref
1435    // drops elsewhere - no resource leak, just delayed release.
1436    let is_last_process_ref = Arc::strong_count(&task.process) == 1;
1437    if !is_last_process_ref {
1438        return;
1439    }
1440
1441    unsafe {
1442        (&mut *task.process.fd_table.get()).close_all();
1443        let capabilities = (&mut *task.process.capabilities.get()).take_all();
1444        for capability in &capabilities {
1445            crate::capability::release_capability(capability, Some(task.id));
1446        }
1447    }
1448
1449    let as_ref = task.process.address_space_arc();
1450    if !as_ref.is_kernel() && Arc::strong_count(&as_ref) == 1 {
1451        as_ref.unmap_all_user_regions();
1452    }
1453}
1454
1455/// Trace putc routed to the arch serial backend (replaces port 0xE9 debug).
1456#[inline]
1457pub(crate) fn debug_trace_putc(c: u8) {
1458    crate::arch::serial::_print(format_args!("{}", c as char));
1459}