diff --git a/src/process/foreign/fork.rs b/src/process/foreign/fork.rs index feed314d9..b5ec85e6d 100644 --- a/src/process/foreign/fork.rs +++ b/src/process/foreign/fork.rs @@ -20,10 +20,18 @@ use super::peer_guard::pid_arg; use crate::process::core::ProcessState; use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_NOENT, ERRNO_PERM}; +/// The last address of the user half. +const USER_VA_MAX: u64 = 0x0000_7FFF_FFFF_FFFF; + /// `MkForeignFork`: a second process holding the first one's register state, /// with zero in its return register so the two can tell each other apart, -/// which is the whole of fork's contract to the program. -pub fn sys_foreign_fork(pid: u64) -> i64 { +/// which is the whole of fork's contract to the program. A non-zero `rsp` is +/// the stack the child starts on instead of its parent's, as a clone that +/// names a stack gives it; it must lie in the user half. +pub fn sys_foreign_fork(pid: u64, rsp: u64) -> i64 { + if rsp > USER_VA_MAX { + return ERRNO_INVAL; + } let Some(caller) = crate::process::current_pid() else { return ERRNO_INVAL; }; @@ -34,8 +42,8 @@ pub fn sys_foreign_fork(pid: u64) -> i64 { if super::registry::supervisor_of(parent) != Some(caller) { return ERRNO_PERM; } - let Some(state) = saved_state(parent) else { - // A guest that is not parked inside a syscall has no frame to copy. + let Some(state) = super::trap_frame::parked_frame(parent) else { + /* A guest that is not parked inside a syscall has no frame to copy. */ return ERRNO_NOENT; }; let child = match super::spawn::empty_guest(caller, b"fork") { @@ -44,9 +52,14 @@ pub fn sys_foreign_fork(pid: u64) -> i64 { }; let mut frame = state; frame.rax = 0; - // The thread pointer is a register the frame does not carry, so the child - // takes its forking thread's, read from that thread's PCB. Without this a - // fork from a thread that set its own FS would give the child a zero one. + if rsp != 0 { + frame.rsp = rsp; + } + /* + * The thread pointer is a register the frame does not carry, so the child + * takes its forking thread's, read from that thread's PCB. Without this a + * fork from a thread that set its own FS would give the child a zero one. + */ let parent_tls = crate::process::with_process(parent, |pcb| pcb.get_tls_base()).unwrap_or(0); crate::process::with_process(child, |pcb| { *pcb.saved_user_context.lock() = Some(frame); @@ -57,7 +70,3 @@ pub fn sys_foreign_fork(pid: u64) -> i64 { }); child as i64 } - -fn saved_state(pid: u32) -> Option { - super::trap_frame::parked_frame(pid) -} diff --git a/src/syscall/microkernel/dispatch/process.rs b/src/syscall/microkernel/dispatch/process.rs index dddf6a4a9..bc9223609 100644 --- a/src/syscall/microkernel/dispatch/process.rs +++ b/src/syscall/microkernel/dispatch/process.rs @@ -97,7 +97,7 @@ pub(super) fn handle(nr: u64, a: Args) -> Option { SYS_PEER_PROTECT => sys_peer_protect(a.a0, a.a1, a.a2, a.a3), SYS_FOREIGN_THREAD => sys_foreign_thread(a.a0, a.a1, a.a2, a.a3, a.a4), SYS_PEER_TLS => sys_peer_tls(a.a0, a.a1), - SYS_FOREIGN_FORK => sys_foreign_fork(a.a0), + SYS_FOREIGN_FORK => sys_foreign_fork(a.a0, a.a1), SYS_PEER_UNMAP => sys_peer_unmap(a.a0, a.a1, a.a2), SYS_FOREIGN_EXEC => sys_foreign_exec(a.a0, a.a1, a.a2), SYS_LOCAL_SIGN => sys_local_sign(a.a0, a.a1, a.a2, a.a3, a.a4), diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs index 68e04090a..5641aa16d 100644 --- a/userland/capsule_linux/src/linux/abi/mod.rs +++ b/userland/capsule_linux/src/linux/abi/mod.rs @@ -23,4 +23,5 @@ pub mod name; pub mod nr; pub mod nr_path; pub mod nr_high; +pub mod nr_sig; pub mod nr_sched; diff --git a/userland/capsule_linux/src/linux/abi/nr_sig.rs b/userland/capsule_linux/src/linux/abi/nr_sig.rs new file mode 100644 index 000000000..6f5df6545 --- /dev/null +++ b/userland/capsule_linux/src/linux/abi/nr_sig.rs @@ -0,0 +1,39 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Syscall numbers for process lifecycle, signals and timers, transcribed +//! from the x86_64 table. + +pub const PAUSE: u64 = 34; +pub const GETITIMER: u64 = 36; +pub const ALARM: u64 = 37; +pub const SETITIMER: u64 = 38; +pub const KILL: u64 = 62; +pub const RT_SIGPENDING: u64 = 127; +pub const RT_SIGTIMEDWAIT: u64 = 128; +pub const RT_SIGQUEUEINFO: u64 = 129; +pub const RT_SIGSUSPEND: u64 = 130; +pub const TKILL: u64 = 200; +pub const TIMER_CREATE: u64 = 222; +pub const TIMER_SETTIME: u64 = 223; +pub const TIMER_GETTIME: u64 = 224; +pub const TIMER_GETOVERRUN: u64 = 225; +pub const TIMER_DELETE: u64 = 226; +pub const TGKILL: u64 = 234; +pub const WAITID: u64 = 247; +pub const RT_TGSIGQUEUEINFO: u64 = 297; +pub const SIGNALFD: u64 = 282; +pub const SIGNALFD4: u64 = 289; diff --git a/userland/capsule_linux/src/linux/call/io.rs b/userland/capsule_linux/src/linux/call/io.rs index f5145519b..973662bf0 100644 --- a/userland/capsule_linux/src/linux/call/io.rs +++ b/userland/capsule_linux/src/linux/call/io.rs @@ -37,6 +37,7 @@ pub fn write(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { Some(Kind::Unix) => crate::linux::unix::send(guest, fd, buf, len), Some(Kind::Pipe) => super::pipe_write(guest, fd, buf, len), Some(Kind::Event) => file::event_write(guest, fd, buf, len), + Some(Kind::Device) => file::dev_write(guest, fd, len), Some(Kind::Resolver) => net::dns::query(guest, fd, buf, len, LOOPBACK_53), Some(Kind::Dir) => errno::fail(errno::EISDIR), _ => errno::fail(errno::EBADF), @@ -52,6 +53,8 @@ pub fn read(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { Some(Kind::Unix) => crate::linux::unix::recv(guest, fd, buf, len), Some(Kind::Pipe) => super::pipe_read(guest, fd, buf, len), Some(Kind::Event) => file::event_read(guest, fd, buf, len), + Some(Kind::Device) => file::dev_read(guest, fd, buf, len), + Some(Kind::Signal) => super::signalfd_now(guest, fd, buf, len), Some(Kind::Resolver) => net::dns::answer_out(guest, fd, buf, len).0, Some(Kind::Dir) => errno::fail(errno::EISDIR), _ => errno::fail(errno::EBADF), diff --git a/userland/capsule_linux/src/linux/call/life.rs b/userland/capsule_linux/src/linux/call/life.rs index 10f50ec2b..c0b015e8e 100644 --- a/userland/capsule_linux/src/linux/call/life.rs +++ b/userland/capsule_linux/src/linux/call/life.rs @@ -14,8 +14,10 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Ending a guest. The call never returns to the guest, so the answer -//! handed back is only what parks it until the supervisor tears it down. +//! Ending a thread or a guest. The call never returns to the guest, so the +//! answer handed back is only what parks it until the supervisor tears it +//! down. A process's end is kept as Linux's wait status: an exit's code in +//! the second byte, or the number of the signal that ended it in the first. use nonos_libc::mk_kill; @@ -58,7 +60,15 @@ pub fn set_tid_address(guest: &mut Guest, tid: u32, word: u64) -> Answer { Answer::value(u64::from(tid)) } +/// exit_group: the whole process ends with `code`. pub fn exit(guest: &mut Guest, code: u64) -> u64 { - guest.exited = Some(code as i32); + guest.exited = Some(((code & 0xff) << 8) as i32); errno::ok(0) } + +/// The process ends on `signum`, unless something already ended it. +pub fn killed(guest: &mut Guest, signum: u8) { + if guest.exited.is_none() { + guest.exited = Some(i32::from(signum & 0x7f)); + } +} diff --git a/userland/capsule_linux/src/linux/call/life_one.rs b/userland/capsule_linux/src/linux/call/life_one.rs new file mode 100644 index 000000000..4b1df4086 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/life_one.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A plain exit, which ends the calling thread only. The process ends with +//! the last of its threads, and that thread's code is its status. + +use super::life::{exit, exit_thread}; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +/// A plain exit ends the calling thread only; the process ends with the last +/// of its threads, and that thread's code is its status. The leader cannot be +/// killed while others run: its pid is the address space every peer call +/// names. So it stays parked inside its exit, a zombie that runs nothing, +/// until the process ends and takes it. +pub fn exit_one(guest: &mut Guest, tid: u32, code: u64) -> Answer { + if tid == guest.pid { + if let Some(at) = guest.clear_tids.iter().position(|(t, _)| *t == tid) { + let (_, word) = guest.clear_tids.remove(at); + if guest.write(word, &0u32.to_le_bytes()) == 4 { + guest.wake(word, 1); + } + } + guest.signals.leader_gone = true; + } else { + let _ = exit_thread(guest, tid); + } + guest.forget_thread(tid); + if guest.live_threads().is_empty() { + let _ = exit(guest, code); + } + Answer::Park +} diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index ab445d17c..53010e330 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -29,6 +29,7 @@ mod io; mod ioctl; mod io_socket; mod life; +mod life_one; mod limits; mod limits_table; mod glibc; @@ -43,12 +44,32 @@ mod pipe_read; mod sched; mod session; pub mod sigframe; +pub mod sigframe_build; +mod sigframe_read; mod signal; +mod signal_its; +mod signalfd; +mod signalfd_info; +mod signalfd_read; +mod signalfd_take; +mod signal_itv; +mod signal_mask; +mod signal_post; +mod signal_queue; +mod signal_real; mod signal_send; +mod signal_stack; +mod signal_timedwait; +mod signal_timer; +mod signal_wait; mod sigreturn; mod sleep; mod spawn; mod thread; +mod timer_create; +mod timer_ops; +mod timer_set; +mod timer_sigev; mod timeops; mod umask; mod uname; @@ -61,7 +82,8 @@ pub use cwd::{chdir, fchdir, getcwd}; pub use futex::futex; pub use ident::{getppid, setuid}; pub use io::{close, read, write}; -pub use life::{exit, exit_thread, set_tid_address}; +pub use life::{exit, exit_thread, killed, set_tid_address}; +pub use life_one::exit_one; pub use limits::{getrlimit, prlimit64}; pub use glibc::prctl; pub use glibc_sched::{clone3, getcpu, membarrier, sched_getaffinity}; @@ -76,14 +98,27 @@ pub use sched::{ sched_setscheduler, }; pub use session::{getpgid, getsid, setpgid, setsid}; -pub use signal::{rt_sigaction, rt_sigprocmask, sigaltstack}; +pub use signal::rt_sigaction; +pub use signal_mask::rt_sigprocmask; +pub use signal_queue::{rt_sigqueueinfo, rt_tgsigqueueinfo}; +pub use signalfd::signalfd4; +pub use signalfd_read::signalfd_read; +pub use signalfd_take::{signalfd_bits, signalfd_now, signalfd_take}; +pub use signal_send::{kill, kill_from, sigpipe, tgkill_from}; +pub use signal_stack::sigaltstack; +pub use signal_timer::{alarm, getitimer, setitimer}; +pub use signal_timedwait::rt_sigtimedwait; +pub use signal_wait::{pause, rt_sigpending, rt_sigsuspend}; pub use sigreturn::rt_sigreturn; -pub use signal_send::kill; pub use sleep::{clock_nanosleep, nanosleep}; -pub use spawn::{clone, execve, fork, reap_one, wait4}; +pub use spawn::{clone, clone_process, execve, fork, vfork, wait4, wait4_usage, waitid}; +pub use spawn::{WALL, WCLONE, WEXITED, WNOHANG, WNOWAIT}; pub use clock::{clock_getres, clock_gettime, now_ms}; pub use epoch::{family_ms, mark_start}; pub use thread::{arch_prctl, getrandom}; +pub use timer_create::timer_create; +pub use timer_ops::{timer_delete, timer_getoverrun, timer_gettime}; +pub use timer_set::timer_settime; pub use timeops::{gettimeofday, time}; pub use umask::{umask, DEFAULT_UMASK}; pub use uname::uname; diff --git a/userland/capsule_linux/src/linux/call/sigframe.rs b/userland/capsule_linux/src/linux/call/sigframe.rs index f6bdd2e71..cd833ba4f 100644 --- a/userland/capsule_linux/src/linux/call/sigframe.rs +++ b/userland/capsule_linux/src/linux/call/sigframe.rs @@ -14,62 +14,35 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! The `rt_sigframe` x86-64 puts on a thread's stack to enter a signal handler, -//! and where to read it back on return. Pure, so the layout is checked against -//! a round trip without a guest. It matches Linux `struct rt_sigframe`: -//! pretcode u64, ucontext at +8, siginfo at +312; the ucontext's sigcontext -//! holds the 18 words `mk_foreign_context` uses, in that order. +//! The `rt_sigframe` x86-64 puts on a thread's stack to enter a signal handler: +//! its layout, and what a handler is entered with (sigframe_build writes it, +//! sigframe_read reads it back). Pure, so the layout is checked against a +//! round trip without a guest. It matches Linux `struct rt_sigframe`: +//! pretcode u64, then the ucontext at +8 (uc_flags, uc_link, the 24-byte +//! uc_stack, the 256-byte sigcontext at +40, uc_sigmask at +296), then the +//! 128-byte siginfo at +312. The sigcontext starts with the 18 words +//! `mk_foreign_context` uses, in that order. -use alloc::vec::Vec; +pub const WORDS: usize = 18; /* r8..r15,rdi,rsi,rbp,rbx,rdx,rax,rcx,rsp,rip,rflags */ +pub const FRAME_SIZE: usize = 440; +pub const UC_OFF: usize = 8; +pub const STACK_OFF: usize = 16; /* uc_stack within the ucontext */ +pub const SIGCONTEXT_OFF: usize = 40; /* uc_mcontext within the ucontext */ +pub const OLDMASK_WORD: usize = 21; /* sigcontext.oldmask, after cs/gs/fs/ss, err, trapno */ +pub const SIGMASK_OFF: usize = 296; /* uc_sigmask within the ucontext */ +pub const INFO_OFF: usize = 312; +pub const INFO_LEN: usize = 128; -pub const WORDS: usize = 18; // r8..r15,rdi,rsi,rbp,rbx,rdx,rax,rcx,rsp,rip,rflags -const FRAME_SIZE: usize = 440; -const UC_OFF: usize = 8; -pub const SIGCONTEXT_OFF: usize = 48; // sigcontext within the ucontext -const INFO_OFF: usize = 312; -const REDZONE: u64 = 128; // the System V red zone below rsp - -fn put(buf: &mut [u8], at: usize, v: u64) { - buf[at..at + 8].copy_from_slice(&v.to_le_bytes()); -} -/// Where the frame lands, the bytes to write there, and the registers that -/// enter the handler. `None` if the stack is too low to hold a frame. -pub fn build( - regs: &[u64; WORDS], - handler: u64, - restorer: u64, - signum: u32, - blocked: u64, -) -> Option<(u64, Vec, [u64; WORDS])> { - // Below the red zone, 16-aligned, then down 8 so the handler sees rsp+8 - // aligned as a call would leave it. - let frame = - (regs[15].checked_sub(REDZONE)?.checked_sub(FRAME_SIZE as u64)? & !15u64).checked_sub(8)?; - let mut buf = alloc::vec![0u8; FRAME_SIZE]; - put(&mut buf, 0, restorer); - let mc = UC_OFF + SIGCONTEXT_OFF; // the 18 words, then the sigmask - for (i, w) in regs.iter().enumerate() { - put(&mut buf, mc + i * 8, *w); - } - put(&mut buf, UC_OFF + SIGCONTEXT_OFF + WORDS * 8, blocked); - put(&mut buf, INFO_OFF, u64::from(signum)); // siginfo: si_signo - let mut out = [0u64; WORDS]; - out[8] = u64::from(signum); // rdi - out[9] = frame + INFO_OFF as u64; // rsi, &siginfo - out[12] = frame + UC_OFF as u64; // rdx, &ucontext - out[15] = frame; // rsp at the frame; rax stays 0, no vector registers - out[16] = handler; // rip - out[17] = regs[17]; // rflags, the kernel masks it - Some((frame, buf, out)) -} - -/// The 18 words a returning frame carries, from the ucontext the guest's rsp -/// points at: the trampoline's `ret` left rsp there. -pub fn returned(uc: &[u8]) -> Option<[u64; WORDS]> { - let mut out = [0u64; WORDS]; - for (i, slot) in out.iter_mut().enumerate() { - let at = SIGCONTEXT_OFF + i * 8; - *slot = u64::from_le_bytes(uc.get(at..at + 8)?.try_into().ok()?); - } - Some(out) +/// What a handler is entered with, beyond the registers it interrupts. +pub struct Entry<'a> { + pub handler: u64, + pub restorer: u64, + pub signum: u32, + /// The mask the thread had, restored by rt_sigreturn. + pub blocked: u64, + /// The top of the alternate stack when the frame goes there. + pub alt_top: Option, + /// uc_stack as Linux saves it: ss_sp, ss_flags, ss_size. + pub stack: [u64; 3], + pub info: &'a [u8; INFO_LEN], } diff --git a/userland/capsule_linux/src/linux/call/sigframe_build.rs b/userland/capsule_linux/src/linux/call/sigframe_build.rs new file mode 100644 index 000000000..191fc5a51 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/sigframe_build.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Building the `rt_sigframe` a handler is entered through: the bytes, where +//! on the stack they go, and the registers the handler starts with. Pure, so +//! the frame is checked without a guest. + +use alloc::vec::Vec; + +use super::sigframe::{Entry, FRAME_SIZE, INFO_OFF, OLDMASK_WORD, SIGCONTEXT_OFF}; +use super::sigframe::{SIGMASK_OFF, STACK_OFF, UC_OFF, WORDS}; + +const REDZONE: u64 = 128; /* the System V red zone below rsp */ +/* The flags a handler is entered with cleared: direction, resume, trap. */ +const ENTRY_CLEARS: u64 = 0x400 | 0x1_0000 | 0x100; + +fn put(buf: &mut [u8], at: usize, v: u64) { + buf[at..at + 8].copy_from_slice(&v.to_le_bytes()); +} + +/// Where the frame lands, the bytes to write there, and the registers that +/// enter the handler. `None` if the stack is too low to hold a frame. +pub fn build(regs: &[u64; WORDS], e: &Entry) -> Option<(u64, Vec, [u64; WORDS])> { + /* + * Below the red zone, or at the top of the alternate stack; 16-aligned, + * then down 8 so the handler sees rsp+8 aligned as a call would leave it. + */ + let top = match e.alt_top { + Some(top) => top, + None => regs[15].checked_sub(REDZONE)?, + }; + let frame = (top.checked_sub(FRAME_SIZE as u64)? & !15u64).checked_sub(8)?; + let mut buf = alloc::vec![0u8; FRAME_SIZE]; + put(&mut buf, 0, e.restorer); + let (st, mc) = (UC_OFF + STACK_OFF, UC_OFF + SIGCONTEXT_OFF); + put(&mut buf, st, e.stack[0]); + buf[st + 8..st + 12].copy_from_slice(&(e.stack[1] as u32).to_le_bytes()); + put(&mut buf, st + 16, e.stack[2]); + for (i, w) in regs.iter().enumerate() { + put(&mut buf, mc + i * 8, *w); + } + put(&mut buf, mc + OLDMASK_WORD * 8, e.blocked); + put(&mut buf, UC_OFF + SIGMASK_OFF, e.blocked); + buf[INFO_OFF..].copy_from_slice(e.info); + /* The rest of the registers are the interrupted ones, as Linux leaves them. */ + let mut out = *regs; + out[8] = u64::from(e.signum); /* rdi */ + out[9] = frame + INFO_OFF as u64; /* rsi, &siginfo */ + out[12] = frame + UC_OFF as u64; /* rdx, &ucontext */ + out[13] = 0; /* rax: no vector registers passed */ + out[15] = frame; + out[16] = e.handler; + out[17] = regs[17] & !ENTRY_CLEARS; + Some((frame, buf, out)) +} diff --git a/userland/capsule_linux/src/linux/call/sigframe_read.rs b/userland/capsule_linux/src/linux/call/sigframe_read.rs new file mode 100644 index 000000000..8e71733af --- /dev/null +++ b/userland/capsule_linux/src/linux/call/sigframe_read.rs @@ -0,0 +1,47 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Reading back the frame `sigframe` built, as rt_sigreturn does: the +//! registers, the mask and the alternate stack the handler returns to. Pure, +//! and any bytes at all may sit where the guest's rsp points, so every read is +//! checked and none can panic. + +use super::sigframe::{SIGCONTEXT_OFF, SIGMASK_OFF, STACK_OFF, WORDS}; + +fn word(uc: &[u8], at: usize) -> Option { + Some(u64::from_le_bytes(uc.get(at..at + 8)?.try_into().ok()?)) +} + +/// The 18 words a returning frame carries, from the ucontext the guest's rsp +/// points at: the trampoline's `ret` left rsp there. +pub fn returned(uc: &[u8]) -> Option<[u64; WORDS]> { + let mut out = [0u64; WORDS]; + for (i, slot) in out.iter_mut().enumerate() { + *slot = word(uc, SIGCONTEXT_OFF + i * 8)?; + } + Some(out) +} + +/// The mask a returning frame restores, from uc_sigmask. +pub fn returned_mask(uc: &[u8]) -> Option { + word(uc, SIGMASK_OFF) +} + +/// uc_stack as the handler left it: ss_sp, ss_flags, ss_size. +pub fn returned_stack(uc: &[u8]) -> Option<[u64; 3]> { + let flags = u32::from_le_bytes(uc.get(STACK_OFF + 8..STACK_OFF + 12)?.try_into().ok()?); + Some([word(uc, STACK_OFF)?, u64::from(flags), word(uc, STACK_OFF + 16)?]) +} diff --git a/userland/capsule_linux/src/linux/call/signal.rs b/userland/capsule_linux/src/linux/call/signal.rs index a0f5698c0..459d8fa66 100644 --- a/userland/capsule_linux/src/linux/call/signal.rs +++ b/userland/capsule_linux/src/linux/call/signal.rs @@ -14,32 +14,39 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Signal dispositions, recorded here and delivered on the return path in -//! `serve::deliver`: a handler is kept with its flags, restorer and mask. +//! Signal dispositions and masks, recorded here and acted on in +//! `serve::deliver`: a handler is kept with its flags, restorer and mask, and +//! each thread has the mask of signals it holds back (signal_mask). use crate::linux::abi::errno; -use crate::linux::guest::sigstate::{SigAction, NSIG}; +use crate::linux::guest::sigstate::{SigAction, NSIG, SIGKILL, SIGSTOP}; use crate::linux::guest::Guest; -// SIGKILL/SIGSTOP cannot be caught; `struct sigaction` is 32 bytes. -const SIGKILL: u64 = 9; -const SIGSTOP: u64 = 19; +/* `struct sigaction` is 32 bytes; a kernel sigset_t is 8. */ const SIGACTION_LEN: usize = 32; +pub const SIGSET_LEN: u64 = 8; -pub fn rt_sigaction(guest: &mut Guest, signum: u64, act: u64, old: u64) -> u64 { - if signum == 0 || signum > NSIG as u64 || signum == SIGKILL || signum == SIGSTOP { +pub fn rt_sigaction(guest: &mut Guest, signum: u64, act: u64, old: u64, size: u64) -> u64 { + let unchangeable = signum == u64::from(SIGKILL) || signum == u64::from(SIGSTOP); + if size != SIGSET_LEN || signum == 0 || signum > NSIG as u64 || (act != 0 && unchangeable) { return errno::fail(errno::EINVAL); } let n = signum as usize; - if old != 0 - && guest.write(old, &encode(guest.signals.action(n).unwrap_or_default())) - < SIGACTION_LEN as i64 - { + let was = guest.signals.action(n).unwrap_or_default(); + let new = match act { + 0 => None, + at => match guest.read(at, SIGACTION_LEN) { + Some(raw) => Some(decode(&raw)), + None => return errno::fail(errno::EFAULT), + }, + }; + if old != 0 && guest.write(old, &encode(was)) < SIGACTION_LEN as i64 { return errno::fail(errno::EFAULT); } - if act != 0 { - match guest.read(act, SIGACTION_LEN) { - Some(raw) => guest.signals.set(n, decode(&raw)), - None => return errno::fail(errno::EFAULT), + if let Some(a) = new { + guest.signals.set(n, a); + /* A signal set to be ignored is dropped where it already waits. */ + if guest.signals.discards(n as u8) { + guest.signals.discard(n as u8); } } errno::ok(0) @@ -57,19 +64,3 @@ fn encode(a: SigAction) -> [u8; SIGACTION_LEN] { b[24..32].copy_from_slice(&a.mask.to_le_bytes()); b } - -/// The old mask reads back empty: nothing is held back, delivery ignores it. -pub fn rt_sigprocmask(guest: &Guest, old: u64) -> u64 { - if old != 0 && guest.write(old, &[0u8; 8]) < 8 { - return errno::fail(errno::EFAULT); - } - errno::ok(0) -} - -/// The old alternate stack reads back unset; a handler uses the own stack. -pub fn sigaltstack(guest: &Guest, old: u64) -> u64 { - if old != 0 && guest.write(old, &[0u8; 24]) < 24 { - return errno::fail(errno::EFAULT); - } - errno::ok(0) -} diff --git a/userland/capsule_linux/src/linux/call/signal_its.rs b/userland/capsule_linux/src/linux/call/signal_its.rs new file mode 100644 index 000000000..59500b6c2 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_its.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `struct itimerspec` in and out: interval, then value, each seconds and +//! nanoseconds. Nanoseconds round up to the millisecond the clock keeps. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +const NSEC: u64 = 1_000_000_000; +const MS: u64 = 1_000_000; + +/// (interval, value) in milliseconds, or the errno a bad one earns. +pub fn read_spec(guest: &Guest, at: u64) -> Result<(u64, u64), u64> { + let raw = guest.read(at, 32).ok_or(errno::fail(errno::EFAULT))?; + let w = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + let ms = |s: u64, ns: u64| match ns < NSEC && (s as i64) >= 0 { + true => Ok(s.saturating_mul(1000).saturating_add(ns.div_ceil(MS))), + false => Err(errno::fail(errno::EINVAL)), + }; + Ok((ms(w(0), w(8))?, ms(w(16), w(24))?)) +} + +/// Write (interval, value) at `at`, if the caller gave somewhere to write. +pub fn write_spec(guest: &Guest, at: u64, interval: u64, value: u64) -> u64 { + if at == 0 { + return errno::ok(0); + } + let mut b = [0u8; 32]; + for (i, ms) in [interval, value].iter().enumerate() { + b[i * 16..i * 16 + 8].copy_from_slice(&(ms / 1000).to_le_bytes()); + b[i * 16 + 8..i * 16 + 16].copy_from_slice(&((ms % 1000) * MS).to_le_bytes()); + } + match guest.write(at, &b) < 32 { + true => errno::fail(errno::EFAULT), + false => errno::ok(0), + } +} diff --git a/userland/capsule_linux/src/linux/call/signal_itv.rs b/userland/capsule_linux/src/linux/call/signal_itv.rs new file mode 100644 index 000000000..edaad03b2 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_itv.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `struct itimerval` in and out: interval, then value, each seconds and +//! microseconds. Microseconds round up to the millisecond the clock keeps. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +const USEC: u64 = 1_000_000; + +/// (interval, value) in milliseconds, or the errno a bad one earns. +pub fn read_val(guest: &Guest, at: u64) -> Result<(u64, u64), u64> { + let raw = guest.read(at, 32).ok_or(errno::fail(errno::EFAULT))?; + let w = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + let ms = |s: u64, us: u64| match us < USEC && (s as i64) >= 0 { + true => Ok(s.saturating_mul(1000).saturating_add(us.div_ceil(1000))), + false => Err(errno::fail(errno::EINVAL)), + }; + Ok((ms(w(0), w(8))?, ms(w(16), w(24))?)) +} + +/// Write (interval, value) at `at`, if the caller gave somewhere to write. +pub fn put(guest: &Guest, at: u64, interval: u64, value: u64) -> u64 { + if at == 0 { + return errno::ok(0); + } + let mut b = [0u8; 32]; + for (i, ms) in [interval, value].iter().enumerate() { + b[i * 16..i * 16 + 8].copy_from_slice(&(ms / 1000).to_le_bytes()); + b[i * 16 + 8..i * 16 + 16].copy_from_slice(&((ms % 1000) * 1000).to_le_bytes()); + } + match guest.write(at, &b) < 32 { + true => errno::fail(errno::EFAULT), + false => errno::ok(0), + } +} diff --git a/userland/capsule_linux/src/linux/call/signal_mask.rs b/userland/capsule_linux/src/linux/call/signal_mask.rs new file mode 100644 index 000000000..d6cf92b97 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_mask.rs @@ -0,0 +1,52 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! rt_sigprocmask: each thread's mask of the signals it holds back, which +//! `serve::deliver` reads before a signal is taken. + +use super::signal::SIGSET_LEN; +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +const SIG_BLOCK: u64 = 0; +const SIG_UNBLOCK: u64 = 1; +const SIG_SETMASK: u64 = 2; + +/// The calling thread's mask: added to, taken from or replaced. The old mask +/// is written after the change is made, as Linux orders it. +pub fn rt_sigprocmask(guest: &mut Guest, tid: u32, how: u64, set: u64, old: u64, size: u64) -> u64 { + if size != SIGSET_LEN { + return errno::fail(errno::EINVAL); + } + let was = guest.signals.blocked(tid); + if set != 0 { + let Some(raw) = guest.read(set, 8) else { + return errno::fail(errno::EFAULT); + }; + let new = u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8])); + let mask = match how { + SIG_BLOCK => was | new, + SIG_UNBLOCK => was & !new, + SIG_SETMASK => new, + _ => return errno::fail(errno::EINVAL), + }; + guest.signals.set_blocked(tid, mask); + } + if old != 0 && guest.write(old, &was.to_le_bytes()) < 8 { + return errno::fail(errno::EFAULT); + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/signal_post.rs b/userland/capsule_linux/src/linux/call/signal_post.rs new file mode 100644 index 000000000..db511df84 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_post.rs @@ -0,0 +1,45 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a sent signal goes: queued here for the caller's own process, or +//! through the outbox with the caller parked, for the family to route and +//! answer once it knows whether anyone was there. + +use crate::linux::abi::errno; +use crate::linux::guest::siginfo::SigInfo; +use crate::linux::guest::sigwaits::{Outbound, Target}; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +/// Queue `info` for `to`: here when it is this process, else through the +/// family. A signal number of 0 only asks whether the target exists. +pub fn post(guest: &mut Guest, tid: u32, to: Target, info: SigInfo) -> Answer { + let here = match to { + Target::Process(p) if guest.owns(p) => Some(0), + Target::Thread(g, t) if guest.owns(t) && (g == 0 || g == guest.pid) => Some(t), + _ => None, + }; + let Some(t) = here else { + guest.signals.outbox.push(Outbound { from: tid, to, info }); + return Answer::Park; + }; + /* A realtime signal past the queue limit is refused, as Linux does. */ + let s = info.signo; + if s != 0 && !guest.signals.raise(t, info) && s >= 32 && !guest.signals.discards(s) { + return Answer::value(errno::fail(errno::EAGAIN)); + } + Answer::value(errno::ok(0)) +} diff --git a/userland/capsule_linux/src/linux/call/signal_queue.rs b/userland/capsule_linux/src/linux/call/signal_queue.rs new file mode 100644 index 000000000..1aadf65c4 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_queue.rs @@ -0,0 +1,72 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `rt_sigqueueinfo` and `rt_tgsigqueueinfo`: a signal with the siginfo the +//! caller wrote, which is how sigqueue hands a value along. As on Linux, only +//! a process signalling itself may claim a kernel code (si_code 0 or above) or +//! SI_TKILL; the signal number is the call's, whatever the siginfo says. + +use crate::linux::abi::errno; +use crate::linux::guest::siginfo::{SigInfo, SI_TKILL}; +use crate::linux::guest::sigstate::NSIG; +use crate::linux::guest::sigwaits::Target; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +pub fn rt_sigqueueinfo(guest: &mut Guest, tid: u32, tgid: u64, sig: u64, uinfo: u64) -> Answer { + queue(guest, tid, Target::Process(tgid as u32), tgid, sig, uinfo) +} + +pub fn rt_tgsigqueueinfo( + guest: &mut Guest, + tid: u32, + tgid: u64, + target: u64, + sig: u64, + uinfo: u64, +) -> Answer { + if (tgid as i64) <= 0 || (target as i64) <= 0 { + return Answer::value(errno::fail(errno::EINVAL)); + } + queue(guest, tid, Target::Thread(tgid as u32, target as u32), tgid, sig, uinfo) +} + +fn queue(guest: &mut Guest, tid: u32, to: Target, tgid: u64, sig: u64, uinfo: u64) -> Answer { + if sig > NSIG as u64 { + return Answer::value(errno::fail(errno::EINVAL)); + } + let Some(raw) = guest.read(uinfo, 32) else { + return Answer::value(errno::fail(errno::EFAULT)); + }; + let word = |i: usize| u32::from_le_bytes(raw[i..i + 4].try_into().unwrap_or([0; 4])); + let code = word(8) as i32; + /* Linux compares the caller's own thread with the one it names. */ + let named = match to { + Target::Thread(_, t) => t, + _ => tgid as u32, + }; + if (code >= 0 || code == SI_TKILL) && named != tid { + return Answer::value(errno::fail(errno::EPERM)); + } + let info = SigInfo { + signo: sig as u8, + code, + pid: crate::linux::serve::kernel_pid(word(16)).unwrap_or(0), + timer: None, + value: u64::from_le_bytes(raw[24..32].try_into().unwrap_or([0; 8])), + }; + super::signal_post::post(guest, tid, to, info) +} diff --git a/userland/capsule_linux/src/linux/call/signal_real.rs b/userland/capsule_linux/src/linux/call/signal_real.rs new file mode 100644 index 000000000..9e878eece --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_real.rs @@ -0,0 +1,38 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! ITIMER_REAL on the family's monotonic clock: arming it, and what it has +//! left. alarm and setitimer both come here. + +use crate::linux::call::now_ms; +use crate::linux::guest::sigtimer::Itimer; +use crate::linux::guest::Guest; + +const CLOCK_MONOTONIC: u64 = 1; + +/// Arm ITIMER_REAL for `value` ms, repeating every `interval`; 0 disarms. +/// Answers what it had left and its old interval. +pub fn arm(guest: &mut Guest, value: u64, interval: u64) -> (u64, u64) { + let was = remaining(guest); + let now = now_ms(CLOCK_MONOTONIC).unwrap_or(0); + guest.signals.real = (value != 0).then(|| Itimer { due: now.saturating_add(value), interval }); + was +} + +pub fn remaining(guest: &Guest) -> (u64, u64) { + let now = now_ms(CLOCK_MONOTONIC).unwrap_or(0); + guest.signals.real.map_or((0, 0), |t| (t.due.saturating_sub(now).max(1), t.interval)) +} diff --git a/userland/capsule_linux/src/linux/call/signal_send.rs b/userland/capsule_linux/src/linux/call/signal_send.rs index 7a9d7f0cb..3b96aa524 100644 --- a/userland/capsule_linux/src/linux/call/signal_send.rs +++ b/userland/capsule_linux/src/linux/call/signal_send.rs @@ -14,50 +14,61 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `kill`, `tkill` and `tgkill`, for the guest's own threads and children. -//! A signal the process catches is queued and delivered on that thread's next -//! return; one whose default is to ignore is dropped; a fatal default ends it. - -use nonos_libc::mk_kill; +//! `kill`, `tkill` and `tgkill`, and SIGPIPE. A signal for the caller's own +//! process is queued here and taken as `serve::deliver` decides; one for +//! another process of the family leaves through the outbox with the caller +//! parked, and the family answers it once it knows whether anyone was there. +//! A guest reaches only its own family: pid_map refuses any other number. +use super::signal_post::post; use crate::linux::abi::errno; -use crate::linux::guest::sigstate::NSIG; +use crate::linux::guest::siginfo::{SigInfo, SI_TKILL, SI_USER}; +use crate::linux::guest::sigstate::{NSIG, SIGPIPE}; +use crate::linux::guest::sigwaits::Target; use crate::linux::guest::Guest; +use crate::linux::serve::Answer; -/// Signals whose default action is to be ignored: child status, urgent data, -/// window size, and a continue with nothing stopped. -const IGNORED_DEFAULT: [u64; 4] = [17, 23, 28, 18]; +/// kill(pid, sig) from thread `tid`, which parks when the family must answer. +pub fn kill_from(guest: &mut Guest, tid: u32, pid: u64, signo: u64) -> Answer { + let to = match pid as i64 { + -1 => Target::All, + 0 => Target::Group(guest.pgid), + p if p < 0 => Target::Group(p.unsigned_abs() as u32), + p => Target::Process(p as u32), + }; + send(guest, tid, to, signo, SI_USER) +} +/// kill for a caller that cannot park: a signal for another process still +/// goes to the family, with no one to answer, so a target that is gone reads +/// as 0 here rather than ESRCH. pub fn kill(guest: &mut Guest, pid: u64, signo: u64) -> u64 { - let target = pid as u32; - // A guest may signal itself, its threads and its children, nothing else. - if !guest.owns(target) && !guest.children.contains(&target) { - return errno::fail(errno::ESRCH); + match kill_from(guest, 0, pid, signo) { + Answer::Reply(v) => v, + Answer::Park => errno::ok(0), } - if signo == 0 { - return errno::ok(0); // an existence check, not a signal +} + +/// tkill(tid, sig), and tgkill(tgid, tid, sig) with `tgid` non-zero. +pub fn tgkill_from(guest: &mut Guest, tid: u32, tgid: u64, target: u64, signo: u64) -> Answer { + if (tgid as i64) < 0 || (target as i64) <= 0 { + return Answer::value(errno::fail(errno::EINVAL)); } + send(guest, tid, Target::Thread(tgid as u32, target as u32), signo, SI_TKILL) +} + +fn send(guest: &mut Guest, tid: u32, to: Target, signo: u64, code: i32) -> Answer { if signo > NSIG as u64 { - return errno::fail(errno::EINVAL); - } - let act = guest.signals.action(signo as usize).unwrap_or_default(); - if act.catches() { - guest.signals.raise(target, signo as u8); - return errno::ok(0); + return Answer::value(errno::fail(errno::EINVAL)); } - if act.ignores() || IGNORED_DEFAULT.contains(&signo) { - return errno::ok(0); - } - terminate(guest, target, signo) + let info = SigInfo::from(signo as u8, code, guest.pid); + post(guest, tid, to, info) } -/// The default action of an uncaught, non-ignored signal is to end the thread. -fn terminate(guest: &mut Guest, target: u32, signo: u64) -> u64 { - guest.forget_waits(target); - guest.threads.retain(|t| *t != target); - guest.signals.forget(target); - match mk_kill(target as u64, signo) { - n if n < 0 => errno::fail(errno::EPERM), - _ => errno::ok(0), - } +/// A write to a pipe no process can read raises SIGPIPE at the writing +/// thread, as Linux's pipe_write does, before the write answers EPIPE: caught, +/// the handler runs over that EPIPE; ignored, only EPIPE is seen; at its +/// default, the process ends. `tid` 0 raises it at the process. +pub fn sigpipe(guest: &mut Guest, tid: u32) { + let _ = guest.signals.raise(tid, SigInfo::from(SIGPIPE, SI_USER, guest.pid)); } diff --git a/userland/capsule_linux/src/linux/call/signal_stack.rs b/userland/capsule_linux/src/linux/call/signal_stack.rs new file mode 100644 index 000000000..5c4aee2f5 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_stack.rs @@ -0,0 +1,69 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sigaltstack`: the calling thread's alternate signal stack, where a +//! handler asking for SA_ONSTACK runs. `stack_t` is ss_sp, a 4-byte ss_flags +//! and its padding, then ss_size. + +use nonos_libc::{mk_foreign_context, ForeignRegs}; + +use crate::linux::abi::errno; +use crate::linux::guest::sigthread::{SS_DISABLE, SS_ONSTACK}; +use crate::linux::guest::Guest; + +const STACK_T_LEN: usize = 24; +/// MINSIGSTKSZ on x86-64. +const MINSIGSTKSZ: u64 = 2048; +/// SS_AUTODISARM, accepted and kept as Linux accepts it. +const SS_AUTODISARM: u64 = 1 << 31; +const RSP: usize = 15; + +pub fn sigaltstack(guest: &mut Guest, tid: u32, ss: u64, old: u64) -> u64 { + let [sp, flags, size] = guest.signals.thread(tid).alt; + let mut regs: ForeignRegs = [0; 18]; + let rsp = if mk_foreign_context(tid, &mut regs) == 0 { regs[RSP] } else { 0 }; + let on = flags & SS_DISABLE == 0 && rsp.wrapping_sub(sp) < size; + if old != 0 { + let now = if on { SS_ONSTACK } else { flags & (SS_DISABLE | SS_AUTODISARM) }; + let mut b = [0u8; STACK_T_LEN]; + b[..8].copy_from_slice(&sp.to_le_bytes()); + b[8..12].copy_from_slice(&(now as u32).to_le_bytes()); + b[16..].copy_from_slice(&size.to_le_bytes()); + if guest.write(old, &b) < STACK_T_LEN as i64 { + return errno::fail(errno::EFAULT); + } + } + if ss == 0 { + return errno::ok(0); + } + let Some(raw) = guest.read(ss, STACK_T_LEN) else { + return errno::fail(errno::EFAULT); + }; + let word = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + let new_flags = u64::from(u32::from_le_bytes(raw[8..12].try_into().unwrap_or([0; 4]))); + if on { + return errno::fail(errno::EPERM); + } + /* SS_ONSTACK asks for the same as 0: an enabled stack. */ + let alt = match new_flags & !(SS_AUTODISARM | SS_ONSTACK) { + SS_DISABLE => [0, SS_DISABLE, 0], + 0 if word(16) < MINSIGSTKSZ => return errno::fail(errno::ENOMEM), + 0 => [word(0), new_flags & SS_AUTODISARM, word(16)], + _ => return errno::fail(errno::EINVAL), + }; + guest.signals.thread(tid).alt = alt; + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/signal_timedwait.rs b/userland/capsule_linux/src/linux/call/signal_timedwait.rs new file mode 100644 index 000000000..43236207a --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_timedwait.rs @@ -0,0 +1,75 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! rt_sigtimedwait takes a pending signal of its set without running a +//! handler, or parks until one comes or its time runs out with EAGAIN. + +use super::signal_wait::{read_set, SIGSET_LEN}; +use crate::linux::abi::errno; +use crate::linux::call::now_ms; +use crate::linux::guest::sigwaits::SigWait; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +const CLOCK_MONOTONIC: u64 = 1; +const NSEC: u64 = 1_000_000_000; + +pub fn rt_sigtimedwait( + guest: &mut Guest, + tid: u32, + set: u64, + info: u64, + ts: u64, + size: u64, +) -> Answer { + if size != SIGSET_LEN { + return Answer::value(errno::fail(errno::EINVAL)); + } + let Some(want) = read_set(guest, set) else { + return Answer::value(errno::fail(errno::EFAULT)); + }; + let due = match ts { + 0 => None, + at => match wait_ms(guest, at) { + Ok(ms) => now_ms(CLOCK_MONOTONIC).map(|now| now.saturating_add(ms)), + Err(e) => return Answer::value(e), + }, + }; + /* A signal of the set already waiting is taken now, whatever the timeout. */ + if let Some(got) = guest.signals.take(tid, want) { + let bytes = got.bytes(); + if info != 0 && guest.write(info, &bytes) < bytes.len() as i64 { + return Answer::value(errno::fail(errno::EFAULT)); + } + return Answer::value(errno::ok(u64::from(got.signo))); + } + if due.is_some_and(|d| now_ms(CLOCK_MONOTONIC).is_some_and(|now| d <= now)) { + return Answer::value(errno::fail(errno::EAGAIN)); + } + guest.signals.sigwaits.push(SigWait { tid, set: want, info, due, records: 0 }); + Answer::Park +} + +/// A relative timespec, in whole milliseconds rounded up. +fn wait_ms(guest: &Guest, at: u64) -> Result { + let raw = guest.read(at, 16).ok_or(errno::fail(errno::EFAULT))?; + let secs = u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8])); + let nanos = u64::from_le_bytes(raw[8..].try_into().unwrap_or([0; 8])); + if nanos >= NSEC || secs > i64::MAX as u64 { + return Err(errno::fail(errno::EINVAL)); + } + Ok(secs.saturating_mul(1000).saturating_add(nanos.div_ceil(1_000_000))) +} diff --git a/userland/capsule_linux/src/linux/call/signal_timer.rs b/userland/capsule_linux/src/linux/call/signal_timer.rs new file mode 100644 index 000000000..e4fa90156 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_timer.rs @@ -0,0 +1,73 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! alarm, setitimer and getitimer. ITIMER_REAL counts on the monotonic clock +//! and raises SIGALRM at the process when it runs out; exec keeps it, fork +//! does not. ITIMER_VIRTUAL and ITIMER_PROF count the process's CPU time, +//! which the kernel does not report to a supervisor: arming one is refused by +//! name, and reading one reports it disarmed, which it always is. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::signal_itv::{put, read_val}; +use super::signal_real::{arm, remaining}; + +const ITIMER_REAL: u64 = 0; +const ITIMER_PROF: u64 = 2; + +/// alarm(seconds): a one-shot ITIMER_REAL, answering the whole seconds the +/// last one had left, rounded as Linux rounds them. +pub fn alarm(guest: &mut Guest, secs: u64) -> u64 { + let (was, _) = arm(guest, secs.min(u64::from(u32::MAX)).saturating_mul(1000), 0); + let (whole, ms) = (was / 1000, was % 1000); + errno::ok(whole + u64::from((whole == 0 && ms > 0) || ms >= 500)) +} + +pub fn setitimer(guest: &mut Guest, which: u64, new: u64, old: u64) -> u64 { + if which > ITIMER_PROF { + return errno::fail(errno::EINVAL); + } + let (interval, value) = match new { + 0 => (0, 0), + at => match read_val(guest, at) { + Ok(v) => v, + Err(e) => return e, + }, + }; + if which != ITIMER_REAL { + if value == 0 { + return put(guest, old, 0, 0); + } + let _ = nonos_libc::mk_debug(UNSERVED.as_ptr(), UNSERVED.len()); + return errno::fail(errno::ENOSYS); + } + let (was, every) = arm(guest, value, interval); + put(guest, old, every, was) +} + +const UNSERVED: &[u8] = b"[LINUX] unserved setitimer: no CPU-time clock for VIRTUAL or PROF\n"; + +pub fn getitimer(guest: &mut Guest, which: u64, out: u64) -> u64 { + if which > ITIMER_PROF { + return errno::fail(errno::EINVAL); + } + let (left, every) = match which { + ITIMER_REAL => remaining(guest), + _ => (0, 0), + }; + put(guest, out, every, left) +} diff --git a/userland/capsule_linux/src/linux/call/signal_wait.rs b/userland/capsule_linux/src/linux/call/signal_wait.rs new file mode 100644 index 000000000..412bfb21f --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_wait.rs @@ -0,0 +1,60 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Waiting for a signal. pause and rt_sigsuspend park until a handler runs, +//! and answer EINTR through it; sigsuspend waits under the mask it was given +//! and the handler's return puts the old one back. rt_sigpending names what +//! waits; rt_sigtimedwait is in signal_timedwait. + +use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::SigWait; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +pub const SIGSET_LEN: u64 = 8; + +pub fn pause(guest: &mut Guest, tid: u32) -> Answer { + guest.signals.sigwaits.push(SigWait { tid, set: 0, info: 0, due: None, records: 0 }); + Answer::Park +} + +pub fn rt_sigsuspend(guest: &mut Guest, tid: u32, set: u64, size: u64) -> Answer { + if size != SIGSET_LEN { + return Answer::value(errno::fail(errno::EINVAL)); + } + let Some(mask) = read_set(guest, set) else { + return Answer::value(errno::fail(errno::EFAULT)); + }; + let old = guest.signals.blocked(tid); + guest.signals.set_blocked(tid, mask); + guest.signals.thread(tid).saved = Some(old); + pause(guest, tid) +} + +pub fn rt_sigpending(guest: &mut Guest, tid: u32, out: u64, size: u64) -> u64 { + if size > SIGSET_LEN { + return errno::fail(errno::EINVAL); + } + let held = guest.signals.pending_for(tid) & guest.signals.blocked(tid); + match guest.write(out, &held.to_le_bytes()[..size as usize]) < size as i64 { + true => errno::fail(errno::EFAULT), + false => errno::ok(0), + } +} + +pub fn read_set(guest: &Guest, at: u64) -> Option { + guest.read(at, 8).map(|raw| u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8]))) +} diff --git a/userland/capsule_linux/src/linux/call/signalfd.rs b/userland/capsule_linux/src/linux/call/signalfd.rs new file mode 100644 index 000000000..6d72f8c22 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signalfd.rs @@ -0,0 +1,60 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `signalfd4` and `signalfd`: a descriptor that reads the signals of a mask +//! pending for the reading thread or its process, as `signalfd_siginfo` +//! records, instead of their handlers running. A read with none pending +//! answers EAGAIN on a non-blocking descriptor and otherwise parks the +//! thread until one comes, as Linux's does. + +use crate::linux::abi::errno; +use crate::linux::file::install; +use crate::linux::guest::sigstate::blockable; +use crate::linux::guest::{Fd, Guest, Kind}; + +const SFD_NONBLOCK: u64 = 0o4000; +const SFD_CLOEXEC: u64 = 0o2000000; +const SIGSET_LEN: u64 = 8; + +pub fn signalfd4(guest: &mut Guest, fd: u64, mask: u64, size: u64, flags: u64) -> u64 { + if size != SIGSET_LEN || flags & !(SFD_NONBLOCK | SFD_CLOEXEC) != 0 { + return errno::fail(errno::EINVAL); + } + let Some(raw) = guest.read(mask, 8) else { + return errno::fail(errno::EFAULT); + }; + let set = blockable(u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8]))); + if fd as i64 != -1 { + /* An existing signalfd takes the new mask; any other descriptor is refused. */ + return match guest.fds.get(fd as usize).filter(|f| f.kind == Kind::Signal) { + Some(f) => { + let at = f.handle as usize; + guest.signals.sigfds[at] = set; + errno::ok(fd) + } + None => errno::fail(errno::EINVAL), + }; + } + guest.signals.sigfds.push(set); + let mut new = Fd::empty(Kind::Signal); + new.handle = (guest.signals.sigfds.len() - 1) as u32; + new.nonblock = flags & SFD_NONBLOCK != 0; + new.cloexec = flags & SFD_CLOEXEC != 0; + match install(guest, new) { + Some(n) => errno::ok(n), + None => errno::fail(errno::EMFILE), + } +} diff --git a/userland/capsule_linux/src/linux/call/signalfd_info.rs b/userland/capsule_linux/src/linux/call/signalfd_info.rs new file mode 100644 index 000000000..3c56a9c7e --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signalfd_info.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A signal as a signalfd hands it over: the 128-byte `signalfd_siginfo`, +//! laid out as Linux's, with the fields the siginfo carries moved to their +//! own places in it. + +use crate::linux::guest::siginfo::{SigInfo, CLD_EXITED, CLD_KILLED}; +use crate::linux::guest::sigstate::SIGCHLD; + +pub const SFD_INFO_LEN: usize = 128; + +impl SigInfo { + pub fn fd_bytes(&self) -> [u8; SFD_INFO_LEN] { + let mut b = [0u8; SFD_INFO_LEN]; + let mut put32 = |at: usize, v: u32| b[at..at + 4].copy_from_slice(&v.to_le_bytes()); + put32(0, u32::from(self.signo)); + put32(8, self.code as u32); + let pid = match self.pid { + 0 => 0, + k => crate::linux::serve::guest_pid(k), + }; + let child = self.signo == SIGCHLD && matches!(self.code, CLD_EXITED | CLD_KILLED); + match self.timer { + /* ssi_tid is the timer's id, and ssi_overrun its overruns. */ + Some((id, overrun)) => { + put32(24, id as u32); + put32(32, overrun as u32); + } + None => put32(12, pid), + } + if child { + put32(40, self.value as u32); + } else { + put32(44, self.value as u32); + } + if !child { + b[48..56].copy_from_slice(&self.value.to_le_bytes()); + } + b + } +} diff --git a/userland/capsule_linux/src/linux/call/signalfd_read.rs b/userland/capsule_linux/src/linux/call/signalfd_read.rs new file mode 100644 index 000000000..77657847b --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signalfd_read.rs @@ -0,0 +1,47 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A read of a signalfd by one thread: the signals of its mask pending for +//! that thread or its process, EAGAIN on a non-blocking descriptor when there +//! are none, or a wait until one comes. + +use super::signalfd_info::SFD_INFO_LEN; +use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::SigWait; +use crate::linux::guest::{Guest, Kind}; +use crate::linux::serve::Answer; + +use super::signalfd_take::signalfd_take; + +/// A read by thread `tid`: what is pending now, EAGAIN, or a parked wait. +pub fn signalfd_read(guest: &mut Guest, tid: u32, fd: u64, buf: u64, len: u64) -> Answer { + let Some(f) = guest.fds.get(fd as usize).filter(|f| f.kind == Kind::Signal) else { + return Answer::value(errno::fail(errno::EBADF)); + }; + let (set, nonblock) = (guest.signals.sigfds[f.handle as usize], f.nonblock); + let records = len / SFD_INFO_LEN as u64; + if records == 0 { + return Answer::value(errno::fail(errno::EINVAL)); + } + if let Some(n) = signalfd_take(guest, tid, set, buf, records) { + return Answer::value(n); + } + if nonblock { + return Answer::value(errno::fail(errno::EAGAIN)); + } + guest.signals.sigwaits.push(SigWait { tid, set, info: buf, due: None, records }); + Answer::Park +} diff --git a/userland/capsule_linux/src/linux/call/signalfd_take.rs b/userland/capsule_linux/src/linux/call/signalfd_take.rs new file mode 100644 index 000000000..2b8ca6aa2 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signalfd_take.rs @@ -0,0 +1,69 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Taking signals for a signalfd read: as many pending signals of the mask as +//! records fit, each written where the next record goes. None when nothing is +//! pending, so the read can wait or answer EAGAIN. + +use super::signalfd_info::SFD_INFO_LEN; +use crate::linux::abi::errno; +use crate::linux::guest::{Guest, Kind}; + +pub fn signalfd_take(guest: &mut Guest, tid: u32, set: u64, buf: u64, records: u64) -> Option { + let mut n = 0; + while n < records { + let Some(info) = guest.signals.take(tid, set) else { + break; + }; + let at = buf + n * SFD_INFO_LEN as u64; + if guest.write(at, &info.fd_bytes()) < SFD_INFO_LEN as i64 { + return Some(if n == 0 { errno::fail(errno::EFAULT) } else { n * SFD_INFO_LEN as u64 }); + } + n += 1; + } + (n > 0).then_some(n * SFD_INFO_LEN as u64) +} + +/// poll's POLLIN when a signal of the mask waits for the process or its +/// first thread; a read by another thread may find its own too. +pub fn signalfd_bits(guest: &Guest, fd: u64) -> u16 { + const POLLIN: u16 = 0x001; + let Some(f) = guest.fds.get(fd as usize).filter(|f| f.kind == Kind::Signal) else { + return 0; + }; + let set = guest.signals.sigfds.get(f.handle as usize).copied().unwrap_or(0); + let who = guest.live_threads().first().copied().unwrap_or(guest.pid); + if guest.signals.pending_for(who) & set != 0 { + POLLIN + } else { + 0 + } +} + +/// A read that reaches a signalfd another way than read(2), readv: what is +/// pending for the process or its first thread now, else EAGAIN. +pub fn signalfd_now(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { + let Some(f) = guest.fds.get(fd as usize).filter(|f| f.kind == Kind::Signal) else { + return errno::fail(errno::EBADF); + }; + let set = guest.signals.sigfds.get(f.handle as usize).copied().unwrap_or(0); + let who = guest.live_threads().first().copied().unwrap_or(guest.pid); + let records = len / SFD_INFO_LEN as u64; + match records { + 0 => errno::fail(errno::EINVAL), + n => signalfd_take(guest, who, set, buf, n).unwrap_or(errno::fail(errno::EAGAIN)), + } +} diff --git a/userland/capsule_linux/src/linux/call/sigreturn.rs b/userland/capsule_linux/src/linux/call/sigreturn.rs index fa0dae276..fb2d28ae3 100644 --- a/userland/capsule_linux/src/linux/call/sigreturn.rs +++ b/userland/capsule_linux/src/linux/call/sigreturn.rs @@ -15,32 +15,49 @@ // along with this program. If not, see . //! `rt_sigreturn`: a thread leaving a signal handler. Its rsp points at the -//! ucontext the frame carried, so the saved registers are read back from -//! there and the kernel resumes the thread into them. Nothing is replied: the -//! thread is no longer in the syscall, it is back where the signal interrupted. +//! ucontext the frame carried, so the saved registers, the mask and the +//! alternate stack are read back from there and the kernel resumes the thread +//! into them. Nothing is replied: the thread is no longer in the syscall, it +//! is back where the signal interrupted. A frame that cannot be read ends the +//! process with SIGSEGV, as Linux's does. use nonos_libc::{mk_foreign_context, mk_foreign_signal, ForeignRegs, SIGNAL_RETURN}; -use super::sigframe::{returned, SIGCONTEXT_OFF, WORDS}; +use super::sigframe::{SIGMASK_OFF, WORDS}; +use super::sigframe_read::{returned, returned_mask, returned_stack}; +use crate::linux::guest::sigstate::SIGSEGV; +use crate::linux::guest::sigthread::{SS_DISABLE, SS_ONSTACK}; use crate::linux::guest::Guest; use crate::linux::serve::Answer; /// rsp in the register word order. const RSP: usize = 15; -pub fn rt_sigreturn(guest: &Guest, tid: u32) -> Answer { +pub fn rt_sigreturn(guest: &mut Guest, tid: u32) -> Answer { let mut regs: ForeignRegs = [0; WORDS]; if mk_foreign_context(tid, &mut regs) != 0 { return Answer::Park; } - // The trampoline's `ret` left rsp at the ucontext; the sigcontext follows. - let want = SIGCONTEXT_OFF + WORDS * 8; - let Some(bytes) = guest.read(regs[RSP], want) else { + /* The trampoline's `ret` left rsp at the ucontext; the mask ends it. */ + let read = guest.read(regs[RSP], SIGMASK_OFF + 8); + let back = read + .as_deref() + .and_then(|uc| Some((returned(uc)?, returned_mask(uc)?, returned_stack(uc)?))); + let Some((restored, mask, stack)) = back else { + super::killed(guest, SIGSEGV); return Answer::Park; }; - let Some(restored) = returned(&bytes) else { - return Answer::Park; - }; - let _ = mk_foreign_signal(tid, &restored, SIGNAL_RETURN); + guest.signals.set_blocked(tid, mask); + let t = guest.signals.thread(tid); + let on_now = t.alt[1] & SS_DISABLE == 0 && restored[RSP].wrapping_sub(t.alt[0]) < t.alt[2]; + if !on_now { + t.alt = match stack[1] & SS_DISABLE { + 0 => [stack[0], stack[1] & !SS_ONSTACK, stack[2]], + _ => [0, SS_DISABLE, 0], + }; + } + if mk_foreign_signal(tid, &restored, SIGNAL_RETURN) != 0 { + super::killed(guest, SIGSEGV); + } Answer::Park } diff --git a/userland/capsule_linux/src/linux/call/spawn/clone.rs b/userland/capsule_linux/src/linux/call/spawn/clone.rs index 96d316879..284914e46 100644 --- a/userland/capsule_linux/src/linux/call/spawn/clone.rs +++ b/userland/capsule_linux/src/linux/call/spawn/clone.rs @@ -62,6 +62,7 @@ pub fn clone(guest: &mut Guest, frame: &ForeignFrame) -> Answer { } let tid = tid as u32; guest.threads.push(tid); + guest.signals.born(frame.pid, tid); /* its creator's mask, as clone gives */ // Linux writes the new tid where the caller asked, and ignores a word it // cannot write; musl keeps the parent's copy as the thread's own tid. if flags & CLONE_PARENT_SETTID != 0 { diff --git a/userland/capsule_linux/src/linux/call/spawn/exec.rs b/userland/capsule_linux/src/linux/call/spawn/exec.rs index 825ff2ae0..d79cab699 100644 --- a/userland/capsule_linux/src/linux/call/spawn/exec.rs +++ b/userland/capsule_linux/src/linux/call/spawn/exec.rs @@ -47,7 +47,20 @@ pub fn execve(guest: &mut Guest, pid: u32, path: u64, argv: u64, envp: u64) -> A super::exec_threads::reap(guest, pid); clear(guest); match load_over(guest, pid, &program, &env) { - Some(()) => Answer::Park, + Some(()) => { + released(guest, pid); + Answer::Park + } None => Answer::value(errno::fail(errno::ENOEXEC)), } } + +/// The new program keeps what Linux keeps of the old one's signals, and a +/// vfork parent waiting on this exec is let go with the child's pid. +fn released(guest: &mut Guest, pid: u32) { + guest.signals.exec_reset(pid); + if let Some(parent) = guest.signals.vfork.take() { + let child = u64::from(crate::linux::serve::guest_pid(guest.pid)); + let _ = nonos_libc::mk_foreign_reply(parent, child); + } +} diff --git a/userland/capsule_linux/src/linux/call/spawn/exec_threads.rs b/userland/capsule_linux/src/linux/call/spawn/exec_threads.rs index 4ad723091..ac9efeaa8 100644 --- a/userland/capsule_linux/src/linux/call/spawn/exec_threads.rs +++ b/userland/capsule_linux/src/linux/call/spawn/exec_threads.rs @@ -34,7 +34,7 @@ pub fn reap(guest: &mut Guest, caller: u32) { * can be ended: the kill marks it, but nothing collects a thread that * is still waiting for an answer. */ - guest.forget_waits(tid); + guest.forget_thread(tid); let _ = mk_foreign_reply(tid, errno::fail(errno::EINTR)); let _ = mk_kill(tid as u64, SIGKILL); } diff --git a/userland/capsule_linux/src/linux/call/spawn/fork.rs b/userland/capsule_linux/src/linux/call/spawn/fork.rs index 55502a762..37c2c624b 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork.rs @@ -14,40 +14,18 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `fork`. - -use nonos_libc::{mk_foreign_fork, mk_foreign_resume}; +//! `fork`: a copy of the calling process that raises SIGCHLD when it ends. use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SIGCHLD; use crate::linux::guest::Guest; use crate::linux::serve::Answer; -use super::fork_copy::copy_spans; +use super::fork_child::fork_child; pub fn fork(guest: &mut Guest, caller: u32) -> Answer { - let child = mk_foreign_fork(caller); - if child < 0 { - return Answer::value(errno::fail(errno::ENOMEM)); - } - let child = child as u32; - if !copy_spans(guest, child) { - return Answer::value(errno::fail(errno::ENOMEM)); - } - /* - * The thread pointer is a register, not memory, so copying the spans does - * not carry it. The kernel fork carries the forking thread's own FS to the - * child, which is right whichever thread forked; the personality's single - * fs_base is only the last thread to set one and would be wrong here. - */ - /* - * The child's state goes to the serve loop before the child runs, so its - * first trap finds a guest that owns it. - */ - guest.forked.push(guest.fork_state(child)); - if mk_foreign_resume(child) < 0 { - guest.forked.pop(); - return Answer::value(errno::fail(errno::ENOMEM)); + match fork_child(guest, caller, 0, SIGCHLD, |_| {}) { + Ok(child) => Answer::value(errno::ok(child as u64)), + Err(e) => Answer::value(e), } - guest.children.push(child); - Answer::value(errno::ok(child as u64)) } diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_child.rs b/userland/capsule_linux/src/linux/call/spawn/fork_child.rs new file mode 100644 index 000000000..85d9212c1 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/fork_child.rs @@ -0,0 +1,71 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The copy every new process starts as: fork's, vfork's and a clone that +//! makes a process all come here. + +use nonos_libc::{mk_foreign_fork_at, mk_foreign_resume}; + +use super::fork_copy::copy_spans; +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SIGCHLD; +use crate::linux::guest::Guest; + +/// A new process copied from this one, running, and adopted by the family +/// once this answer is given. It raises `exit_signal` at its parent when it +/// ends. Its signal state is the forking thread's, as Linux's fork gives it. +/// It starts on `stack` when that is not zero, as a clone naming a stack asks. +/// `prep` runs on the child before it runs at all. +pub(super) fn fork_child( + guest: &mut Guest, + caller: u32, + stack: u64, + exit_signal: u8, + prep: impl FnOnce(&mut Guest), +) -> Result { + let child = mk_foreign_fork_at(caller, stack); + if child < 0 { + return Err(errno::fail(errno::ENOMEM)); + } + let child = child as u32; + if !copy_spans(guest, child) { + return Err(errno::fail(errno::ENOMEM)); + } + /* + * The thread pointer is a register, not memory, so copying the spans does + * not carry it. The kernel fork carries the forking thread's own FS to the + * child, which is right whichever thread forked; the personality's single + * fs_base is only the last thread to set one and would be wrong here. + */ + /* + * The child's state goes to the serve loop before the child runs, so its + * first trap finds a guest that owns it. + */ + let mut state = guest.fork_state(child); + state.signals = guest.signals.forked(caller, child); + state.signals.exit_signal = exit_signal; + prep(&mut state); + guest.forked.push(state); + if mk_foreign_resume(child) < 0 { + guest.forked.pop(); + return Err(errno::fail(errno::ENOMEM)); + } + guest.children.push(child); + if exit_signal != SIGCHLD { + guest.signals.clone_kids.push(child); + } + Ok(child) +} diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs index 3cf86191a..afd482196 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs @@ -16,23 +16,34 @@ //! Copying a parent's spans into the child it just made. -use crate::linux::guest::{Guest, Region}; +use crate::linux::guest::{Guest, Region, MAX_SPAN}; use nonos_libc::peer::{mk_peer_map, mk_peer_write, PEER_PROT_EXEC, PEER_PROT_WRITE}; /// Every span, mapped into the child and then filled from the parent. pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool { let spans = guest.regions.clone(); for span in spans { - // An unbacked reservation has no frames to copy; the child reserves it - // the same way, and its own first access faults a page in. + /* + * An unbacked reservation has no frames to copy; the child reserves it + * the same way, and its own first access faults a page in. + */ if !span.backed { continue; } - if mk_peer_map(child, span.at, span.len, prot_of(&span)) < 0 { - return false; - } - if !copy_one(guest, child, span.at, span.len) { - return false; + /* + * The kernel maps and copies at most MAX_SPAN in one peer call, so a + * larger span, which a Go heap is, crosses in pieces. + */ + let mut done = 0; + while done < span.len { + let take = (span.len - done).min(MAX_SPAN); + if mk_peer_map(child, span.at + done, take, prot_of(&span)) < 0 { + return false; + } + if !copy_one(guest, child, span.at + done, take) { + return false; + } + done += take; } } true diff --git a/userland/capsule_linux/src/linux/call/spawn/mod.rs b/userland/capsule_linux/src/linux/call/spawn/mod.rs index 8369b1db4..1a12afc2b 100644 --- a/userland/capsule_linux/src/linux/call/spawn/mod.rs +++ b/userland/capsule_linux/src/linux/call/spawn/mod.rs @@ -25,10 +25,18 @@ mod exec_resolve; mod exec_shebang; mod exec_threads; mod fork; +mod fork_child; mod fork_copy; +mod vfork; +mod vfork_clone; +mod vfork_flags; mod wait; +mod waitid; pub use clone::clone; pub use exec::execve; pub use fork::fork; -pub use wait::{reap_one, wait4}; +pub use vfork::vfork; +pub use vfork_clone::clone_process; +pub use wait::{wait4, wait4_usage, WALL, WCLONE, WEXITED, WNOHANG, WNOWAIT}; +pub use waitid::waitid; diff --git a/userland/capsule_linux/src/linux/call/spawn/vfork.rs b/userland/capsule_linux/src/linux/call/spawn/vfork.rs new file mode 100644 index 000000000..61291606f --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/vfork.rs @@ -0,0 +1,58 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `vfork`, and the start `clone` shares with it when it makes a process +//! rather than a thread (vfork_clone). The child is a copy, as fork's is: +//! vfork's child shares its parent's memory on Linux, but it may only exec or +//! exit, and the parent sleeps until it does, so what either sees is the +//! same. That is what Go's os/exec asks for with clone(CLONE_VFORK|CLONE_VM), +//! and musl's posix_spawn with a stack of the child's own, which the kernel's +//! fork starts it on. The parent parks here and the family answers it with +//! the child's pid when the child's exec succeeds or the child ends. + +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SIGCHLD; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +use super::fork_child::fork_child; + +pub fn vfork(guest: &mut Guest, caller: u32) -> Answer { + start(guest, caller, 0, SIGCHLD, true, |_| {}) +} + +/// Fork a child and answer with its pid, or park the caller until the child +/// execs or ends when `parks`. +pub(super) fn start( + guest: &mut Guest, + caller: u32, + stack: u64, + signal: u8, + parks: bool, + prep: impl FnOnce(&mut Guest), +) -> Answer { + let child = match fork_child(guest, caller, stack, signal, prep) { + Ok(c) => c, + Err(e) => return Answer::value(e), + }; + if !parks { + return Answer::value(errno::ok(u64::from(child))); + } + if let Some(c) = guest.forked.last_mut() { + c.signals.vfork = Some(caller); + } + Answer::Park +} diff --git a/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs b/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs new file mode 100644 index 000000000..4bac79300 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `clone` when it makes a process rather than a thread: a copy, as fork's +//! is, that starts on the stack the caller named, raises the signal it named +//! when it ends, and with CLONE_VFORK parks its parent as vfork does. The tid +//! words it names are written as Linux writes them. + +use super::vfork::start; +use super::vfork_flags::{CLONE_CHILD_CLEARTID, CLONE_CHILD_SETTID, CLONE_PARENT_SETTID}; +use super::vfork_flags::{CLONE_VFORK, CLONE_VM, CSIGNAL, SERVED}; +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::NSIG; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +/// clone(flags, stack, parent_tid, child_tid, tls) without CLONE_THREAD. +pub fn clone_process(guest: &mut Guest, caller: u32, a: [u64; 6]) -> Answer { + let (flags, stack) = (a[0], a[1]); + let why = if flags & !SERVED != 0 { + Some("clone: flags beyond a copied process") + } else if flags & CLONE_VM != 0 && flags & CLONE_VFORK == 0 { + Some("clone: a process sharing its parent's memory") + } else { + None + }; + if let Some(why) = why { + let line = alloc::format!("[LINUX] unserved {why}, flags {flags:#x}\n"); + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + return Answer::value(errno::fail(errno::ENOSYS)); + } + let signal = flags & CSIGNAL; + if signal > NSIG as u64 { + return Answer::value(errno::fail(errno::EINVAL)); + } + /* The child's own copy of its tid is written before it can read it. */ + let prep = move |child: &mut Guest| { + let seen = crate::linux::serve::guest_pid(child.pid).to_le_bytes(); + if flags & CLONE_CHILD_SETTID != 0 { + let _ = child.write(a[3], &seen); + } + if flags & CLONE_CHILD_CLEARTID != 0 { + child.clear_tids.push((child.pid, a[3])); + } + }; + let answer = start(guest, caller, stack, signal as u8, flags & CLONE_VFORK != 0, prep); + if flags & CLONE_PARENT_SETTID != 0 { + /* Only a child made by this call is in `forked`. */ + if let Some(child) = guest.forked.last().map(|c| c.pid) { + let seen = crate::linux::serve::guest_pid(child).to_le_bytes(); + let _ = guest.write(a[2], &seen); + } + } + answer +} diff --git a/userland/capsule_linux/src/linux/call/spawn/vfork_flags.rs b/userland/capsule_linux/src/linux/call/spawn/vfork_flags.rs new file mode 100644 index 000000000..066bea43a --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/vfork_flags.rs @@ -0,0 +1,36 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The clone flags a new process may be made with, as Linux numbers them. + +pub const CSIGNAL: u64 = 0xff; +pub const CLONE_VM: u64 = 0x100; +pub const CLONE_VFORK: u64 = 0x4000; +pub const CLONE_PARENT_SETTID: u64 = 0x10_0000; +pub const CLONE_CHILD_CLEARTID: u64 = 0x20_0000; +pub const CLONE_DETACHED: u64 = 0x40_0000; +pub const CLONE_UNTRACED: u64 = 0x80_0000; +pub const CLONE_CHILD_SETTID: u64 = 0x100_0000; +/// Every flag honoured for a new process; CLONE_DETACHED and CLONE_UNTRACED +/// are ones Linux itself ignores. +pub const SERVED: u64 = CSIGNAL + | CLONE_VM + | CLONE_VFORK + | CLONE_PARENT_SETTID + | CLONE_CHILD_CLEARTID + | CLONE_DETACHED + | CLONE_UNTRACED + | CLONE_CHILD_SETTID; diff --git a/userland/capsule_linux/src/linux/call/spawn/wait.rs b/userland/capsule_linux/src/linux/call/spawn/wait.rs index 9b518d8e0..072d33fdc 100644 --- a/userland/capsule_linux/src/linux/call/spawn/wait.rs +++ b/userland/capsule_linux/src/linux/call/spawn/wait.rs @@ -14,41 +14,54 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `wait4`: a child that has ended, or a wait until one does. +//! `wait4`: a child that has ended, or a wait until one does. The call only +//! checks what it was asked and parks the caller; the family, which sees +//! every process, answers it at once when a child has ended, with 0 under +//! WNOHANG, or with ECHILD when no child fits, and otherwise when one ends. use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::{ChildWait, Which}; use crate::linux::guest::Guest; use crate::linux::serve::Answer; -/// Set by a caller that will not wait. -const WNOHANG: u64 = 1; +pub const WNOHANG: u64 = 1; +pub const WUNTRACED: u64 = 2; +pub const WEXITED: u64 = 4; +pub const WCONTINUED: u64 = 8; +pub const WNOWAIT: u64 = 0x0100_0000; +pub const WNOTHREAD: u64 = 0x2000_0000; +pub const WALL: u64 = 0x4000_0000; +pub const WCLONE: u64 = 0x8000_0000; +const WAIT4_OPTIONS: u64 = WNOHANG | WUNTRACED | WCONTINUED | WNOTHREAD | WALL | WCLONE; pub fn wait4(guest: &mut Guest, want: u64, status: u64, flags: u64, tid: u32) -> Answer { - if guest.children.is_empty() { - return Answer::value(errno::fail(errno::ECHILD)); - } - if let Some(v) = reap_one(guest, want, status) { - return Answer::value(v); - } - // A child still running under WNOHANG is a zero, not an error. - if flags & WNOHANG != 0 { - return Answer::value(errno::ok(0)); - } - guest.waiting = Some((want, status, tid)); - Answer::Park + wait4_usage(guest, want, status, flags, 0, tid) } -/// Take one ended child the caller asked about, write its status, and give -/// the answer wait4 returns. None while no such child has ended. -pub fn reap_one(guest: &mut Guest, want: u64, status: u64) -> Option { - let any = (want as i64) <= 0; - let at = guest.ended.iter().position(|(pid, _)| any || *pid == want as u32)?; - let (pid, code) = guest.ended.remove(at); - guest.children.retain(|p| *p != pid); - // An exit status sits in the second byte, as WEXITSTATUS reads it. - let word = ((code as u32) & 0xff) << 8; - if status != 0 && guest.write(status, &word.to_le_bytes()) < 4 { - return Some(errno::fail(errno::EFAULT)); +/// wait4 with its rusage: written as zeros, since the kernel reports no CPU +/// time for a guest. +pub fn wait4_usage( + guest: &mut Guest, + want: u64, + status: u64, + flags: u64, + rusage: u64, + tid: u32, +) -> Answer { + if flags & !WAIT4_OPTIONS != 0 { + return Answer::value(errno::fail(errno::EINVAL)); } - Some(errno::ok(pid as u64)) + let which = match want as i64 as i32 { + -1 => Which::Any, + 0 => Which::Group(guest.pgid), + p if p < 0 => Which::Group(p.unsigned_abs()), + p => Which::Pid(p as u32), + }; + let options = flags | WEXITED; + park(guest, ChildWait { tid, which, options, out: status, rusage, waitid: false }) +} + +pub(super) fn park(guest: &mut Guest, w: ChildWait) -> Answer { + guest.signals.childwaits.push(w); + Answer::Park } diff --git a/userland/capsule_linux/src/linux/call/spawn/waitid.rs b/userland/capsule_linux/src/linux/call/spawn/waitid.rs new file mode 100644 index 000000000..271523e88 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/waitid.rs @@ -0,0 +1,59 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `waitid`: wait4's question asked by id type, answered in a siginfo, with +//! WNOWAIT to look without reaping. Linux writes the siginfo whatever the +//! outcome, zeros when nothing is reported, so an error here writes it too. + +use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::{ChildWait, Which}; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +use super::wait::{park, WALL, WCLONE, WCONTINUED, WEXITED, WNOHANG, WNOTHREAD, WNOWAIT}; + +const WSTOPPED: u64 = 2; +const P_ALL: u64 = 0; +const P_PID: u64 = 1; +const P_PGID: u64 = 2; +const P_PIDFD: u64 = 3; +const OPTIONS: u64 = + WNOHANG | WEXITED | WSTOPPED | WCONTINUED | WNOWAIT | WNOTHREAD | WALL | WCLONE; +pub const SIGINFO_LEN: usize = 128; + +pub fn waitid(guest: &mut Guest, tid: u32, a: [u64; 6]) -> Answer { + let (idtype, id, info, options, rusage) = (a[0], a[1] as u32, a[2], a[3], a[4]); + if options & !OPTIONS != 0 || options & (WEXITED | WSTOPPED | WCONTINUED) == 0 { + return refuse(guest, info, errno::EINVAL); + } + let which = match idtype { + P_ALL => Which::Any, + P_PID if (id as i32) > 0 => Which::Pid(id), + P_PGID if id == 0 => Which::Group(guest.pgid), + P_PGID if (id as i32) > 0 => Which::Group(id), + /* No descriptor here is ever a pidfd: pidfd_open is not served. */ + P_PIDFD => return refuse(guest, info, errno::EBADF), + _ => return refuse(guest, info, errno::EINVAL), + }; + park(guest, ChildWait { tid, which, options, out: info, rusage, waitid: true }) +} + +fn refuse(guest: &Guest, info: u64, e: i64) -> Answer { + if info != 0 && guest.write(info, &[0u8; SIGINFO_LEN]) < SIGINFO_LEN as i64 { + return Answer::value(errno::fail(errno::EFAULT)); + } + Answer::value(errno::fail(e)) +} diff --git a/userland/capsule_linux/src/linux/call/timer_create.rs b/userland/capsule_linux/src/linux/call/timer_create.rs new file mode 100644 index 000000000..e045d7f43 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/timer_create.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! timer_create: a POSIX timer, disarmed. Ids count from 0 per process, as +//! Linux's do. Its signal goes to the process, or to the one thread +//! SIGEV_THREAD_ID names; a NULL sigevent means SIGALRM with the id as its +//! value. Timers on a CPU-time clock are refused by name: the kernel does not +//! report a guest's CPU time to its supervisor. + +use super::timer_sigev::read_sigevent; +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SIGALRM; +use crate::linux::guest::sigtimer::PosixTimer; +use crate::linux::guest::Guest; + +const TIMERS_MAX: usize = 1024; + +pub fn timer_create(guest: &mut Guest, clock: u64, sevp: u64, out: u64) -> u64 { + match clock { + 0 | 1 | 7..=9 | 11 => {} + 2 | 3 => { + let line = b"[LINUX] unserved timer_create: no CPU-time clock\n"; + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + return errno::fail(errno::ENOTSUP); + } + 4..=6 => return errno::fail(errno::ENOTSUP), + _ => return errno::fail(errno::EINVAL), + } + let id = (0..).find(|i| !guest.signals.timers.iter().any(|t| t.id == *i)).unwrap_or(0); + let mut t = PosixTimer { + id, + clock, + signo: SIGALRM, + tid: 0, + value: id as u64, + due: None, + interval: 0, + overrun: 0, + last_overrun: 0, + queued: false, + }; + if sevp != 0 { + if let Err(e) = read_sigevent(guest, sevp, &mut t) { + return e; + } + } + if guest.signals.timers.len() >= TIMERS_MAX { + return errno::fail(errno::EAGAIN); + } + if guest.write(out, &id.to_le_bytes()) < 4 { + return errno::fail(errno::EFAULT); + } + guest.signals.timers.push(t); + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/timer_ops.rs b/userland/capsule_linux/src/linux/call/timer_ops.rs new file mode 100644 index 000000000..7d6eb947b --- /dev/null +++ b/userland/capsule_linux/src/linux/call/timer_ops.rs @@ -0,0 +1,59 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! timer_gettime, timer_getoverrun and timer_delete; timer_settime is in +//! timer_set. Expiry raises the timer's signal with SI_TIMER, its id and +//! sigev_value; an expiry while that signal still waits counts as an overrun, +//! reported by the signal and by timer_getoverrun. + +use crate::linux::abi::errno; +use crate::linux::call::now_ms; +use crate::linux::guest::Guest; + +use super::signal_its::write_spec; + +const CLOCK_MONOTONIC: u64 = 1; + +pub fn timer_gettime(guest: &mut Guest, id: u64, out: u64) -> u64 { + let Some(at) = find(guest, id) else { + return errno::fail(errno::EINVAL); + }; + let t = guest.signals.timers[at]; + let now = now_ms(CLOCK_MONOTONIC).unwrap_or(0); + let left = t.due.map_or(0, |d| d.saturating_sub(now).max(1)); + write_spec(guest, out, t.interval, left) +} + +pub fn timer_getoverrun(guest: &mut Guest, id: u64) -> u64 { + match find(guest, id) { + Some(at) => errno::ok(guest.signals.timers[at].last_overrun as u64), + None => errno::fail(errno::EINVAL), + } +} + +pub fn timer_delete(guest: &mut Guest, id: u64) -> u64 { + let Some(at) = find(guest, id) else { + return errno::fail(errno::EINVAL); + }; + let t = guest.signals.timers.remove(at); + guest.signals.drop_timer_signal(t.id); + errno::ok(0) +} + +pub fn find(guest: &Guest, id: u64) -> Option { + let id = i32::try_from(id).ok()?; + guest.signals.timers.iter().position(|t| t.id == id) +} diff --git a/userland/capsule_linux/src/linux/call/timer_set.rs b/userland/capsule_linux/src/linux/call/timer_set.rs new file mode 100644 index 000000000..83041f6e4 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/timer_set.rs @@ -0,0 +1,56 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! timer_settime: arm a timer for a relative time, or an absolute one on its +//! own clock, repeating every interval; a zero value disarms it. The old +//! setting is written first when the caller asked for it. + +use super::signal_its::read_spec; +use super::timer_ops::{find, timer_gettime}; +use crate::linux::abi::errno; +use crate::linux::call::now_ms; +use crate::linux::guest::Guest; + +const TIMER_ABSTIME: u64 = 1; +const CLOCK_MONOTONIC: u64 = 1; + +pub fn timer_settime(guest: &mut Guest, id: u64, flags: u64, new: u64, old: u64) -> u64 { + let Some(at) = find(guest, id) else { + return errno::fail(errno::EINVAL); + }; + if new == 0 { + return errno::fail(errno::EINVAL); + } + let (interval, value) = match read_spec(guest, new) { + Ok(v) => v, + Err(e) => return e, + }; + let rc = timer_gettime(guest, id, old); + let t = &mut guest.signals.timers[at]; + let now = now_ms(CLOCK_MONOTONIC).unwrap_or(0); + let wait = match flags & TIMER_ABSTIME { + 0 => value, + _ => value.saturating_sub(now_ms(t.clock).unwrap_or(0)), + }; + t.due = (value != 0).then(|| now.saturating_add(wait)); + t.interval = if value == 0 { 0 } else { interval }; + t.overrun = 0; + if old != 0 { + rc + } else { + errno::ok(0) + } +} diff --git a/userland/capsule_linux/src/linux/call/timer_sigev.rs b/userland/capsule_linux/src/linux/call/timer_sigev.rs new file mode 100644 index 000000000..8be193c15 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/timer_sigev.rs @@ -0,0 +1,54 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The sigevent timer_create reads: which signal a timer raises and with what +//! value, SIGEV_NONE for none, and SIGEV_THREAD_ID for one thread of the +//! process rather than the whole of it. + +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::NSIG; +use crate::linux::guest::sigtimer::PosixTimer; +use crate::linux::guest::Guest; + +const SIGEV_SIGNAL: u32 = 0; +const SIGEV_NONE: u32 = 1; +const SIGEV_THREAD: u32 = 2; +const SIGEV_THREAD_ID: u32 = 4; + +/// Read the sigevent at `sevp` into `t`, or the errno a bad one earns. +pub fn read_sigevent(guest: &Guest, sevp: u64, t: &mut PosixTimer) -> Result<(), u64> { + let Some(raw) = guest.read(sevp, 20) else { + return Err(errno::fail(errno::EFAULT)); + }; + let word = |i: usize| u32::from_le_bytes(raw[i..i + 4].try_into().unwrap_or([0; 4])); + let (signo, notify) = (word(8), word(12)); + t.value = u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8])); + t.signo = if notify == SIGEV_NONE { 0 } else { signo as u8 }; + let bad_signo = notify != SIGEV_NONE && (signo == 0 || signo as usize > NSIG); + match notify { + _ if bad_signo => return Err(errno::fail(errno::EINVAL)), + SIGEV_SIGNAL | SIGEV_NONE | SIGEV_THREAD => {} + n if n == SIGEV_SIGNAL | SIGEV_THREAD_ID => { + let tid = crate::linux::serve::kernel_pid(word(16)); + match tid.filter(|k| guest.live_threads().contains(k)) { + Some(k) => t.tid = k, + None => return Err(errno::fail(errno::EINVAL)), + } + } + _ => return Err(errno::fail(errno::EINVAL)), + } + Ok(()) +} diff --git a/userland/capsule_linux/src/linux/file/dev.rs b/userland/capsule_linux/src/linux/file/dev.rs new file mode 100644 index 000000000..f47420bd1 --- /dev/null +++ b/userland/capsule_linux/src/linux/file/dev.rs @@ -0,0 +1,69 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The character devices every Linux program may assume: /dev/null, /dev/zero, +//! /dev/full, /dev/random and /dev/urandom. They are not files in the store; +//! each is a descriptor this capsule answers itself, with the major and minor +//! numbers Linux gives it, so fstat and stat say character device as Linux's. + +use crate::linux::abi::errno; +use crate::linux::guest::{Fd, Guest, Kind}; + +use super::flags::wants_write; +use super::slot::install; + +/// Path, major, minor. The handle of an open device is its index here. +const DEVICES: [(&[u8], u64, u64); 5] = [ + (b"/dev/null", 1, 3), + (b"/dev/zero", 1, 5), + (b"/dev/full", 1, 7), + (b"/dev/random", 1, 8), + (b"/dev/urandom", 1, 9), +]; +pub const NULL: u32 = 0; +pub const ZERO: u32 = 1; +pub const FULL: u32 = 2; + +/// The device a guest-visible path names, if it names one. +pub fn device_of(full: &[u8]) -> Option { + DEVICES.iter().position(|(p, _, _)| *p == full).map(|i| i as u32) +} + +/// Open the device `full` names; the caller has checked that it names one. +pub fn open_path(guest: &mut Guest, full: &[u8], flags: u64) -> u64 { + let Some(dev) = device_of(full) else { + return errno::fail(errno::ENOENT); + }; + let mut fd = Fd::empty(Kind::Device); + fd.handle = dev; + fd.path = full.to_vec(); + fd.writable = wants_write(flags); + match install(guest, fd) { + Some(n) => errno::ok(n), + None => errno::fail(errno::EMFILE), + } +} + +/// The major and minor numbers of an open device. +pub fn numbers(dev: u32) -> Option<(u64, u64)> { + DEVICES.get(dev as usize).map(|&(_, major, minor)| (major, minor)) +} + +/// Whether epoll may watch it: null, zero and full have no poll on Linux and +/// are refused with EPERM; the random devices can be waited on. +pub fn polls(dev: u32) -> bool { + dev > FULL +} diff --git a/userland/capsule_linux/src/linux/file/dev_io.rs b/userland/capsule_linux/src/linux/file/dev_io.rs new file mode 100644 index 000000000..17847c6a1 --- /dev/null +++ b/userland/capsule_linux/src/linux/file/dev_io.rs @@ -0,0 +1,64 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Reading and writing the character devices, as Linux's drivers answer: +//! /dev/null reads end of file and takes every write, /dev/zero reads zeros +//! and takes every write, /dev/full reads zeros and refuses every write with +//! ENOSPC, and the two random devices read the kernel's random bytes and take +//! writes without keeping them, as an unprivileged write to them does. + +use alloc::vec; + +use crate::linux::abi::errno; +use crate::linux::guest::{Guest, MAX_SPAN}; + +use super::dev::{FULL, NULL, ZERO}; + +/// Linux's random devices hand at most this much to one read. +const RANDOM_MAX: u64 = 32 << 20; + +fn device(guest: &Guest, fd: u64) -> u32 { + guest.fds.get(fd as usize).map_or(u32::MAX, |f| f.handle) +} + +pub fn read(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { + let dev = device(guest, fd); + if dev == NULL { + return errno::ok(0); + } + /* One step at a time, so a large read never holds a large buffer. */ + let want = if dev == ZERO || dev == FULL { len } else { len.min(RANDOM_MAX) }; + let mut done = 0; + while done < want { + let take = (want - done).min(MAX_SPAN) as usize; + let mut bytes = vec![0u8; take]; + if dev != ZERO && dev != FULL && nonos_libc::crypto_random(bytes.as_mut_ptr(), take) < 0 { + break; + } + if guest.write(buf + done, &bytes) < take as i64 { + return if done == 0 { errno::fail(errno::EFAULT) } else { errno::ok(done) }; + } + done += take as u64; + } + errno::ok(done) +} + +pub fn write(guest: &mut Guest, fd: u64, len: u64) -> u64 { + match device(guest, fd) { + FULL => errno::fail(errno::ENOSPC), + _ => errno::ok(len), + } +} diff --git a/userland/capsule_linux/src/linux/file/dev_stat.rs b/userland/capsule_linux/src/linux/file/dev_stat.rs new file mode 100644 index 000000000..f64043710 --- /dev/null +++ b/userland/capsule_linux/src/linux/file/dev_stat.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What stat, fstat and statx say about a character device: S_IFCHR with +//! read and write for everyone, as Linux's devtmpfs makes them, and the +//! device's major and minor numbers. + +use super::dev::numbers; + +const MODE: u32 = 0o020666; +/// st_mode and st_rdev in a `struct stat`. +const STAT_MODE: usize = 24; +const STAT_RDEV: usize = 40; +/// stx_mode, stx_rdev_major and stx_rdev_minor in a `struct statx`. +const STATX_MODE: usize = 28; +const STATX_RDEV: usize = 128; + +/// A `struct stat` made for a file, turned into the device's. st_rdev is +/// Linux's encoding of the two numbers. +pub fn as_device(stat: &mut [u8], dev: u32) { + let Some((major, minor)) = numbers(dev) else { + return; + }; + let rdev = (minor & 0xff) | ((major & 0xfff) << 8) | ((minor & !0xff) << 12); + stat[STAT_MODE..STAT_MODE + 4].copy_from_slice(&MODE.to_le_bytes()); + stat[STAT_RDEV..STAT_RDEV + 8].copy_from_slice(&rdev.to_le_bytes()); +} + +/// The same for a `struct statx`, which keeps the two numbers apart. +pub fn statx_device(buf: &mut [u8], dev: u32) { + let Some((major, minor)) = numbers(dev) else { + return; + }; + buf[STATX_MODE..STATX_MODE + 2].copy_from_slice(&(MODE as u16).to_le_bytes()); + buf[STATX_RDEV..STATX_RDEV + 4].copy_from_slice(&(major as u32).to_le_bytes()); + buf[STATX_RDEV + 4..STATX_RDEV + 8].copy_from_slice(&(minor as u32).to_le_bytes()); +} diff --git a/userland/capsule_linux/src/linux/file/epoll.rs b/userland/capsule_linux/src/linux/file/epoll.rs index 51435cfbd..8c56745bf 100644 --- a/userland/capsule_linux/src/linux/file/epoll.rs +++ b/userland/capsule_linux/src/linux/file/epoll.rs @@ -54,7 +54,9 @@ pub fn epoll_ctl(guest: &mut Guest, ep: u64, op: u64, fd: u64, event: u64) -> u6 if !open(ep) || !open(fd) { return errno::fail(errno::EBADF); } - if guest.fds.get(fd as usize).is_some_and(|f| matches!(f.kind, Kind::File | Kind::Dir)) { + let unpollable = |f: &Fd| f.kind == Kind::Device && !super::dev::polls(f.handle); + let refused = |f: &Fd| matches!(f.kind, Kind::File | Kind::Dir) || unpollable(f); + if guest.fds.get(fd as usize).is_some_and(refused) { return errno::fail(errno::EPERM); } let Some(list) = guest.fds.get_mut(ep as usize).filter(|f| f.kind == Kind::Epoll) else { diff --git a/userland/capsule_linux/src/linux/file/meta/stat.rs b/userland/capsule_linux/src/linux/file/meta/stat.rs index 90e0ee284..f8f032ac2 100644 --- a/userland/capsule_linux/src/linux/file/meta/stat.rs +++ b/userland/capsule_linux/src/linux/file/meta/stat.rs @@ -19,6 +19,7 @@ use crate::linux::abi::errno; use crate::linux::guest::{Guest, Kind}; +use super::super::dev::device_of; use super::super::flags::AT_FDCWD; use super::super::{path, resolve, store}; use super::statbuf::{build, inode, STAT_LEN}; @@ -27,6 +28,7 @@ use super::statbuf::{build, inode, STAT_LEN}; /// absent. `full` is guest-visible and is confined here. pub fn look(full: &[u8]) -> Option<(u64, bool)> { match store::stat_full(&resolve::key(full)) { + _ if device_of(full).is_some() => Some((0, false)), Ok((size, is_dir, _, _)) => Some((size, is_dir)), Err(_) => None, } @@ -43,7 +45,8 @@ pub fn fstat(guest: &mut Guest, fd: u64, out: u64) -> u64 { _ => (0, false), }; let ino = inode(&entry.path); - write_out(guest, out, size, is_dir, ino) + let dev = (entry.kind == Kind::Device).then_some(entry.handle); + write_out(guest, out, size, is_dir, ino, dev) } pub fn newfstatat(guest: &mut Guest, dirfd: u64, path_ptr: u64, out: u64) -> u64 { @@ -55,13 +58,17 @@ pub fn newfstatat(guest: &mut Guest, dirfd: u64, path_ptr: u64, out: u64) -> u64 } let full = guest.links.follow(resolve::visible(&guest.cwd, &name), true); match look(&full) { - Some((size, is_dir)) => write_out(guest, out, size, is_dir, inode(&full)), + Some((size, is_dir)) => write_out(guest, out, size, is_dir, inode(&full), device_of(&full)), None => errno::fail(errno::ENOENT), } } -fn write_out(guest: &Guest, out: u64, size: u64, is_dir: bool, ino: u64) -> u64 { - if guest.write(out, &build(size, is_dir, ino)) < STAT_LEN as i64 { +fn write_out(guest: &Guest, out: u64, size: u64, is_dir: bool, ino: u64, dev: Option) -> u64 { + let mut stat = build(size, is_dir, ino); + if let Some(d) = dev { + super::super::dev_stat::as_device(&mut stat, d); + } + if guest.write(out, &stat) < STAT_LEN as i64 { return errno::fail(errno::EFAULT); } errno::ok(0) diff --git a/userland/capsule_linux/src/linux/file/meta/statx.rs b/userland/capsule_linux/src/linux/file/meta/statx.rs index d1375b89e..0ba7c4558 100644 --- a/userland/capsule_linux/src/linux/file/meta/statx.rs +++ b/userland/capsule_linux/src/linux/file/meta/statx.rs @@ -41,7 +41,10 @@ pub fn statx(guest: &Guest, dirfd: u64, path: u64, out: u64) -> u64 { let Some(at) = resolve_at(guest, dirfd, path) else { return errno::fail(errno::EFAULT); }; - let Ok((size, is_dir, _, readonly)) = store::stat_full(&key(&at)) else { + let dev = super::super::dev::device_of(&at); + let Some((size, is_dir, _, readonly)) = + store::stat_full(&key(&at)).ok().or(dev.map(|_| (0, false, 0, false))) + else { return errno::fail(errno::ENOENT); }; let mode = if is_dir { S_IFDIR } else { S_IFREG } | if readonly { 0o555 } else { 0o755 }; @@ -53,6 +56,9 @@ pub fn statx(guest: &Guest, dirfd: u64, path: u64, out: u64) -> u64 { buf[32..40].copy_from_slice(&inode(&at).to_le_bytes()); // stx_ino buf[40..48].copy_from_slice(&size.to_le_bytes()); // stx_size buf[48..56].copy_from_slice(&size.div_ceil(512).to_le_bytes()); // stx_blocks + if let Some(d) = dev { + super::super::dev_stat::statx_device(&mut buf, d); + } match guest.write(out, &buf) { n if n < 0 => errno::fail(errno::EFAULT), _ => errno::ok(0), diff --git a/userland/capsule_linux/src/linux/file/mod.rs b/userland/capsule_linux/src/linux/file/mod.rs index 51bf8ea96..c9db777f0 100644 --- a/userland/capsule_linux/src/linux/file/mod.rs +++ b/userland/capsule_linux/src/linux/file/mod.rs @@ -20,6 +20,9 @@ mod at; mod clamp; pub(super) mod close; mod cstr; +mod dev; +mod dev_io; +mod dev_stat; mod dir; mod dir_children; mod dirent; @@ -59,6 +62,7 @@ mod write; pub use close::close; pub use cstr::read_cstr; +pub use dev_io::{read as dev_read, write as dev_write}; pub use dirents::getdents64; pub use dirops::{mkdirat, rmdir, unlinkat}; pub use epoll::{epoll_create, epoll_ctl}; diff --git a/userland/capsule_linux/src/linux/file/open.rs b/userland/capsule_linux/src/linux/file/open.rs index a36d29a0b..5ef6051a9 100644 --- a/userland/capsule_linux/src/linux/file/open.rs +++ b/userland/capsule_linux/src/linux/file/open.rs @@ -22,7 +22,7 @@ use crate::linux::abi::errno; use crate::linux::guest::{Guest, Kind}; use super::flags::{wants_write, AT_FDCWD, O_CLOEXEC, O_CREAT, O_DIRECTORY}; -use super::{dir, path, regular, resolve, store}; +use super::{dev, dir, path, regular, resolve, store}; pub fn openat(guest: &mut Guest, dirfd: u64, path_ptr: u64, flags: u64) -> u64 { let Some(name) = path::read_path(guest, path_ptr) else { @@ -34,6 +34,7 @@ pub fn openat(guest: &mut Guest, dirfd: u64, path_ptr: u64, flags: u64) -> u64 { }; let full = guest.links.follow(resolve::visible(&base, &name), true); let got = match store::stat(&resolve::key(&full)).ok() { + _ if dev::device_of(&full).is_some() => dev::open_path(guest, &full, flags), Some((_, true)) => dir::open(guest, full), Some((_, false)) if flags & O_DIRECTORY != 0 => errno::fail(errno::ENOTDIR), Some((size, false)) => regular::open(guest, full, size, flags), diff --git a/userland/capsule_linux/src/linux/file/seek.rs b/userland/capsule_linux/src/linux/file/seek.rs index 140e5610d..66002939f 100644 --- a/userland/capsule_linux/src/linux/file/seek.rs +++ b/userland/capsule_linux/src/linux/file/seek.rs @@ -14,7 +14,6 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - //! `lseek`. The position is this capsule's, not the server's: a read takes //! a window at an offset, so the descriptor's offset is the whole of it. @@ -29,6 +28,10 @@ pub fn lseek(guest: &mut Guest, fd: u64, offset: u64, whence: u64) -> u64 { let Some(entry) = guest.fds.get_mut(fd as usize) else { return errno::fail(errno::EBADF); }; + /* A character device has no position to move, and Linux answers 0. */ + if entry.kind == Kind::Device { + return errno::ok(0); + } if entry.kind != Kind::File { // A pipe or a console has no position, which Linux calls ESPIPE. return errno::fail(errno::ESPIPE); diff --git a/userland/capsule_linux/src/linux/guest/fd_kind.rs b/userland/capsule_linux/src/linux/guest/fd_kind.rs index 6888e6dd0..4519659a5 100644 --- a/userland/capsule_linux/src/linux/guest/fd_kind.rs +++ b/userland/capsule_linux/src/linux/guest/fd_kind.rs @@ -44,4 +44,8 @@ pub enum Kind { Resolver, /// An eventfd: a counter one thread adds to and another takes from. Event, + /// A character device this capsule answers: /dev/null and its kin. + Device, + /// A signalfd: signals of a mask, read as records. + Signal, } diff --git a/userland/capsule_linux/src/linux/guest/handle.rs b/userland/capsule_linux/src/linux/guest/handle.rs index 1f12ca784..a5d0c8bca 100644 --- a/userland/capsule_linux/src/linux/guest/handle.rs +++ b/userland/capsule_linux/src/linux/guest/handle.rs @@ -78,8 +78,6 @@ pub struct Guest { pub forked: Vec, /// Children that have ended, with their exit codes, until waited for. pub ended: Vec<(u32, i32)>, - /// A parked wait4: the pid it wants, where the status goes, the caller. - pub waiting: Option<(u64, u64, u32)>, /// Threads parked in a sleep: the monotonic deadline, and who. pub sleepers: Vec<(u64, u32)>, /// Calls parked until a descriptor they wait on is ready. diff --git a/userland/capsule_linux/src/linux/guest/handle_new.rs b/userland/capsule_linux/src/linux/guest/handle_new.rs index 0d8b798db..b290a2c50 100644 --- a/userland/capsule_linux/src/linux/guest/handle_new.rs +++ b/userland/capsule_linux/src/linux/guest/handle_new.rs @@ -57,7 +57,6 @@ impl Guest { umask: crate::linux::call::DEFAULT_UMASK, forked: Vec::new(), ended: Vec::new(), - waiting: None, sleepers: Vec::new(), blocked: Vec::new(), links: Default::default(), diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index 514a26bdd..66a818d87 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -27,8 +27,21 @@ mod fd_make; mod fork_state; mod handle; mod handle_new; +pub mod sigdefault; +pub mod siginfo; +mod siglive; +pub mod sigpark; pub mod sigqueue; +mod sigqueue_new; +mod sigqueue_ops; +mod sigraise; pub mod sigstate; +mod sigtake; +pub mod sigthread; +mod sigthread_copy; +pub mod sigtimer; +mod sigtimer_rearm; +pub mod sigwaits; mod layout; mod links; mod links_add; diff --git a/userland/capsule_linux/src/linux/guest/sigdefault.rs b/userland/capsule_linux/src/linux/guest/sigdefault.rs new file mode 100644 index 000000000..f6910fdc3 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigdefault.rs @@ -0,0 +1,40 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What Linux does with a signal no handler takes, when its disposition is +//! SIG_DFL. The table is Linux's, transcribed. + +/// What an uncaught signal does when its disposition is SIG_DFL. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Default { + Ignore, + Terminate, + Stop, + Continue, +} + +/// Linux's table: child status, urgent data and window size are ignored, +/// SIGCONT continues, the four stop signals stop, everything else ends the +/// process. The core-dumping ones end it too: no core is ever written, so the +/// status carries no core flag, as on Linux with a zero core limit. +pub fn default_of(signum: u8) -> Default { + match signum { + 17 | 23 | 28 => Default::Ignore, + 18 => Default::Continue, + 19..=22 => Default::Stop, + _ => Default::Terminate, + } +} diff --git a/userland/capsule_linux/src/linux/guest/siginfo.rs b/userland/capsule_linux/src/linux/guest/siginfo.rs new file mode 100644 index 000000000..da0a7118b --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/siginfo.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What a signal carries: the 128-byte siginfo Linux hands a handler, a +//! sigtimedwait or a waitid. A pid is kept in kernel terms and becomes the +//! guest's own number only as the bytes are written, so no guest sees a +//! kernel pid in one. + +pub const SI_USER: i32 = 0; +pub const SI_KERNEL: i32 = 0x80; +pub const SI_TIMER: i32 = -2; +pub const SI_TKILL: i32 = -6; +pub const CLD_EXITED: i32 = 1; +pub const CLD_KILLED: i32 = 2; +pub const INFO_LEN: usize = 128; + +#[derive(Clone, Copy, Default)] +pub struct SigInfo { + pub signo: u8, + pub code: i32, + /// The sending process, or the child that ended; zero names none. + pub pid: u32, + /// si_timerid and si_overrun, for a timer's signal, which has no pid. + pub timer: Option<(i32, i32)>, + /// si_value for a queued or timer signal; si_status for SIGCHLD. + pub value: u64, +} + +impl SigInfo { + /// A signal sent by a process, as kill, tkill and sigqueue send one. + pub fn from(signo: u8, code: i32, pid: u32) -> Self { + SigInfo { signo, code, pid, ..SigInfo::default() } + } + + /// Linux's layout: signo, errno, code, then at +16 either the pid and + /// uid or the timer id and overrun, then the value or status at +24. + pub fn bytes(&self) -> [u8; INFO_LEN] { + let mut b = [0u8; INFO_LEN]; + b[0..4].copy_from_slice(&i32::from(self.signo).to_le_bytes()); + b[8..12].copy_from_slice(&self.code.to_le_bytes()); + let (at16, at20) = match self.timer { + Some((id, overrun)) => (id, overrun), + None => (guest_number(self.pid), 0), /* uid 0: every guest's */ + }; + b[16..20].copy_from_slice(&at16.to_le_bytes()); + b[20..24].copy_from_slice(&at20.to_le_bytes()); + b[24..32].copy_from_slice(&self.value.to_le_bytes()); + b + } +} + +fn guest_number(pid: u32) -> i32 { + match pid { + 0 => 0, + k => crate::linux::serve::guest_pid(k) as i32, + } +} diff --git a/userland/capsule_linux/src/linux/guest/siglive.rs b/userland/capsule_linux/src/linux/guest/siglive.rs new file mode 100644 index 000000000..ef6627502 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/siglive.rs @@ -0,0 +1,41 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which threads of a process can take a signal, and a thread that has gone +//! taking its signal state and its waits with it. + +use alloc::vec::Vec; + +use super::handle::Guest; + +impl Guest { + /// The leader, unless it has made a plain exit, and every other thread. + pub fn live_threads(&self) -> Vec { + let mut all = self.threads.clone(); + if !self.signals.leader_gone { + all.insert(0, self.pid); + } + all + } + + /// A thread that has gone takes its signal state and its waits with it. + pub fn forget_thread(&mut self, tid: u32) { + let _ = self.leave_waits(tid); + self.signals.threads.retain(|t| t.tid != tid); + self.signals.pending.retain(|(t, _)| *t != tid); + self.signals.timers_requeue(); + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigpark.rs b/userland/capsule_linux/src/linux/guest/sigpark.rs new file mode 100644 index 000000000..ac2afeada --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigpark.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Taking a thread out of the wait it is parked in, so a caught signal can end +//! that wait as Linux does: the wait is dropped here and never answered; the +//! handler's frame is the answer. + +use super::handle::Guest; + +/// Parked, and in which kind of wait. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Parked { + /// A timed sleep, with its monotonic deadline. + Sleep(u64), + /// Any other wait a signal ends. + Wait, +} + +impl Guest { + /// The wait `tid` is parked in, if a signal can end it. + pub fn parked(&self, tid: u32) -> Option { + if let Some(&(due, _)) = self.sleepers.iter().find(|(_, t)| *t == tid) { + return Some(Parked::Sleep(due)); + } + let waiting = self.waits.iter().any(|(t, _)| *t == tid) + || self.blocked.iter().any(|w| w.tid == tid) + || self.signals.sigwaits.iter().any(|w| w.tid == tid) + || self.signals.childwaits.iter().any(|w| w.tid == tid); + waiting.then_some(Parked::Wait) + } + + /// Take `tid` out of every wait it is parked in, without answering it. + pub fn leave_waits(&mut self, tid: u32) -> Option { + let was = self.parked(tid)?; + self.sleepers.retain(|(_, t)| *t != tid); + /* The futex waits, their timeouts and every descriptor wait. */ + self.forget_waits(tid); + self.signals.sigwaits.retain(|w| w.tid != tid); + self.signals.childwaits.retain(|w| w.tid != tid); + Some(was) + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigqueue.rs b/userland/capsule_linux/src/linux/guest/sigqueue.rs index 03593c493..ca46189d0 100644 --- a/userland/capsule_linux/src/linux/guest/sigqueue.rs +++ b/userland/capsule_linux/src/linux/guest/sigqueue.rs @@ -14,58 +14,47 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Signals raised against a process's threads and not yet delivered, with the -//! disposition of each. The queue names the thread a signal is for. +//! Signals raised against a process and not yet taken, with the disposition +//! of each. An entry names the thread it is for, or 0 for the process as a +//! whole, which any thread not blocking it may take, as on Linux. use alloc::vec::Vec; +use super::siginfo::SigInfo; use super::sigstate::{SigAction, NSIG}; +use super::sigthread::ThreadSig; +use super::sigtimer::{Itimer, PosixTimer}; +use super::sigwaits::{ChildWait, Outbound, SigWait}; + +/// Linux's default RLIMIT_SIGPENDING for a small machine: how many queued +/// realtime signals a process may hold before sigqueue answers EAGAIN. +pub const QUEUE_MAX: usize = 1024; #[derive(Clone)] pub struct Signals { - actions: [SigAction; NSIG], - pending: Vec<(u32, u8)>, -} - -impl Default for Signals { - fn default() -> Self { - Self { actions: [SigAction::default(); NSIG], pending: Vec::new() } - } -} - -impl Signals { - /// Record a disposition; `signum` is 1..=NSIG. - pub fn set(&mut self, signum: usize, act: SigAction) { - if (1..=NSIG).contains(&signum) { - self.actions[signum - 1] = act; - } - } - - pub fn action(&self, signum: usize) -> Option { - (1..=NSIG).contains(&signum).then(|| self.actions[signum - 1]) - } - - /// Queue a signal against a thread. A standard signal already pending is - /// not queued twice, as Linux coalesces non-realtime signals. - pub fn raise(&mut self, tid: u32, signum: u8) { - if !self.pending.iter().any(|p| *p == (tid, signum)) { - self.pending.push((tid, signum)); - } - } - - /// The next signal for `tid` its disposition catches, removed. Signals - /// with no handler are left for the caller to default. - pub fn take_caught(&mut self, tid: u32) -> Option<(u8, SigAction)> { - let at = self - .pending - .iter() - .position(|(t, s)| *t == tid && self.actions[*s as usize - 1].catches())?; - let signum = self.pending.remove(at).1; - Some((signum, self.actions[signum as usize - 1])) - } - - /// Drop every signal pending for a thread that has gone. - pub fn forget(&mut self, tid: u32) { - self.pending.retain(|(t, _)| *t != tid); - } + pub(super) actions: [SigAction; NSIG], + pub(super) pending: Vec<(u32, SigInfo)>, + /// Each thread's mask, alternate stack and suspended mask. + pub(super) threads: Vec, + /// Signals for other processes of the family, and who sent each. + pub outbox: Vec, + /// ITIMER_REAL, and the POSIX timers timer_create made. + pub real: Option, + pub timers: Vec, + /// Threads parked in pause, sigsuspend or sigtimedwait. + pub sigwaits: Vec, + /// Threads parked in wait4 or waitid. + pub childwaits: Vec, + /// Set once the leader has made a plain exit while other threads run on. + pub leader_gone: bool, + /// A vfork parent's thread, parked until this child execs or ends. + pub vfork: Option, + /// What this process raises at its parent when it ends; clone names it. + pub exit_signal: u8, + /// Children that raise something other than SIGCHLD, for __WCLONE. + pub clone_kids: Vec, + /// The process group each ended child was in, for a wait by group. + pub kid_groups: Vec<(u32, u32)>, + /// The mask of each signalfd, named by its descriptor's handle. + pub sigfds: Vec, } diff --git a/userland/capsule_linux/src/linux/guest/sigqueue_new.rs b/userland/capsule_linux/src/linux/guest/sigqueue_new.rs new file mode 100644 index 000000000..a1afd5ee6 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigqueue_new.rs @@ -0,0 +1,44 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A process's signal state as it starts: every disposition at its default, +//! nothing pending, no timers, no waits, and SIGCHLD as its exit signal. + +use alloc::vec::Vec; + +use super::sigqueue::Signals; +use super::sigstate::{SigAction, NSIG}; + +impl Default for Signals { + fn default() -> Self { + Signals { + actions: [SigAction::default(); NSIG], + pending: Vec::new(), + threads: Vec::new(), + outbox: Vec::new(), + real: None, + timers: Vec::new(), + sigwaits: Vec::new(), + childwaits: Vec::new(), + leader_gone: false, + vfork: None, + exit_signal: super::sigstate::SIGCHLD, + clone_kids: Vec::new(), + kid_groups: Vec::new(), + sigfds: Vec::new(), + } + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs new file mode 100644 index 000000000..d475cf10b --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs @@ -0,0 +1,49 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Reading and recording a process's signal state: the disposition of each +//! signal, the nearest deadline its timers and waits have, and what is +//! pending for a thread. + +use super::sigqueue::Signals; +use super::sigstate::{bit, SigAction, NSIG}; + +impl Signals { + /// Record a disposition; `signum` is 1..=NSIG. + pub fn set(&mut self, signum: usize, act: SigAction) { + if (1..=NSIG).contains(&signum) { + self.actions[signum - 1] = act; + } + } + + pub fn action(&self, signum: usize) -> Option { + (1..=NSIG).contains(&signum).then(|| self.actions[signum - 1]) + } + + /// The nearest deadline of a timer or a sigtimedwait, for the serve loop's + /// wait to end by. + pub fn next_due(&self) -> Option { + let timers = self.timers.iter().filter_map(|t| t.due); + let waits = self.sigwaits.iter().filter_map(|w| w.due); + self.real.map(|t| t.due).into_iter().chain(timers).chain(waits).min() + } + + /// Every signal pending for `tid` or for its process, as a mask. + pub fn pending_for(&self, tid: u32) -> u64 { + let mine = self.pending.iter().filter(|(t, _)| *t == tid || *t == 0); + mine.fold(0, |m, (_, i)| m | bit(i.signo)) + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigraise.rs b/userland/capsule_linux/src/linux/guest/sigraise.rs new file mode 100644 index 000000000..598a1e005 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigraise.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Raising a signal, in Linux's order: a standard signal already pending for +//! the same target is not queued again, a realtime one queues each time until +//! the queue is full; taking one is in sigtake. + +use super::sigdefault::{default_of, Default}; +use super::siginfo::SigInfo; +use super::sigqueue::{Signals, QUEUE_MAX}; +use super::sigstate::bit; + +/// The first realtime signal: from here up each raise is queued. +const SIGRTMIN: u8 = 32; + +impl Signals { + /// Queue `info` for thread `tid`, or for the process when `tid` is 0. + /// False when nothing was queued: an ignored signal no thread blocks is + /// discarded as it is sent, as Linux does, and so is a coalesced one. + pub fn raise(&mut self, tid: u32, info: SigInfo) -> bool { + let s = info.signo; + if self.discards(s) && !self.threads.iter().any(|t| t.blocked & bit(s) != 0) { + return false; + } + if s < SIGRTMIN && self.pending.iter().any(|(t, i)| *t == tid && i.signo == s) { + return false; + } + if s >= SIGRTMIN && self.pending.len() >= QUEUE_MAX { + return false; + } + self.pending.push((tid, info)); + true + } + + /// True when taking `signum` would do nothing: ignored by its handler, + /// or by a default of ignore or continue with nothing ever stopped. + pub fn discards(&self, signum: u8) -> bool { + let act = self.actions[signum as usize - 1]; + act.ignores() + || (!act.catches() && matches!(default_of(signum), Default::Ignore | Default::Continue)) + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigstate.rs b/userland/capsule_linux/src/linux/guest/sigstate.rs index ae31b5025..a14a7e0b2 100644 --- a/userland/capsule_linux/src/linux/guest/sigstate.rs +++ b/userland/capsule_linux/src/linux/guest/sigstate.rs @@ -14,12 +14,36 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! A signal's disposition: what the guest asked to happen when it fires. -//! Process-wide, as on Linux. +//! A signal's disposition, its number and its bit in a mask; what Linux does +//! with a signal no handler takes is in sigdefault. Dispositions are +//! process-wide, as on Linux; the numbers are transcribed. /// The largest signal Linux defines. pub const NSIG: usize = 64; +pub const SIGKILL: u8 = 9; +pub const SIGSEGV: u8 = 11; +pub const SIGPIPE: u8 = 13; +pub const SIGALRM: u8 = 14; +pub const SIGCHLD: u8 = 17; +pub const SIGSTOP: u8 = 19; + +pub const SA_NOCLDWAIT: u64 = 2; +pub const SA_ONSTACK: u64 = 0x0800_0000; +pub const SA_RESTART: u64 = 0x1000_0000; +pub const SA_NODEFER: u64 = 0x4000_0000; +pub const SA_RESETHAND: u64 = 0x8000_0000; + +/// The one bit a signal has in a mask. +pub fn bit(signum: u8) -> u64 { + 1u64 << (signum - 1) +} + +/// SIGKILL and SIGSTOP are never blocked, whatever a mask asks for. +pub fn blockable(mask: u64) -> u64 { + mask & !(bit(SIGKILL) | bit(SIGSTOP)) +} + /// `struct sigaction` as the guest passes it: handler, flags, restorer, mask. #[derive(Clone, Copy, Default)] pub struct SigAction { diff --git a/userland/capsule_linux/src/linux/guest/sigtake.rs b/userland/capsule_linux/src/linux/guest/sigtake.rs new file mode 100644 index 000000000..9ee8588c7 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigtake.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Taking a signal, in Linux's order: the lowest-numbered first, the thread's +//! own before the process's; and dropping what a deleted timer or a newly +//! ignored signal leaves pending. + +use super::siginfo::SigInfo; +use super::sigqueue::Signals; +use super::sigstate::bit; + +impl Signals { + /// Take the next signal `tid` may take: one of `allow`, its own before the + /// process's, lowest first. A timer's signal carries the overruns counted + /// while it waited, and its timer is free to queue again. + pub fn take(&mut self, tid: u32, allow: u64) -> Option { + let lowest = |want: u32| { + let each = self.pending.iter().enumerate(); + let fit = each.filter(|(_, (t, i))| *t == want && allow & bit(i.signo) != 0); + fit.min_by_key(|(_, (_, i))| i.signo).map(|(at, _)| at) + }; + let at = lowest(tid).or_else(|| lowest(0))?; + let mut info = self.pending.remove(at).1; + if let Some((id, _)) = info.timer { + if let Some(t) = self.timers.iter_mut().find(|t| t.id == id) { + info.timer = Some((id, t.overrun)); + t.last_overrun = t.overrun; + t.overrun = 0; + t.queued = false; + } + } + Some(info) + } + + /// Drop the signal a deleted timer left queued. + pub fn drop_timer_signal(&mut self, id: i32) { + self.pending.retain(|(_, i)| i.timer.is_none_or(|(t, _)| t != id)); + } + + /// Drop every pending `signum`, as setting SIG_IGN on it does. + pub fn discard(&mut self, signum: u8) { + self.pending.retain(|(_, i)| i.signo != signum); + self.timers_requeue(); + } + + /// A timer whose queued signal was dropped may queue again. + pub fn timers_requeue(&mut self) { + for t in self.timers.iter_mut().filter(|t| t.queued) { + t.queued = self.pending.iter().any(|(_, i)| i.timer.is_some_and(|(id, _)| id == t.id)); + } + } + + /// Each pending signal with the thread it is for, 0 for the process. + pub fn queued(&self) -> alloc::vec::Vec<(u32, u8)> { + self.pending.iter().map(|(t, i)| (*t, i.signo)).collect() + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigthread.rs b/userland/capsule_linux/src/linux/guest/sigthread.rs new file mode 100644 index 000000000..a4527f11e --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigthread.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What each thread has of its own: the signals it blocks, its alternate +//! stack, and the mask a sigsuspend put aside until a handler returns. + +use super::sigqueue::Signals; +use super::sigstate::blockable; + +/// ss_flags: running on the alternate stack, and no alternate stack set. +pub const SS_ONSTACK: u64 = 1; +pub const SS_DISABLE: u64 = 2; + +#[derive(Clone, Copy)] +pub struct ThreadSig { + pub tid: u32, + pub blocked: u64, + /// ss_sp, ss_flags, ss_size. + pub alt: [u64; 3], + /// The mask sigsuspend replaced, which the handler's frame restores. + pub saved: Option, +} + +impl ThreadSig { + pub(super) fn new(tid: u32, blocked: u64) -> Self { + ThreadSig { tid, blocked, alt: [0, SS_DISABLE, 0], saved: None } + } +} + +impl Signals { + pub fn thread(&mut self, tid: u32) -> &mut ThreadSig { + let at = match self.threads.iter().position(|t| t.tid == tid) { + Some(at) => at, + None => { + self.threads.push(ThreadSig::new(tid, 0)); + self.threads.len() - 1 + } + }; + &mut self.threads[at] + } + + pub fn blocked(&self, tid: u32) -> u64 { + self.threads.iter().find(|t| t.tid == tid).map_or(0, |t| t.blocked) + } + + pub fn set_blocked(&mut self, tid: u32, mask: u64) { + self.thread(tid).blocked = blockable(mask); + } + + /// A new thread starts with its creator's mask and no alternate stack, + /// as clone(CLONE_VM) leaves it. + pub fn born(&mut self, parent: u32, child: u32) { + let blocked = self.blocked(parent); + self.threads.retain(|t| t.tid != child); + self.threads.push(ThreadSig::new(child, blocked)); + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigthread_copy.rs b/userland/capsule_linux/src/linux/guest/sigthread_copy.rs new file mode 100644 index 000000000..439074569 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigthread_copy.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The signal state a new program starts with: a forked child's, copied from +//! the forking thread, and what exec leaves of a process's own. + +use super::sigqueue::Signals; +use super::sigstate::SigAction; +use super::sigthread::ThreadSig; + +impl Signals { + /// A forked child: the dispositions, and the forking thread's mask and + /// alternate stack for its one thread. Nothing pending, no timers. + pub fn forked(&self, caller: u32, child: u32) -> Signals { + /* A signalfd is a descriptor, and the child holds a copy of each. */ + let sigfds = self.sigfds.clone(); + let mut s = Signals { actions: self.actions, sigfds, ..Signals::default() }; + let mine = self.threads.iter().find(|t| t.tid == caller).copied(); + let mut t = mine.unwrap_or(ThreadSig::new(child, 0)); + t.tid = child; + t.saved = None; + s.threads.push(t); + s + } + + /// Exec keeps the mask, the pending signals and ITIMER_REAL; caught + /// signals go back to their default, ignored ones stay ignored; the + /// alternate stack and the POSIX timers are gone. + pub fn exec_reset(&mut self, tid: u32) { + for act in self.actions.iter_mut().filter(|a| a.catches()) { + *act = SigAction::default(); + } + self.timers.clear(); + self.pending.retain(|(_, i)| i.timer.is_none()); + let blocked = self.blocked(tid); + self.threads = alloc::vec![ThreadSig::new(tid, blocked)]; + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigtimer.rs b/userland/capsule_linux/src/linux/guest/sigtimer.rs new file mode 100644 index 000000000..141f93242 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigtimer.rs @@ -0,0 +1,49 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The timers that end in a signal: ITIMER_REAL, which alarm and setitimer +//! set, and the POSIX timers timer_create makes. Deadlines are milliseconds of +//! the family's monotonic clock, the finest step it keeps. + +/// ITIMER_REAL: when it next fires, and its period, 0 for once. +#[derive(Clone, Copy)] +pub struct Itimer { + pub due: u64, + pub interval: u64, +} + +/// One timer_create timer. +#[derive(Clone, Copy)] +pub struct PosixTimer { + /// The id the guest was given, from 0 up, as Linux numbers them. + pub id: i32, + /// The clock an absolute time is read on. + pub clock: u64, + /// The signal it raises, 0 for SIGEV_NONE. + pub signo: u8, + /// The thread SIGEV_THREAD_ID named, 0 for the process. + pub tid: u32, + /// sigev_value, handed back in si_value. + pub value: u64, + pub due: Option, + pub interval: u64, + /// Expiries while its signal was still queued, and the count the last + /// taken signal carried, which timer_getoverrun reports. + pub overrun: i32, + pub last_overrun: i32, + /// Its signal is queued and not yet taken. + pub queued: bool, +} diff --git a/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs b/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs new file mode 100644 index 000000000..d1ac5df70 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs @@ -0,0 +1,53 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! How a timer that ends in a signal moves on once it fires: a periodic one +//! past every period that ended unseen, counted as overruns for a POSIX +//! timer; a one-shot one stops. Pure, so the arithmetic is checked without a +//! guest. + +use super::sigtimer::{Itimer, PosixTimer}; + +impl Itimer { + /// Move past every period that has ended by `now`; false for a one-shot. + pub fn rearm(&mut self, now: u64) -> bool { + if self.interval == 0 { + return false; + } + while self.due <= now { + self.due = self.due.saturating_add(self.interval); + } + true + } +} + +impl PosixTimer { + /// Fire at `now`: the periods that passed unseen are overruns, as Linux + /// counts them. False when the timer does not fire again. + pub fn rearm(&mut self, now: u64) -> bool { + let Some(due) = self.due else { + return false; + }; + if self.interval == 0 { + self.due = None; + return false; + } + let missed = now.saturating_sub(due) / self.interval; + self.overrun = self.overrun.saturating_add(missed.min(i32::MAX as u64) as i32); + self.due = Some(due + (missed + 1) * self.interval); + true + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigwaits.rs b/userland/capsule_linux/src/linux/guest/sigwaits.rs new file mode 100644 index 000000000..b07e0d3c1 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigwaits.rs @@ -0,0 +1,74 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Threads parked until a signal or a child says something, and signals on +//! their way to another process of the family. + +use super::siginfo::SigInfo; + +/// A thread in pause, sigsuspend or sigtimedwait. +#[derive(Clone, Copy)] +pub struct SigWait { + pub tid: u32, + /// sigtimedwait's set: a signal in it is taken, not handled. 0 for pause + /// and sigsuspend, which only a handler ends. + pub set: u64, + /// Where sigtimedwait writes the siginfo, 0 for nowhere. + pub info: u64, + /// When sigtimedwait gives up with EAGAIN. + pub due: Option, + /// For a signalfd read, how many records fit where `info` points; 0 else. + pub records: u64, +} + +/// Which children a wait4 or waitid asks about. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Which { + Any, + Pid(u32), + Group(u32), +} + +/// A thread in wait4 or waitid, and where its answer goes. +#[derive(Clone, Copy)] +pub struct ChildWait { + pub tid: u32, + pub which: Which, + pub options: u64, + /// wait4's status word, or waitid's siginfo; and the rusage, 0 for none. + pub out: u64, + pub rusage: u64, + pub waitid: bool, +} + +/// Who a signal leaving this process is for. +#[derive(Clone, Copy)] +pub enum Target { + Process(u32), + /// A thread, and the process it must belong to, 0 for any. + Thread(u32, u32), + Group(u32), + /// kill(-1): every process of the family but the sender. + All, +} + +/// A signal for another process, and the thread parked until it is sent. +#[derive(Clone, Copy)] +pub struct Outbound { + pub from: u32, + pub to: Target, + pub info: SigInfo, +} diff --git a/userland/capsule_linux/src/linux/net/poll.rs b/userland/capsule_linux/src/linux/net/poll.rs index 55b30b9a4..08db2b49f 100644 --- a/userland/capsule_linux/src/linux/net/poll.rs +++ b/userland/capsule_linux/src/linux/net/poll.rs @@ -38,6 +38,7 @@ pub fn ready(guest: &Guest, fd: u64) -> u16 { Some(Kind::Timer) => crate::linux::file::timer_bits(guest, fd), Some(Kind::Pipe) => crate::linux::call::pipe_bits(guest, fd), Some(Kind::Event) => crate::linux::file::event_bits(guest, fd), + Some(Kind::Signal) => crate::linux::call::signalfd_bits(guest, fd), Some(Kind::Resolver) => resolver_bits(guest, fd), Some(_) => POLLIN | POLLOUT, } diff --git a/userland/capsule_linux/src/linux/serve/deliver.rs b/userland/capsule_linux/src/linux/serve/deliver.rs index e5ed46059..ffbf471f6 100644 --- a/userland/capsule_linux/src/linux/serve/deliver.rs +++ b/userland/capsule_linux/src/linux/serve/deliver.rs @@ -14,48 +14,60 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Delivering a caught signal to the thread returning from a syscall. The -//! handler is entered with the interrupted syscall's return value already in -//! rax, so when it returns through rt_sigreturn the program sees that value. -//! Only the trapping thread is delivered to here; a signal raised against a -//! thread parked elsewhere waits in the queue until that thread next traps. +//! A thread taking the signals it may take. A caught one enters its handler +//! through Linux's rt_sigframe; an uncaught one does what its default says, +//! and a default of ending the process ends all of it. A thread takes them +//! when it returns from a call, with the call's value in rax (maybe_deliver); +//! when a signal ends the wait it is parked in (deliver_wait); and when the +//! kernel stops it running, on the registers it stopped with (deliver_on). -use nonos_libc::{mk_foreign_context, mk_foreign_signal, ForeignRegs, SIGNAL_DELIVER}; +use nonos_libc::{mk_foreign_context, ForeignRegs}; -use crate::linux::call::sigframe::build; +use super::deliver_enter::enter; +use super::deliver_say::stop_unserved; +use crate::linux::call::killed; +use crate::linux::guest::sigdefault::{default_of, Default}; use crate::linux::guest::Guest; /// rax in the register word order. const RAX: usize = 13; -/// True when a handler was entered, so the caller must not also reply. +/// True when the thread was answered, into a handler or by its process +/// ending, so the caller must not also reply. pub fn maybe_deliver(guest: &mut Guest, tid: u32, reply: u64) -> bool { - let Some((signum, act)) = guest.signals.take_caught(tid) else { - return false; - }; - let mut regs: ForeignRegs = [0; 18]; - let built = (mk_foreign_context(tid, &mut regs) == 0).then(|| { - regs[RAX] = reply; - build(®s, act.handler, act.restorer, u32::from(signum), 0) - }); - let Some(Some((_, buf, enter))) = built else { - // Could not read the thread or shape a frame: keep the signal pending. - guest.signals.raise(tid, signum); - return false; - }; - if guest.write(enter[15], &buf) < buf.len() as i64 { - guest.signals.raise(tid, signum); + if guest.signals.pending_for(tid) & !guest.signals.blocked(tid) == 0 { return false; } - if mk_foreign_signal(tid, &enter, SIGNAL_DELIVER) != 0 { - guest.signals.raise(tid, signum); + let mut regs: ForeignRegs = [0; 18]; + if mk_foreign_context(tid, &mut regs) != 0 { return false; } - say(tid, signum, act.handler); - true + regs[RAX] = reply; + deliver_on(guest, tid, regs) } -fn say(tid: u32, signum: u8, handler: u64) { - let line = alloc::format!("[LINUX] signal {signum} to tid {tid}, handler {handler:#x}\n"); - let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); +/// Take what `tid` may take and act on it, over `regs` as they stand. For a +/// frame the kernel stopped while running as much as for a returning call. +pub fn deliver_on(guest: &mut Guest, tid: u32, regs: ForeignRegs) -> bool { + loop { + let allow = !guest.signals.blocked(tid); + let Some(info) = guest.signals.take(tid, allow) else { + return false; + }; + let act = guest.signals.action(info.signo as usize).unwrap_or_default(); + if act.catches() { + return enter(guest, tid, ®s, info); + } + if act.ignores() { + continue; + } + match default_of(info.signo) { + Default::Terminate => { + killed(guest, info.signo); + return true; + } + Default::Stop => stop_unserved(info.signo), + Default::Ignore | Default::Continue => {} + } + } } diff --git a/userland/capsule_linux/src/linux/serve/deliver_enter.rs b/userland/capsule_linux/src/linux/serve/deliver_enter.rs new file mode 100644 index 000000000..cd85d9a12 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_enter.rs @@ -0,0 +1,74 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Entering a handler: the frame on the thread's stack, or at the top of its +//! alternate stack when the handler asked for SA_ONSTACK and the thread is not +//! already on it; the mask while the handler runs is the thread's own, the +//! handler's sa_mask, and the signal itself unless SA_NODEFER. + +use nonos_libc::{mk_foreign_signal, ForeignRegs, SIGNAL_DELIVER}; + +use super::deliver_say::say; +use super::deliver_stack::placement; +use crate::linux::call::killed; +use crate::linux::call::sigframe::Entry; +use crate::linux::call::sigframe_build::build; +use crate::linux::guest::siginfo::SigInfo; +use crate::linux::guest::sigstate::{bit, SigAction, SA_NODEFER, SA_RESETHAND, SIGSEGV}; +use crate::linux::guest::Guest; + +const RSP: usize = 15; + +/// True once the thread is answered: in its handler, or its process ended +/// because no frame could be written, which Linux answers with SIGSEGV. +pub fn enter(guest: &mut Guest, tid: u32, regs: &ForeignRegs, info: SigInfo) -> bool { + let act = guest.signals.action(info.signo as usize).unwrap_or_default(); + let t = *guest.signals.thread(tid); + let (alt_top, stack) = placement(t.alt, act.flags, regs[RSP]); + let saved = t.saved.unwrap_or(t.blocked); + let bytes = info.bytes(); + let entry = Entry { + handler: act.handler, + restorer: act.restorer, + signum: u32::from(info.signo), + blocked: saved, + alt_top, + stack, + info: &bytes, + }; + let Some((at, buf, into)) = build(regs, &entry).filter(|_| act.restorer != 0) else { + killed(guest, SIGSEGV); + return true; + }; + if guest.write(at, &buf) < buf.len() as i64 { + killed(guest, SIGSEGV); + return true; + } + if mk_foreign_signal(tid, &into, SIGNAL_DELIVER) != 0 { + /* Not parked after all: the signal waits for its next return. */ + let _ = guest.signals.raise(tid, info); + return false; + } + let during = + t.blocked | act.mask | if act.flags & SA_NODEFER != 0 { 0 } else { bit(info.signo) }; + guest.signals.set_blocked(tid, during); + guest.signals.thread(tid).saved = None; + if act.flags & SA_RESETHAND != 0 { + guest.signals.set(info.signo as usize, SigAction::default()); + } + say(tid, info.signo, act.handler); + true +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_interrupt.rs b/userland/capsule_linux/src/linux/serve/deliver_interrupt.rs new file mode 100644 index 000000000..004b59736 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_interrupt.rs @@ -0,0 +1,48 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A caught signal ending a parked thread's wait: the handler is entered over +//! the parked call, whose answer is EINTR or a restart as deliver_restart +//! decides, and a relative sleep writes what it had left first. + +use nonos_libc::{mk_foreign_context, mk_foreign_reply, ForeignRegs}; + +use super::deliver_enter::enter; +use super::deliver_rem::write_rem; +use super::deliver_restart::rewind; +use crate::linux::guest::siginfo::SigInfo; +use crate::linux::guest::Guest; + +/// End `tid`'s wait with a handler entered over its parked call. +pub fn interrupt(guest: &mut Guest, tid: u32, info: SigInfo) { + let mut regs: ForeignRegs = [0; 18]; + if mk_foreign_context(tid, &mut regs) != 0 { + let _ = guest.signals.raise(tid, info); + return; + } + let Some(parked) = guest.leave_waits(tid) else { + let _ = guest.signals.raise(tid, info); + return; + }; + let act = guest.signals.action(info.signo as usize).unwrap_or_default(); + write_rem(guest, ®s, parked); + rewind(&mut regs, act.flags); + if !enter(guest, tid, ®s, info) { + /* Out of its wait with no handler entered: it must still be answered. */ + let _ = + mk_foreign_reply(tid, crate::linux::abi::errno::fail(crate::linux::abi::errno::EINTR)); + } +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_pipe.rs b/userland/capsule_linux/src/linux/serve/deliver_pipe.rs new file mode 100644 index 000000000..636d2e27a --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_pipe.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! SIGPIPE at the thread whose write found no reader. Linux raises it in the +//! write itself; here the write answers EPIPE and the answer is where the +//! signal is raised, before the reply, so the thread takes it on its way out +//! as Linux's does. A send with MSG_NOSIGNAL raises nothing. + +use crate::linux::abi::errno; +use crate::linux::call::sigpipe; +use crate::linux::guest::Guest; + +const WRITE: u64 = 1; +const WRITEV: u64 = 20; +const SENDTO: u64 = 44; +const SENDMSG: u64 = 46; +const MSG_NOSIGNAL: u64 = 0x4000; + +/// Raise SIGPIPE at `tid` when call `nr` with arguments `a` answered EPIPE. +pub fn broken_pipe(g: &mut Guest, tid: u32, nr: u64, a: [u64; 6], value: u64) { + if value != errno::fail(errno::EPIPE) { + return; + } + let quiet = match nr { + WRITE | WRITEV => false, + SENDTO => a[3] & MSG_NOSIGNAL != 0, + SENDMSG => a[2] & MSG_NOSIGNAL != 0, + _ => return, + }; + if !quiet { + sigpipe(g, tid); + } +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_rem.rs b/userland/capsule_linux/src/linux/serve/deliver_rem.rs new file mode 100644 index 000000000..5b646cb2f --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_rem.rs @@ -0,0 +1,54 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A relative sleep cut short by a handler writes the time it had left, as +//! Linux's nanosleep and relative clock_nanosleep do; an absolute one writes +//! nothing, as on Linux. + +use nonos_libc::ForeignRegs; + +use super::deliver_restart::{R10, RAX, RSI}; +use crate::linux::call::now_ms; +use crate::linux::guest::sigpark::Parked; +use crate::linux::guest::Guest; + +const CLOCK_MONOTONIC: u64 = 1; +const TIMER_ABSTIME: u64 = 1; +const NANOSLEEP: u64 = 35; +const CLOCK_NANOSLEEP: u64 = 230; + +/// A relative sleep cut short writes what was left of it where it was asked. +pub fn write_rem(guest: &Guest, regs: &ForeignRegs, parked: Parked) { + let Parked::Sleep(due) = parked else { + return; + }; + let rem = match regs[RAX] { + NANOSLEEP => regs[RSI], + CLOCK_NANOSLEEP if regs[RSI] & TIMER_ABSTIME == 0 => regs[R10], + _ => 0, + }; + let Some(now) = now_ms(CLOCK_MONOTONIC) else { + return; + }; + if rem == 0 { + return; + } + let left = due.saturating_sub(now); + let mut spec = [0u8; 16]; + spec[..8].copy_from_slice(&(left / 1000).to_le_bytes()); + spec[8..].copy_from_slice(&((left % 1000) * 1_000_000).to_le_bytes()); + let _ = guest.write(rem, &spec); +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_restart.rs b/userland/capsule_linux/src/linux/serve/deliver_restart.rs new file mode 100644 index 000000000..48b61266b --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_restart.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What an interrupted call answers, as Linux decides it. Reads and writes, +//! the socket calls, wait4, waitid and an untimed futex wait end with +//! ERESTARTSYS: under SA_RESTART the call runs again once the handler returns, +//! otherwise EINTR. Sleeps, timed waits, poll, select, epoll, pause and the +//! signal waits always answer EINTR, and a relative nanosleep writes the time +//! it had left (deliver_rem). + +use nonos_libc::ForeignRegs; + +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SA_RESTART; + +pub const R10: usize = 2; +pub const RSI: usize = 9; +pub const RAX: usize = 13; +const RIP: usize = 16; +/// The length of `syscall`, which a restart steps back over. +const SYSCALL_LEN: u64 = 2; + +/// read, write, readv, writev, accept, sendto, recvfrom, sendmsg, recvmsg, +/// wait4, waitid, accept4. +const RESTARTS: [u64; 12] = [0, 1, 19, 20, 43, 44, 45, 46, 47, 61, 247, 288]; +const FUTEX: u64 = 202; + +/// Set rax to what the handler returns into: the call again, or EINTR. The +/// parked frame's rax is still the call's number. +pub fn rewind(regs: &mut ForeignRegs, flags: u64) { + let nr = regs[RAX]; + let untimed_futex = nr == FUTEX && regs[R10] == 0; + if flags & SA_RESTART != 0 && (RESTARTS.contains(&nr) || untimed_futex) { + regs[RIP] = regs[RIP].wrapping_sub(SYSCALL_LEN); + return; + } + regs[RAX] = errno::fail(errno::EINTR); +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_say.rs b/userland/capsule_linux/src/linux/serve/deliver_say.rs new file mode 100644 index 000000000..fa0e34f92 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_say.rs @@ -0,0 +1,30 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What delivering a signal says in the log: each handler entered, and each +//! stop that is not served. + +/// No guest is ever stopped: job control needs the kernel to hold every +/// thread of a process still, which it does not offer a supervisor. +pub fn stop_unserved(signum: u8) { + let line = alloc::format!("[LINUX] unserved stop: signal {signum} does not stop a guest\n"); + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); +} + +pub fn say(tid: u32, signum: u8, handler: u64) { + let line = alloc::format!("[LINUX] signal {signum} to tid {tid}, handler {handler:#x}\n"); + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs b/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs new file mode 100644 index 000000000..8c840cf16 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs @@ -0,0 +1,48 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A thread in sigtimedwait takes a pending signal of its set without any +//! handler running, and its call answers with that signal's number, the +//! siginfo written where it asked. + +use nonos_libc::mk_foreign_reply; + +use crate::linux::guest::Guest; + +/// A thread in sigtimedwait takes a pending signal of its set, and its call +/// answers with that signal's number and siginfo. +pub fn taken_by_sigtimedwait(guest: &mut Guest) { + for w in guest.signals.sigwaits.clone().into_iter().filter(|w| w.set != 0) { + if w.records != 0 { + if let Some(n) = + crate::linux::call::signalfd_take(guest, w.tid, w.set, w.info, w.records) + { + guest.signals.sigwaits.retain(|x| x.tid != w.tid); + let _ = mk_foreign_reply(w.tid, n); + } + continue; + } + let Some(info) = guest.signals.take(w.tid, w.set) else { + continue; + }; + guest.signals.sigwaits.retain(|x| x.tid != w.tid); + let value = match w.info != 0 && guest.write(w.info, &info.bytes()) < 128 { + true => crate::linux::abi::errno::fail(crate::linux::abi::errno::EFAULT), + false => u64::from(info.signo), + }; + let _ = mk_foreign_reply(w.tid, value); + } +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_stack.rs b/userland/capsule_linux/src/linux/serve/deliver_stack.rs new file mode 100644 index 000000000..01937e05d --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_stack.rs @@ -0,0 +1,40 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a handler's frame goes: on the thread's stack, or at the top of its +//! alternate stack when the handler asked for SA_ONSTACK and the thread is not +//! already on it; and the uc_stack the frame records, with ss_flags as +//! sigaltstack would report them at the moment the signal came. + +use crate::linux::guest::sigstate::SA_ONSTACK; +use crate::linux::guest::sigthread::{SS_DISABLE, SS_ONSTACK}; + +/// The top of the alternate stack when the frame goes there, and uc_stack: +/// ss_sp, ss_flags, ss_size. `alt` is the thread's, `rsp` where it stopped. +pub fn placement(alt: [u64; 3], act_flags: u64, rsp: u64) -> (Option, [u64; 3]) { + let [sp, flags, size] = alt; + let on_alt = flags & SS_DISABLE == 0 && rsp.wrapping_sub(sp) < size; + let alt_top = (act_flags & SA_ONSTACK != 0 && flags & SS_DISABLE == 0 && !on_alt) + .then(|| sp.saturating_add(size)); + let ss_flags = if flags & SS_DISABLE != 0 { + SS_DISABLE + } else if on_alt { + SS_ONSTACK + } else { + 0 + }; + (alt_top, [sp, ss_flags, size]) +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_wait.rs b/userland/capsule_linux/src/linux/serve/deliver_wait.rs new file mode 100644 index 000000000..3cd8d2bbc --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_wait.rs @@ -0,0 +1,61 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Signals that reach a process whose threads are parked, not returning. +//! Run after every answer: a sigtimedwait takes a signal in its set; an +//! uncaught signal's default acts at once, as Linux's does when it is sent; +//! and a caught signal ends the wait of a thread that does not block it, with +//! EINTR, or restarts the call under SA_RESTART where Linux restarts it. + +use super::deliver_interrupt::interrupt; +use super::deliver_say::stop_unserved; +use super::deliver_sigwait::taken_by_sigtimedwait; +use crate::linux::call::killed; +use crate::linux::guest::sigdefault::{default_of, Default}; +use crate::linux::guest::sigstate::bit; +use crate::linux::guest::Guest; + +pub fn settle(guest: &mut Guest) { + taken_by_sigtimedwait(guest); + for (tid, signo) in guest.signals.queued() { + if guest.exited.is_some() { + return; + } + let act = guest.signals.action(signo as usize).unwrap_or_default(); + let takers: alloc::vec::Vec = match tid { + 0 => guest.live_threads(), + t => alloc::vec![t], + }; + let free = |g: &Guest, t: &u32| g.signals.blocked(*t) & bit(signo) == 0; + if !act.catches() { + if takers.iter().any(|t| free(guest, t)) { + let _ = guest.signals.take(takers[0], bit(signo)); + match (act.ignores(), default_of(signo)) { + (false, Default::Terminate) => killed(guest, signo), + (false, Default::Stop) => stop_unserved(signo), + _ => {} + } + } + continue; + } + let parked = takers.iter().find(|t| free(guest, t) && guest.parked(**t).is_some()); + if let Some(&t) = parked { + if let Some(info) = guest.signals.take(t, bit(signo)) { + interrupt(guest, t, info); + } + } + } +} diff --git a/userland/capsule_linux/src/linux/serve/family.rs b/userland/capsule_linux/src/linux/serve/family.rs index 49ea32f48..92cb042e0 100644 --- a/userland/capsule_linux/src/linux/serve/family.rs +++ b/userland/capsule_linux/src/linux/serve/family.rs @@ -26,7 +26,7 @@ use core::mem; use nonos_libc::{mk_foreign_reply, ForeignFrame, FOREIGN_NR_DIED}; use super::answer::Answer; -use super::dispatch::answer; +use super::route_life::answer; use super::pid_map::frame_in; use super::pid_ns::PidNs; use super::pid_out::value_out; @@ -69,6 +69,7 @@ impl Family { let born = mem::take(&mut g.forked); if let Answer::Reply(value) = got { // A caught signal for this thread is delivered in place of the reply. + super::deliver_pipe::broken_pipe(g, frame.pid, frame.nr, frame.args(), value); let out = value_out(&mut self.ns, frame.nr, value); if !super::deliver::maybe_deliver(g, frame.pid, out) { let _ = mk_foreign_reply(frame.pid, out); @@ -79,15 +80,13 @@ impl Family { /// A guest thread ended on a signal. On Linux that ends the thread group, /// so the guest exits; reap then kills its other threads and answers any - /// waiter. The status carries the signal in the shell's 128+signo form. + /// waiter. The status is Linux's wait status: the signal's number. fn thread_died(&mut self, pid: u32, code: i32) { let Some(g) = self.guests.iter_mut().find(|g| g.owns(pid)) else { return; }; g.threads.retain(|t| *t != pid); - if g.exited.is_none() { - g.exited = Some(128 + signo_of(code)); - } + crate::linux::call::killed(g, signo_of(code) as u8); let line = alloc::format!("[LINUX] guest thread {pid} ended on a signal; ending the process\n"); crate::linux::start::say(line.as_bytes()); diff --git a/userland/capsule_linux/src/linux/serve/family_reap.rs b/userland/capsule_linux/src/linux/serve/family_reap.rs index 371e1c3c0..9e035f5bf 100644 --- a/userland/capsule_linux/src/linux/serve/family_reap.rs +++ b/userland/capsule_linux/src/linux/serve/family_reap.rs @@ -14,50 +14,49 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Ending what has exited, and telling each parent. - -use nonos_libc::{mk_foreign_reply, mk_kill}; +//! Ending what has exited, and telling each parent: the ended child waits as +//! a zombie until it is waited for, and its exit signal, SIGCHLD unless clone +//! named another, is raised at the parent with CLD_EXITED or CLD_KILLED. A +//! parent that ignores SIGCHLD, or asked for SA_NOCLDWAIT, has its children +//! reaped as they end, as Linux does. use super::family::Family; -use crate::linux::call::reap_one; - -const SIGKILL: u64 = 9; +use crate::linux::guest::siginfo::{SigInfo, CLD_EXITED, CLD_KILLED}; +use crate::linux::guest::sigstate::{SA_NOCLDWAIT, SIGCHLD}; +use crate::linux::guest::Guest; impl Family { - /// End every process that asked to, and tell its parent. + /// End every process that asked to, answer the waits and signals that + /// follows, and go again while that ends anything more. pub fn reap(&mut self) { - while let Some(i) = self.guests.iter().position(|g| g.exited.is_some()) { - let gone = self.guests.remove(i); - let code = gone.exited.unwrap_or(0); - for tid in gone.threads.iter().chain([gone.pid].iter()) { - let rc = mk_kill(*tid as u64, SIGKILL); - if rc < 0 { - // Refused, it runs on after its process ended. - let line = alloc::format!( - "[LINUX] kill refused: pid {tid} outlives its process, errno {}\n", - -rc - ); - let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); - } - } - if gone.pid == self.root { - self.root_code = code; - } - let Some(p) = self.guests.iter_mut().find(|g| g.children.contains(&gone.pid)) else { - continue; - }; - p.ended.push((gone.pid, code)); - if let Some((want, status, tid)) = p.waiting { - if let Some(value) = reap_one(p, want, status) { - p.waiting = None; - let value = super::pid_out::value_out( - &mut self.ns, - crate::linux::abi::nr::WAIT4, - value, - ); - let _ = mk_foreign_reply(tid, value); - } + loop { + let ended = self.end_exited(); + self.settle_child_waits(); + self.settle_signals(); + if !ended && !self.guests.iter().any(|g| g.exited.is_some()) { + return; } } } } + +/// Tell `p` that `gone` ended with `status`: kept as a zombie or reaped at +/// once, and its exit signal raised with CLD_EXITED or CLD_KILLED. +pub fn tell_parent(p: &mut Guest, gone: &Guest, status: i32) { + let sig = gone.signals.exit_signal; + let chld = p.signals.action(SIGCHLD as usize).unwrap_or_default(); + if sig == SIGCHLD && (chld.ignores() || chld.flags & SA_NOCLDWAIT != 0) { + p.children.retain(|c| *c != gone.pid); + } else { + p.ended.push((gone.pid, status)); + p.signals.kid_groups.push((gone.pid, gone.pgid)); + } + if sig != 0 { + let (code, value) = match status & 0x7f { + 0 => (CLD_EXITED, (status >> 8) & 0xff), + s => (CLD_KILLED, s), + }; + let info = SigInfo { value: value as u64, ..SigInfo::from(sig, code, gone.pid) }; + let _ = p.signals.raise(0, info); + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_reap_end.rs b/userland/capsule_linux/src/linux/serve/family_reap_end.rs new file mode 100644 index 000000000..e7dbeea23 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_reap_end.rs @@ -0,0 +1,67 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Ending each process that has exited: its threads are killed, its end is +//! said in the log, a vfork parent waiting on it is let go, and its parent is +//! told as family_reap tells it. + +use nonos_libc::mk_kill; + +use super::family::Family; +use super::family_reap::tell_parent; +use super::family_wait::answer; + +const SIGKILL: u64 = 9; + +impl Family { + pub(super) fn end_exited(&mut self) -> bool { + let mut any = false; + while let Some(i) = self.guests.iter().position(|g| g.exited.is_some()) { + any = true; + let gone = self.guests.remove(i); + let status = gone.exited.unwrap_or(0); + for tid in gone.threads.iter().chain([gone.pid].iter()) { + let rc = mk_kill(*tid as u64, SIGKILL); + if rc < 0 { + /* Refused, it runs on after its process ended. */ + let line = alloc::format!( + "[LINUX] kill refused: pid {tid} outlives its process, errno {}\n", + -rc + ); + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + } + } + let (code, signo) = ((status >> 8) & 0xff, status & 0x7f); + let shown = crate::linux::serve::guest_pid(gone.pid); + let line = match signo { + 0 => alloc::format!("[LINUX] process {shown} exited, status {code}\n"), + s => alloc::format!("[LINUX] process {shown} ended by signal {s}\n"), + }; + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + if gone.pid == self.root { + self.root_code = if signo == 0 { code } else { 128 + signo }; + } + let Some(p) = self.guests.iter_mut().find(|g| g.children.contains(&gone.pid)) else { + continue; + }; + if let Some(t) = gone.signals.vfork { + answer(p, t, u64::from(shown)); + } + tell_parent(p, &gone, status); + } + any + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_signal.rs b/userland/capsule_linux/src/linux/serve/family_signal.rs new file mode 100644 index 000000000..2ff67d604 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_signal.rs @@ -0,0 +1,35 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Signals between processes of the family, and the timers that raise them, +//! settled after every answer: the outbox is routed, due timers fire, and +//! each process's parked threads take what now reaches them. + +use super::family::Family; +use super::family_signal_fire::fire; +use crate::linux::call::now_ms; + +const CLOCK_MONOTONIC: u64 = 1; + +impl Family { + pub(super) fn settle_signals(&mut self) { + self.route_outbox(); + if let Some(now) = now_ms(CLOCK_MONOTONIC) { + self.guests.iter_mut().for_each(|g| fire(g, now)); + } + self.guests.iter_mut().for_each(super::deliver_wait::settle); + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_signal_fire.rs b/userland/capsule_linux/src/linux/serve/family_signal_fire.rs new file mode 100644 index 000000000..a91dad558 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_signal_fire.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The timers of a process that are due: ITIMER_REAL raises SIGALRM, a POSIX +//! timer raises its own signal or counts an overrun while that still waits, +//! and a sigtimedwait whose time is up answers EAGAIN. + +use super::family_wait::answer; +use crate::linux::abi::errno; +use crate::linux::guest::siginfo::{SigInfo, SI_KERNEL, SI_TIMER}; +use crate::linux::guest::sigstate::SIGALRM; +use crate::linux::guest::Guest; + +/// Every timer of `g` due by `now`, fired, and every sigtimedwait whose time +/// is up answered EAGAIN. +pub fn fire(g: &mut Guest, now: u64) { + if let Some(mut t) = g.signals.real.filter(|t| t.due <= now) { + let _ = g.signals.raise(0, SigInfo::from(SIGALRM, SI_KERNEL, 0)); + g.signals.real = t.rearm(now).then_some(t); + } + for i in 0..g.signals.timers.len() { + let mut t = g.signals.timers[i]; + if t.due.is_none_or(|d| d > now) { + continue; + } + if t.queued { + t.overrun = t.overrun.saturating_add(1); + } else if t.signo != 0 { + let info = SigInfo { + timer: Some((t.id, 0)), + value: t.value, + ..SigInfo::from(t.signo, SI_TIMER, 0) + }; + t.queued = g.signals.raise(t.tid, info); + } + let _ = t.rearm(now); + g.signals.timers[i] = t; + } + let late: alloc::vec::Vec = g + .signals + .sigwaits + .iter() + .filter(|w| w.due.is_some_and(|d| d <= now)) + .map(|w| w.tid) + .collect(); + for tid in late { + g.signals.sigwaits.retain(|w| w.tid != tid); + answer(g, tid, errno::fail(errno::EAGAIN)); + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_signal_route.rs b/userland/capsule_linux/src/linux/serve/family_signal_route.rs new file mode 100644 index 000000000..4e5268510 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_signal_route.rs @@ -0,0 +1,61 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A signal from the outbox reaches every process its target names, and its +//! sender is answered 0, or ESRCH when it named no one, as kill answers; an +//! ended child not yet waited for counts, as a zombie does on Linux. + +use super::family::Family; +use super::family_wait::answer; +use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::{Outbound, Target}; +use crate::linux::guest::Guest; + +impl Family { + pub(super) fn route_outbox(&mut self) { + let mut out = alloc::vec::Vec::new(); + for g in self.guests.iter_mut() { + out.extend(core::mem::take(&mut g.signals.outbox).into_iter().map(|o| (g.pid, o))); + } + for (sender, o) in out { + let zombie = |pid: u32| self.guests.iter().any(|g| g.ended.iter().any(|e| e.0 == pid)); + let mut reached = + matches!(o.to, Target::Process(p) | Target::Thread(_, p) if zombie(p)); + for g in self.guests.iter_mut() { + if let Some(t) = taker(g, sender, &o) { + reached = true; + if o.info.signo != 0 && g.exited.is_none() { + let _ = g.signals.raise(t, o.info); + } + } + } + let value = if reached { 0 } else { errno::fail(errno::ESRCH) }; + if let Some(g) = self.guests.iter_mut().find(|g| g.owns(o.from)) { + answer(g, o.from, value); + } + } + } +} + +/// The thread of `g` a signal is for, 0 for the whole process, or None. +fn taker(g: &Guest, sender: u32, o: &Outbound) -> Option { + match o.to { + Target::Process(p) => g.owns(p).then_some(0), + Target::Thread(tgid, t) => (g.owns(t) && (tgid == 0 || tgid == g.pid)).then_some(t), + Target::Group(pg) => (g.pgid == pg).then_some(0), + Target::All => (g.pid != sender).then_some(0), + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_sleep.rs b/userland/capsule_linux/src/linux/serve/family_sleep.rs index 2f0e73985..8dfa0dbbc 100644 --- a/userland/capsule_linux/src/linux/serve/family_sleep.rs +++ b/userland/capsule_linux/src/linux/serve/family_sleep.rs @@ -47,8 +47,10 @@ impl Family { let sleeper = self .guests .iter() - .flat_map(|g| g.sleepers.iter().chain(g.futex_until.iter())) - .map(|&(d, _)| d.saturating_sub(now)) + .flat_map(|g| g.sleepers.iter().chain(g.futex_until.iter()).map(|&(d, _)| d)) + /* A timer or a sigtimedwait is due too. */ + .chain(self.guests.iter().filter_map(|g| g.signals.next_due())) + .map(|d| d.saturating_sub(now)) .min(); [sleeper, self.next_wait_ms(now)].into_iter().flatten().min() } diff --git a/userland/capsule_linux/src/linux/serve/family_wait.rs b/userland/capsule_linux/src/linux/serve/family_wait.rs new file mode 100644 index 000000000..812885a08 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_wait.rs @@ -0,0 +1,45 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Answering wait4 and waitid. A child that has ended and fits is reported, +//! and reaped unless WNOWAIT; with none ended, WNOHANG answers 0, and a caller +//! with no child that could ever fit gets ECHILD. The family answers, since a +//! wait by group needs every child's group, which only it can see. + +use nonos_libc::mk_foreign_reply; + +use super::family::Family; +use crate::linux::guest::Guest; + +/// Reply to a parked thread, or enter the handler of a signal it now takes. +pub fn answer(g: &mut Guest, tid: u32, value: u64) { + if !super::deliver::maybe_deliver(g, tid, value) { + let _ = mk_foreign_reply(tid, value); + } +} + +impl Family { + pub(super) fn settle_child_waits(&mut self) { + for i in 0..self.guests.len() { + for w in core::mem::take(&mut self.guests[i].signals.childwaits) { + match self.try_wait(i, &w) { + Some(v) => answer(&mut self.guests[i], w.tid, v), + None => self.guests[i].signals.childwaits.push(w), + } + } + } + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_wait_report.rs b/userland/capsule_linux/src/linux/serve/family_wait_report.rs new file mode 100644 index 000000000..2d8599184 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_wait_report.rs @@ -0,0 +1,65 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What a finished wait writes back: waitid's siginfo or wait4's status word, +//! and struct rusage. The kernel reports no CPU time for a guest, so rusage is +//! written as all zeros; Go always passes one, so it is still written, and the +//! log says once per family run that its zeros are not a measurement. + +use core::sync::atomic::{AtomicBool, Ordering}; + +use crate::linux::abi::errno; +use crate::linux::guest::siginfo::{SigInfo, CLD_EXITED, CLD_KILLED}; +use crate::linux::guest::sigstate::SIGCHLD; +use crate::linux::guest::sigwaits::ChildWait; +use crate::linux::guest::Guest; + +/// struct rusage: no CPU time is reported for a guest, so all of it is zero. +const RUSAGE_LEN: usize = 144; + +/// Set once the zero rusage has been said in the log. +static RUSAGE_SAID: AtomicBool = AtomicBool::new(false); + +/// Write what the caller asked for and give the value its call returns. +pub fn report(p: &Guest, w: &ChildWait, child: Option<(u32, i32)>, value: u64) -> u64 { + let wrote = match (w.waitid, child) { + (true, Some((pid, status))) => { + let (code, st) = match status & 0x7f { + 0 => (CLD_EXITED, (status >> 8) & 0xff), + s => (CLD_KILLED, s), + }; + let info = SigInfo { value: st as u64, ..SigInfo::from(SIGCHLD, code, pid) }; + w.out == 0 || p.write(w.out, &info.bytes()) == 128 + } + (true, None) => w.out == 0 || p.write(w.out, &[0u8; 128]) == 128, + (false, Some((_, status))) => w.out == 0 || p.write(w.out, &status.to_le_bytes()) == 4, + (false, None) => true, + }; + let usage = child.is_none() || w.rusage == 0 || zero_rusage(p, w.rusage); + match wrote && usage { + true => value, + false => errno::fail(errno::EFAULT), + } +} + +/// Write an all-zero rusage at `at`, saying so the first time in a family run. +fn zero_rusage(p: &Guest, at: u64) -> bool { + if !RUSAGE_SAID.swap(true, Ordering::Relaxed) { + let line = b"[LINUX] rusage: no CPU time is reported for a guest, written as zero\n"; + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + } + p.write(at, &[0u8; RUSAGE_LEN]) > 0 +} diff --git a/userland/capsule_linux/src/linux/serve/family_wait_try.rs b/userland/capsule_linux/src/linux/serve/family_wait_try.rs new file mode 100644 index 000000000..073caf39c --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_wait_try.rs @@ -0,0 +1,66 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! One parked wait4 or waitid tried against its caller's children: the first +//! ended child that fits is reported, or the call answers at once with none, +//! or it stays parked until a child ends. + +use super::family::Family; +use super::family_wait_report::report; +use crate::linux::abi::errno; +use crate::linux::call::{WALL, WCLONE, WEXITED, WNOHANG, WNOWAIT}; +use crate::linux::guest::sigwaits::{ChildWait, Which}; + +impl Family { + pub(super) fn try_wait(&mut self, i: usize, w: &ChildWait) -> Option { + let group_of = |pid: u32| { + let live = self.guests.iter().find(|g| g.pid == pid).map(|g| g.pgid); + live.or_else(|| { + self.guests[i].signals.kid_groups.iter().find(|k| k.0 == pid).map(|k| k.1) + }) + }; + let p = &self.guests[i]; + let fits = |pid: u32| { + let clone = p.signals.clone_kids.contains(&pid); + let kind = w.options & WALL != 0 || (w.options & WCLONE != 0) == clone; + kind && match w.which { + Which::Any => true, + Which::Pid(want) => pid == want, + Which::Group(g) => group_of(pid) == Some(g), + } + }; + let exits = w.options & WEXITED != 0; + let done = p.ended.iter().position(|(pid, _)| exits && fits(*pid)); + let any = p.children.iter().any(|c| fits(*c)); + let p = &mut self.guests[i]; + let Some(at) = done else { + return match (any, w.options & WNOHANG != 0) { + (false, _) => Some(report(p, w, None, errno::fail(errno::ECHILD))), + (true, true) => Some(report(p, w, None, 0)), + (true, false) => None, + }; + }; + let (pid, status) = p.ended[at]; + if w.options & WNOWAIT == 0 { + p.ended.remove(at); + p.children.retain(|c| *c != pid); + p.signals.clone_kids.retain(|c| *c != pid); + p.signals.kid_groups.retain(|k| k.0 != pid); + } + let shown = u64::from(crate::linux::serve::guest_pid(pid)); + Some(report(p, w, Some((pid, status)), if w.waitid { 0 } else { shown })) + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_waits.rs b/userland/capsule_linux/src/linux/serve/family_waits.rs index 735f19a44..6983f8be0 100644 --- a/userland/capsule_linux/src/linux/serve/family_waits.rs +++ b/userland/capsule_linux/src/linux/serve/family_waits.rs @@ -55,6 +55,7 @@ impl Family { } }; // A caught signal for this thread is delivered in place of the reply. + super::deliver_pipe::broken_pipe(g, wait.tid, wait.nr, wait.args, value); if !super::deliver::maybe_deliver(g, wait.tid, value) { let _ = mk_foreign_reply(wait.tid, value); } diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index 700f5b2dd..917245ecf 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -18,24 +18,43 @@ mod answer; mod deliver; +mod deliver_enter; +mod deliver_pipe; +mod deliver_interrupt; +mod deliver_rem; +mod deliver_restart; +mod deliver_say; +mod deliver_sigwait; +mod deliver_stack; +mod deliver_wait; mod dispatch; mod family; mod family_futex; mod family_lend; mod family_reap; +mod family_reap_end; +mod family_signal; +mod family_signal_fire; +mod family_signal_route; mod family_sleep; +mod family_wait; +mod family_wait_report; +mod family_wait_try; mod family_waits; mod loop_impl; mod pid_map; mod pid_ns; +mod pid_space; mod refused; mod pid_out; +mod route_life; mod table; mod table_file; mod table_link; mod table_mem; mod table_net; mod table_proc; +mod table_sig; mod tally; mod unserved; mod waits; @@ -44,3 +63,4 @@ mod waits_time; pub use answer::Answer; pub use loop_impl::serve; +pub use pid_space::{inward as kernel_pid, outward as guest_pid}; diff --git a/userland/capsule_linux/src/linux/serve/pid_map.rs b/userland/capsule_linux/src/linux/serve/pid_map.rs index 894b1cb81..b42464139 100644 --- a/userland/capsule_linux/src/linux/serve/pid_map.rs +++ b/userland/capsule_linux/src/linux/serve/pid_map.rs @@ -22,7 +22,7 @@ use nonos_libc::ForeignFrame; -use crate::linux::abi::{errno, nr, nr_path as np}; +use crate::linux::abi::{errno, nr, nr_path as np, nr_sig as ns}; use super::pid_ns::PidNs; @@ -41,9 +41,12 @@ pub fn frame_in(ns: &PidNs, frame: &ForeignFrame) -> Option { fn args_in(ns: &PidNs, call: u64, a: &mut [u64; 6]) -> Result<(), u64> { let (slots, missing): (&[usize], i64) = match call { nr::WAIT4 => (&[0], errno::ECHILD), + /* waitid names a pid or a group only for P_PID and P_PGID. */ + ns::WAITID if matches!(a[0], 1 | 2) && a[1] != 0 => (&[1], errno::ECHILD), + ns::RT_SIGQUEUEINFO => (&[0], errno::ESRCH), np::KILL | np::TKILL | np::GETPGID | np::GETSID => (&[0], errno::ESRCH), // The thread group, then the thread: both are numbers the guest was given. - np::TGKILL => (&[0, 1], errno::ESRCH), + np::TGKILL | ns::RT_TGSIGQUEUEINFO => (&[0, 1], errno::ESRCH), nr::SCHED_SETPARAM | nr::SCHED_GETPARAM | nr::SCHED_SETSCHEDULER diff --git a/userland/capsule_linux/src/linux/serve/pid_ns.rs b/userland/capsule_linux/src/linux/serve/pid_ns.rs index c3c9501f1..4f3ad801a 100644 --- a/userland/capsule_linux/src/linux/serve/pid_ns.rs +++ b/userland/capsule_linux/src/linux/serve/pid_ns.rs @@ -21,36 +21,30 @@ //! it sees these instead: the personality is 1, the program it started is 2, //! and each process or thread after takes the next number. None is reused //! while the family lives, so a stale number never reaches a newer process. +//! +//! A personality hosts one family, so the numbers are kept in one place and +//! any handler can write a guest's number into what it hands back: a siginfo +//! or a waitid answer carries a pid, not only a return value. -use alloc::vec::Vec; +use super::pid_space::{inward, outward, SPACE}; -pub struct PidNs { - map: Vec<(u32, u32)>, - next: u32, -} +/// The family's numbering. Made once, when the family is. +pub struct PidNs; impl PidNs { pub fn new(personality: u32, first: u32) -> Self { - PidNs { map: alloc::vec![(personality, 1), (first, 2)], next: 3 } + *SPACE.0.borrow_mut() = (alloc::vec![(personality, 1), (first, 2)], 3); + PidNs } /// The number a guest sees for kernel pid `k`, given on first sight. /// Zero once the space is spent, which no caller reads as a process. pub fn outward(&mut self, k: u32) -> u32 { - if let Some(&(_, g)) = self.map.iter().find(|(kp, _)| *kp == k) { - return g; - } - let Some(after) = self.next.checked_add(1) else { - return 0; - }; - let g = self.next; - self.next = after; - self.map.push((k, g)); - g + outward(k) } /// The kernel pid behind a guest's number, if it names one of this family. pub fn inward(&self, g: u32) -> Option { - self.map.iter().find(|(_, gp)| *gp == g).map(|(k, _)| *k) + inward(g) } } diff --git a/userland/capsule_linux/src/linux/serve/pid_space.rs b/userland/capsule_linux/src/linux/serve/pid_space.rs new file mode 100644 index 000000000..89ad08e2d --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/pid_space.rs @@ -0,0 +1,47 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a family's pid numbers are kept. A personality hosts one family, so +//! they live in one static any handler can reach without the family at hand. + +use alloc::vec::Vec; +use core::cell::RefCell; + +pub(super) struct Space(pub(super) RefCell<(Vec<(u32, u32)>, u32)>); +/* SAFETY: eK@nonos.systems - only the personality's single serve loop touches it. */ +unsafe impl Sync for Space {} +pub(super) static SPACE: Space = Space(RefCell::new((Vec::new(), 0))); + +/// `PidNs::inward`, for a handler with no family at hand. +pub fn inward(g: u32) -> Option { + let space = SPACE.0.borrow(); + space.0.iter().find(|(_, gp)| *gp == g).map(|(k, _)| *k) +} + +/// `PidNs::outward`, for a handler with no family at hand. +pub fn outward(k: u32) -> u32 { + let mut space = SPACE.0.borrow_mut(); + if let Some(&(_, g)) = space.0.iter().find(|(kp, _)| *kp == k) { + return g; + } + let g = space.1; + let Some(after) = g.checked_add(1) else { + return 0; + }; + space.1 = after; + space.0.push((k, g)); + g +} diff --git a/userland/capsule_linux/src/linux/serve/route_life.rs b/userland/capsule_linux/src/linux/serve/route_life.rs new file mode 100644 index 000000000..0a2b06b1e --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/route_life.rs @@ -0,0 +1,74 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The calls of process lifecycle and signals that can leave their caller +//! parked: a plain exit, a new process, a wait for a child or a signal, and a +//! signal sent where only the family can say whether anyone received it. +//! Asked before `dispatch`, so these are answered here whatever it holds. + +use nonos_libc::ForeignFrame; + +use super::answer::Answer; +use crate::linux::abi::errno; +use crate::linux::abi::{nr, nr_sig as ns}; +use crate::linux::call; +use crate::linux::guest::Guest; + +/// clone's CLONE_THREAD: without it, clone makes a process. +const CLONE_THREAD: u64 = 0x10000; + +/// One trap answered: here when it is one of these calls, else by dispatch. +pub fn answer(guest: &mut Guest, frame: &ForeignFrame) -> Answer { + match first(guest, frame) { + Some(got) => { + super::tally::call(); + got + } + None => super::dispatch::answer(guest, frame), + } +} + +fn first(guest: &mut Guest, frame: &ForeignFrame) -> Option { + let (a, tid) = (frame.args(), frame.pid); + Some(match frame.nr { + nr::EXIT => call::exit_one(guest, tid, a[0]), + nr::EXIT_GROUP => { + let _ = call::exit(guest, a[0]); + Answer::Park + } + nr::CLONE if a[0] & CLONE_THREAD == 0 => call::clone_process(guest, tid, a), + nr::VFORK => call::vfork(guest, tid), + nr::WAIT4 => call::wait4_usage(guest, a[0], a[1], a[2], a[3], tid), + ns::WAITID => call::waitid(guest, tid, a), + ns::KILL => call::kill_from(guest, tid, a[0], a[1]), + ns::TKILL => call::tgkill_from(guest, tid, 0, a[0], a[1]), + ns::TGKILL if (a[0] as i64) <= 0 => Answer::value(errno::fail(errno::EINVAL)), + ns::TGKILL => call::tgkill_from(guest, tid, a[0], a[1], a[2]), + ns::RT_SIGQUEUEINFO => call::rt_sigqueueinfo(guest, tid, a[0], a[1], a[2]), + ns::RT_TGSIGQUEUEINFO => call::rt_tgsigqueueinfo(guest, tid, a[0], a[1], a[2], a[3]), + ns::PAUSE => call::pause(guest, tid), + ns::SIGNALFD4 => Answer::value(call::signalfd4(guest, a[0], a[1], a[2], a[3])), + ns::SIGNALFD => Answer::value(call::signalfd4(guest, a[0], a[1], a[2], 0)), + nr::READ if signalfd(guest, a[0]) => call::signalfd_read(guest, tid, a[0], a[1], a[2]), + ns::RT_SIGSUSPEND => call::rt_sigsuspend(guest, tid, a[0], a[1]), + ns::RT_SIGTIMEDWAIT => call::rt_sigtimedwait(guest, tid, a[0], a[1], a[2], a[3]), + _ => return None, + }) +} + +fn signalfd(guest: &Guest, fd: u64) -> bool { + guest.fds.get(fd as usize).is_some_and(|f| f.kind == crate::linux::guest::Kind::Signal) +} diff --git a/userland/capsule_linux/src/linux/serve/table.rs b/userland/capsule_linux/src/linux/serve/table.rs index 160338d7f..52b67fe8c 100644 --- a/userland/capsule_linux/src/linux/serve/table.rs +++ b/userland/capsule_linux/src/linux/serve/table.rs @@ -25,6 +25,7 @@ use super::table_link::link_ops; use super::table_mem::mem_ops; use super::table_net::net_ops; use super::table_proc::proc_ops; +use super::table_sig::sig_ops; pub fn plain(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> u64 { if let Some(v) = file_ops(guest, tid, nr, a) { @@ -42,6 +43,9 @@ pub fn plain(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> u64 { if let Some(v) = proc_ops(guest, nr, a) { return v; } + if let Some(v) = sig_ops(guest, tid, nr, a) { + return v; + } rest(guest, tid, nr, a) } @@ -53,9 +57,6 @@ fn rest(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> u64 { np::GETRLIMIT => call::getrlimit(guest, a[0], a[1]), np::UMASK => call::umask(guest, a[0]), np::PRLIMIT64 => call::prlimit64(guest, a[1], a[2], a[3]), - nr::RT_SIGACTION => call::rt_sigaction(guest, a[0], a[1], a[2]), - nr::RT_SIGPROCMASK => call::rt_sigprocmask(guest, a[2]), - nr::SIGALTSTACK => call::sigaltstack(guest, a[1]), nr::RSEQ | nr::SET_ROBUST_LIST => errno::ok(0), nr::ARCH_PRCTL => call::arch_prctl(guest, tid, a[0], a[1]), nr::GETRANDOM => call::getrandom(guest, a[0], a[1], a[2]), diff --git a/userland/capsule_linux/src/linux/serve/table_sig.rs b/userland/capsule_linux/src/linux/serve/table_sig.rs new file mode 100644 index 000000000..640c966ad --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/table_sig.rs @@ -0,0 +1,40 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Signal dispositions, masks and stacks, and the timers that end in a +//! signal: the calls of that family that answer at once. + +use crate::linux::abi::{nr, nr_sig as ns}; +use crate::linux::call; +use crate::linux::guest::Guest; + +pub fn sig_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option { + Some(match nr { + nr::RT_SIGACTION => call::rt_sigaction(guest, a[0], a[1], a[2], a[3]), + nr::RT_SIGPROCMASK => call::rt_sigprocmask(guest, tid, a[0], a[1], a[2], a[3]), + nr::SIGALTSTACK => call::sigaltstack(guest, tid, a[0], a[1]), + ns::RT_SIGPENDING => call::rt_sigpending(guest, tid, a[0], a[1]), + ns::ALARM => call::alarm(guest, a[0]), + ns::SETITIMER => call::setitimer(guest, a[0], a[1], a[2]), + ns::GETITIMER => call::getitimer(guest, a[0], a[1]), + ns::TIMER_CREATE => call::timer_create(guest, a[0], a[1], a[2]), + ns::TIMER_SETTIME => call::timer_settime(guest, a[0], a[1], a[2], a[3]), + ns::TIMER_GETTIME => call::timer_gettime(guest, a[0], a[1]), + ns::TIMER_GETOVERRUN => call::timer_getoverrun(guest, a[0]), + ns::TIMER_DELETE => call::timer_delete(guest, a[0]), + _ => return None, + }) +} diff --git a/userland/capsule_linux_proofs/src/calls.rs b/userland/capsule_linux_proofs/src/calls.rs new file mode 100644 index 000000000..7fb933908 --- /dev/null +++ b/userland/capsule_linux_proofs/src/calls.rs @@ -0,0 +1,38 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The capsule's pure call code, mounted from its tree as it ships: the +//! signal frame built and read back, the arithmetic of the timers that end in +//! a signal, and the reading of exec's shebang line. None of it names the +//! capsule's crate, so it runs here without a guest. + +#[path = "../../capsule_linux/src/linux/call/sigframe.rs"] +pub mod sigframe; + +#[path = "../../capsule_linux/src/linux/call/sigframe_build.rs"] +pub mod sigframe_build; + +#[path = "../../capsule_linux/src/linux/call/sigframe_read.rs"] +pub mod sigframe_read; + +#[path = "../../capsule_linux/src/linux/guest/sigtimer.rs"] +pub mod sigtimer; + +#[path = "../../capsule_linux/src/linux/guest/sigtimer_rearm.rs"] +pub mod sigtimer_rearm; + +#[path = "../../capsule_linux/src/linux/call/spawn/exec_shebang.rs"] +pub mod exec_shebang; diff --git a/userland/capsule_linux_proofs/src/lib.rs b/userland/capsule_linux_proofs/src/lib.rs index caa4499bf..0bbaa500e 100644 --- a/userland/capsule_linux_proofs/src/lib.rs +++ b/userland/capsule_linux_proofs/src/lib.rs @@ -51,8 +51,10 @@ pub mod statbuf; #[path = "../../capsule_linux/src/linux/net/host_body.rs"] pub mod host_body; -// net.sockets' own reader for a connect-by-host body, mounted at the crate -// paths it names, so the capsule's encoder is held to the real parser. +/* + * net.sockets' own reader for a connect-by-host body, mounted at the crate + * paths it names, so the capsule's encoder is held to the real parser. + */ #[path = "../../capsule_net_sockets/src/protocol/errno.rs"] pub mod protocol; pub mod server; @@ -60,11 +62,8 @@ pub mod server; #[path = "../../capsule_linux/src/linux/net/route.rs"] pub mod route; -#[path = "../../capsule_linux/src/linux/call/sigframe.rs"] -pub mod sigframe; - -#[path = "../../capsule_linux/src/linux/call/spawn/exec_shebang.rs"] -pub mod exec_shebang; +pub mod calls; +pub use calls::{exec_shebang, sigframe, sigframe_build, sigframe_read, sigtimer}; #[cfg(test)] pub mod image; diff --git a/userland/capsule_linux_proofs/src/tests.rs b/userland/capsule_linux_proofs/src/tests.rs index a5860bdd6..c28d4a409 100644 --- a/userland/capsule_linux_proofs/src/tests.rs +++ b/userland/capsule_linux_proofs/src/tests.rs @@ -40,7 +40,10 @@ mod pacman_desc_tests; mod pacman_rsa_tests; mod resolve_tests; mod route_tests; +mod sigframe_layout_tests; +mod sigframe_mutation_tests; mod sigframe_tests; +mod sigtimer_tests; mod service; mod stack_words_tests; mod stat_tests; diff --git a/userland/capsule_linux_proofs/src/tests/mutation_tests.rs b/userland/capsule_linux_proofs/src/tests/mutation_tests.rs index 0b5254e0c..8d3477f42 100644 --- a/userland/capsule_linux_proofs/src/tests/mutation_tests.rs +++ b/userland/capsule_linux_proofs/src/tests/mutation_tests.rs @@ -28,7 +28,7 @@ use crate::install::unpacked::unpacked; use super::mutation::damage; -const ROUNDS: usize = 1500; +pub(super) const ROUNDS: usize = 1500; #[test] fn damaged_debs_never_panic() { @@ -64,19 +64,3 @@ fn damaged_indexes_never_panic() { } } } - -#[test] -fn a_damaged_signal_frame_never_panics_returning() { - use crate::sigframe::{build, returned}; - let mut s = 0x516E_A100u64; - let mut base = [0u64; 18]; - base[15] = 0x7fff_ff00_0000; - let (_, buf, _) = build(&base, 0x4000, 0x4008, 11, 0).expect("frame"); - for _ in 0..ROUNDS { - let mut v = buf.clone(); - damage(&mut s, &mut v); - // rt_sigreturn reads the ucontext at the guest's rsp: any bytes there. - let _ = returned(&v); - let _ = v.first().map(|_| returned(&v[v.len().min(8)..])); - } -} diff --git a/userland/capsule_linux_proofs/src/tests/sigframe_layout_tests.rs b/userland/capsule_linux_proofs/src/tests/sigframe_layout_tests.rs new file mode 100644 index 000000000..898fed910 --- /dev/null +++ b/userland/capsule_linux_proofs/src/tests/sigframe_layout_tests.rs @@ -0,0 +1,59 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The frame byte for byte against Linux's `struct rt_sigframe`, and where a +//! frame goes: below the alternate stack's top, or nowhere when the stack is +//! too low to hold one. A handler that reads its ucontext (Go's does, to +//! preempt) finds each field where Linux puts it. + +use super::sigframe_tests::{at, entry, regs, INFO, RSP}; +use crate::sigframe::SIGCONTEXT_OFF; +use crate::sigframe_build::build; + +#[test] +fn the_frame_is_linux_rt_sigframe_byte_for_byte() { + /* + * pretcode at 0; ucontext at 8 with uc_mcontext at +40, oldmask its 22nd + * word, uc_sigmask at +296; siginfo at 312. + */ + assert_eq!(SIGCONTEXT_OFF, 40); + let saved = regs(); + let (_, buf, _) = build(&saved, &entry(1, 0xca11, 3, 0xabcd)).expect("frame"); + assert_eq!(at(&buf, 0), 0xca11); + assert_eq!(at(&buf, 8 + 40), saved[0]); /* r8 */ + assert_eq!(at(&buf, 8 + 40 + 16 * 8), saved[16]); /* rip */ + assert_eq!(at(&buf, 8 + 40 + 21 * 8), 0xabcd); /* oldmask */ + assert_eq!(at(&buf, 8 + 296), 0xabcd); /* uc_sigmask */ + assert_eq!(&buf[312..440], &INFO[..]); + assert_eq!(buf[8 + 24], 2); /* uc_stack.ss_flags: SS_DISABLE */ +} + +#[test] +fn a_frame_for_the_alternate_stack_sits_below_its_top() { + let saved = regs(); + let mut e = entry(1, 2, 3, 0); + e.alt_top = Some(0x5000_0000); + let (frame, _, enter) = build(&saved, &e).expect("frame"); + assert!(frame < 0x5000_0000 && frame > 0x5000_0000 - 1024); + assert_eq!(enter[RSP], frame); +} + +#[test] +fn a_stack_too_low_to_hold_a_frame_is_refused() { + let mut low = regs(); + low[RSP] = 64; /* below the red zone plus a frame */ + assert!(build(&low, &entry(1, 2, 3, 0)).is_none()); +} diff --git a/userland/capsule_linux_proofs/src/tests/sigframe_mutation_tests.rs b/userland/capsule_linux_proofs/src/tests/sigframe_mutation_tests.rs new file mode 100644 index 000000000..24a8ef39f --- /dev/null +++ b/userland/capsule_linux_proofs/src/tests/sigframe_mutation_tests.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Seeded mutation of the signal frame a handler returns through: whatever +//! bytes sit where the guest's rsp points, rt_sigreturn's reader must return, +//! in a debug build, without a panic. + +use super::mutation::damage; +use super::mutation_tests::ROUNDS; +use crate::sigframe::Entry; +use crate::sigframe_build::build; +use crate::sigframe_read::returned; + +#[test] +fn a_damaged_signal_frame_never_panics_returning() { + let mut s = 0x516E_A100u64; + let mut base = [0u64; 18]; + base[15] = 0x7fff_ff00_0000; + let info = [0u8; 128]; + let e = Entry { + handler: 0x4000, + restorer: 0x4008, + signum: 11, + blocked: 0, + alt_top: None, + stack: [0, 2, 0], + info: &info, + }; + let (_, buf, _) = build(&base, &e).expect("frame"); + for _ in 0..ROUNDS { + let mut v = buf.clone(); + damage(&mut s, &mut v); + /* rt_sigreturn reads the ucontext at the guest's rsp: any bytes there. */ + let _ = returned(&v); + let _ = v.first().map(|_| returned(&v[v.len().min(8)..])); + } +} diff --git a/userland/capsule_linux_proofs/src/tests/sigframe_tests.rs b/userland/capsule_linux_proofs/src/tests/sigframe_tests.rs index aaa75bcfb..2cca213f5 100644 --- a/userland/capsule_linux_proofs/src/tests/sigframe_tests.rs +++ b/userland/capsule_linux_proofs/src/tests/sigframe_tests.rs @@ -17,33 +17,50 @@ //! The signal frame a handler enters through, and the frame it returns from, //! are the same layout read two ways: build one, then read the sigcontext //! back the way rt_sigreturn does, and the registers must be identical. This -//! is what lets a program's handler run and return to where it was. +//! is what lets a program's handler run and return to where it was. The +//! offsets are Linux's `struct rt_sigframe`, which a handler that reads its +//! ucontext (Go's does, to preempt) depends on byte for byte. -use crate::sigframe::{build, returned, SIGCONTEXT_OFF, WORDS}; +use crate::sigframe::{Entry, WORDS}; +use crate::sigframe_build::build; +use crate::sigframe_read::returned; -const RSP: usize = 15; +pub(super) const RSP: usize = 15; const RAX: usize = 13; -fn regs() -> [u64; WORDS] { +pub(super) fn regs() -> [u64; WORDS] { let mut r = [0u64; WORDS]; for (i, w) in r.iter_mut().enumerate() { - *w = 0x1111_0000 + i as u64; // a distinct value per register + *w = 0x1111_0000 + i as u64; /* a distinct value per register */ } - r[RSP] = 0x7fff_ffe0_0000; // a plausible stack pointer, page aligned + r[RSP] = 0x7fff_ffe0_0000; /* a plausible stack pointer, page aligned */ r } +pub(super) const INFO: [u8; 128] = [0x5a; 128]; + +pub(super) fn entry(handler: u64, restorer: u64, signum: u32, blocked: u64) -> Entry<'static> { + Entry { handler, restorer, signum, blocked, alt_top: None, stack: [0, 2, 0], info: &INFO } +} + +pub(super) fn at(buf: &[u8], off: usize) -> u64 { + u64::from_le_bytes(buf[off..off + 8].try_into().unwrap()) +} + #[test] fn a_returning_frame_restores_the_registers_the_handler_was_entered_over() { let saved = regs(); - let (frame, buf, enter) = build(&saved, 0xdead_beef, 0xca11, 11, 0x1234).expect("frame"); - // The handler is entered at the frame, below the old stack, 16-byte down 8. + let (frame, buf, enter) = + build(&saved, &entry(0xdead_beef, 0xca11, 11, 0x1234)).expect("frame"); + /* The handler is entered at the frame, below the old stack, 16-byte down 8. */ assert!(frame < saved[RSP] - 128); + assert_eq!(frame % 16, 8); assert_eq!(enter[RSP], frame); - assert_eq!(enter[16], 0xdead_beef); // rip = handler - assert_eq!(enter[8], 11); // rdi = signum - assert_eq!(enter[12], frame + 8); // rdx = &ucontext - // rt_sigreturn reads the ucontext the guest's rsp points at: frame + 8. + assert_eq!(enter[16], 0xdead_beef); /* rip = handler */ + assert_eq!(enter[8], 11); /* rdi = signum */ + assert_eq!(enter[9], frame + 312); /* rsi = &siginfo */ + assert_eq!(enter[12], frame + 8); /* rdx = &ucontext */ + /* rt_sigreturn reads the ucontext the guest's rsp points at: frame + 8. */ let uc = &buf[8..]; assert_eq!(returned(uc).unwrap(), saved); } @@ -51,25 +68,7 @@ fn a_returning_frame_restores_the_registers_the_handler_was_entered_over() { #[test] fn the_syscall_return_value_rides_in_the_saved_rax() { let mut saved = regs(); - saved[RAX] = 0; // as the thread trapped, before we set the reply - saved[RAX] = 42; // the value the interrupted syscall returns - let (_, buf, _) = build(&saved, 1, 2, 3, 0).expect("frame"); + saved[RAX] = 42; /* the value the interrupted syscall returns */ + let (_, buf, _) = build(&saved, &entry(1, 2, 3, 0)).expect("frame"); assert_eq!(returned(&buf[8..]).unwrap()[RAX], 42); } - -#[test] -fn a_stack_too_low_to_hold_a_frame_is_refused() { - let mut low = regs(); - low[RSP] = 64; // below the red zone plus a frame - assert!(build(&low, 1, 2, 3, 0).is_none()); -} - -#[test] -fn the_sigcontext_sits_where_the_ucontext_says() { - // uc_mcontext is at SIGCONTEXT_OFF within the ucontext, which is at frame+8. - let saved = regs(); - let (_, buf, _) = build(&saved, 1, 2, 3, 0).expect("frame"); - let at = 8 + SIGCONTEXT_OFF; - let r8 = u64::from_le_bytes(buf[at..at + 8].try_into().unwrap()); - assert_eq!(r8, saved[0]); -} diff --git a/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs b/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs new file mode 100644 index 000000000..4183ff3a3 --- /dev/null +++ b/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs @@ -0,0 +1,73 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The arithmetic of the timers that end in a signal: a periodic timer moves +//! past every period that ended while it was not looked at, a one-shot one +//! stops, and a POSIX timer counts the periods it missed as overruns, as +//! Linux's timer_getoverrun reports them. + +use crate::sigtimer::{Itimer, PosixTimer}; + +fn posix(due: u64, interval: u64) -> PosixTimer { + PosixTimer { + id: 0, + clock: 1, + signo: 34, + tid: 0, + value: 0, + due: Some(due), + interval, + overrun: 0, + last_overrun: 0, + queued: false, + } +} + +#[test] +fn a_periodic_itimer_moves_past_every_period_that_ended() { + let mut t = Itimer { due: 1000, interval: 200 }; + assert!(t.rearm(1450)); + assert_eq!(t.due, 1600); +} + +#[test] +fn a_one_shot_itimer_does_not_fire_again() { + let mut t = Itimer { due: 1000, interval: 0 }; + assert!(!t.rearm(1000)); +} + +#[test] +fn a_posix_timer_fired_late_counts_the_periods_it_missed() { + let mut t = posix(1000, 100); + assert!(t.rearm(1350)); + assert_eq!(t.overrun, 3); + assert_eq!(t.due, Some(1400)); +} + +#[test] +fn a_posix_timer_fired_on_time_counts_nothing() { + let mut t = posix(1000, 100); + assert!(t.rearm(1000)); + assert_eq!(t.overrun, 0); + assert_eq!(t.due, Some(1100)); +} + +#[test] +fn a_one_shot_posix_timer_disarms() { + let mut t = posix(1000, 0); + assert!(!t.rearm(1200)); + assert_eq!(t.due, None); +} diff --git a/userland/libc/src/foreign.rs b/userland/libc/src/foreign.rs index a50956d75..4bb9c833f 100644 --- a/userland/libc/src/foreign.rs +++ b/userland/libc/src/foreign.rs @@ -18,8 +18,7 @@ //! kernel refuses on its behalf. use crate::syscall::{ - call_raw, N_MK_FOREIGN_EXEC, N_MK_FOREIGN_FORK, N_MK_FOREIGN_REPLY, N_MK_FOREIGN_SPAWN, - N_MK_FOREIGN_START, + call_raw, N_MK_FOREIGN_EXEC, N_MK_FOREIGN_REPLY, N_MK_FOREIGN_SPAWN, N_MK_FOREIGN_START, N_MK_FOREIGN_THREAD, N_MK_FOREIGN_WAIT, }; @@ -42,12 +41,6 @@ pub fn mk_foreign_resume(pid: u32) -> i64 { call_raw(N_MK_FOREIGN_START, [pid as u64, 0, 0, 0, 0, 0]) } -/// A second process holding a guest's register state, with zero in its return -/// register. -pub fn mk_foreign_fork(pid: u32) -> i64 { - call_raw(N_MK_FOREIGN_FORK, [pid as u64, 0, 0, 0, 0, 0]) -} - /// Replace the program a parked guest is running. pub fn mk_foreign_exec(pid: u32, entry: u64, rsp: u64) -> i64 { call_raw(N_MK_FOREIGN_EXEC, [pid as u64, entry, rsp, 0, 0, 0]) diff --git a/userland/libc/src/foreign_fork.rs b/userland/libc/src/foreign_fork.rs new file mode 100644 index 000000000..93942203f --- /dev/null +++ b/userland/libc/src/foreign_fork.rs @@ -0,0 +1,32 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Forking a guest: a second process holding its register state, with zero +//! in its return register, on its parent's stack or one the caller names. + +use crate::syscall::{call_raw, N_MK_FOREIGN_FORK}; + +/// A second process holding a guest's register state, with zero in its return +/// register. +pub fn mk_foreign_fork(pid: u32) -> i64 { + mk_foreign_fork_at(pid, 0) +} + +/// A fork whose child starts on `rsp` rather than its parent's stack pointer, +/// as a clone that names a stack asks; zero keeps the parent's. +pub fn mk_foreign_fork_at(pid: u32, rsp: u64) -> i64 { + call_raw(N_MK_FOREIGN_FORK, [pid as u64, rsp, 0, 0, 0, 0]) +} diff --git a/userland/libc/src/lib.rs b/userland/libc/src/lib.rs index a1176472b..535b61f9a 100644 --- a/userland/libc/src/lib.rs +++ b/userland/libc/src/lib.rs @@ -27,6 +27,7 @@ pub mod capsule_verify; pub mod crypto; pub mod debug; pub mod foreign; +pub mod foreign_fork; pub mod foreign_frame; pub mod foreign_signal; pub mod graphics; @@ -82,9 +83,10 @@ pub use consent::{ mk_local_restore, }; pub use foreign::{ - mk_foreign_exec, mk_foreign_fork, mk_foreign_reply, mk_foreign_resume, mk_foreign_spawn, - mk_foreign_start, mk_foreign_thread, mk_foreign_wait, + mk_foreign_exec, mk_foreign_reply, mk_foreign_resume, mk_foreign_spawn, mk_foreign_start, + mk_foreign_thread, mk_foreign_wait, }; +pub use foreign_fork::{mk_foreign_fork, mk_foreign_fork_at}; pub use foreign_frame::{ForeignFrame, FOREIGN_NR_DIED}; pub use foreign_signal::{ mk_foreign_context, mk_foreign_signal, ForeignRegs, SIGNAL_DELIVER, SIGNAL_RETURN, diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index c42c8e873..9499ab0b1 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -54,10 +54,12 @@ nonos-mk-check-linux-guest-$(1)-keys: \ $(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_ed25519.pub \ $(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_mldsa65.pub LINUX_GUEST_STORE_DEPS += $$(linux-guest-$(1)_ARTIFACTS) $$(linux-guest-$(1)_ATTESTATION) -LINUX_GUEST_STORE_ENTRIES += --entry /linux$(or $(5),/bin/$(1))=$$(linux-guest-$(1)_BIN) \ +LINUX_GUEST_ENTRIES_$(1) := --entry /linux$(or $(5),/bin/$(1))=$$(linux-guest-$(1)_BIN) \ --entry /linux$(or $(5),/bin/$(1)).nonos_id_cert.bin=$$(linux-guest-$(1)_CERT) \ --entry /linux$(or $(5),/bin/$(1)).manifest.bin=$$(linux-guest-$(1)_MANIFEST) \ --entry /linux$(or $(5),/bin/$(1)).zk_trailer.bin=$$(linux-guest-$(1)_ATTESTATION) +# Every guest, or only those LINUX_GUEST_SET names: vfs loads 16 MiB at most. +LINUX_GUEST_STORE_ENTRIES += $$(if $$(filter $(1),$$(or $$(LINUX_GUEST_SET),$(1))),$$(LINUX_GUEST_ENTRIES_$(1))) endef $(eval $(call LINUX_GUEST,suite,4950,4951)) @@ -113,6 +115,8 @@ $(LINUX_GUESTS_C)/cthreads: $(LINUX_GUESTS_DIR)/c/cthreads.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cthreads,4974,4975,$(LINUX_GUESTS_C)/cthreads)) +# Process lifecycle and signals, each against Linux; see LifeGuests.mk. +include $(LINUX_GUESTS_DIR)/LifeGuests.mk # Waiting as Linux waits: futex timeouts and requeue, eventfd, epoll_wait's # timeout and wake, a non-blocking pipe, a full pipe, and edge-triggered epoll. $(LINUX_GUESTS_C)/cwait: $(LINUX_GUESTS_DIR)/c/cwait.c diff --git a/userland/linux_guests/LifeGuests.mk b/userland/linux_guests/LifeGuests.mk new file mode 100644 index 000000000..69917cf01 --- /dev/null +++ b/userland/linux_guests/LifeGuests.mk @@ -0,0 +1,20 @@ +# Guests for process lifecycle and signals. Included by Guests.mk. +# +# Process lifecycle and signals, each against Linux: a leader's plain exit +# that leaves its worker running, os/exec from Go through clone(CLONE_VFORK), +# SIGCHLD with waitid and wait4, SIGPIPE on a widowed pipe, and SIGALRM ending +# a sleep, with the timer and signal-wait calls. +$(LINUX_GUESTS_C)/leaderexit: $(LINUX_GUESTS_DIR)/c/leaderexit.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,leaderexit,4990,4991,$(LINUX_GUESTS_C)/leaderexit)) +$(GO_OUT)/exec: $(addprefix $(LINUX_GUESTS_DIR)/go/exec/,forkexec.go devices.go) +$(eval $(call LINUX_GUEST,goexec,4992,4993,$(GO_OUT)/exec)) +$(LINUX_GUESTS_C)/sigchld: $(addprefix $(LINUX_GUESTS_DIR)/c/,sigchld.c sigchld_wait.c sigchld_spawn.c sigchld.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $(filter %.c,$^) +$(eval $(call LINUX_GUEST,sigchld,4994,4995,$(LINUX_GUESTS_C)/sigchld)) +$(LINUX_GUESTS_C)/sigpipe: $(LINUX_GUESTS_DIR)/c/sigpipe.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,sigpipe,4996,4997,$(LINUX_GUESTS_C)/sigpipe)) +$(LINUX_GUESTS_C)/alarm: $(addprefix $(LINUX_GUESTS_DIR)/c/,alarm.c alarm_wait.c alarm_timer.c alarm_sigfd.c alarm.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $(filter %.c,$^) +$(eval $(call LINUX_GUEST,alarm,4998,4999,$(LINUX_GUESTS_C)/alarm)) diff --git a/userland/linux_guests/c/alarm.c b/userland/linux_guests/c/alarm.c new file mode 100644 index 000000000..665491878 --- /dev/null +++ b/userland/linux_guests/c/alarm.c @@ -0,0 +1,75 @@ +/* + * Signals that arrive while a thread waits, and the calls that wait for them. + * alarm(1) then sleep(10): on Linux SIGALRM ends the sleep early with its + * handler run, well under two seconds in. Then setitimer and getitimer with + * pause; the signal waits, POSIX timers and signalfd follow in their own + * files. Every part prints its line. + */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include "alarm.h" +volatile sig_atomic_t alarms; +static int failed, parts; + +static void on_alrm(int sig) { (void)sig; alarms++; } + +void part(int ok, const char *what, long n) { + printf("[C] alarm %s: %s (%ld)\n", ok ? "ok" : "FAIL", what, n); + fflush(stdout); + failed += !ok; + parts++; +} + +long ms_since(struct timespec *t0) { + struct timespec t; + clock_gettime(CLOCK_MONOTONIC, &t); + return (t.tv_sec - t0->tv_sec) * 1000 + (t.tv_nsec - t0->tv_nsec) / 1000000; +} + +void catch(int sig, void (*fn)(int)) { + struct sigaction sa; + memset(&sa, 0, sizeof sa); + sa.sa_handler = fn; + sigaction(sig, &sa, 0); +} + +int verdict(void) { + printf("[C] alarm %s: %d of %d parts held\n", failed ? "FAIL" : "PASS", parts - failed, parts); + fflush(stdout); + return failed != 0; +} + +int main(void) { + struct timespec t0; + catch(SIGALRM, on_alrm); + clock_gettime(CLOCK_MONOTONIC, &t0); + alarm(1); + unsigned left = sleep(10); + long ms = ms_since(&t0); + part(alarms == 1 && ms < 2000 && left >= 8, "alarm(1) ended sleep(10) early, ms", ms); + struct itimerval it = {{0, 200000}, {0, 200000}}, now; + alarms = 0; + clock_gettime(CLOCK_MONOTONIC, &t0); + setitimer(ITIMER_REAL, &it, 0); + /* Bounded: where pause fails at once instead of waiting, this still ends. */ + for (int i = 0; alarms < 3 && i < 50; i++) { + pause(); + } + ms = ms_since(&t0); + getitimer(ITIMER_REAL, &now); + part(ms >= 550 && ms < 1500 && now.it_interval.tv_usec == 200000 && + now.it_value.tv_usec <= 200000, + "setitimer 200 ms interval: three SIGALRMs through pause, ms", ms); + memset(&it, 0, sizeof it); + setitimer(ITIMER_REAL, &it, &now); + getitimer(ITIMER_REAL, &now); + part(now.it_value.tv_sec == 0 && now.it_value.tv_usec == 0, "a disarmed timer reads zero", 0); + part(alarm(0) == 0, "alarm(0) with nothing armed answers 0", 0); + signal_waits(); + posix_timer(); + signal_fd(); + return verdict(); +} diff --git a/userland/linux_guests/c/alarm.h b/userland/linux_guests/c/alarm.h new file mode 100644 index 000000000..75fd94886 --- /dev/null +++ b/userland/linux_guests/c/alarm.h @@ -0,0 +1,17 @@ +/* + * What the alarm guest's two halves share: its SIGALRM and SIGUSR1 counts, + * what the SIGUSR1 handler read from its ucontext, and the part line every + * check prints. + */ +#include +#include + +extern volatile sig_atomic_t alarms, usr1, usr1_rip_in_text, usr1_saved_blocked; + +void part(int ok, const char *what, long n); +long ms_since(struct timespec *t0); +void catch(int sig, void (*fn)(int)); +void signal_waits(void); +void posix_timer(void); +void signal_fd(void); +int verdict(void); diff --git a/userland/linux_guests/c/alarm_sigfd.c b/userland/linux_guests/c/alarm_sigfd.c new file mode 100644 index 000000000..15a6f78a1 --- /dev/null +++ b/userland/linux_guests/c/alarm_sigfd.c @@ -0,0 +1,44 @@ +/* + * signalfd, checked against Linux: a blocked SIGUSR1 raised at the process + * reads back from a signalfd as a record with its number and the sender's + * pid, a non-blocking signalfd with nothing pending answers EAGAIN, poll + * reports it readable once a signal waits, and a sigqueue value arrives in + * ssi_int through a blocking read. + */ +#define _GNU_SOURCE +#include +#include +#include +#include + +#include "alarm.h" + +void signal_fd(void) { + sigset_t set; + sigemptyset(&set); + sigaddset(&set, SIGUSR1); + sigprocmask(SIG_BLOCK, &set, 0); + int fd = signalfd(-1, &set, SFD_NONBLOCK | SFD_CLOEXEC); + struct signalfd_siginfo si; + errno = 0; + ssize_t n = read(fd, &si, sizeof si); + part(fd >= 0 && n == -1 && errno == EAGAIN, "signalfd with nothing pending: EAGAIN", fd); + struct pollfd p = {fd, POLLIN, 0}; + int before = poll(&p, 1, 0); + kill(getpid(), SIGUSR1); + int after = poll(&p, 1, 0); + part(before == 0 && after == 1 && (p.revents & POLLIN), "poll: readable once SIGUSR1 waits", + after); + n = read(fd, &si, sizeof si); + part(n == sizeof si && si.ssi_signo == SIGUSR1 && si.ssi_pid == (unsigned)getpid(), + "signalfd reads SIGUSR1 with the sender's pid", si.ssi_signo); + close(fd); + sigaddset(&set, SIGUSR2); + int blocking = signalfd(-1, &set, 0); + union sigval v = {.sival_int = 4321}; + sigqueue(getpid(), SIGUSR2, v); + n = read(blocking, &si, sizeof si); + part(n == sizeof si && si.ssi_signo == SIGUSR2 && si.ssi_int == 4321, + "a blocking signalfd read takes sigqueue's value", si.ssi_int); + close(blocking); +} diff --git a/userland/linux_guests/c/alarm_timer.c b/userland/linux_guests/c/alarm_timer.c new file mode 100644 index 000000000..13b5954dd --- /dev/null +++ b/userland/linux_guests/c/alarm_timer.c @@ -0,0 +1,40 @@ +/* + * A POSIX timer, checked against Linux: timer_create on the monotonic clock + * with SIGEV_SIGNAL, armed for 100 ms with a 100 ms period, gives three + * SI_TIMER signals carrying its value through sigtimedwait; timer_gettime + * reads its period back, timer_getoverrun answers, timer_delete ends it. + */ +#define _GNU_SOURCE +#include + +#include "alarm.h" + +void posix_timer(void) { + sigset_t rt; + sigemptyset(&rt); + sigaddset(&rt, SIGRTMIN); + sigprocmask(SIG_BLOCK, &rt, 0); + struct sigevent ev; + memset(&ev, 0, sizeof ev); + ev.sigev_notify = SIGEV_SIGNAL; + ev.sigev_signo = SIGRTMIN; + ev.sigev_value.sival_int = 77; + timer_t id; + int rc = timer_create(CLOCK_MONOTONIC, &ev, &id); + struct itimerspec ts = {{0, 100000000}, {0, 100000000}}, got; + timer_settime(id, 0, &ts, 0); + siginfo_t si; + struct timespec wait = {1, 0}; + int fired = 0; + for (int i = 0; i < 3; i++) { + if (sigtimedwait(&rt, &si, &wait) == SIGRTMIN && si.si_code == SI_TIMER && + si.si_value.sival_int == 77) { + fired++; + } + } + timer_gettime(id, &got); + int over = timer_getoverrun(id); + part(rc == 0 && fired == 3 && got.it_interval.tv_nsec == 100000000 && over >= 0, + "timer_create 100 ms: three SI_TIMER signals, value 77", fired); + part(timer_delete(id) == 0, "timer_delete", 0); +} diff --git a/userland/linux_guests/c/alarm_wait.c b/userland/linux_guests/c/alarm_wait.c new file mode 100644 index 000000000..087339ac9 --- /dev/null +++ b/userland/linux_guests/c/alarm_wait.c @@ -0,0 +1,70 @@ +/* + * The signal waits, checked against Linux: a blocked SIGUSR1 shows in + * sigpending, sigsuspend runs its handler and answers EINTR with the mask put + * back, sigqueue hands a value to sigtimedwait, and a sigtimedwait with + * nothing pending times out. The SIGUSR1 handler reads its own ucontext: the + * interrupted rip must lie in this program's text and the saved mask must be + * the one sigsuspend replaced, which holds only at Linux's offsets. + */ +#define _GNU_SOURCE +#include +#include +#include +#include + +#include "alarm.h" + +extern char __executable_start[], etext[]; +volatile sig_atomic_t usr1, usr1_rip_in_text, usr1_saved_blocked; + +static void on_usr1(int sig, siginfo_t *info, void *ctx) { + (void)sig; + (void)info; + ucontext_t *uc = ctx; + unsigned long rip = (unsigned long)uc->uc_mcontext.gregs[REG_RIP]; + usr1_rip_in_text = rip >= (unsigned long)__executable_start && rip < (unsigned long)etext; + usr1_saved_blocked = sigismember(&uc->uc_sigmask, SIGUSR1); + usr1++; +} + +void signal_waits(void) { + struct sigaction sa; + memset(&sa, 0, sizeof sa); + sa.sa_sigaction = on_usr1; + sa.sa_flags = SA_SIGINFO; + sigaction(SIGUSR1, &sa, 0); + sigset_t block, none, pending, after, two; + sigemptyset(&block); + sigaddset(&block, SIGUSR1); + sigprocmask(SIG_BLOCK, &block, 0); + raise(SIGUSR1); + sigpending(&pending); + part(usr1 == 0 && sigismember(&pending, SIGUSR1), "a blocked SIGUSR1 waits, pending", 0); + sigemptyset(&none); + errno = 0; + int rc = sigsuspend(&none), e = errno; + sigprocmask(SIG_BLOCK, 0, &after); + part(rc == -1 && e == EINTR && usr1 == 1 && sigismember(&after, SIGUSR1), + "sigsuspend ran the handler, EINTR, mask restored", usr1); + part(usr1_rip_in_text && usr1_saved_blocked, + "the handler's ucontext: rip in the program's text, uc_sigmask the saved mask", 0); + + sigemptyset(&two); + sigaddset(&two, SIGUSR2); + sigprocmask(SIG_BLOCK, &two, 0); + union sigval v = {.sival_int = 1234}; + sigqueue(getpid(), SIGUSR2, v); + siginfo_t si; + struct timespec wait = {1, 0}, brief = {0, 100000000}, t0; + rc = sigtimedwait(&two, &si, &wait); + part(rc == SIGUSR2 && si.si_code == SI_QUEUE && si.si_value.sival_int == 1234 && + si.si_pid == getpid(), + "sigqueue into sigtimedwait: SI_QUEUE, value", si.si_value.sival_int); + clock_gettime(CLOCK_MONOTONIC, &t0); + errno = 0; + rc = sigtimedwait(&two, &si, &brief); + e = errno; + long ms = ms_since(&t0); + part(rc == -1 && e == EAGAIN && ms >= 90 && ms < 1000, "sigtimedwait timed out: EAGAIN, ms", + ms); +} diff --git a/userland/linux_guests/c/leaderexit.c b/userland/linux_guests/c/leaderexit.c new file mode 100644 index 000000000..c59e8ad33 --- /dev/null +++ b/userland/linux_guests/c/leaderexit.c @@ -0,0 +1,36 @@ +/* + * The leader ends with a plain exit, not exit_group. On Linux that ends only + * the leader: the process lives on in its other threads, and its status is + * the exit code of the last thread to end. The worker prints its line a second + * after the leader has gone and ends with 42, so a log with the PASS line and + * a status of 42 is Linux's behaviour; a process that ended at the leader's + * exit never prints the line and reports 0. + */ +#include +#include +#include +#include +#include + +static void *work(void *arg) { + (void)arg; + struct timespec t = {1, 0}; + nanosleep(&t, 0); + printf("[C] leaderexit PASS: worker ran 1 s after the leader's exit\n"); + fflush(stdout); + syscall(SYS_exit, 42); + return 0; +} + +int main(void) { + pthread_t t; + if (pthread_create(&t, 0, work, 0) != 0) { + printf("[C] leaderexit FAIL: create\n"); + fflush(stdout); + return 1; + } + printf("[C] leaderexit: leader exits, worker runs on\n"); + fflush(stdout); + syscall(SYS_exit, 0); + return 1; +} diff --git a/userland/linux_guests/c/sigchld.c b/userland/linux_guests/c/sigchld.c new file mode 100644 index 000000000..ed98cae08 --- /dev/null +++ b/userland/linux_guests/c/sigchld.c @@ -0,0 +1,72 @@ +/* + * A parent told about its children as Linux tells it. A child's exit raises + * SIGCHLD at a parent that catches it, with the child's pid, CLD_EXITED and + * its code in the siginfo; waitid(P_ALL, WEXITED) then reports the same + * child. The wait4 and waitid checks follow in sigchld_wait.c and musl's ways + * to start a program in sigchld_spawn.c. Every part runs and prints its line; + * the last line is PASS only if all of them held. + */ +#include +#include +#include +#include +#include +#include +#include "sigchld.h" + +static volatile sig_atomic_t got, got_code, got_status; +static volatile pid_t got_pid; +static int failed, parts; + +static void on_chld(int sig, siginfo_t *info, void *uc) { + (void)uc; + got = sig; + got_pid = info->si_pid; + got_code = info->si_code; + got_status = info->si_status; +} + +void part(int ok, const char *what) { + printf("[C] sigchld %s: %s\n", ok ? "ok" : "FAIL", what); + fflush(stdout); + failed += !ok; + parts++; +} + +pid_t child_exiting(int code, unsigned delay_ms) { + pid_t pid = fork(); + if (pid == 0) { + struct timespec t = {delay_ms / 1000, (delay_ms % 1000) * 1000000L}; + nanosleep(&t, 0); + _exit(code); + } + return pid; +} + +int main(void) { + struct sigaction sa; + memset(&sa, 0, sizeof sa); + sa.sa_sigaction = on_chld; + sa.sa_flags = SA_SIGINFO | SA_RESTART; + sigaction(SIGCHLD, &sa, 0); + pid_t a = child_exiting(7, 0); + /* Sleep in short steps until the handler has run or three seconds pass. */ + for (int i = 0; i < 300 && !got; i++) { + struct timespec t = {0, 10 * 1000 * 1000}; + nanosleep(&t, 0); + } + part(got == SIGCHLD && got_pid == a && got_code == CLD_EXITED && got_status == 7, + "SIGCHLD caught with the child's pid, CLD_EXITED and status 7"); + siginfo_t si; + memset(&si, 0, sizeof si); + int rc = waitid(P_ALL, 0, &si, WEXITED); + part(rc == 0 && si.si_pid == a && si.si_signo == SIGCHLD && si.si_code == CLD_EXITED && + si.si_status == 7, + "waitid(P_ALL, WEXITED) reports the child, CLD_EXITED, status 7"); + waits(); + spawns(); + printf("[C] sigchld %s: %d of %d parts held\n", failed ? "FAIL" : "PASS", parts - failed, + parts); + fflush(stdout); + return failed != 0; +} diff --git a/userland/linux_guests/c/sigchld.h b/userland/linux_guests/c/sigchld.h new file mode 100644 index 000000000..3c2a8b72a --- /dev/null +++ b/userland/linux_guests/c/sigchld.h @@ -0,0 +1,10 @@ +/* + * What the sigchld guest's files share: the part line every check prints, + * a child that exits after a delay, and the checks each file holds. + */ +#include + +void part(int ok, const char *what); +pid_t child_exiting(int code, unsigned delay_ms); +void waits(void); +void spawns(void); diff --git a/userland/linux_guests/c/sigchld_spawn.c b/userland/linux_guests/c/sigchld_spawn.c new file mode 100644 index 000000000..24b070be0 --- /dev/null +++ b/userland/linux_guests/c/sigchld_spawn.c @@ -0,0 +1,39 @@ +/* + * musl's own ways to start a program, which clone a vfork child onto a stack + * of its own: posix_spawn, system through the shell, and popen reading the + * child's output through a pipe. Then, with every child reaped, wait4 and + * waitid answer ECHILD. + */ +#include +#include +#include +#include +#include +#include +#include +#include "sigchld.h" + +extern char **environ; + +void spawns(void) { + pid_t d = -1; + int st = -1; + char *argv[] = {"/bin/cthreads", 0}; + int rc = posix_spawn(&d, "/bin/cthreads", 0, 0, argv, environ); + part(rc == 0 && waitpid(d, &st, 0) == d && WIFEXITED(st) && WEXITSTATUS(st) == 0, + "posix_spawn runs cthreads, status 0"); + st = system("/bin/leaderexit"); + part(WIFEXITED(st) && WEXITSTATUS(st) == 42, "system runs leaderexit through sh, status 42"); + char line[128] = {0}; + FILE *f = popen("/bin/cthreads", "r"); + int got_line = f && fgets(line, sizeof line, f) != 0; + int closed = f ? pclose(f) : -1; + part(got_line && strstr(line, "cthreads PASS") && WIFEXITED(closed) && WEXITSTATUS(closed) == 0, + "popen reads cthreads' PASS line through a pipe"); + siginfo_t si; + errno = 0; + part(wait4(-1, &st, 0, 0) == -1 && errno == ECHILD, "wait4 with no child left: ECHILD"); + errno = 0; + part(waitid(P_ALL, 0, &si, WEXITED) == -1 && errno == ECHILD, + "waitid with no child left: ECHILD"); +} diff --git a/userland/linux_guests/c/sigchld_wait.c b/userland/linux_guests/c/sigchld_wait.c new file mode 100644 index 000000000..93ec344d6 --- /dev/null +++ b/userland/linux_guests/c/sigchld_wait.c @@ -0,0 +1,39 @@ +/* + * wait4 and waitid against Linux: WNOHANG on a running child, waitid WNOWAIT + * looking without reaping, WUNTRACED, the status word of an exit and of a + * death by signal, and EINVAL for an option Linux does not know. + */ +#include +#include +#include +#include +#include +#include "sigchld.h" + +void waits(void) { + siginfo_t si; + pid_t b = child_exiting(3, 300); + int st = -1; + part(wait4(b, &st, WNOHANG, 0) == 0, "wait4 WNOHANG on a running child answers 0"); + memset(&si, 0, sizeof si); + si.si_pid = 1; + int rc = waitid(P_PID, b, &si, WEXITED | WNOHANG); + part(rc == 0 && si.si_pid == 0, "waitid WNOHANG on a running child answers 0, si_pid 0"); + memset(&si, 0, sizeof si); + rc = waitid(P_PID, b, &si, WEXITED | WNOWAIT); + part(rc == 0 && si.si_pid == b && si.si_status == 3, "waitid WNOWAIT reports and leaves it"); + rc = wait4(b, &st, WUNTRACED, 0); + part(rc == b && WIFEXITED(st) && WEXITSTATUS(st) == 3, + "wait4 WUNTRACED then reaps it: WIFEXITED, status 3"); + errno = 0; + part(wait4(-1, &st, 0x100, 0) == -1 && errno == EINVAL, "wait4 with an unknown option: EINVAL"); + pid_t c = fork(); + if (c == 0) { + signal(SIGTERM, SIG_DFL); + kill(getpid(), SIGTERM); + for (;;) pause(); + } + rc = wait4(c, &st, 0, 0); + part(rc == c && WIFSIGNALED(st) && WTERMSIG(st) == SIGTERM && !WIFEXITED(st), + "a child ended by SIGTERM: WIFSIGNALED, WTERMSIG 15"); +} diff --git a/userland/linux_guests/c/sigpipe.c b/userland/linux_guests/c/sigpipe.c new file mode 100644 index 000000000..020dc3238 --- /dev/null +++ b/userland/linux_guests/c/sigpipe.c @@ -0,0 +1,73 @@ +/* + * A write to a pipe no one can read raises SIGPIPE on the writing thread, as + * Linux does. Caught, the handler runs and the write returns EPIPE; ignored, + * the write returns EPIPE and nothing else happens; left at its default, it + * ends the process, which a parent sees as a death by signal 13. + */ +#include +#include +#include +#include +#include +#include + +static volatile sig_atomic_t caught; +static int failed, parts; + +static void on_pipe(int sig) { caught = sig; } + +static void part(int ok, const char *what) { + printf("[C] sigpipe %s: %s\n", ok ? "ok" : "FAIL", what); + fflush(stdout); + failed += !ok; + parts++; +} + +/* A pipe whose read end is already closed; the write end is returned. */ +static int widowed(void) { + int p[2]; + if (pipe(p) != 0) { + return -1; + } + close(p[0]); + return p[1]; +} + +int main(void) { + struct sigaction sa; + memset(&sa, 0, sizeof sa); + sa.sa_handler = on_pipe; + sigaction(SIGPIPE, &sa, 0); + int w = widowed(); + errno = 0; + ssize_t n = write(w, "x", 1); + int e = errno; + part(caught == SIGPIPE && n == -1 && e == EPIPE, "caught: the handler ran, write gave EPIPE"); + close(w); + + signal(SIGPIPE, SIG_IGN); + caught = 0; + w = widowed(); + errno = 0; + n = write(w, "x", 1); + e = errno; + part(caught == 0 && n == -1 && e == EPIPE, "ignored: write gave EPIPE, no handler"); + close(w); + + pid_t c = fork(); + if (c == 0) { + signal(SIGPIPE, SIG_DFL); + int cw = widowed(); + write(cw, "x", 1); + _exit(0); + } + int st = 0; + pid_t got = waitpid(c, &st, 0); + part(got == c && WIFSIGNALED(st) && WTERMSIG(st) == SIGPIPE, + "default: the writer died of signal 13"); + + printf("[C] sigpipe %s: %d of %d parts held\n", failed ? "FAIL" : "PASS", parts - failed, + parts); + fflush(stdout); + return failed != 0; +} diff --git a/userland/linux_guests/go/exec/devices.go b/userland/linux_guests/go/exec/devices.go new file mode 100644 index 000000000..45c1bcf0e --- /dev/null +++ b/userland/linux_guests/go/exec/devices.go @@ -0,0 +1,33 @@ +/* + * The character devices, as Go's os package reaches them: /dev/null stats as + * a character device and reads end of file, /dev/zero reads zeros, and a + * write to /dev/full fails with ENOSPC. + */ +package main + +import ( + "errors" + "os" + "syscall" +) + +func deviceParts() { + st, err := os.Stat("/dev/null") + part(err == nil && st.Mode()&os.ModeCharDevice != 0, "/dev/null stats as a character device") + null, err := os.ReadFile("/dev/null") + part(err == nil && len(null) == 0, "/dev/null reads end of file") + z, err := os.Open("/dev/zero") + buf := make([]byte, 4096) + n := 0 + if err == nil { + n, err = z.Read(buf) + z.Close() + } + zeros := n == len(buf) + for _, b := range buf { + zeros = zeros && b == 0 + } + part(err == nil && zeros, "/dev/zero reads 4096 zeros") + err = os.WriteFile("/dev/full", []byte("x"), 0) + part(errors.Is(err, syscall.ENOSPC), "a write to /dev/full fails with ENOSPC") +} diff --git a/userland/linux_guests/go/exec/forkexec.go b/userland/linux_guests/go/exec/forkexec.go new file mode 100644 index 000000000..91da4afe9 --- /dev/null +++ b/userland/linux_guests/go/exec/forkexec.go @@ -0,0 +1,39 @@ +/* + * The syscall.ForkExec half of goexec: Go's own clone(CLONE_VFORK|CLONE_VM| + * SIGCHLD) and execve, with the child's exec error sent back through a pipe, + * then Wait4 for each child's status. Nothing here needs the runtime's poller. + */ +package main + +import ( + "errors" + "fmt" + "syscall" +) + +/* run starts path with syscall.ForkExec and waits for it with Wait4. */ +func run(path string) (syscall.WaitStatus, error) { + attr := &syscall.ProcAttr{Files: []uintptr{0, 1, 2}} + pid, err := syscall.ForkExec(path, []string{path}, attr) + if err != nil { + return 0, err + } + var ws syscall.WaitStatus + got, err := syscall.Wait4(pid, &ws, 0, nil) + if err == nil && got != pid { + err = fmt.Errorf("wait4 answered %d for child %d", got, pid) + } + return ws, err +} + +func forkExecParts() { + ws, err := run("/bin/cthreads") + part(err == nil && ws.Exited() && ws.ExitStatus() == 0, + fmt.Sprintf("ForkExec cthreads: status %d, err %v", ws.ExitStatus(), err)) + ws, err = run("/bin/leaderexit") + part(err == nil && ws.Exited() && ws.ExitStatus() == 42, + fmt.Sprintf("ForkExec leaderexit: status %d, err %v", ws.ExitStatus(), err)) + _, err = run("/bin/no-such-program") + part(errors.Is(err, syscall.ENOENT), fmt.Sprintf("ForkExec a missing program: %v", err)) + fmt.Printf("[GO] goexec: ForkExec parts done, %d of %d held\n", parts-failed, parts) +} diff --git a/userland/linux_guests/go/exec/go.mod b/userland/linux_guests/go/exec/go.mod new file mode 100644 index 000000000..04b7c7215 --- /dev/null +++ b/userland/linux_guests/go/exec/go.mod @@ -0,0 +1,3 @@ +module nonos/guest/exec + +go 1.24 diff --git a/userland/linux_guests/go/exec/main.go b/userland/linux_guests/go/exec/main.go new file mode 100644 index 000000000..915d42c4e --- /dev/null +++ b/userland/linux_guests/go/exec/main.go @@ -0,0 +1,66 @@ +/* + * Starting programs from Go. syscall.ForkExec is Go's own clone(CLONE_VFORK| + * CLONE_VM|SIGCHLD) and execve, with the child's exec error sent back through + * a pipe; Wait4 then reads each child's status. The children are the pthreads + * guest, a program that ends with status 42, and a path that does not exist. + * Then os/exec does the same through cmd.Output, which also needs the + * runtime's poller (eventfd, epoll) for its pipes, and opens /dev/null for + * every stream left nil; those parts run last, after the devices' own checks. + */ +package main + +import ( + "errors" + "fmt" + "io/fs" + "os" + "os/exec" + "strings" + "syscall" +) + +var failed, parts int + +func part(ok bool, what string) { + mark := "ok" + if !ok { + mark = "FAIL" + failed++ + } + parts++ + fmt.Printf("[GO] goexec %s: %s\n", mark, what) +} + +func main() { + forkExecParts() + deviceParts() + + cmd := exec.Command("/bin/cthreads") + out, err := cmd.Output() + line := strings.TrimSpace(string(out)) + code := -1 + if cmd.ProcessState != nil { + code = cmd.ProcessState.ExitCode() + } + part(err == nil && code == 0 && strings.Contains(line, "cthreads PASS"), + fmt.Sprintf("os/exec cthreads Output: status %d, err %v, output %q", code, err, line)) + + err = exec.Command("/bin/leaderexit").Run() + var ee *exec.ExitError + part(errors.As(err, &ee) && ee.ExitCode() == 42, + fmt.Sprintf("os/exec leaderexit's status came back: %v", err)) + + err = exec.Command("/bin/no-such-program").Run() + var pe *fs.PathError + part(errors.As(err, &pe) && pe.Op == "fork/exec" && errors.Is(pe.Err, syscall.ENOENT), + fmt.Sprintf("os/exec a missing program: %v", err)) + + verdict := "PASS" + if failed != 0 { + verdict = "FAIL" + } + fmt.Printf("[GO] goexec %s: %d of %d parts held\n", verdict, parts-failed, parts) + if failed != 0 { + os.Exit(1) + } +}