From dff31857f04b5d6eabe1d890355abe42f11d8ddd Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:02:08 +0000 Subject: [PATCH 01/26] linux: the signal frame is linux's rt_sigframe byte for byte The ucontext in the frame a handler is entered through put the sigcontext at offset 48 where Linux has it at 40, and the mask right after the 18 register words where Linux has uc_sigmask at 296. A handler that only returned through rt_sigreturn never noticed, since the frame was read back with the same wrong offsets. A handler that reads its ucontext did: Go's preemption handler reads the interrupted rip and rsp from it and saw every register one word off. The frame is now struct rt_sigframe as Linux lays it out: pretcode, then the ucontext with uc_stack at +16, the sigcontext at +40 with oldmask as its 22nd word, uc_sigmask at +296, then the 128-byte siginfo at +312. The handler is entered with the interrupted registers left as they were, rax zero and the direction, resume and trap flags cleared. sigframe holds the layout and the Entry a handler is entered with, which can also place the frame at the top of an alternate stack and carries a whole siginfo; sigframe_build writes the frame and sigframe_read reads it back. The host proofs mount all three and check the offsets against Linux's. --- userland/capsule_linux/src/linux/call/mod.rs | 2 + .../capsule_linux/src/linux/call/sigframe.rs | 85 +++++++------------ .../src/linux/call/sigframe_build.rs | 68 +++++++++++++++ .../src/linux/call/sigframe_read.rs | 35 ++++++++ .../capsule_linux/src/linux/call/sigreturn.rs | 3 +- .../capsule_linux/src/linux/serve/deliver.rs | 17 +++- userland/capsule_linux_proofs/src/calls.rs | 31 +++++++ userland/capsule_linux_proofs/src/lib.rs | 13 ++- userland/capsule_linux_proofs/src/tests.rs | 2 + .../src/tests/mutation_tests.rs | 18 +--- .../src/tests/sigframe_layout_tests.rs | 59 +++++++++++++ .../src/tests/sigframe_mutation_tests.rs | 50 +++++++++++ .../src/tests/sigframe_tests.rs | 63 +++++++------- 13 files changed, 331 insertions(+), 115 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/sigframe_build.rs create mode 100644 userland/capsule_linux/src/linux/call/sigframe_read.rs create mode 100644 userland/capsule_linux_proofs/src/calls.rs create mode 100644 userland/capsule_linux_proofs/src/tests/sigframe_layout_tests.rs create mode 100644 userland/capsule_linux_proofs/src/tests/sigframe_mutation_tests.rs diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index fcc45d321..8ffbc7cb2 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -39,6 +39,8 @@ mod pipe_read; mod pipe_wait; mod session; pub mod sigframe; +pub mod sigframe_build; +mod sigframe_read; mod signal; mod signal_send; mod sigreturn; diff --git a/userland/capsule_linux/src/linux/call/sigframe.rs b/userland/capsule_linux/src/linux/call/sigframe.rs index f6bdd2e71..cd833ba4f 100644 --- a/userland/capsule_linux/src/linux/call/sigframe.rs +++ b/userland/capsule_linux/src/linux/call/sigframe.rs @@ -14,62 +14,35 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! The `rt_sigframe` x86-64 puts on a thread's stack to enter a signal handler, -//! and where to read it back on return. Pure, so the layout is checked against -//! a round trip without a guest. It matches Linux `struct rt_sigframe`: -//! pretcode u64, ucontext at +8, siginfo at +312; the ucontext's sigcontext -//! holds the 18 words `mk_foreign_context` uses, in that order. +//! The `rt_sigframe` x86-64 puts on a thread's stack to enter a signal handler: +//! its layout, and what a handler is entered with (sigframe_build writes it, +//! sigframe_read reads it back). Pure, so the layout is checked against a +//! round trip without a guest. It matches Linux `struct rt_sigframe`: +//! pretcode u64, then the ucontext at +8 (uc_flags, uc_link, the 24-byte +//! uc_stack, the 256-byte sigcontext at +40, uc_sigmask at +296), then the +//! 128-byte siginfo at +312. The sigcontext starts with the 18 words +//! `mk_foreign_context` uses, in that order. -use alloc::vec::Vec; +pub const WORDS: usize = 18; /* r8..r15,rdi,rsi,rbp,rbx,rdx,rax,rcx,rsp,rip,rflags */ +pub const FRAME_SIZE: usize = 440; +pub const UC_OFF: usize = 8; +pub const STACK_OFF: usize = 16; /* uc_stack within the ucontext */ +pub const SIGCONTEXT_OFF: usize = 40; /* uc_mcontext within the ucontext */ +pub const OLDMASK_WORD: usize = 21; /* sigcontext.oldmask, after cs/gs/fs/ss, err, trapno */ +pub const SIGMASK_OFF: usize = 296; /* uc_sigmask within the ucontext */ +pub const INFO_OFF: usize = 312; +pub const INFO_LEN: usize = 128; -pub const WORDS: usize = 18; // r8..r15,rdi,rsi,rbp,rbx,rdx,rax,rcx,rsp,rip,rflags -const FRAME_SIZE: usize = 440; -const UC_OFF: usize = 8; -pub const SIGCONTEXT_OFF: usize = 48; // sigcontext within the ucontext -const INFO_OFF: usize = 312; -const REDZONE: u64 = 128; // the System V red zone below rsp - -fn put(buf: &mut [u8], at: usize, v: u64) { - buf[at..at + 8].copy_from_slice(&v.to_le_bytes()); -} -/// Where the frame lands, the bytes to write there, and the registers that -/// enter the handler. `None` if the stack is too low to hold a frame. -pub fn build( - regs: &[u64; WORDS], - handler: u64, - restorer: u64, - signum: u32, - blocked: u64, -) -> Option<(u64, Vec, [u64; WORDS])> { - // Below the red zone, 16-aligned, then down 8 so the handler sees rsp+8 - // aligned as a call would leave it. - let frame = - (regs[15].checked_sub(REDZONE)?.checked_sub(FRAME_SIZE as u64)? & !15u64).checked_sub(8)?; - let mut buf = alloc::vec![0u8; FRAME_SIZE]; - put(&mut buf, 0, restorer); - let mc = UC_OFF + SIGCONTEXT_OFF; // the 18 words, then the sigmask - for (i, w) in regs.iter().enumerate() { - put(&mut buf, mc + i * 8, *w); - } - put(&mut buf, UC_OFF + SIGCONTEXT_OFF + WORDS * 8, blocked); - put(&mut buf, INFO_OFF, u64::from(signum)); // siginfo: si_signo - let mut out = [0u64; WORDS]; - out[8] = u64::from(signum); // rdi - out[9] = frame + INFO_OFF as u64; // rsi, &siginfo - out[12] = frame + UC_OFF as u64; // rdx, &ucontext - out[15] = frame; // rsp at the frame; rax stays 0, no vector registers - out[16] = handler; // rip - out[17] = regs[17]; // rflags, the kernel masks it - Some((frame, buf, out)) -} - -/// The 18 words a returning frame carries, from the ucontext the guest's rsp -/// points at: the trampoline's `ret` left rsp there. -pub fn returned(uc: &[u8]) -> Option<[u64; WORDS]> { - let mut out = [0u64; WORDS]; - for (i, slot) in out.iter_mut().enumerate() { - let at = SIGCONTEXT_OFF + i * 8; - *slot = u64::from_le_bytes(uc.get(at..at + 8)?.try_into().ok()?); - } - Some(out) +/// What a handler is entered with, beyond the registers it interrupts. +pub struct Entry<'a> { + pub handler: u64, + pub restorer: u64, + pub signum: u32, + /// The mask the thread had, restored by rt_sigreturn. + pub blocked: u64, + /// The top of the alternate stack when the frame goes there. + pub alt_top: Option, + /// uc_stack as Linux saves it: ss_sp, ss_flags, ss_size. + pub stack: [u64; 3], + pub info: &'a [u8; INFO_LEN], } diff --git a/userland/capsule_linux/src/linux/call/sigframe_build.rs b/userland/capsule_linux/src/linux/call/sigframe_build.rs new file mode 100644 index 000000000..191fc5a51 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/sigframe_build.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Building the `rt_sigframe` a handler is entered through: the bytes, where +//! on the stack they go, and the registers the handler starts with. Pure, so +//! the frame is checked without a guest. + +use alloc::vec::Vec; + +use super::sigframe::{Entry, FRAME_SIZE, INFO_OFF, OLDMASK_WORD, SIGCONTEXT_OFF}; +use super::sigframe::{SIGMASK_OFF, STACK_OFF, UC_OFF, WORDS}; + +const REDZONE: u64 = 128; /* the System V red zone below rsp */ +/* The flags a handler is entered with cleared: direction, resume, trap. */ +const ENTRY_CLEARS: u64 = 0x400 | 0x1_0000 | 0x100; + +fn put(buf: &mut [u8], at: usize, v: u64) { + buf[at..at + 8].copy_from_slice(&v.to_le_bytes()); +} + +/// Where the frame lands, the bytes to write there, and the registers that +/// enter the handler. `None` if the stack is too low to hold a frame. +pub fn build(regs: &[u64; WORDS], e: &Entry) -> Option<(u64, Vec, [u64; WORDS])> { + /* + * Below the red zone, or at the top of the alternate stack; 16-aligned, + * then down 8 so the handler sees rsp+8 aligned as a call would leave it. + */ + let top = match e.alt_top { + Some(top) => top, + None => regs[15].checked_sub(REDZONE)?, + }; + let frame = (top.checked_sub(FRAME_SIZE as u64)? & !15u64).checked_sub(8)?; + let mut buf = alloc::vec![0u8; FRAME_SIZE]; + put(&mut buf, 0, e.restorer); + let (st, mc) = (UC_OFF + STACK_OFF, UC_OFF + SIGCONTEXT_OFF); + put(&mut buf, st, e.stack[0]); + buf[st + 8..st + 12].copy_from_slice(&(e.stack[1] as u32).to_le_bytes()); + put(&mut buf, st + 16, e.stack[2]); + for (i, w) in regs.iter().enumerate() { + put(&mut buf, mc + i * 8, *w); + } + put(&mut buf, mc + OLDMASK_WORD * 8, e.blocked); + put(&mut buf, UC_OFF + SIGMASK_OFF, e.blocked); + buf[INFO_OFF..].copy_from_slice(e.info); + /* The rest of the registers are the interrupted ones, as Linux leaves them. */ + let mut out = *regs; + out[8] = u64::from(e.signum); /* rdi */ + out[9] = frame + INFO_OFF as u64; /* rsi, &siginfo */ + out[12] = frame + UC_OFF as u64; /* rdx, &ucontext */ + out[13] = 0; /* rax: no vector registers passed */ + out[15] = frame; + out[16] = e.handler; + out[17] = regs[17] & !ENTRY_CLEARS; + Some((frame, buf, out)) +} diff --git a/userland/capsule_linux/src/linux/call/sigframe_read.rs b/userland/capsule_linux/src/linux/call/sigframe_read.rs new file mode 100644 index 000000000..dd1aac153 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/sigframe_read.rs @@ -0,0 +1,35 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Reading back the frame `sigframe` built, as rt_sigreturn does: the +//! registers the handler returns to. Pure, and any bytes at all may sit where +//! the guest's rsp points, so every read is checked and none can panic. + +use super::sigframe::{SIGCONTEXT_OFF, WORDS}; + +fn word(uc: &[u8], at: usize) -> Option { + Some(u64::from_le_bytes(uc.get(at..at + 8)?.try_into().ok()?)) +} + +/// The 18 words a returning frame carries, from the ucontext the guest's rsp +/// points at: the trampoline's `ret` left rsp there. +pub fn returned(uc: &[u8]) -> Option<[u64; WORDS]> { + let mut out = [0u64; WORDS]; + for (i, slot) in out.iter_mut().enumerate() { + *slot = word(uc, SIGCONTEXT_OFF + i * 8)?; + } + Some(out) +} diff --git a/userland/capsule_linux/src/linux/call/sigreturn.rs b/userland/capsule_linux/src/linux/call/sigreturn.rs index fa0dae276..09b598c6f 100644 --- a/userland/capsule_linux/src/linux/call/sigreturn.rs +++ b/userland/capsule_linux/src/linux/call/sigreturn.rs @@ -21,7 +21,8 @@ use nonos_libc::{mk_foreign_context, mk_foreign_signal, ForeignRegs, SIGNAL_RETURN}; -use super::sigframe::{returned, SIGCONTEXT_OFF, WORDS}; +use super::sigframe::{SIGCONTEXT_OFF, WORDS}; +use super::sigframe_read::returned; use crate::linux::guest::Guest; use crate::linux::serve::Answer; diff --git a/userland/capsule_linux/src/linux/serve/deliver.rs b/userland/capsule_linux/src/linux/serve/deliver.rs index e5ed46059..0af49d080 100644 --- a/userland/capsule_linux/src/linux/serve/deliver.rs +++ b/userland/capsule_linux/src/linux/serve/deliver.rs @@ -22,7 +22,8 @@ use nonos_libc::{mk_foreign_context, mk_foreign_signal, ForeignRegs, SIGNAL_DELIVER}; -use crate::linux::call::sigframe::build; +use crate::linux::call::sigframe::{Entry, INFO_LEN}; +use crate::linux::call::sigframe_build::build; use crate::linux::guest::Guest; /// rax in the register word order. @@ -36,7 +37,19 @@ pub fn maybe_deliver(guest: &mut Guest, tid: u32, reply: u64) -> bool { let mut regs: ForeignRegs = [0; 18]; let built = (mk_foreign_context(tid, &mut regs) == 0).then(|| { regs[RAX] = reply; - build(®s, act.handler, act.restorer, u32::from(signum), 0) + /* The siginfo carries the number; uc_stack is SS_DISABLE, no alternate stack. */ + let mut info = [0u8; INFO_LEN]; + info[..4].copy_from_slice(&i32::from(signum).to_le_bytes()); + let entry = Entry { + handler: act.handler, + restorer: act.restorer, + signum: u32::from(signum), + blocked: 0, + alt_top: None, + stack: [0, 2, 0], + info: &info, + }; + build(®s, &entry) }); let Some(Some((_, buf, enter))) = built else { // Could not read the thread or shape a frame: keep the signal pending. diff --git a/userland/capsule_linux_proofs/src/calls.rs b/userland/capsule_linux_proofs/src/calls.rs new file mode 100644 index 000000000..5eb542041 --- /dev/null +++ b/userland/capsule_linux_proofs/src/calls.rs @@ -0,0 +1,31 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The capsule's pure call code, mounted from its tree as it ships: the +//! signal frame built and read back, and the reading of exec's shebang line. +//! None of it names the capsule's crate, so it runs here without a guest. + +#[path = "../../capsule_linux/src/linux/call/sigframe.rs"] +pub mod sigframe; + +#[path = "../../capsule_linux/src/linux/call/sigframe_build.rs"] +pub mod sigframe_build; + +#[path = "../../capsule_linux/src/linux/call/sigframe_read.rs"] +pub mod sigframe_read; + +#[path = "../../capsule_linux/src/linux/call/spawn/exec_shebang.rs"] +pub mod exec_shebang; diff --git a/userland/capsule_linux_proofs/src/lib.rs b/userland/capsule_linux_proofs/src/lib.rs index caa4499bf..5aef0ee56 100644 --- a/userland/capsule_linux_proofs/src/lib.rs +++ b/userland/capsule_linux_proofs/src/lib.rs @@ -51,8 +51,10 @@ pub mod statbuf; #[path = "../../capsule_linux/src/linux/net/host_body.rs"] pub mod host_body; -// net.sockets' own reader for a connect-by-host body, mounted at the crate -// paths it names, so the capsule's encoder is held to the real parser. +/* + * net.sockets' own reader for a connect-by-host body, mounted at the crate + * paths it names, so the capsule's encoder is held to the real parser. + */ #[path = "../../capsule_net_sockets/src/protocol/errno.rs"] pub mod protocol; pub mod server; @@ -60,11 +62,8 @@ pub mod server; #[path = "../../capsule_linux/src/linux/net/route.rs"] pub mod route; -#[path = "../../capsule_linux/src/linux/call/sigframe.rs"] -pub mod sigframe; - -#[path = "../../capsule_linux/src/linux/call/spawn/exec_shebang.rs"] -pub mod exec_shebang; +pub mod calls; +pub use calls::{exec_shebang, sigframe, sigframe_build, sigframe_read}; #[cfg(test)] pub mod image; diff --git a/userland/capsule_linux_proofs/src/tests.rs b/userland/capsule_linux_proofs/src/tests.rs index a5860bdd6..498353d57 100644 --- a/userland/capsule_linux_proofs/src/tests.rs +++ b/userland/capsule_linux_proofs/src/tests.rs @@ -40,6 +40,8 @@ mod pacman_desc_tests; mod pacman_rsa_tests; mod resolve_tests; mod route_tests; +mod sigframe_layout_tests; +mod sigframe_mutation_tests; mod sigframe_tests; mod service; mod stack_words_tests; diff --git a/userland/capsule_linux_proofs/src/tests/mutation_tests.rs b/userland/capsule_linux_proofs/src/tests/mutation_tests.rs index 0b5254e0c..8d3477f42 100644 --- a/userland/capsule_linux_proofs/src/tests/mutation_tests.rs +++ b/userland/capsule_linux_proofs/src/tests/mutation_tests.rs @@ -28,7 +28,7 @@ use crate::install::unpacked::unpacked; use super::mutation::damage; -const ROUNDS: usize = 1500; +pub(super) const ROUNDS: usize = 1500; #[test] fn damaged_debs_never_panic() { @@ -64,19 +64,3 @@ fn damaged_indexes_never_panic() { } } } - -#[test] -fn a_damaged_signal_frame_never_panics_returning() { - use crate::sigframe::{build, returned}; - let mut s = 0x516E_A100u64; - let mut base = [0u64; 18]; - base[15] = 0x7fff_ff00_0000; - let (_, buf, _) = build(&base, 0x4000, 0x4008, 11, 0).expect("frame"); - for _ in 0..ROUNDS { - let mut v = buf.clone(); - damage(&mut s, &mut v); - // rt_sigreturn reads the ucontext at the guest's rsp: any bytes there. - let _ = returned(&v); - let _ = v.first().map(|_| returned(&v[v.len().min(8)..])); - } -} diff --git a/userland/capsule_linux_proofs/src/tests/sigframe_layout_tests.rs b/userland/capsule_linux_proofs/src/tests/sigframe_layout_tests.rs new file mode 100644 index 000000000..898fed910 --- /dev/null +++ b/userland/capsule_linux_proofs/src/tests/sigframe_layout_tests.rs @@ -0,0 +1,59 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The frame byte for byte against Linux's `struct rt_sigframe`, and where a +//! frame goes: below the alternate stack's top, or nowhere when the stack is +//! too low to hold one. A handler that reads its ucontext (Go's does, to +//! preempt) finds each field where Linux puts it. + +use super::sigframe_tests::{at, entry, regs, INFO, RSP}; +use crate::sigframe::SIGCONTEXT_OFF; +use crate::sigframe_build::build; + +#[test] +fn the_frame_is_linux_rt_sigframe_byte_for_byte() { + /* + * pretcode at 0; ucontext at 8 with uc_mcontext at +40, oldmask its 22nd + * word, uc_sigmask at +296; siginfo at 312. + */ + assert_eq!(SIGCONTEXT_OFF, 40); + let saved = regs(); + let (_, buf, _) = build(&saved, &entry(1, 0xca11, 3, 0xabcd)).expect("frame"); + assert_eq!(at(&buf, 0), 0xca11); + assert_eq!(at(&buf, 8 + 40), saved[0]); /* r8 */ + assert_eq!(at(&buf, 8 + 40 + 16 * 8), saved[16]); /* rip */ + assert_eq!(at(&buf, 8 + 40 + 21 * 8), 0xabcd); /* oldmask */ + assert_eq!(at(&buf, 8 + 296), 0xabcd); /* uc_sigmask */ + assert_eq!(&buf[312..440], &INFO[..]); + assert_eq!(buf[8 + 24], 2); /* uc_stack.ss_flags: SS_DISABLE */ +} + +#[test] +fn a_frame_for_the_alternate_stack_sits_below_its_top() { + let saved = regs(); + let mut e = entry(1, 2, 3, 0); + e.alt_top = Some(0x5000_0000); + let (frame, _, enter) = build(&saved, &e).expect("frame"); + assert!(frame < 0x5000_0000 && frame > 0x5000_0000 - 1024); + assert_eq!(enter[RSP], frame); +} + +#[test] +fn a_stack_too_low_to_hold_a_frame_is_refused() { + let mut low = regs(); + low[RSP] = 64; /* below the red zone plus a frame */ + assert!(build(&low, &entry(1, 2, 3, 0)).is_none()); +} diff --git a/userland/capsule_linux_proofs/src/tests/sigframe_mutation_tests.rs b/userland/capsule_linux_proofs/src/tests/sigframe_mutation_tests.rs new file mode 100644 index 000000000..24a8ef39f --- /dev/null +++ b/userland/capsule_linux_proofs/src/tests/sigframe_mutation_tests.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Seeded mutation of the signal frame a handler returns through: whatever +//! bytes sit where the guest's rsp points, rt_sigreturn's reader must return, +//! in a debug build, without a panic. + +use super::mutation::damage; +use super::mutation_tests::ROUNDS; +use crate::sigframe::Entry; +use crate::sigframe_build::build; +use crate::sigframe_read::returned; + +#[test] +fn a_damaged_signal_frame_never_panics_returning() { + let mut s = 0x516E_A100u64; + let mut base = [0u64; 18]; + base[15] = 0x7fff_ff00_0000; + let info = [0u8; 128]; + let e = Entry { + handler: 0x4000, + restorer: 0x4008, + signum: 11, + blocked: 0, + alt_top: None, + stack: [0, 2, 0], + info: &info, + }; + let (_, buf, _) = build(&base, &e).expect("frame"); + for _ in 0..ROUNDS { + let mut v = buf.clone(); + damage(&mut s, &mut v); + /* rt_sigreturn reads the ucontext at the guest's rsp: any bytes there. */ + let _ = returned(&v); + let _ = v.first().map(|_| returned(&v[v.len().min(8)..])); + } +} diff --git a/userland/capsule_linux_proofs/src/tests/sigframe_tests.rs b/userland/capsule_linux_proofs/src/tests/sigframe_tests.rs index aaa75bcfb..2cca213f5 100644 --- a/userland/capsule_linux_proofs/src/tests/sigframe_tests.rs +++ b/userland/capsule_linux_proofs/src/tests/sigframe_tests.rs @@ -17,33 +17,50 @@ //! The signal frame a handler enters through, and the frame it returns from, //! are the same layout read two ways: build one, then read the sigcontext //! back the way rt_sigreturn does, and the registers must be identical. This -//! is what lets a program's handler run and return to where it was. +//! is what lets a program's handler run and return to where it was. The +//! offsets are Linux's `struct rt_sigframe`, which a handler that reads its +//! ucontext (Go's does, to preempt) depends on byte for byte. -use crate::sigframe::{build, returned, SIGCONTEXT_OFF, WORDS}; +use crate::sigframe::{Entry, WORDS}; +use crate::sigframe_build::build; +use crate::sigframe_read::returned; -const RSP: usize = 15; +pub(super) const RSP: usize = 15; const RAX: usize = 13; -fn regs() -> [u64; WORDS] { +pub(super) fn regs() -> [u64; WORDS] { let mut r = [0u64; WORDS]; for (i, w) in r.iter_mut().enumerate() { - *w = 0x1111_0000 + i as u64; // a distinct value per register + *w = 0x1111_0000 + i as u64; /* a distinct value per register */ } - r[RSP] = 0x7fff_ffe0_0000; // a plausible stack pointer, page aligned + r[RSP] = 0x7fff_ffe0_0000; /* a plausible stack pointer, page aligned */ r } +pub(super) const INFO: [u8; 128] = [0x5a; 128]; + +pub(super) fn entry(handler: u64, restorer: u64, signum: u32, blocked: u64) -> Entry<'static> { + Entry { handler, restorer, signum, blocked, alt_top: None, stack: [0, 2, 0], info: &INFO } +} + +pub(super) fn at(buf: &[u8], off: usize) -> u64 { + u64::from_le_bytes(buf[off..off + 8].try_into().unwrap()) +} + #[test] fn a_returning_frame_restores_the_registers_the_handler_was_entered_over() { let saved = regs(); - let (frame, buf, enter) = build(&saved, 0xdead_beef, 0xca11, 11, 0x1234).expect("frame"); - // The handler is entered at the frame, below the old stack, 16-byte down 8. + let (frame, buf, enter) = + build(&saved, &entry(0xdead_beef, 0xca11, 11, 0x1234)).expect("frame"); + /* The handler is entered at the frame, below the old stack, 16-byte down 8. */ assert!(frame < saved[RSP] - 128); + assert_eq!(frame % 16, 8); assert_eq!(enter[RSP], frame); - assert_eq!(enter[16], 0xdead_beef); // rip = handler - assert_eq!(enter[8], 11); // rdi = signum - assert_eq!(enter[12], frame + 8); // rdx = &ucontext - // rt_sigreturn reads the ucontext the guest's rsp points at: frame + 8. + assert_eq!(enter[16], 0xdead_beef); /* rip = handler */ + assert_eq!(enter[8], 11); /* rdi = signum */ + assert_eq!(enter[9], frame + 312); /* rsi = &siginfo */ + assert_eq!(enter[12], frame + 8); /* rdx = &ucontext */ + /* rt_sigreturn reads the ucontext the guest's rsp points at: frame + 8. */ let uc = &buf[8..]; assert_eq!(returned(uc).unwrap(), saved); } @@ -51,25 +68,7 @@ fn a_returning_frame_restores_the_registers_the_handler_was_entered_over() { #[test] fn the_syscall_return_value_rides_in_the_saved_rax() { let mut saved = regs(); - saved[RAX] = 0; // as the thread trapped, before we set the reply - saved[RAX] = 42; // the value the interrupted syscall returns - let (_, buf, _) = build(&saved, 1, 2, 3, 0).expect("frame"); + saved[RAX] = 42; /* the value the interrupted syscall returns */ + let (_, buf, _) = build(&saved, &entry(1, 2, 3, 0)).expect("frame"); assert_eq!(returned(&buf[8..]).unwrap()[RAX], 42); } - -#[test] -fn a_stack_too_low_to_hold_a_frame_is_refused() { - let mut low = regs(); - low[RSP] = 64; // below the red zone plus a frame - assert!(build(&low, 1, 2, 3, 0).is_none()); -} - -#[test] -fn the_sigcontext_sits_where_the_ucontext_says() { - // uc_mcontext is at SIGCONTEXT_OFF within the ucontext, which is at frame+8. - let saved = regs(); - let (_, buf, _) = build(&saved, 1, 2, 3, 0).expect("frame"); - let at = 8 + SIGCONTEXT_OFF; - let r8 = u64::from_le_bytes(buf[at..at + 8].try_into().unwrap()); - assert_eq!(r8, saved[0]); -} From 1b2ea57962767a82b7188e1885c1f613ac8a4986 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:02:32 +0000 Subject: [PATCH 02/26] linux: any handler can turn a kernel pid into the guest's number The numbers a guest sees for its processes and threads were kept inside the family's PidNs, which only the serve loop holds. A handler writing a pid into memory rather than returning it, as a siginfo or a waitid answer does, had no way to translate it, and would have handed the guest a kernel pid it cannot name. A personality hosts one family, so the numbering now lives in one place, pid_space, that PidNs fronts, and serve exports guest_pid and kernel_pid for any handler to use. The numbers given are the same as before. The two exports have no caller yet; the siginfo and waitid work that follows uses them. --- userland/capsule_linux/src/linux/serve/mod.rs | 2 + .../capsule_linux/src/linux/serve/pid_ns.rs | 28 +++++------ .../src/linux/serve/pid_space.rs | 47 +++++++++++++++++++ 3 files changed, 60 insertions(+), 17 deletions(-) create mode 100644 userland/capsule_linux/src/linux/serve/pid_space.rs diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index b75373ce9..d9fc6b295 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -26,6 +26,7 @@ mod family_sleep; mod loop_impl; mod pid_map; mod pid_ns; +mod pid_space; mod refused; mod pid_out; mod table; @@ -39,3 +40,4 @@ mod unserved; pub use answer::Answer; pub use loop_impl::serve; +pub use pid_space::{inward as kernel_pid, outward as guest_pid}; diff --git a/userland/capsule_linux/src/linux/serve/pid_ns.rs b/userland/capsule_linux/src/linux/serve/pid_ns.rs index c3c9501f1..4f3ad801a 100644 --- a/userland/capsule_linux/src/linux/serve/pid_ns.rs +++ b/userland/capsule_linux/src/linux/serve/pid_ns.rs @@ -21,36 +21,30 @@ //! it sees these instead: the personality is 1, the program it started is 2, //! and each process or thread after takes the next number. None is reused //! while the family lives, so a stale number never reaches a newer process. +//! +//! A personality hosts one family, so the numbers are kept in one place and +//! any handler can write a guest's number into what it hands back: a siginfo +//! or a waitid answer carries a pid, not only a return value. -use alloc::vec::Vec; +use super::pid_space::{inward, outward, SPACE}; -pub struct PidNs { - map: Vec<(u32, u32)>, - next: u32, -} +/// The family's numbering. Made once, when the family is. +pub struct PidNs; impl PidNs { pub fn new(personality: u32, first: u32) -> Self { - PidNs { map: alloc::vec![(personality, 1), (first, 2)], next: 3 } + *SPACE.0.borrow_mut() = (alloc::vec![(personality, 1), (first, 2)], 3); + PidNs } /// The number a guest sees for kernel pid `k`, given on first sight. /// Zero once the space is spent, which no caller reads as a process. pub fn outward(&mut self, k: u32) -> u32 { - if let Some(&(_, g)) = self.map.iter().find(|(kp, _)| *kp == k) { - return g; - } - let Some(after) = self.next.checked_add(1) else { - return 0; - }; - let g = self.next; - self.next = after; - self.map.push((k, g)); - g + outward(k) } /// The kernel pid behind a guest's number, if it names one of this family. pub fn inward(&self, g: u32) -> Option { - self.map.iter().find(|(_, gp)| *gp == g).map(|(k, _)| *k) + inward(g) } } diff --git a/userland/capsule_linux/src/linux/serve/pid_space.rs b/userland/capsule_linux/src/linux/serve/pid_space.rs new file mode 100644 index 000000000..89ad08e2d --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/pid_space.rs @@ -0,0 +1,47 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a family's pid numbers are kept. A personality hosts one family, so +//! they live in one static any handler can reach without the family at hand. + +use alloc::vec::Vec; +use core::cell::RefCell; + +pub(super) struct Space(pub(super) RefCell<(Vec<(u32, u32)>, u32)>); +/* SAFETY: eK@nonos.systems - only the personality's single serve loop touches it. */ +unsafe impl Sync for Space {} +pub(super) static SPACE: Space = Space(RefCell::new((Vec::new(), 0))); + +/// `PidNs::inward`, for a handler with no family at hand. +pub fn inward(g: u32) -> Option { + let space = SPACE.0.borrow(); + space.0.iter().find(|(_, gp)| *gp == g).map(|(k, _)| *k) +} + +/// `PidNs::outward`, for a handler with no family at hand. +pub fn outward(k: u32) -> u32 { + let mut space = SPACE.0.borrow_mut(); + if let Some(&(_, g)) = space.0.iter().find(|(kp, _)| *kp == k) { + return g; + } + let g = space.1; + let Some(after) = g.checked_add(1) else { + return 0; + }; + space.1 = after; + space.0.push((k, g)); + g +} From 3bd641ce6796dddf3d6d7ab673c474a30d65dcd8 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:02:35 +0000 Subject: [PATCH 03/26] linux: signals honour per-thread masks, alternate stacks and defaults rt_sigprocmask ignored the mask it was given and read back an empty one, and sigaltstack ignored the stack. A handler was entered on the thread's own stack whatever SA_ONSTACK said, which crashes Go as soon as a signal lands on a small goroutine stack. sa_mask, SA_NODEFER and SA_RESETHAND were ignored, so a handler could be re-entered by its own signal. The siginfo carried only the signal's number, and rt_sigreturn restored the registers but not the mask. A signal that could not be delivered was raised again and again, and one the process did not catch waited forever. Each thread now has its own mask and alternate stack. A new thread's mask is its creator's, a forked child's is the forking thread's, and exec resets caught handlers to their default while keeping the mask and what is pending. A handler is entered on the alternate stack when it asked for one, with the thread's mask plus sa_mask plus the signal itself unless SA_NODEFER, and SA_RESETHAND puts the default back. The frame carries a full siginfo with the sender's pid in the guest's numbering. rt_sigreturn restores the mask and the alternate stack from the ucontext, and a frame that cannot be written or read back ends the process with SIGSEGV, as on Linux. A signal with no handler does its Linux default: ignore, or end the whole process. Standard signals coalesce, realtime ones queue, and the lowest number is taken first. Signals keeps the state the later calls need, so some fields and types are written but not yet read. born has its caller in clone, which is kept as it is here. --- userland/capsule_linux/src/linux/call/life.rs | 8 ++ userland/capsule_linux/src/linux/call/mod.rs | 8 +- .../src/linux/call/sigframe_read.rs | 18 +++- .../capsule_linux/src/linux/call/signal.rs | 55 +++++------- .../src/linux/call/signal_mask.rs | 52 ++++++++++++ .../src/linux/call/signal_send.rs | 6 +- .../src/linux/call/signal_stack.rs | 69 +++++++++++++++ .../capsule_linux/src/linux/call/sigreturn.rs | 42 ++++++--- .../src/linux/call/spawn/exec.rs | 10 ++- .../src/linux/call/spawn/exec_threads.rs | 2 +- .../src/linux/call/spawn/fork.rs | 4 +- userland/capsule_linux/src/linux/guest/mod.rs | 12 +++ .../src/linux/guest/sigdefault.rs | 40 +++++++++ .../capsule_linux/src/linux/guest/siginfo.rs | 65 ++++++++++++++ .../capsule_linux/src/linux/guest/siglive.rs | 28 ++++++ .../capsule_linux/src/linux/guest/sigpark.rs | 56 ++++++++++++ .../capsule_linux/src/linux/guest/sigqueue.rs | 81 ++++++++---------- .../src/linux/guest/sigqueue_new.rs | 43 ++++++++++ .../src/linux/guest/sigqueue_ops.rs | 40 +++++++++ .../capsule_linux/src/linux/guest/sigraise.rs | 55 ++++++++++++ .../capsule_linux/src/linux/guest/sigstate.rs | 24 +++++- .../capsule_linux/src/linux/guest/sigtake.rs | 42 +++++++++ .../src/linux/guest/sigthread.rs | 70 +++++++++++++++ .../src/linux/guest/sigthread_copy.rs | 49 +++++++++++ .../capsule_linux/src/linux/guest/sigtimer.rs | 49 +++++++++++ .../capsule_linux/src/linux/guest/sigwaits.rs | 72 ++++++++++++++++ .../capsule_linux/src/linux/serve/deliver.rs | 85 +++++++++---------- .../src/linux/serve/deliver_enter.rs | 74 ++++++++++++++++ .../src/linux/serve/deliver_say.rs | 30 +++++++ .../src/linux/serve/deliver_stack.rs | 40 +++++++++ userland/capsule_linux/src/linux/serve/mod.rs | 4 + .../capsule_linux/src/linux/serve/table.rs | 7 +- .../src/linux/serve/table_sig.rs | 31 +++++++ 33 files changed, 1120 insertions(+), 151 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/signal_mask.rs create mode 100644 userland/capsule_linux/src/linux/call/signal_stack.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigdefault.rs create mode 100644 userland/capsule_linux/src/linux/guest/siginfo.rs create mode 100644 userland/capsule_linux/src/linux/guest/siglive.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigpark.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigqueue_new.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigqueue_ops.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigraise.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigtake.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigthread.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigthread_copy.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigtimer.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigwaits.rs create mode 100644 userland/capsule_linux/src/linux/serve/deliver_enter.rs create mode 100644 userland/capsule_linux/src/linux/serve/deliver_say.rs create mode 100644 userland/capsule_linux/src/linux/serve/deliver_stack.rs create mode 100644 userland/capsule_linux/src/linux/serve/table_sig.rs diff --git a/userland/capsule_linux/src/linux/call/life.rs b/userland/capsule_linux/src/linux/call/life.rs index 10f50ec2b..740a4d539 100644 --- a/userland/capsule_linux/src/linux/call/life.rs +++ b/userland/capsule_linux/src/linux/call/life.rs @@ -62,3 +62,11 @@ pub fn exit(guest: &mut Guest, code: u64) -> u64 { guest.exited = Some(code as i32); errno::ok(0) } + +/// The process ends on `signum`, unless something already ended it. Kept in +/// the shell's 128+signo form, as a thread's fatal fault is. +pub fn killed(guest: &mut Guest, signum: u8) { + if guest.exited.is_none() { + guest.exited = Some(128 + i32::from(signum & 0x7f)); + } +} diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index 8ffbc7cb2..f01f4ccd1 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -42,7 +42,9 @@ pub mod sigframe; pub mod sigframe_build; mod sigframe_read; mod signal; +mod signal_mask; mod signal_send; +mod signal_stack; mod sigreturn; mod sleep; mod spawn; @@ -58,7 +60,7 @@ pub use cwd::{chdir, fchdir, getcwd}; pub use futex::futex; pub use ident::{getppid, setuid}; pub use io::{close, read, write}; -pub use life::{exit, exit_thread, set_tid_address}; +pub use life::{exit, exit_thread, killed, set_tid_address}; pub use limits::{getrlimit, prlimit64}; pub use glibc::prctl; pub use glibc_sched::{clone3, getcpu, membarrier, sched_getaffinity}; @@ -69,7 +71,9 @@ pub use pipe_io::write as pipe_write; pub use pipe_read::read as pipe_read; pub use pipe_wait::{is_pipe, read_or_park as pipe_read_or_park}; pub use session::{getpgid, getsid, setpgid, setsid}; -pub use signal::{rt_sigaction, rt_sigprocmask, sigaltstack}; +pub use signal::rt_sigaction; +pub use signal_mask::rt_sigprocmask; +pub use signal_stack::sigaltstack; pub use sigreturn::rt_sigreturn; pub use signal_send::kill; pub use sleep::{clock_nanosleep, nanosleep}; diff --git a/userland/capsule_linux/src/linux/call/sigframe_read.rs b/userland/capsule_linux/src/linux/call/sigframe_read.rs index dd1aac153..8e71733af 100644 --- a/userland/capsule_linux/src/linux/call/sigframe_read.rs +++ b/userland/capsule_linux/src/linux/call/sigframe_read.rs @@ -15,10 +15,11 @@ // along with this program. If not, see . //! Reading back the frame `sigframe` built, as rt_sigreturn does: the -//! registers the handler returns to. Pure, and any bytes at all may sit where -//! the guest's rsp points, so every read is checked and none can panic. +//! registers, the mask and the alternate stack the handler returns to. Pure, +//! and any bytes at all may sit where the guest's rsp points, so every read is +//! checked and none can panic. -use super::sigframe::{SIGCONTEXT_OFF, WORDS}; +use super::sigframe::{SIGCONTEXT_OFF, SIGMASK_OFF, STACK_OFF, WORDS}; fn word(uc: &[u8], at: usize) -> Option { Some(u64::from_le_bytes(uc.get(at..at + 8)?.try_into().ok()?)) @@ -33,3 +34,14 @@ pub fn returned(uc: &[u8]) -> Option<[u64; WORDS]> { } Some(out) } + +/// The mask a returning frame restores, from uc_sigmask. +pub fn returned_mask(uc: &[u8]) -> Option { + word(uc, SIGMASK_OFF) +} + +/// uc_stack as the handler left it: ss_sp, ss_flags, ss_size. +pub fn returned_stack(uc: &[u8]) -> Option<[u64; 3]> { + let flags = u32::from_le_bytes(uc.get(STACK_OFF + 8..STACK_OFF + 12)?.try_into().ok()?); + Some([word(uc, STACK_OFF)?, u64::from(flags), word(uc, STACK_OFF + 16)?]) +} diff --git a/userland/capsule_linux/src/linux/call/signal.rs b/userland/capsule_linux/src/linux/call/signal.rs index a0f5698c0..459d8fa66 100644 --- a/userland/capsule_linux/src/linux/call/signal.rs +++ b/userland/capsule_linux/src/linux/call/signal.rs @@ -14,32 +14,39 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Signal dispositions, recorded here and delivered on the return path in -//! `serve::deliver`: a handler is kept with its flags, restorer and mask. +//! Signal dispositions and masks, recorded here and acted on in +//! `serve::deliver`: a handler is kept with its flags, restorer and mask, and +//! each thread has the mask of signals it holds back (signal_mask). use crate::linux::abi::errno; -use crate::linux::guest::sigstate::{SigAction, NSIG}; +use crate::linux::guest::sigstate::{SigAction, NSIG, SIGKILL, SIGSTOP}; use crate::linux::guest::Guest; -// SIGKILL/SIGSTOP cannot be caught; `struct sigaction` is 32 bytes. -const SIGKILL: u64 = 9; -const SIGSTOP: u64 = 19; +/* `struct sigaction` is 32 bytes; a kernel sigset_t is 8. */ const SIGACTION_LEN: usize = 32; +pub const SIGSET_LEN: u64 = 8; -pub fn rt_sigaction(guest: &mut Guest, signum: u64, act: u64, old: u64) -> u64 { - if signum == 0 || signum > NSIG as u64 || signum == SIGKILL || signum == SIGSTOP { +pub fn rt_sigaction(guest: &mut Guest, signum: u64, act: u64, old: u64, size: u64) -> u64 { + let unchangeable = signum == u64::from(SIGKILL) || signum == u64::from(SIGSTOP); + if size != SIGSET_LEN || signum == 0 || signum > NSIG as u64 || (act != 0 && unchangeable) { return errno::fail(errno::EINVAL); } let n = signum as usize; - if old != 0 - && guest.write(old, &encode(guest.signals.action(n).unwrap_or_default())) - < SIGACTION_LEN as i64 - { + let was = guest.signals.action(n).unwrap_or_default(); + let new = match act { + 0 => None, + at => match guest.read(at, SIGACTION_LEN) { + Some(raw) => Some(decode(&raw)), + None => return errno::fail(errno::EFAULT), + }, + }; + if old != 0 && guest.write(old, &encode(was)) < SIGACTION_LEN as i64 { return errno::fail(errno::EFAULT); } - if act != 0 { - match guest.read(act, SIGACTION_LEN) { - Some(raw) => guest.signals.set(n, decode(&raw)), - None => return errno::fail(errno::EFAULT), + if let Some(a) = new { + guest.signals.set(n, a); + /* A signal set to be ignored is dropped where it already waits. */ + if guest.signals.discards(n as u8) { + guest.signals.discard(n as u8); } } errno::ok(0) @@ -57,19 +64,3 @@ fn encode(a: SigAction) -> [u8; SIGACTION_LEN] { b[24..32].copy_from_slice(&a.mask.to_le_bytes()); b } - -/// The old mask reads back empty: nothing is held back, delivery ignores it. -pub fn rt_sigprocmask(guest: &Guest, old: u64) -> u64 { - if old != 0 && guest.write(old, &[0u8; 8]) < 8 { - return errno::fail(errno::EFAULT); - } - errno::ok(0) -} - -/// The old alternate stack reads back unset; a handler uses the own stack. -pub fn sigaltstack(guest: &Guest, old: u64) -> u64 { - if old != 0 && guest.write(old, &[0u8; 24]) < 24 { - return errno::fail(errno::EFAULT); - } - errno::ok(0) -} diff --git a/userland/capsule_linux/src/linux/call/signal_mask.rs b/userland/capsule_linux/src/linux/call/signal_mask.rs new file mode 100644 index 000000000..d6cf92b97 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_mask.rs @@ -0,0 +1,52 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! rt_sigprocmask: each thread's mask of the signals it holds back, which +//! `serve::deliver` reads before a signal is taken. + +use super::signal::SIGSET_LEN; +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +const SIG_BLOCK: u64 = 0; +const SIG_UNBLOCK: u64 = 1; +const SIG_SETMASK: u64 = 2; + +/// The calling thread's mask: added to, taken from or replaced. The old mask +/// is written after the change is made, as Linux orders it. +pub fn rt_sigprocmask(guest: &mut Guest, tid: u32, how: u64, set: u64, old: u64, size: u64) -> u64 { + if size != SIGSET_LEN { + return errno::fail(errno::EINVAL); + } + let was = guest.signals.blocked(tid); + if set != 0 { + let Some(raw) = guest.read(set, 8) else { + return errno::fail(errno::EFAULT); + }; + let new = u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8])); + let mask = match how { + SIG_BLOCK => was | new, + SIG_UNBLOCK => was & !new, + SIG_SETMASK => new, + _ => return errno::fail(errno::EINVAL), + }; + guest.signals.set_blocked(tid, mask); + } + if old != 0 && guest.write(old, &was.to_le_bytes()) < 8 { + return errno::fail(errno::EFAULT); + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/signal_send.rs b/userland/capsule_linux/src/linux/call/signal_send.rs index f7606f427..20979e1aa 100644 --- a/userland/capsule_linux/src/linux/call/signal_send.rs +++ b/userland/capsule_linux/src/linux/call/signal_send.rs @@ -21,6 +21,7 @@ use nonos_libc::mk_kill; use crate::linux::abi::errno; +use crate::linux::guest::siginfo::{SigInfo, SI_USER}; use crate::linux::guest::sigstate::NSIG; use crate::linux::guest::Guest; @@ -42,7 +43,7 @@ pub fn kill(guest: &mut Guest, pid: u64, signo: u64) -> u64 { } let act = guest.signals.action(signo as usize).unwrap_or_default(); if act.catches() { - guest.signals.raise(target, signo as u8); + let _ = guest.signals.raise(target, SigInfo::from(signo as u8, SI_USER, guest.pid)); return errno::ok(0); } if act.ignores() || IGNORED_DEFAULT.contains(&signo) { @@ -53,9 +54,8 @@ pub fn kill(guest: &mut Guest, pid: u64, signo: u64) -> u64 { /// The default action of an uncaught, non-ignored signal is to end the thread. fn terminate(guest: &mut Guest, target: u32, signo: u64) -> u64 { - guest.waits.retain(|(w, _)| *w != target); + guest.forget_thread(target); guest.threads.retain(|t| *t != target); - guest.signals.forget(target); match mk_kill(target as u64, signo) { n if n < 0 => errno::fail(errno::EPERM), _ => errno::ok(0), diff --git a/userland/capsule_linux/src/linux/call/signal_stack.rs b/userland/capsule_linux/src/linux/call/signal_stack.rs new file mode 100644 index 000000000..5c4aee2f5 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_stack.rs @@ -0,0 +1,69 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sigaltstack`: the calling thread's alternate signal stack, where a +//! handler asking for SA_ONSTACK runs. `stack_t` is ss_sp, a 4-byte ss_flags +//! and its padding, then ss_size. + +use nonos_libc::{mk_foreign_context, ForeignRegs}; + +use crate::linux::abi::errno; +use crate::linux::guest::sigthread::{SS_DISABLE, SS_ONSTACK}; +use crate::linux::guest::Guest; + +const STACK_T_LEN: usize = 24; +/// MINSIGSTKSZ on x86-64. +const MINSIGSTKSZ: u64 = 2048; +/// SS_AUTODISARM, accepted and kept as Linux accepts it. +const SS_AUTODISARM: u64 = 1 << 31; +const RSP: usize = 15; + +pub fn sigaltstack(guest: &mut Guest, tid: u32, ss: u64, old: u64) -> u64 { + let [sp, flags, size] = guest.signals.thread(tid).alt; + let mut regs: ForeignRegs = [0; 18]; + let rsp = if mk_foreign_context(tid, &mut regs) == 0 { regs[RSP] } else { 0 }; + let on = flags & SS_DISABLE == 0 && rsp.wrapping_sub(sp) < size; + if old != 0 { + let now = if on { SS_ONSTACK } else { flags & (SS_DISABLE | SS_AUTODISARM) }; + let mut b = [0u8; STACK_T_LEN]; + b[..8].copy_from_slice(&sp.to_le_bytes()); + b[8..12].copy_from_slice(&(now as u32).to_le_bytes()); + b[16..].copy_from_slice(&size.to_le_bytes()); + if guest.write(old, &b) < STACK_T_LEN as i64 { + return errno::fail(errno::EFAULT); + } + } + if ss == 0 { + return errno::ok(0); + } + let Some(raw) = guest.read(ss, STACK_T_LEN) else { + return errno::fail(errno::EFAULT); + }; + let word = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + let new_flags = u64::from(u32::from_le_bytes(raw[8..12].try_into().unwrap_or([0; 4]))); + if on { + return errno::fail(errno::EPERM); + } + /* SS_ONSTACK asks for the same as 0: an enabled stack. */ + let alt = match new_flags & !(SS_AUTODISARM | SS_ONSTACK) { + SS_DISABLE => [0, SS_DISABLE, 0], + 0 if word(16) < MINSIGSTKSZ => return errno::fail(errno::ENOMEM), + 0 => [word(0), new_flags & SS_AUTODISARM, word(16)], + _ => return errno::fail(errno::EINVAL), + }; + guest.signals.thread(tid).alt = alt; + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/sigreturn.rs b/userland/capsule_linux/src/linux/call/sigreturn.rs index 09b598c6f..fb2d28ae3 100644 --- a/userland/capsule_linux/src/linux/call/sigreturn.rs +++ b/userland/capsule_linux/src/linux/call/sigreturn.rs @@ -15,33 +15,49 @@ // along with this program. If not, see . //! `rt_sigreturn`: a thread leaving a signal handler. Its rsp points at the -//! ucontext the frame carried, so the saved registers are read back from -//! there and the kernel resumes the thread into them. Nothing is replied: the -//! thread is no longer in the syscall, it is back where the signal interrupted. +//! ucontext the frame carried, so the saved registers, the mask and the +//! alternate stack are read back from there and the kernel resumes the thread +//! into them. Nothing is replied: the thread is no longer in the syscall, it +//! is back where the signal interrupted. A frame that cannot be read ends the +//! process with SIGSEGV, as Linux's does. use nonos_libc::{mk_foreign_context, mk_foreign_signal, ForeignRegs, SIGNAL_RETURN}; -use super::sigframe::{SIGCONTEXT_OFF, WORDS}; -use super::sigframe_read::returned; +use super::sigframe::{SIGMASK_OFF, WORDS}; +use super::sigframe_read::{returned, returned_mask, returned_stack}; +use crate::linux::guest::sigstate::SIGSEGV; +use crate::linux::guest::sigthread::{SS_DISABLE, SS_ONSTACK}; use crate::linux::guest::Guest; use crate::linux::serve::Answer; /// rsp in the register word order. const RSP: usize = 15; -pub fn rt_sigreturn(guest: &Guest, tid: u32) -> Answer { +pub fn rt_sigreturn(guest: &mut Guest, tid: u32) -> Answer { let mut regs: ForeignRegs = [0; WORDS]; if mk_foreign_context(tid, &mut regs) != 0 { return Answer::Park; } - // The trampoline's `ret` left rsp at the ucontext; the sigcontext follows. - let want = SIGCONTEXT_OFF + WORDS * 8; - let Some(bytes) = guest.read(regs[RSP], want) else { + /* The trampoline's `ret` left rsp at the ucontext; the mask ends it. */ + let read = guest.read(regs[RSP], SIGMASK_OFF + 8); + let back = read + .as_deref() + .and_then(|uc| Some((returned(uc)?, returned_mask(uc)?, returned_stack(uc)?))); + let Some((restored, mask, stack)) = back else { + super::killed(guest, SIGSEGV); return Answer::Park; }; - let Some(restored) = returned(&bytes) else { - return Answer::Park; - }; - let _ = mk_foreign_signal(tid, &restored, SIGNAL_RETURN); + guest.signals.set_blocked(tid, mask); + let t = guest.signals.thread(tid); + let on_now = t.alt[1] & SS_DISABLE == 0 && restored[RSP].wrapping_sub(t.alt[0]) < t.alt[2]; + if !on_now { + t.alt = match stack[1] & SS_DISABLE { + 0 => [stack[0], stack[1] & !SS_ONSTACK, stack[2]], + _ => [0, SS_DISABLE, 0], + }; + } + if mk_foreign_signal(tid, &restored, SIGNAL_RETURN) != 0 { + super::killed(guest, SIGSEGV); + } Answer::Park } diff --git a/userland/capsule_linux/src/linux/call/spawn/exec.rs b/userland/capsule_linux/src/linux/call/spawn/exec.rs index 825ff2ae0..42b7b82f7 100644 --- a/userland/capsule_linux/src/linux/call/spawn/exec.rs +++ b/userland/capsule_linux/src/linux/call/spawn/exec.rs @@ -47,7 +47,15 @@ pub fn execve(guest: &mut Guest, pid: u32, path: u64, argv: u64, envp: u64) -> A super::exec_threads::reap(guest, pid); clear(guest); match load_over(guest, pid, &program, &env) { - Some(()) => Answer::Park, + Some(()) => { + released(guest, pid); + Answer::Park + } None => Answer::value(errno::fail(errno::ENOEXEC)), } } + +/// The new program keeps what Linux keeps of the old one's signals. +fn released(guest: &mut Guest, pid: u32) { + guest.signals.exec_reset(pid); +} diff --git a/userland/capsule_linux/src/linux/call/spawn/exec_threads.rs b/userland/capsule_linux/src/linux/call/spawn/exec_threads.rs index c2a423772..ac9efeaa8 100644 --- a/userland/capsule_linux/src/linux/call/spawn/exec_threads.rs +++ b/userland/capsule_linux/src/linux/call/spawn/exec_threads.rs @@ -34,7 +34,7 @@ pub fn reap(guest: &mut Guest, caller: u32) { * can be ended: the kill marks it, but nothing collects a thread that * is still waiting for an answer. */ - guest.waits.retain(|(w, _)| *w != tid); + guest.forget_thread(tid); let _ = mk_foreign_reply(tid, errno::fail(errno::EINTR)); let _ = mk_kill(tid as u64, SIGKILL); } diff --git a/userland/capsule_linux/src/linux/call/spawn/fork.rs b/userland/capsule_linux/src/linux/call/spawn/fork.rs index 55502a762..f60069907 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork.rs @@ -43,7 +43,9 @@ pub fn fork(guest: &mut Guest, caller: u32) -> Answer { * The child's state goes to the serve loop before the child runs, so its * first trap finds a guest that owns it. */ - guest.forked.push(guest.fork_state(child)); + let mut state = guest.fork_state(child); + state.signals = guest.signals.forked(caller, child); + guest.forked.push(state); if mk_foreign_resume(child) < 0 { guest.forked.pop(); return Answer::value(errno::fail(errno::ENOMEM)); diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index bf0b7b03c..b26a52198 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -25,8 +25,20 @@ mod fd_make; mod fork_state; mod handle; mod handle_new; +pub mod sigdefault; +pub mod siginfo; +mod siglive; +mod sigpark; pub mod sigqueue; +mod sigqueue_new; +mod sigqueue_ops; +mod sigraise; pub mod sigstate; +mod sigtake; +pub mod sigthread; +mod sigthread_copy; +pub mod sigtimer; +pub mod sigwaits; mod layout; mod links; mod links_add; diff --git a/userland/capsule_linux/src/linux/guest/sigdefault.rs b/userland/capsule_linux/src/linux/guest/sigdefault.rs new file mode 100644 index 000000000..f6910fdc3 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigdefault.rs @@ -0,0 +1,40 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What Linux does with a signal no handler takes, when its disposition is +//! SIG_DFL. The table is Linux's, transcribed. + +/// What an uncaught signal does when its disposition is SIG_DFL. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Default { + Ignore, + Terminate, + Stop, + Continue, +} + +/// Linux's table: child status, urgent data and window size are ignored, +/// SIGCONT continues, the four stop signals stop, everything else ends the +/// process. The core-dumping ones end it too: no core is ever written, so the +/// status carries no core flag, as on Linux with a zero core limit. +pub fn default_of(signum: u8) -> Default { + match signum { + 17 | 23 | 28 => Default::Ignore, + 18 => Default::Continue, + 19..=22 => Default::Stop, + _ => Default::Terminate, + } +} diff --git a/userland/capsule_linux/src/linux/guest/siginfo.rs b/userland/capsule_linux/src/linux/guest/siginfo.rs new file mode 100644 index 000000000..7c6188ea3 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/siginfo.rs @@ -0,0 +1,65 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What a signal carries: the 128-byte siginfo Linux hands a handler, a +//! sigtimedwait or a waitid. A pid is kept in kernel terms and becomes the +//! guest's own number only as the bytes are written, so no guest sees a +//! kernel pid in one. + +pub const SI_USER: i32 = 0; +pub const INFO_LEN: usize = 128; + +#[derive(Clone, Copy, Default)] +pub struct SigInfo { + pub signo: u8, + pub code: i32, + /// The sending process, or the child that ended; zero names none. + pub pid: u32, + /// si_timerid and si_overrun, for a timer's signal, which has no pid. + pub timer: Option<(i32, i32)>, + /// si_value for a queued or timer signal; si_status for SIGCHLD. + pub value: u64, +} + +impl SigInfo { + /// A signal sent by a process, as kill, tkill and sigqueue send one. + pub fn from(signo: u8, code: i32, pid: u32) -> Self { + SigInfo { signo, code, pid, ..SigInfo::default() } + } + + /// Linux's layout: signo, errno, code, then at +16 either the pid and + /// uid or the timer id and overrun, then the value or status at +24. + pub fn bytes(&self) -> [u8; INFO_LEN] { + let mut b = [0u8; INFO_LEN]; + b[0..4].copy_from_slice(&i32::from(self.signo).to_le_bytes()); + b[8..12].copy_from_slice(&self.code.to_le_bytes()); + let (at16, at20) = match self.timer { + Some((id, overrun)) => (id, overrun), + None => (guest_number(self.pid), 0), /* uid 0: every guest's */ + }; + b[16..20].copy_from_slice(&at16.to_le_bytes()); + b[20..24].copy_from_slice(&at20.to_le_bytes()); + b[24..32].copy_from_slice(&self.value.to_le_bytes()); + b + } +} + +fn guest_number(pid: u32) -> i32 { + match pid { + 0 => 0, + k => crate::linux::serve::guest_pid(k) as i32, + } +} diff --git a/userland/capsule_linux/src/linux/guest/siglive.rs b/userland/capsule_linux/src/linux/guest/siglive.rs new file mode 100644 index 000000000..ab7e784a6 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/siglive.rs @@ -0,0 +1,28 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A thread that has gone, taking its signal state and its waits with it. + +use super::handle::Guest; + +impl Guest { + /// A thread that has gone takes its signal state and its waits with it. + pub fn forget_thread(&mut self, tid: u32) { + let _ = self.leave_waits(tid); + self.signals.threads.retain(|t| t.tid != tid); + self.signals.pending.retain(|(t, _)| *t != tid); + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigpark.rs b/userland/capsule_linux/src/linux/guest/sigpark.rs new file mode 100644 index 000000000..f9912efdd --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigpark.rs @@ -0,0 +1,56 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which wait a thread is parked in, and taking it out of every one, so a +//! thread that has gone leaves no wait behind to be answered. + +use super::handle::Guest; + +/// Parked, and in which kind of wait. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Parked { + /// A timed sleep, with its monotonic deadline. + Sleep(u64), + /// Any other wait a signal ends. + Wait, +} + +impl Guest { + /// The wait `tid` is parked in, if a signal can end it. + pub fn parked(&self, tid: u32) -> Option { + if let Some(&(due, _)) = self.sleepers.iter().find(|(_, t)| *t == tid) { + return Some(Parked::Sleep(due)); + } + let waiting = self.waits.iter().any(|(t, _)| *t == tid) + || self.pipe_wait.is_some_and(|w| w.3 == tid) + || self.signals.sigwaits.iter().any(|w| w.tid == tid) + || self.signals.childwaits.iter().any(|w| w.tid == tid); + waiting.then_some(Parked::Wait) + } + + /// Take `tid` out of every wait it is parked in, without answering it. + pub fn leave_waits(&mut self, tid: u32) -> Option { + let was = self.parked(tid)?; + self.sleepers.retain(|(_, t)| *t != tid); + self.waits.retain(|(t, _)| *t != tid); + if self.pipe_wait.is_some_and(|w| w.3 == tid) { + self.pipe_wait = None; + } + self.signals.sigwaits.retain(|w| w.tid != tid); + self.signals.childwaits.retain(|w| w.tid != tid); + Some(was) + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigqueue.rs b/userland/capsule_linux/src/linux/guest/sigqueue.rs index 03593c493..56db59eb6 100644 --- a/userland/capsule_linux/src/linux/guest/sigqueue.rs +++ b/userland/capsule_linux/src/linux/guest/sigqueue.rs @@ -14,58 +14,45 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Signals raised against a process's threads and not yet delivered, with the -//! disposition of each. The queue names the thread a signal is for. +//! Signals raised against a process and not yet taken, with the disposition +//! of each. An entry names the thread it is for, or 0 for the process as a +//! whole, which any thread not blocking it may take, as on Linux. use alloc::vec::Vec; +use super::siginfo::SigInfo; use super::sigstate::{SigAction, NSIG}; +use super::sigthread::ThreadSig; +use super::sigtimer::{Itimer, PosixTimer}; +use super::sigwaits::{ChildWait, Outbound, SigWait}; + +/// Linux's default RLIMIT_SIGPENDING for a small machine: how many queued +/// realtime signals a process may hold before sigqueue answers EAGAIN. +pub const QUEUE_MAX: usize = 1024; #[derive(Clone)] pub struct Signals { - actions: [SigAction; NSIG], - pending: Vec<(u32, u8)>, -} - -impl Default for Signals { - fn default() -> Self { - Self { actions: [SigAction::default(); NSIG], pending: Vec::new() } - } -} - -impl Signals { - /// Record a disposition; `signum` is 1..=NSIG. - pub fn set(&mut self, signum: usize, act: SigAction) { - if (1..=NSIG).contains(&signum) { - self.actions[signum - 1] = act; - } - } - - pub fn action(&self, signum: usize) -> Option { - (1..=NSIG).contains(&signum).then(|| self.actions[signum - 1]) - } - - /// Queue a signal against a thread. A standard signal already pending is - /// not queued twice, as Linux coalesces non-realtime signals. - pub fn raise(&mut self, tid: u32, signum: u8) { - if !self.pending.iter().any(|p| *p == (tid, signum)) { - self.pending.push((tid, signum)); - } - } - - /// The next signal for `tid` its disposition catches, removed. Signals - /// with no handler are left for the caller to default. - pub fn take_caught(&mut self, tid: u32) -> Option<(u8, SigAction)> { - let at = self - .pending - .iter() - .position(|(t, s)| *t == tid && self.actions[*s as usize - 1].catches())?; - let signum = self.pending.remove(at).1; - Some((signum, self.actions[signum as usize - 1])) - } - - /// Drop every signal pending for a thread that has gone. - pub fn forget(&mut self, tid: u32) { - self.pending.retain(|(t, _)| *t != tid); - } + pub(super) actions: [SigAction; NSIG], + pub(super) pending: Vec<(u32, SigInfo)>, + /// Each thread's mask, alternate stack and suspended mask. + pub(super) threads: Vec, + /// Signals for other processes of the family, and who sent each. + pub outbox: Vec, + /// ITIMER_REAL, and the POSIX timers timer_create made. + pub real: Option, + pub timers: Vec, + /// Threads parked in pause, sigsuspend or sigtimedwait. + pub sigwaits: Vec, + /// Threads parked in wait4 or waitid. + pub childwaits: Vec, + /// Set once the leader has made a plain exit while other threads run on. + pub leader_gone: bool, + /// A vfork parent's thread, parked until this child execs or ends. + pub vfork: Option, + /// What this process raises at its parent when it ends; clone names it. + pub exit_signal: u8, + /// Children that raise something other than SIGCHLD, for __WCLONE. + pub clone_kids: Vec, + /// The process group each ended child was in, for a wait by group. + pub kid_groups: Vec<(u32, u32)>, } diff --git a/userland/capsule_linux/src/linux/guest/sigqueue_new.rs b/userland/capsule_linux/src/linux/guest/sigqueue_new.rs new file mode 100644 index 000000000..183333960 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigqueue_new.rs @@ -0,0 +1,43 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A process's signal state as it starts: every disposition at its default, +//! nothing pending, no timers, no waits, and SIGCHLD as its exit signal. + +use alloc::vec::Vec; + +use super::sigqueue::Signals; +use super::sigstate::{SigAction, NSIG}; + +impl Default for Signals { + fn default() -> Self { + Signals { + actions: [SigAction::default(); NSIG], + pending: Vec::new(), + threads: Vec::new(), + outbox: Vec::new(), + real: None, + timers: Vec::new(), + sigwaits: Vec::new(), + childwaits: Vec::new(), + leader_gone: false, + vfork: None, + exit_signal: super::sigstate::SIGCHLD, + clone_kids: Vec::new(), + kid_groups: Vec::new(), + } + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs new file mode 100644 index 000000000..9eec3ad8d --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs @@ -0,0 +1,40 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Reading and recording a process's signal state: the disposition of each +//! signal, and what is pending for a thread. + +use super::sigqueue::Signals; +use super::sigstate::{bit, SigAction, NSIG}; + +impl Signals { + /// Record a disposition; `signum` is 1..=NSIG. + pub fn set(&mut self, signum: usize, act: SigAction) { + if (1..=NSIG).contains(&signum) { + self.actions[signum - 1] = act; + } + } + + pub fn action(&self, signum: usize) -> Option { + (1..=NSIG).contains(&signum).then(|| self.actions[signum - 1]) + } + + /// Every signal pending for `tid` or for its process, as a mask. + pub fn pending_for(&self, tid: u32) -> u64 { + let mine = self.pending.iter().filter(|(t, _)| *t == tid || *t == 0); + mine.fold(0, |m, (_, i)| m | bit(i.signo)) + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigraise.rs b/userland/capsule_linux/src/linux/guest/sigraise.rs new file mode 100644 index 000000000..598a1e005 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigraise.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Raising a signal, in Linux's order: a standard signal already pending for +//! the same target is not queued again, a realtime one queues each time until +//! the queue is full; taking one is in sigtake. + +use super::sigdefault::{default_of, Default}; +use super::siginfo::SigInfo; +use super::sigqueue::{Signals, QUEUE_MAX}; +use super::sigstate::bit; + +/// The first realtime signal: from here up each raise is queued. +const SIGRTMIN: u8 = 32; + +impl Signals { + /// Queue `info` for thread `tid`, or for the process when `tid` is 0. + /// False when nothing was queued: an ignored signal no thread blocks is + /// discarded as it is sent, as Linux does, and so is a coalesced one. + pub fn raise(&mut self, tid: u32, info: SigInfo) -> bool { + let s = info.signo; + if self.discards(s) && !self.threads.iter().any(|t| t.blocked & bit(s) != 0) { + return false; + } + if s < SIGRTMIN && self.pending.iter().any(|(t, i)| *t == tid && i.signo == s) { + return false; + } + if s >= SIGRTMIN && self.pending.len() >= QUEUE_MAX { + return false; + } + self.pending.push((tid, info)); + true + } + + /// True when taking `signum` would do nothing: ignored by its handler, + /// or by a default of ignore or continue with nothing ever stopped. + pub fn discards(&self, signum: u8) -> bool { + let act = self.actions[signum as usize - 1]; + act.ignores() + || (!act.catches() && matches!(default_of(signum), Default::Ignore | Default::Continue)) + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigstate.rs b/userland/capsule_linux/src/linux/guest/sigstate.rs index ae31b5025..2fcad0711 100644 --- a/userland/capsule_linux/src/linux/guest/sigstate.rs +++ b/userland/capsule_linux/src/linux/guest/sigstate.rs @@ -14,12 +14,32 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! A signal's disposition: what the guest asked to happen when it fires. -//! Process-wide, as on Linux. +//! A signal's disposition, its number and its bit in a mask; what Linux does +//! with a signal no handler takes is in sigdefault. Dispositions are +//! process-wide, as on Linux; the numbers are transcribed. /// The largest signal Linux defines. pub const NSIG: usize = 64; +pub const SIGKILL: u8 = 9; +pub const SIGSEGV: u8 = 11; +pub const SIGCHLD: u8 = 17; +pub const SIGSTOP: u8 = 19; + +pub const SA_ONSTACK: u64 = 0x0800_0000; +pub const SA_NODEFER: u64 = 0x4000_0000; +pub const SA_RESETHAND: u64 = 0x8000_0000; + +/// The one bit a signal has in a mask. +pub fn bit(signum: u8) -> u64 { + 1u64 << (signum - 1) +} + +/// SIGKILL and SIGSTOP are never blocked, whatever a mask asks for. +pub fn blockable(mask: u64) -> u64 { + mask & !(bit(SIGKILL) | bit(SIGSTOP)) +} + /// `struct sigaction` as the guest passes it: handler, flags, restorer, mask. #[derive(Clone, Copy, Default)] pub struct SigAction { diff --git a/userland/capsule_linux/src/linux/guest/sigtake.rs b/userland/capsule_linux/src/linux/guest/sigtake.rs new file mode 100644 index 000000000..8a30681a0 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigtake.rs @@ -0,0 +1,42 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Taking a signal, in Linux's order: the lowest-numbered first, the thread's +//! own before the process's; and dropping what a newly ignored signal leaves +//! pending. + +use super::siginfo::SigInfo; +use super::sigqueue::Signals; +use super::sigstate::bit; + +impl Signals { + /// Take the next signal `tid` may take: one of `allow`, its own before the + /// process's, lowest first. + pub fn take(&mut self, tid: u32, allow: u64) -> Option { + let lowest = |want: u32| { + let each = self.pending.iter().enumerate(); + let fit = each.filter(|(_, (t, i))| *t == want && allow & bit(i.signo) != 0); + fit.min_by_key(|(_, (_, i))| i.signo).map(|(at, _)| at) + }; + let at = lowest(tid).or_else(|| lowest(0))?; + Some(self.pending.remove(at).1) + } + + /// Drop every pending `signum`, as setting SIG_IGN on it does. + pub fn discard(&mut self, signum: u8) { + self.pending.retain(|(_, i)| i.signo != signum); + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigthread.rs b/userland/capsule_linux/src/linux/guest/sigthread.rs new file mode 100644 index 000000000..a4527f11e --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigthread.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What each thread has of its own: the signals it blocks, its alternate +//! stack, and the mask a sigsuspend put aside until a handler returns. + +use super::sigqueue::Signals; +use super::sigstate::blockable; + +/// ss_flags: running on the alternate stack, and no alternate stack set. +pub const SS_ONSTACK: u64 = 1; +pub const SS_DISABLE: u64 = 2; + +#[derive(Clone, Copy)] +pub struct ThreadSig { + pub tid: u32, + pub blocked: u64, + /// ss_sp, ss_flags, ss_size. + pub alt: [u64; 3], + /// The mask sigsuspend replaced, which the handler's frame restores. + pub saved: Option, +} + +impl ThreadSig { + pub(super) fn new(tid: u32, blocked: u64) -> Self { + ThreadSig { tid, blocked, alt: [0, SS_DISABLE, 0], saved: None } + } +} + +impl Signals { + pub fn thread(&mut self, tid: u32) -> &mut ThreadSig { + let at = match self.threads.iter().position(|t| t.tid == tid) { + Some(at) => at, + None => { + self.threads.push(ThreadSig::new(tid, 0)); + self.threads.len() - 1 + } + }; + &mut self.threads[at] + } + + pub fn blocked(&self, tid: u32) -> u64 { + self.threads.iter().find(|t| t.tid == tid).map_or(0, |t| t.blocked) + } + + pub fn set_blocked(&mut self, tid: u32, mask: u64) { + self.thread(tid).blocked = blockable(mask); + } + + /// A new thread starts with its creator's mask and no alternate stack, + /// as clone(CLONE_VM) leaves it. + pub fn born(&mut self, parent: u32, child: u32) { + let blocked = self.blocked(parent); + self.threads.retain(|t| t.tid != child); + self.threads.push(ThreadSig::new(child, blocked)); + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigthread_copy.rs b/userland/capsule_linux/src/linux/guest/sigthread_copy.rs new file mode 100644 index 000000000..0b1977b7d --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigthread_copy.rs @@ -0,0 +1,49 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The signal state a new program starts with: a forked child's, copied from +//! the forking thread, and what exec leaves of a process's own. + +use super::sigqueue::Signals; +use super::sigstate::SigAction; +use super::sigthread::ThreadSig; + +impl Signals { + /// A forked child: the dispositions, and the forking thread's mask and + /// alternate stack for its one thread. Nothing pending, no timers. + pub fn forked(&self, caller: u32, child: u32) -> Signals { + let mut s = Signals { actions: self.actions, ..Signals::default() }; + let mine = self.threads.iter().find(|t| t.tid == caller).copied(); + let mut t = mine.unwrap_or(ThreadSig::new(child, 0)); + t.tid = child; + t.saved = None; + s.threads.push(t); + s + } + + /// Exec keeps the mask, the pending signals and ITIMER_REAL; caught + /// signals go back to their default, ignored ones stay ignored; the + /// alternate stack and the POSIX timers are gone. + pub fn exec_reset(&mut self, tid: u32) { + for act in self.actions.iter_mut().filter(|a| a.catches()) { + *act = SigAction::default(); + } + self.timers.clear(); + self.pending.retain(|(_, i)| i.timer.is_none()); + let blocked = self.blocked(tid); + self.threads = alloc::vec![ThreadSig::new(tid, blocked)]; + } +} diff --git a/userland/capsule_linux/src/linux/guest/sigtimer.rs b/userland/capsule_linux/src/linux/guest/sigtimer.rs new file mode 100644 index 000000000..141f93242 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigtimer.rs @@ -0,0 +1,49 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The timers that end in a signal: ITIMER_REAL, which alarm and setitimer +//! set, and the POSIX timers timer_create makes. Deadlines are milliseconds of +//! the family's monotonic clock, the finest step it keeps. + +/// ITIMER_REAL: when it next fires, and its period, 0 for once. +#[derive(Clone, Copy)] +pub struct Itimer { + pub due: u64, + pub interval: u64, +} + +/// One timer_create timer. +#[derive(Clone, Copy)] +pub struct PosixTimer { + /// The id the guest was given, from 0 up, as Linux numbers them. + pub id: i32, + /// The clock an absolute time is read on. + pub clock: u64, + /// The signal it raises, 0 for SIGEV_NONE. + pub signo: u8, + /// The thread SIGEV_THREAD_ID named, 0 for the process. + pub tid: u32, + /// sigev_value, handed back in si_value. + pub value: u64, + pub due: Option, + pub interval: u64, + /// Expiries while its signal was still queued, and the count the last + /// taken signal carried, which timer_getoverrun reports. + pub overrun: i32, + pub last_overrun: i32, + /// Its signal is queued and not yet taken. + pub queued: bool, +} diff --git a/userland/capsule_linux/src/linux/guest/sigwaits.rs b/userland/capsule_linux/src/linux/guest/sigwaits.rs new file mode 100644 index 000000000..6dde91516 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigwaits.rs @@ -0,0 +1,72 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Threads parked until a signal or a child says something, and signals on +//! their way to another process of the family. + +use super::siginfo::SigInfo; + +/// A thread in pause, sigsuspend or sigtimedwait. +#[derive(Clone, Copy)] +pub struct SigWait { + pub tid: u32, + /// sigtimedwait's set: a signal in it is taken, not handled. 0 for pause + /// and sigsuspend, which only a handler ends. + pub set: u64, + /// Where sigtimedwait writes the siginfo, 0 for nowhere. + pub info: u64, + /// When sigtimedwait gives up with EAGAIN. + pub due: Option, +} + +/// Which children a wait4 or waitid asks about. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Which { + Any, + Pid(u32), + Group(u32), +} + +/// A thread in wait4 or waitid, and where its answer goes. +#[derive(Clone, Copy)] +pub struct ChildWait { + pub tid: u32, + pub which: Which, + pub options: u64, + /// wait4's status word, or waitid's siginfo; and the rusage, 0 for none. + pub out: u64, + pub rusage: u64, + pub waitid: bool, +} + +/// Who a signal leaving this process is for. +#[derive(Clone, Copy)] +pub enum Target { + Process(u32), + /// A thread, and the process it must belong to, 0 for any. + Thread(u32, u32), + Group(u32), + /// kill(-1): every process of the family but the sender. + All, +} + +/// A signal for another process, and the thread parked until it is sent. +#[derive(Clone, Copy)] +pub struct Outbound { + pub from: u32, + pub to: Target, + pub info: SigInfo, +} diff --git a/userland/capsule_linux/src/linux/serve/deliver.rs b/userland/capsule_linux/src/linux/serve/deliver.rs index 0af49d080..97800066a 100644 --- a/userland/capsule_linux/src/linux/serve/deliver.rs +++ b/userland/capsule_linux/src/linux/serve/deliver.rs @@ -14,61 +14,60 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Delivering a caught signal to the thread returning from a syscall. The -//! handler is entered with the interrupted syscall's return value already in -//! rax, so when it returns through rt_sigreturn the program sees that value. -//! Only the trapping thread is delivered to here; a signal raised against a -//! thread parked elsewhere waits in the queue until that thread next traps. +//! A thread taking the signals it may take. A caught one enters its handler +//! through Linux's rt_sigframe; an uncaught one does what its default says, +//! and a default of ending the process ends all of it. A thread takes them +//! when it returns from a call, with the call's value in rax (maybe_deliver), +//! and when the kernel stops it running, on the registers it stopped with +//! (deliver_on). -use nonos_libc::{mk_foreign_context, mk_foreign_signal, ForeignRegs, SIGNAL_DELIVER}; +use nonos_libc::{mk_foreign_context, ForeignRegs}; -use crate::linux::call::sigframe::{Entry, INFO_LEN}; -use crate::linux::call::sigframe_build::build; +use super::deliver_enter::enter; +use super::deliver_say::stop_unserved; +use crate::linux::call::killed; +use crate::linux::guest::sigdefault::{default_of, Default}; use crate::linux::guest::Guest; /// rax in the register word order. const RAX: usize = 13; -/// True when a handler was entered, so the caller must not also reply. +/// True when the thread was answered, into a handler or by its process +/// ending, so the caller must not also reply. pub fn maybe_deliver(guest: &mut Guest, tid: u32, reply: u64) -> bool { - let Some((signum, act)) = guest.signals.take_caught(tid) else { - return false; - }; - let mut regs: ForeignRegs = [0; 18]; - let built = (mk_foreign_context(tid, &mut regs) == 0).then(|| { - regs[RAX] = reply; - /* The siginfo carries the number; uc_stack is SS_DISABLE, no alternate stack. */ - let mut info = [0u8; INFO_LEN]; - info[..4].copy_from_slice(&i32::from(signum).to_le_bytes()); - let entry = Entry { - handler: act.handler, - restorer: act.restorer, - signum: u32::from(signum), - blocked: 0, - alt_top: None, - stack: [0, 2, 0], - info: &info, - }; - build(®s, &entry) - }); - let Some(Some((_, buf, enter))) = built else { - // Could not read the thread or shape a frame: keep the signal pending. - guest.signals.raise(tid, signum); - return false; - }; - if guest.write(enter[15], &buf) < buf.len() as i64 { - guest.signals.raise(tid, signum); + if guest.signals.pending_for(tid) & !guest.signals.blocked(tid) == 0 { return false; } - if mk_foreign_signal(tid, &enter, SIGNAL_DELIVER) != 0 { - guest.signals.raise(tid, signum); + let mut regs: ForeignRegs = [0; 18]; + if mk_foreign_context(tid, &mut regs) != 0 { return false; } - say(tid, signum, act.handler); - true + regs[RAX] = reply; + deliver_on(guest, tid, regs) } -fn say(tid: u32, signum: u8, handler: u64) { - let line = alloc::format!("[LINUX] signal {signum} to tid {tid}, handler {handler:#x}\n"); - let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); +/// Take what `tid` may take and act on it, over `regs` as they stand. For a +/// frame the kernel stopped while running as much as for a returning call. +pub fn deliver_on(guest: &mut Guest, tid: u32, regs: ForeignRegs) -> bool { + loop { + let allow = !guest.signals.blocked(tid); + let Some(info) = guest.signals.take(tid, allow) else { + return false; + }; + let act = guest.signals.action(info.signo as usize).unwrap_or_default(); + if act.catches() { + return enter(guest, tid, ®s, info); + } + if act.ignores() { + continue; + } + match default_of(info.signo) { + Default::Terminate => { + killed(guest, info.signo); + return true; + } + Default::Stop => stop_unserved(info.signo), + Default::Ignore | Default::Continue => {} + } + } } diff --git a/userland/capsule_linux/src/linux/serve/deliver_enter.rs b/userland/capsule_linux/src/linux/serve/deliver_enter.rs new file mode 100644 index 000000000..cd85d9a12 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_enter.rs @@ -0,0 +1,74 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Entering a handler: the frame on the thread's stack, or at the top of its +//! alternate stack when the handler asked for SA_ONSTACK and the thread is not +//! already on it; the mask while the handler runs is the thread's own, the +//! handler's sa_mask, and the signal itself unless SA_NODEFER. + +use nonos_libc::{mk_foreign_signal, ForeignRegs, SIGNAL_DELIVER}; + +use super::deliver_say::say; +use super::deliver_stack::placement; +use crate::linux::call::killed; +use crate::linux::call::sigframe::Entry; +use crate::linux::call::sigframe_build::build; +use crate::linux::guest::siginfo::SigInfo; +use crate::linux::guest::sigstate::{bit, SigAction, SA_NODEFER, SA_RESETHAND, SIGSEGV}; +use crate::linux::guest::Guest; + +const RSP: usize = 15; + +/// True once the thread is answered: in its handler, or its process ended +/// because no frame could be written, which Linux answers with SIGSEGV. +pub fn enter(guest: &mut Guest, tid: u32, regs: &ForeignRegs, info: SigInfo) -> bool { + let act = guest.signals.action(info.signo as usize).unwrap_or_default(); + let t = *guest.signals.thread(tid); + let (alt_top, stack) = placement(t.alt, act.flags, regs[RSP]); + let saved = t.saved.unwrap_or(t.blocked); + let bytes = info.bytes(); + let entry = Entry { + handler: act.handler, + restorer: act.restorer, + signum: u32::from(info.signo), + blocked: saved, + alt_top, + stack, + info: &bytes, + }; + let Some((at, buf, into)) = build(regs, &entry).filter(|_| act.restorer != 0) else { + killed(guest, SIGSEGV); + return true; + }; + if guest.write(at, &buf) < buf.len() as i64 { + killed(guest, SIGSEGV); + return true; + } + if mk_foreign_signal(tid, &into, SIGNAL_DELIVER) != 0 { + /* Not parked after all: the signal waits for its next return. */ + let _ = guest.signals.raise(tid, info); + return false; + } + let during = + t.blocked | act.mask | if act.flags & SA_NODEFER != 0 { 0 } else { bit(info.signo) }; + guest.signals.set_blocked(tid, during); + guest.signals.thread(tid).saved = None; + if act.flags & SA_RESETHAND != 0 { + guest.signals.set(info.signo as usize, SigAction::default()); + } + say(tid, info.signo, act.handler); + true +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_say.rs b/userland/capsule_linux/src/linux/serve/deliver_say.rs new file mode 100644 index 000000000..fa0e34f92 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_say.rs @@ -0,0 +1,30 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What delivering a signal says in the log: each handler entered, and each +//! stop that is not served. + +/// No guest is ever stopped: job control needs the kernel to hold every +/// thread of a process still, which it does not offer a supervisor. +pub fn stop_unserved(signum: u8) { + let line = alloc::format!("[LINUX] unserved stop: signal {signum} does not stop a guest\n"); + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); +} + +pub fn say(tid: u32, signum: u8, handler: u64) { + let line = alloc::format!("[LINUX] signal {signum} to tid {tid}, handler {handler:#x}\n"); + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_stack.rs b/userland/capsule_linux/src/linux/serve/deliver_stack.rs new file mode 100644 index 000000000..01937e05d --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_stack.rs @@ -0,0 +1,40 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a handler's frame goes: on the thread's stack, or at the top of its +//! alternate stack when the handler asked for SA_ONSTACK and the thread is not +//! already on it; and the uc_stack the frame records, with ss_flags as +//! sigaltstack would report them at the moment the signal came. + +use crate::linux::guest::sigstate::SA_ONSTACK; +use crate::linux::guest::sigthread::{SS_DISABLE, SS_ONSTACK}; + +/// The top of the alternate stack when the frame goes there, and uc_stack: +/// ss_sp, ss_flags, ss_size. `alt` is the thread's, `rsp` where it stopped. +pub fn placement(alt: [u64; 3], act_flags: u64, rsp: u64) -> (Option, [u64; 3]) { + let [sp, flags, size] = alt; + let on_alt = flags & SS_DISABLE == 0 && rsp.wrapping_sub(sp) < size; + let alt_top = (act_flags & SA_ONSTACK != 0 && flags & SS_DISABLE == 0 && !on_alt) + .then(|| sp.saturating_add(size)); + let ss_flags = if flags & SS_DISABLE != 0 { + SS_DISABLE + } else if on_alt { + SS_ONSTACK + } else { + 0 + }; + (alt_top, [sp, ss_flags, size]) +} diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index d9fc6b295..50c466107 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -18,6 +18,9 @@ mod answer; mod deliver; +mod deliver_enter; +mod deliver_say; +mod deliver_stack; mod dispatch; mod family; mod family_pipes; @@ -35,6 +38,7 @@ mod table_link; mod table_mem; mod table_net; mod table_proc; +mod table_sig; mod tally; mod unserved; diff --git a/userland/capsule_linux/src/linux/serve/table.rs b/userland/capsule_linux/src/linux/serve/table.rs index 6574bcb25..791be3023 100644 --- a/userland/capsule_linux/src/linux/serve/table.rs +++ b/userland/capsule_linux/src/linux/serve/table.rs @@ -25,6 +25,7 @@ use super::table_link::link_ops; use super::table_mem::mem_ops; use super::table_net::net_ops; use super::table_proc::proc_ops; +use super::table_sig::sig_ops; pub fn plain(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> u64 { if let Some(v) = file_ops(guest, tid, nr, a) { @@ -42,6 +43,9 @@ pub fn plain(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> u64 { if let Some(v) = proc_ops(guest, nr, a) { return v; } + if let Some(v) = sig_ops(guest, tid, nr, a) { + return v; + } rest(guest, tid, nr, a) } @@ -53,9 +57,6 @@ fn rest(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> u64 { np::GETRLIMIT => call::getrlimit(guest, a[0], a[1]), np::UMASK => call::umask(guest, a[0]), np::PRLIMIT64 => call::prlimit64(guest, a[1], a[2], a[3]), - nr::RT_SIGACTION => call::rt_sigaction(guest, a[0], a[1], a[2]), - nr::RT_SIGPROCMASK => call::rt_sigprocmask(guest, a[2]), - nr::SIGALTSTACK => call::sigaltstack(guest, a[1]), nr::RSEQ | nr::SET_ROBUST_LIST => errno::ok(0), nr::ARCH_PRCTL => call::arch_prctl(guest, tid, a[0], a[1]), nr::GETRANDOM => call::getrandom(guest, a[0], a[1], a[2]), diff --git a/userland/capsule_linux/src/linux/serve/table_sig.rs b/userland/capsule_linux/src/linux/serve/table_sig.rs new file mode 100644 index 000000000..8ee668761 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/table_sig.rs @@ -0,0 +1,31 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Signal dispositions, masks and stacks: the calls of that family that +//! answer at once. + +use crate::linux::abi::nr; +use crate::linux::call; +use crate::linux::guest::Guest; + +pub fn sig_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option { + Some(match nr { + nr::RT_SIGACTION => call::rt_sigaction(guest, a[0], a[1], a[2], a[3]), + nr::RT_SIGPROCMASK => call::rt_sigprocmask(guest, tid, a[0], a[1], a[2], a[3]), + nr::SIGALTSTACK => call::sigaltstack(guest, tid, a[0], a[1]), + _ => return None, + }) +} From 3af1d3cc3b936589de08186e4ee754a33e2b88fc Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:02:42 +0000 Subject: [PATCH 04/26] linux: kill reaches every process of the family, not only the caller kill, tkill and tgkill looked up the target's disposition in the calling process and queued the signal in the caller's own queue under the target's pid. A signal sent to a child was therefore judged by the parent's handlers and never seen by the child. A signal whose default is fatal killed only the one thread it named, and left the rest of its process running with the process never marked as ended. A signal for the caller's own process is now queued there, for a thread or for the whole process, and taken by whichever thread may take it. A signal for any other process leaves through an outbox with the caller parked; the family raises it in every process the target names, a pid, a thread, a process group or all of them, and answers the caller 0 or ESRCH, counting an ended child not yet waited for as Linux counts a zombie. The target's own disposition decides what happens, and a fatal default ends the whole process. rt_sigqueueinfo and rt_tgsigqueueinfo carry the caller's siginfo and value, with Linux's rule that only a process signalling itself may claim a kernel si_code. pid_map translates the pids these calls name. The new routes in route_life are not asked yet, so kill keeps a form for callers that cannot park, and tgkill_from and the sigqueue calls have no caller until they are. --- userland/capsule_linux/src/linux/abi/mod.rs | 1 + .../capsule_linux/src/linux/abi/nr_sig.rs | 37 ++++++++++ userland/capsule_linux/src/linux/call/mod.rs | 5 +- .../src/linux/call/signal_post.rs | 45 ++++++++++++ .../src/linux/call/signal_queue.rs | 72 +++++++++++++++++++ .../src/linux/call/signal_send.rs | 71 +++++++++--------- .../capsule_linux/src/linux/guest/siginfo.rs | 1 + .../src/linux/serve/family_reap.rs | 62 +++++++--------- .../src/linux/serve/family_reap_end.rs | 55 ++++++++++++++ .../src/linux/serve/family_signal.rs | 26 +++++++ .../src/linux/serve/family_signal_route.rs | 61 ++++++++++++++++ .../src/linux/serve/family_wait.rs | 29 ++++++++ userland/capsule_linux/src/linux/serve/mod.rs | 5 ++ .../capsule_linux/src/linux/serve/pid_map.rs | 4 +- .../src/linux/serve/route_life.rs | 41 +++++++++++ 15 files changed, 443 insertions(+), 72 deletions(-) create mode 100644 userland/capsule_linux/src/linux/abi/nr_sig.rs create mode 100644 userland/capsule_linux/src/linux/call/signal_post.rs create mode 100644 userland/capsule_linux/src/linux/call/signal_queue.rs create mode 100644 userland/capsule_linux/src/linux/serve/family_reap_end.rs create mode 100644 userland/capsule_linux/src/linux/serve/family_signal.rs create mode 100644 userland/capsule_linux/src/linux/serve/family_signal_route.rs create mode 100644 userland/capsule_linux/src/linux/serve/family_wait.rs create mode 100644 userland/capsule_linux/src/linux/serve/route_life.rs diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs index fe250836a..4968980a8 100644 --- a/userland/capsule_linux/src/linux/abi/mod.rs +++ b/userland/capsule_linux/src/linux/abi/mod.rs @@ -23,3 +23,4 @@ pub mod name; pub mod nr; pub mod nr_path; pub mod nr_high; +pub mod nr_sig; diff --git a/userland/capsule_linux/src/linux/abi/nr_sig.rs b/userland/capsule_linux/src/linux/abi/nr_sig.rs new file mode 100644 index 000000000..9a5f9e65a --- /dev/null +++ b/userland/capsule_linux/src/linux/abi/nr_sig.rs @@ -0,0 +1,37 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Syscall numbers for process lifecycle, signals and timers, transcribed +//! from the x86_64 table. + +pub const PAUSE: u64 = 34; +pub const GETITIMER: u64 = 36; +pub const ALARM: u64 = 37; +pub const SETITIMER: u64 = 38; +pub const KILL: u64 = 62; +pub const RT_SIGPENDING: u64 = 127; +pub const RT_SIGTIMEDWAIT: u64 = 128; +pub const RT_SIGQUEUEINFO: u64 = 129; +pub const RT_SIGSUSPEND: u64 = 130; +pub const TKILL: u64 = 200; +pub const TIMER_CREATE: u64 = 222; +pub const TIMER_SETTIME: u64 = 223; +pub const TIMER_GETTIME: u64 = 224; +pub const TIMER_GETOVERRUN: u64 = 225; +pub const TIMER_DELETE: u64 = 226; +pub const TGKILL: u64 = 234; +pub const WAITID: u64 = 247; +pub const RT_TGSIGQUEUEINFO: u64 = 297; diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index f01f4ccd1..6866e9c55 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -43,6 +43,8 @@ pub mod sigframe_build; mod sigframe_read; mod signal; mod signal_mask; +mod signal_post; +mod signal_queue; mod signal_send; mod signal_stack; mod sigreturn; @@ -73,9 +75,10 @@ pub use pipe_wait::{is_pipe, read_or_park as pipe_read_or_park}; pub use session::{getpgid, getsid, setpgid, setsid}; pub use signal::rt_sigaction; pub use signal_mask::rt_sigprocmask; +pub use signal_queue::{rt_sigqueueinfo, rt_tgsigqueueinfo}; +pub use signal_send::{kill, kill_from, tgkill_from}; pub use signal_stack::sigaltstack; pub use sigreturn::rt_sigreturn; -pub use signal_send::kill; pub use sleep::{clock_nanosleep, nanosleep}; pub use spawn::{clone, execve, fork, reap_one, wait4}; pub use clock::{clock_getres, clock_gettime, now_ms}; diff --git a/userland/capsule_linux/src/linux/call/signal_post.rs b/userland/capsule_linux/src/linux/call/signal_post.rs new file mode 100644 index 000000000..db511df84 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_post.rs @@ -0,0 +1,45 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a sent signal goes: queued here for the caller's own process, or +//! through the outbox with the caller parked, for the family to route and +//! answer once it knows whether anyone was there. + +use crate::linux::abi::errno; +use crate::linux::guest::siginfo::SigInfo; +use crate::linux::guest::sigwaits::{Outbound, Target}; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +/// Queue `info` for `to`: here when it is this process, else through the +/// family. A signal number of 0 only asks whether the target exists. +pub fn post(guest: &mut Guest, tid: u32, to: Target, info: SigInfo) -> Answer { + let here = match to { + Target::Process(p) if guest.owns(p) => Some(0), + Target::Thread(g, t) if guest.owns(t) && (g == 0 || g == guest.pid) => Some(t), + _ => None, + }; + let Some(t) = here else { + guest.signals.outbox.push(Outbound { from: tid, to, info }); + return Answer::Park; + }; + /* A realtime signal past the queue limit is refused, as Linux does. */ + let s = info.signo; + if s != 0 && !guest.signals.raise(t, info) && s >= 32 && !guest.signals.discards(s) { + return Answer::value(errno::fail(errno::EAGAIN)); + } + Answer::value(errno::ok(0)) +} diff --git a/userland/capsule_linux/src/linux/call/signal_queue.rs b/userland/capsule_linux/src/linux/call/signal_queue.rs new file mode 100644 index 000000000..1aadf65c4 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_queue.rs @@ -0,0 +1,72 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `rt_sigqueueinfo` and `rt_tgsigqueueinfo`: a signal with the siginfo the +//! caller wrote, which is how sigqueue hands a value along. As on Linux, only +//! a process signalling itself may claim a kernel code (si_code 0 or above) or +//! SI_TKILL; the signal number is the call's, whatever the siginfo says. + +use crate::linux::abi::errno; +use crate::linux::guest::siginfo::{SigInfo, SI_TKILL}; +use crate::linux::guest::sigstate::NSIG; +use crate::linux::guest::sigwaits::Target; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +pub fn rt_sigqueueinfo(guest: &mut Guest, tid: u32, tgid: u64, sig: u64, uinfo: u64) -> Answer { + queue(guest, tid, Target::Process(tgid as u32), tgid, sig, uinfo) +} + +pub fn rt_tgsigqueueinfo( + guest: &mut Guest, + tid: u32, + tgid: u64, + target: u64, + sig: u64, + uinfo: u64, +) -> Answer { + if (tgid as i64) <= 0 || (target as i64) <= 0 { + return Answer::value(errno::fail(errno::EINVAL)); + } + queue(guest, tid, Target::Thread(tgid as u32, target as u32), tgid, sig, uinfo) +} + +fn queue(guest: &mut Guest, tid: u32, to: Target, tgid: u64, sig: u64, uinfo: u64) -> Answer { + if sig > NSIG as u64 { + return Answer::value(errno::fail(errno::EINVAL)); + } + let Some(raw) = guest.read(uinfo, 32) else { + return Answer::value(errno::fail(errno::EFAULT)); + }; + let word = |i: usize| u32::from_le_bytes(raw[i..i + 4].try_into().unwrap_or([0; 4])); + let code = word(8) as i32; + /* Linux compares the caller's own thread with the one it names. */ + let named = match to { + Target::Thread(_, t) => t, + _ => tgid as u32, + }; + if (code >= 0 || code == SI_TKILL) && named != tid { + return Answer::value(errno::fail(errno::EPERM)); + } + let info = SigInfo { + signo: sig as u8, + code, + pid: crate::linux::serve::kernel_pid(word(16)).unwrap_or(0), + timer: None, + value: u64::from_le_bytes(raw[24..32].try_into().unwrap_or([0; 8])), + }; + super::signal_post::post(guest, tid, to, info) +} diff --git a/userland/capsule_linux/src/linux/call/signal_send.rs b/userland/capsule_linux/src/linux/call/signal_send.rs index 20979e1aa..0b7107973 100644 --- a/userland/capsule_linux/src/linux/call/signal_send.rs +++ b/userland/capsule_linux/src/linux/call/signal_send.rs @@ -14,50 +14,53 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `kill`, `tkill` and `tgkill`, for the guest's own threads and children. -//! A signal the process catches is queued and delivered on that thread's next -//! return; one whose default is to ignore is dropped; a fatal default ends it. - -use nonos_libc::mk_kill; +//! `kill`, `tkill` and `tgkill`. A signal for the caller's own +//! process is queued here and taken as `serve::deliver` decides; one for +//! another process of the family leaves through the outbox with the caller +//! parked, and the family answers it once it knows whether anyone was there. +//! A guest reaches only its own family: pid_map refuses any other number. +use super::signal_post::post; use crate::linux::abi::errno; -use crate::linux::guest::siginfo::{SigInfo, SI_USER}; +use crate::linux::guest::siginfo::{SigInfo, SI_TKILL, SI_USER}; use crate::linux::guest::sigstate::NSIG; +use crate::linux::guest::sigwaits::Target; use crate::linux::guest::Guest; +use crate::linux::serve::Answer; -/// Signals whose default action is to be ignored: child status, urgent data, -/// window size, and a continue with nothing stopped. -const IGNORED_DEFAULT: [u64; 4] = [17, 23, 28, 18]; +/// kill(pid, sig) from thread `tid`, which parks when the family must answer. +pub fn kill_from(guest: &mut Guest, tid: u32, pid: u64, signo: u64) -> Answer { + let to = match pid as i64 { + -1 => Target::All, + 0 => Target::Group(guest.pgid), + p if p < 0 => Target::Group(p.unsigned_abs() as u32), + p => Target::Process(p as u32), + }; + send(guest, tid, to, signo, SI_USER) +} +/// kill for a caller that cannot park: a signal for another process still +/// goes to the family, with no one to answer, so a target that is gone reads +/// as 0 here rather than ESRCH. pub fn kill(guest: &mut Guest, pid: u64, signo: u64) -> u64 { - let target = pid as u32; - // A guest may signal itself, its threads and its children, nothing else. - if !guest.owns(target) && !guest.children.contains(&target) { - return errno::fail(errno::ESRCH); - } - if signo == 0 { - return errno::ok(0); // an existence check, not a signal - } - if signo > NSIG as u64 { - return errno::fail(errno::EINVAL); + match kill_from(guest, 0, pid, signo) { + Answer::Reply(v) => v, + Answer::Park => errno::ok(0), } - let act = guest.signals.action(signo as usize).unwrap_or_default(); - if act.catches() { - let _ = guest.signals.raise(target, SigInfo::from(signo as u8, SI_USER, guest.pid)); - return errno::ok(0); - } - if act.ignores() || IGNORED_DEFAULT.contains(&signo) { - return errno::ok(0); +} + +/// tkill(tid, sig), and tgkill(tgid, tid, sig) with `tgid` non-zero. +pub fn tgkill_from(guest: &mut Guest, tid: u32, tgid: u64, target: u64, signo: u64) -> Answer { + if (tgid as i64) < 0 || (target as i64) <= 0 { + return Answer::value(errno::fail(errno::EINVAL)); } - terminate(guest, target, signo) + send(guest, tid, Target::Thread(tgid as u32, target as u32), signo, SI_TKILL) } -/// The default action of an uncaught, non-ignored signal is to end the thread. -fn terminate(guest: &mut Guest, target: u32, signo: u64) -> u64 { - guest.forget_thread(target); - guest.threads.retain(|t| *t != target); - match mk_kill(target as u64, signo) { - n if n < 0 => errno::fail(errno::EPERM), - _ => errno::ok(0), +fn send(guest: &mut Guest, tid: u32, to: Target, signo: u64, code: i32) -> Answer { + if signo > NSIG as u64 { + return Answer::value(errno::fail(errno::EINVAL)); } + let info = SigInfo::from(signo as u8, code, guest.pid); + post(guest, tid, to, info) } diff --git a/userland/capsule_linux/src/linux/guest/siginfo.rs b/userland/capsule_linux/src/linux/guest/siginfo.rs index 7c6188ea3..936a11da3 100644 --- a/userland/capsule_linux/src/linux/guest/siginfo.rs +++ b/userland/capsule_linux/src/linux/guest/siginfo.rs @@ -20,6 +20,7 @@ //! kernel pid in one. pub const SI_USER: i32 = 0; +pub const SI_TKILL: i32 = -6; pub const INFO_LEN: usize = 128; #[derive(Clone, Copy, Default)] diff --git a/userland/capsule_linux/src/linux/serve/family_reap.rs b/userland/capsule_linux/src/linux/serve/family_reap.rs index 831a95f57..aea13de1a 100644 --- a/userland/capsule_linux/src/linux/serve/family_reap.rs +++ b/userland/capsule_linux/src/linux/serve/family_reap.rs @@ -14,51 +14,41 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Ending what has exited, and telling each parent. +//! Ending what has exited, and telling each parent: the ended child waits +//! until it is waited for, and a wait4 parked for it is answered. -use nonos_libc::{mk_foreign_reply, mk_kill}; +use nonos_libc::mk_foreign_reply; use super::family::Family; +use super::pid_ns::PidNs; +use super::pid_out::value_out; +use crate::linux::abi::nr; use crate::linux::call::reap_one; - -const SIGKILL: u64 = 9; +use crate::linux::guest::Guest; impl Family { - /// End every process that asked to, and tell its parent. + /// End every process that asked to, deliver the signals that follow, + /// and go again while that ends anything more. pub fn reap(&mut self) { self.settle_pipes(); - while let Some(i) = self.guests.iter().position(|g| g.exited.is_some()) { - let gone = self.guests.remove(i); - let code = gone.exited.unwrap_or(0); - for tid in gone.threads.iter().chain([gone.pid].iter()) { - let rc = mk_kill(*tid as u64, SIGKILL); - if rc < 0 { - // Refused, it runs on after its process ended. - let line = alloc::format!( - "[LINUX] kill refused: pid {tid} outlives its process, errno {}\n", - -rc - ); - let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); - } - } - if gone.pid == self.root { - self.root_code = code; - } - let Some(p) = self.guests.iter_mut().find(|g| g.children.contains(&gone.pid)) else { - continue; - }; - p.ended.push((gone.pid, code)); - if let Some((want, status, tid)) = p.waiting { - if let Some(value) = reap_one(p, want, status) { - p.waiting = None; - let value = super::pid_out::value_out( - &mut self.ns, - crate::linux::abi::nr::WAIT4, - value, - ); - let _ = mk_foreign_reply(tid, value); - } + loop { + let ended = self.end_exited(); + self.settle_signals(); + if !ended && !self.guests.iter().any(|g| g.exited.is_some()) { + return; } } } } + +/// Tell `p` that `gone` ended with `code`: kept until it is waited for, and a +/// parked wait4 answered with its pid in the guest's numbering. +pub fn tell_parent(p: &mut Guest, gone: &Guest, code: i32, ns: &mut PidNs) { + p.ended.push((gone.pid, code)); + if let Some((want, status, tid)) = p.waiting { + if let Some(value) = reap_one(p, want, status) { + p.waiting = None; + let _ = mk_foreign_reply(tid, value_out(ns, nr::WAIT4, value)); + } + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_reap_end.rs b/userland/capsule_linux/src/linux/serve/family_reap_end.rs new file mode 100644 index 000000000..47ef2a5c6 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_reap_end.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Ending each process that has exited: its threads are killed and its parent +//! is told as family_reap tells it. + +use nonos_libc::mk_kill; + +use super::family::Family; +use super::family_reap::tell_parent; + +const SIGKILL: u64 = 9; + +impl Family { + pub(super) fn end_exited(&mut self) -> bool { + let mut any = false; + while let Some(i) = self.guests.iter().position(|g| g.exited.is_some()) { + any = true; + let gone = self.guests.remove(i); + let code = gone.exited.unwrap_or(0); + for tid in gone.threads.iter().chain([gone.pid].iter()) { + let rc = mk_kill(*tid as u64, SIGKILL); + if rc < 0 { + /* Refused, it runs on after its process ended. */ + let line = alloc::format!( + "[LINUX] kill refused: pid {tid} outlives its process, errno {}\n", + -rc + ); + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + } + } + if gone.pid == self.root { + self.root_code = code; + } + let Some(p) = self.guests.iter_mut().find(|g| g.children.contains(&gone.pid)) else { + continue; + }; + tell_parent(p, &gone, code, &mut self.ns); + } + any + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_signal.rs b/userland/capsule_linux/src/linux/serve/family_signal.rs new file mode 100644 index 000000000..0b757afc7 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_signal.rs @@ -0,0 +1,26 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Signals between processes of the family, settled after every answer: the +//! outbox is routed to every process each signal names. + +use super::family::Family; + +impl Family { + pub(super) fn settle_signals(&mut self) { + self.route_outbox(); + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_signal_route.rs b/userland/capsule_linux/src/linux/serve/family_signal_route.rs new file mode 100644 index 000000000..4e5268510 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_signal_route.rs @@ -0,0 +1,61 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A signal from the outbox reaches every process its target names, and its +//! sender is answered 0, or ESRCH when it named no one, as kill answers; an +//! ended child not yet waited for counts, as a zombie does on Linux. + +use super::family::Family; +use super::family_wait::answer; +use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::{Outbound, Target}; +use crate::linux::guest::Guest; + +impl Family { + pub(super) fn route_outbox(&mut self) { + let mut out = alloc::vec::Vec::new(); + for g in self.guests.iter_mut() { + out.extend(core::mem::take(&mut g.signals.outbox).into_iter().map(|o| (g.pid, o))); + } + for (sender, o) in out { + let zombie = |pid: u32| self.guests.iter().any(|g| g.ended.iter().any(|e| e.0 == pid)); + let mut reached = + matches!(o.to, Target::Process(p) | Target::Thread(_, p) if zombie(p)); + for g in self.guests.iter_mut() { + if let Some(t) = taker(g, sender, &o) { + reached = true; + if o.info.signo != 0 && g.exited.is_none() { + let _ = g.signals.raise(t, o.info); + } + } + } + let value = if reached { 0 } else { errno::fail(errno::ESRCH) }; + if let Some(g) = self.guests.iter_mut().find(|g| g.owns(o.from)) { + answer(g, o.from, value); + } + } + } +} + +/// The thread of `g` a signal is for, 0 for the whole process, or None. +fn taker(g: &Guest, sender: u32, o: &Outbound) -> Option { + match o.to { + Target::Process(p) => g.owns(p).then_some(0), + Target::Thread(tgid, t) => (g.owns(t) && (tgid == 0 || tgid == g.pid)).then_some(t), + Target::Group(pg) => (g.pgid == pg).then_some(0), + Target::All => (g.pid != sender).then_some(0), + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_wait.rs b/userland/capsule_linux/src/linux/serve/family_wait.rs new file mode 100644 index 000000000..b62e5b3bc --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_wait.rs @@ -0,0 +1,29 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Answering a thread the family held parked: with the value its call +//! returns, or by entering the handler of a signal it now takes. + +use nonos_libc::mk_foreign_reply; + +use crate::linux::guest::Guest; + +/// Reply to a parked thread, or enter the handler of a signal it now takes. +pub fn answer(g: &mut Guest, tid: u32, value: u64) { + if !super::deliver::maybe_deliver(g, tid, value) { + let _ = mk_foreign_reply(tid, value); + } +} diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index 50c466107..6bd4175c0 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -25,13 +25,18 @@ mod dispatch; mod family; mod family_pipes; mod family_reap; +mod family_reap_end; +mod family_signal; +mod family_signal_route; mod family_sleep; +mod family_wait; mod loop_impl; mod pid_map; mod pid_ns; mod pid_space; mod refused; mod pid_out; +mod route_life; mod table; mod table_file; mod table_link; diff --git a/userland/capsule_linux/src/linux/serve/pid_map.rs b/userland/capsule_linux/src/linux/serve/pid_map.rs index 9321d1d6c..2a10005c4 100644 --- a/userland/capsule_linux/src/linux/serve/pid_map.rs +++ b/userland/capsule_linux/src/linux/serve/pid_map.rs @@ -22,7 +22,7 @@ use nonos_libc::ForeignFrame; -use crate::linux::abi::{errno, nr, nr_path as np}; +use crate::linux::abi::{errno, nr, nr_path as np, nr_sig as ns}; use super::pid_ns::PidNs; @@ -41,6 +41,8 @@ pub fn frame_in(ns: &PidNs, frame: &ForeignFrame) -> Option { fn args_in(ns: &PidNs, call: u64, a: &mut [u64; 6]) -> Result<(), u64> { let (slots, missing): (&[usize], i64) = match call { nr::WAIT4 => (&[0], errno::ECHILD), + ns::TGKILL | ns::RT_TGSIGQUEUEINFO => (&[0, 1], errno::ESRCH), + ns::RT_SIGQUEUEINFO => (&[0], errno::ESRCH), np::KILL | np::TKILL | np::GETPGID | np::GETSID => (&[0], errno::ESRCH), np::SETPGID => (&[0, 1], errno::ESRCH), _ => return Ok(()), diff --git a/userland/capsule_linux/src/linux/serve/route_life.rs b/userland/capsule_linux/src/linux/serve/route_life.rs new file mode 100644 index 000000000..50db389f6 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/route_life.rs @@ -0,0 +1,41 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The calls of process lifecycle and signals that can leave their caller +//! parked: a signal sent where only the family can say whether anyone +//! received it. Asked first by `dispatch`, so these are answered here +//! whatever it holds. + +use nonos_libc::ForeignFrame; + +use super::answer::Answer; +use crate::linux::abi::errno; +use crate::linux::abi::nr_sig as ns; +use crate::linux::call; +use crate::linux::guest::Guest; + +pub fn answer(guest: &mut Guest, frame: &ForeignFrame) -> Option { + let (a, tid) = (frame.args(), frame.pid); + Some(match frame.nr { + ns::KILL => call::kill_from(guest, tid, a[0], a[1]), + ns::TKILL => call::tgkill_from(guest, tid, 0, a[0], a[1]), + ns::TGKILL if (a[0] as i64) <= 0 => Answer::value(errno::fail(errno::EINVAL)), + ns::TGKILL => call::tgkill_from(guest, tid, a[0], a[1], a[2]), + ns::RT_SIGQUEUEINFO => call::rt_sigqueueinfo(guest, tid, a[0], a[1], a[2]), + ns::RT_TGSIGQUEUEINFO => call::rt_tgsigqueueinfo(guest, tid, a[0], a[1], a[2], a[3]), + _ => return None, + }) +} From 294461d13b91bbe076fea47b9f6952a3cf29760e Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:02:52 +0000 Subject: [PATCH 05/26] linux: a leader's plain exit ends only the leader, status is linux's A plain exit from the leader was treated as exit_group: the whole process ended with the leader, taking every other thread with it, where Linux ends only the leader and lets the process run until its last thread exits. A program whose main thread leaves with a plain exit while other threads still run was cut short. The status a process ended with was kept as a bare exit code, or as 128 plus the signal, so a waiting parent could not tell an exit from a signal. exit_one ends only the calling thread. The leader cannot be killed while other threads run, since every peer call names its pid as the address space, so it clears its tid word, wakes its joiner and stays parked as a zombie until the last thread exits; that thread's code is the process's status. What a process ended with is now kept as Linux's wait status word: the exit code in the second byte, or the signal's number in the first. The reap logs which it was, and the first guest's code is taken from it, as 128 plus the signal for a signal. A fault reported by the kernel is still kept as 128 plus the signal, which reads as that signal. exit_one is routed from route_life, which dispatch does not ask yet. --- userland/capsule_linux/src/linux/call/life.rs | 14 +++--- .../capsule_linux/src/linux/call/life_one.rs | 46 +++++++++++++++++++ userland/capsule_linux/src/linux/call/mod.rs | 2 + .../src/linux/call/spawn/wait.rs | 6 +-- .../capsule_linux/src/linux/guest/siglive.rs | 14 +++++- .../src/linux/serve/family_reap.rs | 8 ++-- .../src/linux/serve/family_reap_end.rs | 17 +++++-- .../src/linux/serve/route_life.rs | 13 ++++-- 8 files changed, 97 insertions(+), 23 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/life_one.rs diff --git a/userland/capsule_linux/src/linux/call/life.rs b/userland/capsule_linux/src/linux/call/life.rs index 740a4d539..c0b015e8e 100644 --- a/userland/capsule_linux/src/linux/call/life.rs +++ b/userland/capsule_linux/src/linux/call/life.rs @@ -14,8 +14,10 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Ending a guest. The call never returns to the guest, so the answer -//! handed back is only what parks it until the supervisor tears it down. +//! Ending a thread or a guest. The call never returns to the guest, so the +//! answer handed back is only what parks it until the supervisor tears it +//! down. A process's end is kept as Linux's wait status: an exit's code in +//! the second byte, or the number of the signal that ended it in the first. use nonos_libc::mk_kill; @@ -58,15 +60,15 @@ pub fn set_tid_address(guest: &mut Guest, tid: u32, word: u64) -> Answer { Answer::value(u64::from(tid)) } +/// exit_group: the whole process ends with `code`. pub fn exit(guest: &mut Guest, code: u64) -> u64 { - guest.exited = Some(code as i32); + guest.exited = Some(((code & 0xff) << 8) as i32); errno::ok(0) } -/// The process ends on `signum`, unless something already ended it. Kept in -/// the shell's 128+signo form, as a thread's fatal fault is. +/// The process ends on `signum`, unless something already ended it. pub fn killed(guest: &mut Guest, signum: u8) { if guest.exited.is_none() { - guest.exited = Some(128 + i32::from(signum & 0x7f)); + guest.exited = Some(i32::from(signum & 0x7f)); } } diff --git a/userland/capsule_linux/src/linux/call/life_one.rs b/userland/capsule_linux/src/linux/call/life_one.rs new file mode 100644 index 000000000..4b1df4086 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/life_one.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A plain exit, which ends the calling thread only. The process ends with +//! the last of its threads, and that thread's code is its status. + +use super::life::{exit, exit_thread}; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +/// A plain exit ends the calling thread only; the process ends with the last +/// of its threads, and that thread's code is its status. The leader cannot be +/// killed while others run: its pid is the address space every peer call +/// names. So it stays parked inside its exit, a zombie that runs nothing, +/// until the process ends and takes it. +pub fn exit_one(guest: &mut Guest, tid: u32, code: u64) -> Answer { + if tid == guest.pid { + if let Some(at) = guest.clear_tids.iter().position(|(t, _)| *t == tid) { + let (_, word) = guest.clear_tids.remove(at); + if guest.write(word, &0u32.to_le_bytes()) == 4 { + guest.wake(word, 1); + } + } + guest.signals.leader_gone = true; + } else { + let _ = exit_thread(guest, tid); + } + guest.forget_thread(tid); + if guest.live_threads().is_empty() { + let _ = exit(guest, code); + } + Answer::Park +} diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index 6866e9c55..ac8290181 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -26,6 +26,7 @@ mod ident; mod io; mod io_socket; mod life; +mod life_one; mod limits; mod limits_table; mod glibc; @@ -63,6 +64,7 @@ pub use futex::futex; pub use ident::{getppid, setuid}; pub use io::{close, read, write}; pub use life::{exit, exit_thread, killed, set_tid_address}; +pub use life_one::exit_one; pub use limits::{getrlimit, prlimit64}; pub use glibc::prctl; pub use glibc_sched::{clone3, getcpu, membarrier, sched_getaffinity}; diff --git a/userland/capsule_linux/src/linux/call/spawn/wait.rs b/userland/capsule_linux/src/linux/call/spawn/wait.rs index 9b518d8e0..fbe07aeff 100644 --- a/userland/capsule_linux/src/linux/call/spawn/wait.rs +++ b/userland/capsule_linux/src/linux/call/spawn/wait.rs @@ -43,10 +43,10 @@ pub fn wait4(guest: &mut Guest, want: u64, status: u64, flags: u64, tid: u32) -> pub fn reap_one(guest: &mut Guest, want: u64, status: u64) -> Option { let any = (want as i64) <= 0; let at = guest.ended.iter().position(|(pid, _)| any || *pid == want as u32)?; - let (pid, code) = guest.ended.remove(at); + let (pid, kept) = guest.ended.remove(at); guest.children.retain(|p| *p != pid); - // An exit status sits in the second byte, as WEXITSTATUS reads it. - let word = ((code as u32) & 0xff) << 8; + /* Kept as Linux's wait status word: an exit's code in the second byte. */ + let word = kept as u32; if status != 0 && guest.write(status, &word.to_le_bytes()) < 4 { return Some(errno::fail(errno::EFAULT)); } diff --git a/userland/capsule_linux/src/linux/guest/siglive.rs b/userland/capsule_linux/src/linux/guest/siglive.rs index ab7e784a6..bfe811c33 100644 --- a/userland/capsule_linux/src/linux/guest/siglive.rs +++ b/userland/capsule_linux/src/linux/guest/siglive.rs @@ -14,11 +14,23 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! A thread that has gone, taking its signal state and its waits with it. +//! Which threads of a process can take a signal, and a thread that has gone +//! taking its signal state and its waits with it. + +use alloc::vec::Vec; use super::handle::Guest; impl Guest { + /// The leader, unless it has made a plain exit, and every other thread. + pub fn live_threads(&self) -> Vec { + let mut all = self.threads.clone(); + if !self.signals.leader_gone { + all.insert(0, self.pid); + } + all + } + /// A thread that has gone takes its signal state and its waits with it. pub fn forget_thread(&mut self, tid: u32) { let _ = self.leave_waits(tid); diff --git a/userland/capsule_linux/src/linux/serve/family_reap.rs b/userland/capsule_linux/src/linux/serve/family_reap.rs index aea13de1a..3d626953a 100644 --- a/userland/capsule_linux/src/linux/serve/family_reap.rs +++ b/userland/capsule_linux/src/linux/serve/family_reap.rs @@ -41,10 +41,10 @@ impl Family { } } -/// Tell `p` that `gone` ended with `code`: kept until it is waited for, and a -/// parked wait4 answered with its pid in the guest's numbering. -pub fn tell_parent(p: &mut Guest, gone: &Guest, code: i32, ns: &mut PidNs) { - p.ended.push((gone.pid, code)); +/// Tell `p` that `gone` ended with `status`: kept until it is waited for, and +/// a parked wait4 answered with its pid in the guest's numbering. +pub fn tell_parent(p: &mut Guest, gone: &Guest, status: i32, ns: &mut PidNs) { + p.ended.push((gone.pid, status)); if let Some((want, status, tid)) = p.waiting { if let Some(value) = reap_one(p, want, status) { p.waiting = None; diff --git a/userland/capsule_linux/src/linux/serve/family_reap_end.rs b/userland/capsule_linux/src/linux/serve/family_reap_end.rs index 47ef2a5c6..662adcc60 100644 --- a/userland/capsule_linux/src/linux/serve/family_reap_end.rs +++ b/userland/capsule_linux/src/linux/serve/family_reap_end.rs @@ -14,8 +14,8 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Ending each process that has exited: its threads are killed and its parent -//! is told as family_reap tells it. +//! Ending each process that has exited: its threads are killed, its end is +//! said in the log, and its parent is told as family_reap tells it. use nonos_libc::mk_kill; @@ -30,7 +30,7 @@ impl Family { while let Some(i) = self.guests.iter().position(|g| g.exited.is_some()) { any = true; let gone = self.guests.remove(i); - let code = gone.exited.unwrap_or(0); + let status = gone.exited.unwrap_or(0); for tid in gone.threads.iter().chain([gone.pid].iter()) { let rc = mk_kill(*tid as u64, SIGKILL); if rc < 0 { @@ -42,13 +42,20 @@ impl Family { let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); } } + let (code, signo) = ((status >> 8) & 0xff, status & 0x7f); + let shown = crate::linux::serve::guest_pid(gone.pid); + let line = match signo { + 0 => alloc::format!("[LINUX] process {shown} exited, status {code}\n"), + s => alloc::format!("[LINUX] process {shown} ended by signal {s}\n"), + }; + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); if gone.pid == self.root { - self.root_code = code; + self.root_code = if signo == 0 { code } else { 128 + signo }; } let Some(p) = self.guests.iter_mut().find(|g| g.children.contains(&gone.pid)) else { continue; }; - tell_parent(p, &gone, code, &mut self.ns); + tell_parent(p, &gone, status, &mut self.ns); } any } diff --git a/userland/capsule_linux/src/linux/serve/route_life.rs b/userland/capsule_linux/src/linux/serve/route_life.rs index 50db389f6..762892f79 100644 --- a/userland/capsule_linux/src/linux/serve/route_life.rs +++ b/userland/capsule_linux/src/linux/serve/route_life.rs @@ -15,21 +15,26 @@ // along with this program. If not, see . //! The calls of process lifecycle and signals that can leave their caller -//! parked: a signal sent where only the family can say whether anyone -//! received it. Asked first by `dispatch`, so these are answered here -//! whatever it holds. +//! parked: a plain exit, and a signal sent where only the family can say +//! whether anyone received it. Asked first by `dispatch`, so these are +//! answered here whatever it holds. use nonos_libc::ForeignFrame; use super::answer::Answer; use crate::linux::abi::errno; -use crate::linux::abi::nr_sig as ns; +use crate::linux::abi::{nr, nr_sig as ns}; use crate::linux::call; use crate::linux::guest::Guest; pub fn answer(guest: &mut Guest, frame: &ForeignFrame) -> Option { let (a, tid) = (frame.args(), frame.pid); Some(match frame.nr { + nr::EXIT => call::exit_one(guest, tid, a[0]), + nr::EXIT_GROUP => { + let _ = call::exit(guest, a[0]); + Answer::Park + } ns::KILL => call::kill_from(guest, tid, a[0], a[1]), ns::TKILL => call::tgkill_from(guest, tid, 0, a[0], a[1]), ns::TGKILL if (a[0] as i64) <= 0 => Answer::value(errno::fail(errno::EINVAL)), From 88bd9a132fa0a2640c8ed37d1ae1d3e8e06d19e3 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:02:57 +0000 Subject: [PATCH 06/26] linux: fork copies a span larger than 1 MiB in pieces fork mapped and copied each span of the parent into the child with one peer call. The kernel maps and copies at most MAX_SPAN, 1 MiB, in one call and refuses more, so a parent with any larger span could not fork: a Go process, whose heap arenas are far larger, got ENOMEM from fork and from the clone os/exec uses. A span is now mapped and copied in pieces of at most MAX_SPAN each, with the same protection, so a span of any size crosses whole. --- .../src/linux/call/spawn/fork_copy.rs | 27 +++++++++++++------ 1 file changed, 19 insertions(+), 8 deletions(-) diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs index 3cf86191a..afd482196 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs @@ -16,23 +16,34 @@ //! Copying a parent's spans into the child it just made. -use crate::linux::guest::{Guest, Region}; +use crate::linux::guest::{Guest, Region, MAX_SPAN}; use nonos_libc::peer::{mk_peer_map, mk_peer_write, PEER_PROT_EXEC, PEER_PROT_WRITE}; /// Every span, mapped into the child and then filled from the parent. pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool { let spans = guest.regions.clone(); for span in spans { - // An unbacked reservation has no frames to copy; the child reserves it - // the same way, and its own first access faults a page in. + /* + * An unbacked reservation has no frames to copy; the child reserves it + * the same way, and its own first access faults a page in. + */ if !span.backed { continue; } - if mk_peer_map(child, span.at, span.len, prot_of(&span)) < 0 { - return false; - } - if !copy_one(guest, child, span.at, span.len) { - return false; + /* + * The kernel maps and copies at most MAX_SPAN in one peer call, so a + * larger span, which a Go heap is, crosses in pieces. + */ + let mut done = 0; + while done < span.len { + let take = (span.len - done).min(MAX_SPAN); + if mk_peer_map(child, span.at + done, take, prot_of(&span)) < 0 { + return false; + } + if !copy_one(guest, child, span.at + done, take) { + return false; + } + done += take; } } true From 4c5f99d3e3f290d2f2dd103caa592fa176d69ad5 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:03:01 +0000 Subject: [PATCH 07/26] linux: clone can make a process, and vfork holds its parent clone without CLONE_THREAD answered ENOSYS, and vfork was served as a plain fork that let the parent run on at once. Go's os/exec starts every child with clone(CLONE_VFORK|CLONE_VM|SIGCHLD) on no new stack, so no Go program could run another program. clone without CLONE_THREAD now makes a process copied from its parent, as fork does, and honours CSIGNAL, CLONE_VFORK, CLONE_PARENT_SETTID, CLONE_CHILD_SETTID and CLONE_CHILD_CLEARTID, writing the tids in the guest's numbering. A copy stands in for CLONE_VM with CLONE_VFORK, since such a child may only exec or exit and the parent is held until it does, so neither can see the difference. Anything else is refused by name. vfork, and clone with CLONE_VFORK, park the calling thread until the child execs, answered from the successful exec, or ends, answered from the reap, in both cases with the child's pid. A new process takes its signal state from the thread that made it and keeps the signal it raises at its parent when it ends. These routes are in route_life, which dispatch does not ask yet, so they have no caller until it does. --- userland/capsule_linux/src/linux/call/mod.rs | 2 +- .../src/linux/call/spawn/exec.rs | 7 +- .../src/linux/call/spawn/fork.rs | 36 ++-------- .../src/linux/call/spawn/fork_child.rs | 65 +++++++++++++++++ .../capsule_linux/src/linux/call/spawn/mod.rs | 6 ++ .../src/linux/call/spawn/vfork.rs | 56 +++++++++++++++ .../src/linux/call/spawn/vfork_clone.rs | 70 +++++++++++++++++++ .../src/linux/call/spawn/vfork_flags.rs | 36 ++++++++++ .../src/linux/serve/family_reap_end.rs | 7 +- .../src/linux/serve/route_life.rs | 11 ++- 10 files changed, 260 insertions(+), 36 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/spawn/fork_child.rs create mode 100644 userland/capsule_linux/src/linux/call/spawn/vfork.rs create mode 100644 userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs create mode 100644 userland/capsule_linux/src/linux/call/spawn/vfork_flags.rs diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index ac8290181..211f2401b 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -82,7 +82,7 @@ pub use signal_send::{kill, kill_from, tgkill_from}; pub use signal_stack::sigaltstack; pub use sigreturn::rt_sigreturn; pub use sleep::{clock_nanosleep, nanosleep}; -pub use spawn::{clone, execve, fork, reap_one, wait4}; +pub use spawn::{clone, clone_process, execve, fork, reap_one, vfork, wait4}; pub use clock::{clock_getres, clock_gettime, now_ms}; pub use epoch::{family_ms, mark_start}; pub use thread::{arch_prctl, getrandom}; diff --git a/userland/capsule_linux/src/linux/call/spawn/exec.rs b/userland/capsule_linux/src/linux/call/spawn/exec.rs index 42b7b82f7..d79cab699 100644 --- a/userland/capsule_linux/src/linux/call/spawn/exec.rs +++ b/userland/capsule_linux/src/linux/call/spawn/exec.rs @@ -55,7 +55,12 @@ pub fn execve(guest: &mut Guest, pid: u32, path: u64, argv: u64, envp: u64) -> A } } -/// The new program keeps what Linux keeps of the old one's signals. +/// The new program keeps what Linux keeps of the old one's signals, and a +/// vfork parent waiting on this exec is let go with the child's pid. fn released(guest: &mut Guest, pid: u32) { guest.signals.exec_reset(pid); + if let Some(parent) = guest.signals.vfork.take() { + let child = u64::from(crate::linux::serve::guest_pid(guest.pid)); + let _ = nonos_libc::mk_foreign_reply(parent, child); + } } diff --git a/userland/capsule_linux/src/linux/call/spawn/fork.rs b/userland/capsule_linux/src/linux/call/spawn/fork.rs index f60069907..720da83a6 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork.rs @@ -14,42 +14,18 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `fork`. - -use nonos_libc::{mk_foreign_fork, mk_foreign_resume}; +//! `fork`: a copy of the calling process that raises SIGCHLD when it ends. use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SIGCHLD; use crate::linux::guest::Guest; use crate::linux::serve::Answer; -use super::fork_copy::copy_spans; +use super::fork_child::fork_child; pub fn fork(guest: &mut Guest, caller: u32) -> Answer { - let child = mk_foreign_fork(caller); - if child < 0 { - return Answer::value(errno::fail(errno::ENOMEM)); - } - let child = child as u32; - if !copy_spans(guest, child) { - return Answer::value(errno::fail(errno::ENOMEM)); - } - /* - * The thread pointer is a register, not memory, so copying the spans does - * not carry it. The kernel fork carries the forking thread's own FS to the - * child, which is right whichever thread forked; the personality's single - * fs_base is only the last thread to set one and would be wrong here. - */ - /* - * The child's state goes to the serve loop before the child runs, so its - * first trap finds a guest that owns it. - */ - let mut state = guest.fork_state(child); - state.signals = guest.signals.forked(caller, child); - guest.forked.push(state); - if mk_foreign_resume(child) < 0 { - guest.forked.pop(); - return Answer::value(errno::fail(errno::ENOMEM)); + match fork_child(guest, caller, SIGCHLD, |_| {}) { + Ok(child) => Answer::value(errno::ok(child as u64)), + Err(e) => Answer::value(e), } - guest.children.push(child); - Answer::value(errno::ok(child as u64)) } diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_child.rs b/userland/capsule_linux/src/linux/call/spawn/fork_child.rs new file mode 100644 index 000000000..099f8b964 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/fork_child.rs @@ -0,0 +1,65 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The copy every new process starts as: fork's, vfork's and a clone that +//! makes a process all come here. + +use nonos_libc::{mk_foreign_fork, mk_foreign_resume}; + +use super::fork_copy::copy_spans; +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +/// A new process copied from this one, running, and adopted by the family +/// once this answer is given. It raises `exit_signal` at its parent when it +/// ends. Its signal state is the forking thread's, as Linux's fork gives it. +/// `prep` runs on the child before it runs at all. +pub(super) fn fork_child( + guest: &mut Guest, + caller: u32, + exit_signal: u8, + prep: impl FnOnce(&mut Guest), +) -> Result { + let child = mk_foreign_fork(caller); + if child < 0 { + return Err(errno::fail(errno::ENOMEM)); + } + let child = child as u32; + if !copy_spans(guest, child) { + return Err(errno::fail(errno::ENOMEM)); + } + /* + * The thread pointer is a register, not memory, so copying the spans does + * not carry it. The kernel fork carries the forking thread's own FS to the + * child, which is right whichever thread forked; the personality's single + * fs_base is only the last thread to set one and would be wrong here. + */ + /* + * The child's state goes to the serve loop before the child runs, so its + * first trap finds a guest that owns it. + */ + let mut state = guest.fork_state(child); + state.signals = guest.signals.forked(caller, child); + state.signals.exit_signal = exit_signal; + prep(&mut state); + guest.forked.push(state); + if mk_foreign_resume(child) < 0 { + guest.forked.pop(); + return Err(errno::fail(errno::ENOMEM)); + } + guest.children.push(child); + Ok(child) +} diff --git a/userland/capsule_linux/src/linux/call/spawn/mod.rs b/userland/capsule_linux/src/linux/call/spawn/mod.rs index 8369b1db4..e95ebe95d 100644 --- a/userland/capsule_linux/src/linux/call/spawn/mod.rs +++ b/userland/capsule_linux/src/linux/call/spawn/mod.rs @@ -25,10 +25,16 @@ mod exec_resolve; mod exec_shebang; mod exec_threads; mod fork; +mod fork_child; mod fork_copy; +mod vfork; +mod vfork_clone; +mod vfork_flags; mod wait; pub use clone::clone; pub use exec::execve; pub use fork::fork; +pub use vfork::vfork; +pub use vfork_clone::clone_process; pub use wait::{reap_one, wait4}; diff --git a/userland/capsule_linux/src/linux/call/spawn/vfork.rs b/userland/capsule_linux/src/linux/call/spawn/vfork.rs new file mode 100644 index 000000000..6144e076e --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/vfork.rs @@ -0,0 +1,56 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `vfork`, and the start `clone` shares with it when it makes a process +//! rather than a thread (vfork_clone). The child is a copy, as fork's is: +//! vfork's child shares its parent's memory on Linux, but it may only exec or +//! exit, and the parent sleeps until it does, so what either sees is the +//! same. That is what Go's os/exec asks for with clone(CLONE_VFORK|CLONE_VM). +//! The parent parks here and the family answers it with the child's pid when +//! the child's exec succeeds or the child ends. + +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SIGCHLD; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +use super::fork_child::fork_child; + +pub fn vfork(guest: &mut Guest, caller: u32) -> Answer { + start(guest, caller, SIGCHLD, true, |_| {}) +} + +/// Fork a child and answer with its pid, or park the caller until the child +/// execs or ends when `parks`. +pub(super) fn start( + guest: &mut Guest, + caller: u32, + signal: u8, + parks: bool, + prep: impl FnOnce(&mut Guest), +) -> Answer { + let child = match fork_child(guest, caller, signal, prep) { + Ok(c) => c, + Err(e) => return Answer::value(e), + }; + if !parks { + return Answer::value(errno::ok(u64::from(child))); + } + if let Some(c) = guest.forked.last_mut() { + c.signals.vfork = Some(caller); + } + Answer::Park +} diff --git a/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs b/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs new file mode 100644 index 000000000..c2a3d892a --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `clone` when it makes a process rather than a thread: a copy, as fork's +//! is, on its parent's stack, that raises the signal it named when it ends, +//! and with CLONE_VFORK parks its parent as vfork does. The tid words it +//! names are written as Linux writes them. + +use super::vfork::start; +use super::vfork_flags::{CLONE_CHILD_CLEARTID, CLONE_CHILD_SETTID, CLONE_PARENT_SETTID}; +use super::vfork_flags::{CLONE_VFORK, CLONE_VM, CSIGNAL, SERVED}; +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::NSIG; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +/// clone(flags, stack, parent_tid, child_tid, tls) without CLONE_THREAD. +pub fn clone_process(guest: &mut Guest, caller: u32, a: [u64; 6]) -> Answer { + let (flags, stack) = (a[0], a[1]); + let why = if flags & !SERVED != 0 { + Some("clone: flags beyond a copied process") + } else if flags & CLONE_VM != 0 && flags & CLONE_VFORK == 0 { + Some("clone: a process sharing its parent's memory") + } else if stack != 0 { + Some("clone: a new process on a stack of its own") + } else { + None + }; + if let Some(why) = why { + let line = alloc::format!("[LINUX] unserved {why}, flags {flags:#x}\n"); + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + return Answer::value(errno::fail(errno::ENOSYS)); + } + let signal = flags & CSIGNAL; + if signal > NSIG as u64 { + return Answer::value(errno::fail(errno::EINVAL)); + } + /* The child's own copy of its tid is written before it can read it. */ + let prep = move |child: &mut Guest| { + let seen = crate::linux::serve::guest_pid(child.pid).to_le_bytes(); + if flags & CLONE_CHILD_SETTID != 0 { + let _ = child.write(a[3], &seen); + } + if flags & CLONE_CHILD_CLEARTID != 0 { + child.clear_tids.push((child.pid, a[3])); + } + }; + let answer = start(guest, caller, signal as u8, flags & CLONE_VFORK != 0, prep); + if flags & CLONE_PARENT_SETTID != 0 { + /* Only a child made by this call is in `forked`. */ + if let Some(child) = guest.forked.last().map(|c| c.pid) { + let seen = crate::linux::serve::guest_pid(child).to_le_bytes(); + let _ = guest.write(a[2], &seen); + } + } + answer +} diff --git a/userland/capsule_linux/src/linux/call/spawn/vfork_flags.rs b/userland/capsule_linux/src/linux/call/spawn/vfork_flags.rs new file mode 100644 index 000000000..066bea43a --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/vfork_flags.rs @@ -0,0 +1,36 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The clone flags a new process may be made with, as Linux numbers them. + +pub const CSIGNAL: u64 = 0xff; +pub const CLONE_VM: u64 = 0x100; +pub const CLONE_VFORK: u64 = 0x4000; +pub const CLONE_PARENT_SETTID: u64 = 0x10_0000; +pub const CLONE_CHILD_CLEARTID: u64 = 0x20_0000; +pub const CLONE_DETACHED: u64 = 0x40_0000; +pub const CLONE_UNTRACED: u64 = 0x80_0000; +pub const CLONE_CHILD_SETTID: u64 = 0x100_0000; +/// Every flag honoured for a new process; CLONE_DETACHED and CLONE_UNTRACED +/// are ones Linux itself ignores. +pub const SERVED: u64 = CSIGNAL + | CLONE_VM + | CLONE_VFORK + | CLONE_PARENT_SETTID + | CLONE_CHILD_CLEARTID + | CLONE_DETACHED + | CLONE_UNTRACED + | CLONE_CHILD_SETTID; diff --git a/userland/capsule_linux/src/linux/serve/family_reap_end.rs b/userland/capsule_linux/src/linux/serve/family_reap_end.rs index 662adcc60..ecdc224db 100644 --- a/userland/capsule_linux/src/linux/serve/family_reap_end.rs +++ b/userland/capsule_linux/src/linux/serve/family_reap_end.rs @@ -15,12 +15,14 @@ // along with this program. If not, see . //! Ending each process that has exited: its threads are killed, its end is -//! said in the log, and its parent is told as family_reap tells it. +//! said in the log, a vfork parent waiting on it is let go, and its parent is +//! told as family_reap tells it. use nonos_libc::mk_kill; use super::family::Family; use super::family_reap::tell_parent; +use super::family_wait::answer; const SIGKILL: u64 = 9; @@ -55,6 +57,9 @@ impl Family { let Some(p) = self.guests.iter_mut().find(|g| g.children.contains(&gone.pid)) else { continue; }; + if let Some(t) = gone.signals.vfork { + answer(p, t, u64::from(shown)); + } tell_parent(p, &gone, status, &mut self.ns); } any diff --git a/userland/capsule_linux/src/linux/serve/route_life.rs b/userland/capsule_linux/src/linux/serve/route_life.rs index 762892f79..f21833b2a 100644 --- a/userland/capsule_linux/src/linux/serve/route_life.rs +++ b/userland/capsule_linux/src/linux/serve/route_life.rs @@ -15,9 +15,9 @@ // along with this program. If not, see . //! The calls of process lifecycle and signals that can leave their caller -//! parked: a plain exit, and a signal sent where only the family can say -//! whether anyone received it. Asked first by `dispatch`, so these are -//! answered here whatever it holds. +//! parked: a plain exit, a new process, and a signal sent where only the +//! family can say whether anyone received it. Asked first by `dispatch`, so +//! these are answered here whatever it holds. use nonos_libc::ForeignFrame; @@ -27,6 +27,9 @@ use crate::linux::abi::{nr, nr_sig as ns}; use crate::linux::call; use crate::linux::guest::Guest; +/// clone's CLONE_THREAD: without it, clone makes a process. +const CLONE_THREAD: u64 = 0x10000; + pub fn answer(guest: &mut Guest, frame: &ForeignFrame) -> Option { let (a, tid) = (frame.args(), frame.pid); Some(match frame.nr { @@ -35,6 +38,8 @@ pub fn answer(guest: &mut Guest, frame: &ForeignFrame) -> Option { let _ = call::exit(guest, a[0]); Answer::Park } + nr::CLONE if a[0] & CLONE_THREAD == 0 => call::clone_process(guest, tid, a), + nr::VFORK => call::vfork(guest, tid), ns::KILL => call::kill_from(guest, tid, a[0], a[1]), ns::TKILL => call::tgkill_from(guest, tid, 0, a[0], a[1]), ns::TGKILL if (a[0] as i64) <= 0 => Answer::value(errno::fail(errno::EINVAL)), From aaed1932643f7dd6e5c443cb480476e897c3a88b Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:03:05 +0000 Subject: [PATCH 08/26] linux: wait4 and waitid answer as linux's do, and SIGCHLD is raised wait4 held one parked waiter per process, and a second waiting thread replaced the first. It wrote every status as an exit code, so a child ended by a signal read as a normal exit, and it treated a wait by process group as a wait for any child. It never wrote the rusage, ignored its options, and waitid was not served. Nothing was raised at a parent when a child ended, so a SIGCHLD handler never ran. wait4 and waitid now park the calling thread, any number of them, and the family answers each once it can: at once with a child that has ended and fits, 0 under WNOHANG, ECHILD when no child could ever fit, and otherwise when one ends. A wait can name any child, one pid, the caller's own group or another group, with __WALL and __WCLONE choosing by the signal a child raises. The status word is the one the child ended with, waitid reports CLD_EXITED or CLD_KILLED in a siginfo and can look without reaping under WNOWAIT. The rusage is written as zeros, since no CPU time is reported for a guest, and the log says so once per family run in a "[LINUX] rusage: ..." line, so the zeros are not read as a measurement. A child that ends raises its exit signal at its parent, SIGCHLD unless clone named another, with its pid and status, and a parent that ignores SIGCHLD or set SA_NOCLDWAIT has its children reaped as they end. Reaping goes round again while answering ends anything more. wait4 keeps its caller in dispatch. waitid is routed from route_life, which dispatch does not ask yet. The guest's old parked wait4 field is no longer read and is left for a separate change to remove. --- userland/capsule_linux/src/linux/call/mod.rs | 3 +- .../src/linux/call/spawn/fork_child.rs | 4 ++ .../capsule_linux/src/linux/call/spawn/mod.rs | 4 +- .../src/linux/call/spawn/wait.rs | 67 +++++++++++-------- .../src/linux/call/spawn/waitid.rs | 59 ++++++++++++++++ .../capsule_linux/src/linux/guest/siginfo.rs | 2 + .../capsule_linux/src/linux/guest/sigstate.rs | 1 + .../src/linux/serve/family_reap.rs | 47 +++++++------ .../src/linux/serve/family_reap_end.rs | 2 +- .../src/linux/serve/family_wait.rs | 20 +++++- .../src/linux/serve/family_wait_report.rs | 65 ++++++++++++++++++ .../src/linux/serve/family_wait_try.rs | 66 ++++++++++++++++++ userland/capsule_linux/src/linux/serve/mod.rs | 2 + .../capsule_linux/src/linux/serve/pid_map.rs | 2 + .../src/linux/serve/route_life.rs | 8 ++- 15 files changed, 298 insertions(+), 54 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/spawn/waitid.rs create mode 100644 userland/capsule_linux/src/linux/serve/family_wait_report.rs create mode 100644 userland/capsule_linux/src/linux/serve/family_wait_try.rs diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index 211f2401b..0c8824460 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -82,7 +82,8 @@ pub use signal_send::{kill, kill_from, tgkill_from}; pub use signal_stack::sigaltstack; pub use sigreturn::rt_sigreturn; pub use sleep::{clock_nanosleep, nanosleep}; -pub use spawn::{clone, clone_process, execve, fork, reap_one, vfork, wait4}; +pub use spawn::{clone, clone_process, execve, fork, vfork, wait4, wait4_usage, waitid}; +pub use spawn::{WALL, WCLONE, WEXITED, WNOHANG, WNOWAIT}; pub use clock::{clock_getres, clock_gettime, now_ms}; pub use epoch::{family_ms, mark_start}; pub use thread::{arch_prctl, getrandom}; diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_child.rs b/userland/capsule_linux/src/linux/call/spawn/fork_child.rs index 099f8b964..062bd431e 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork_child.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork_child.rs @@ -21,6 +21,7 @@ use nonos_libc::{mk_foreign_fork, mk_foreign_resume}; use super::fork_copy::copy_spans; use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SIGCHLD; use crate::linux::guest::Guest; /// A new process copied from this one, running, and adopted by the family @@ -61,5 +62,8 @@ pub(super) fn fork_child( return Err(errno::fail(errno::ENOMEM)); } guest.children.push(child); + if exit_signal != SIGCHLD { + guest.signals.clone_kids.push(child); + } Ok(child) } diff --git a/userland/capsule_linux/src/linux/call/spawn/mod.rs b/userland/capsule_linux/src/linux/call/spawn/mod.rs index e95ebe95d..1a12afc2b 100644 --- a/userland/capsule_linux/src/linux/call/spawn/mod.rs +++ b/userland/capsule_linux/src/linux/call/spawn/mod.rs @@ -31,10 +31,12 @@ mod vfork; mod vfork_clone; mod vfork_flags; mod wait; +mod waitid; pub use clone::clone; pub use exec::execve; pub use fork::fork; pub use vfork::vfork; pub use vfork_clone::clone_process; -pub use wait::{reap_one, wait4}; +pub use wait::{wait4, wait4_usage, WALL, WCLONE, WEXITED, WNOHANG, WNOWAIT}; +pub use waitid::waitid; diff --git a/userland/capsule_linux/src/linux/call/spawn/wait.rs b/userland/capsule_linux/src/linux/call/spawn/wait.rs index fbe07aeff..072d33fdc 100644 --- a/userland/capsule_linux/src/linux/call/spawn/wait.rs +++ b/userland/capsule_linux/src/linux/call/spawn/wait.rs @@ -14,41 +14,54 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `wait4`: a child that has ended, or a wait until one does. +//! `wait4`: a child that has ended, or a wait until one does. The call only +//! checks what it was asked and parks the caller; the family, which sees +//! every process, answers it at once when a child has ended, with 0 under +//! WNOHANG, or with ECHILD when no child fits, and otherwise when one ends. use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::{ChildWait, Which}; use crate::linux::guest::Guest; use crate::linux::serve::Answer; -/// Set by a caller that will not wait. -const WNOHANG: u64 = 1; +pub const WNOHANG: u64 = 1; +pub const WUNTRACED: u64 = 2; +pub const WEXITED: u64 = 4; +pub const WCONTINUED: u64 = 8; +pub const WNOWAIT: u64 = 0x0100_0000; +pub const WNOTHREAD: u64 = 0x2000_0000; +pub const WALL: u64 = 0x4000_0000; +pub const WCLONE: u64 = 0x8000_0000; +const WAIT4_OPTIONS: u64 = WNOHANG | WUNTRACED | WCONTINUED | WNOTHREAD | WALL | WCLONE; pub fn wait4(guest: &mut Guest, want: u64, status: u64, flags: u64, tid: u32) -> Answer { - if guest.children.is_empty() { - return Answer::value(errno::fail(errno::ECHILD)); - } - if let Some(v) = reap_one(guest, want, status) { - return Answer::value(v); - } - // A child still running under WNOHANG is a zero, not an error. - if flags & WNOHANG != 0 { - return Answer::value(errno::ok(0)); - } - guest.waiting = Some((want, status, tid)); - Answer::Park + wait4_usage(guest, want, status, flags, 0, tid) } -/// Take one ended child the caller asked about, write its status, and give -/// the answer wait4 returns. None while no such child has ended. -pub fn reap_one(guest: &mut Guest, want: u64, status: u64) -> Option { - let any = (want as i64) <= 0; - let at = guest.ended.iter().position(|(pid, _)| any || *pid == want as u32)?; - let (pid, kept) = guest.ended.remove(at); - guest.children.retain(|p| *p != pid); - /* Kept as Linux's wait status word: an exit's code in the second byte. */ - let word = kept as u32; - if status != 0 && guest.write(status, &word.to_le_bytes()) < 4 { - return Some(errno::fail(errno::EFAULT)); +/// wait4 with its rusage: written as zeros, since the kernel reports no CPU +/// time for a guest. +pub fn wait4_usage( + guest: &mut Guest, + want: u64, + status: u64, + flags: u64, + rusage: u64, + tid: u32, +) -> Answer { + if flags & !WAIT4_OPTIONS != 0 { + return Answer::value(errno::fail(errno::EINVAL)); } - Some(errno::ok(pid as u64)) + let which = match want as i64 as i32 { + -1 => Which::Any, + 0 => Which::Group(guest.pgid), + p if p < 0 => Which::Group(p.unsigned_abs()), + p => Which::Pid(p as u32), + }; + let options = flags | WEXITED; + park(guest, ChildWait { tid, which, options, out: status, rusage, waitid: false }) +} + +pub(super) fn park(guest: &mut Guest, w: ChildWait) -> Answer { + guest.signals.childwaits.push(w); + Answer::Park } diff --git a/userland/capsule_linux/src/linux/call/spawn/waitid.rs b/userland/capsule_linux/src/linux/call/spawn/waitid.rs new file mode 100644 index 000000000..271523e88 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/spawn/waitid.rs @@ -0,0 +1,59 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `waitid`: wait4's question asked by id type, answered in a siginfo, with +//! WNOWAIT to look without reaping. Linux writes the siginfo whatever the +//! outcome, zeros when nothing is reported, so an error here writes it too. + +use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::{ChildWait, Which}; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +use super::wait::{park, WALL, WCLONE, WCONTINUED, WEXITED, WNOHANG, WNOTHREAD, WNOWAIT}; + +const WSTOPPED: u64 = 2; +const P_ALL: u64 = 0; +const P_PID: u64 = 1; +const P_PGID: u64 = 2; +const P_PIDFD: u64 = 3; +const OPTIONS: u64 = + WNOHANG | WEXITED | WSTOPPED | WCONTINUED | WNOWAIT | WNOTHREAD | WALL | WCLONE; +pub const SIGINFO_LEN: usize = 128; + +pub fn waitid(guest: &mut Guest, tid: u32, a: [u64; 6]) -> Answer { + let (idtype, id, info, options, rusage) = (a[0], a[1] as u32, a[2], a[3], a[4]); + if options & !OPTIONS != 0 || options & (WEXITED | WSTOPPED | WCONTINUED) == 0 { + return refuse(guest, info, errno::EINVAL); + } + let which = match idtype { + P_ALL => Which::Any, + P_PID if (id as i32) > 0 => Which::Pid(id), + P_PGID if id == 0 => Which::Group(guest.pgid), + P_PGID if (id as i32) > 0 => Which::Group(id), + /* No descriptor here is ever a pidfd: pidfd_open is not served. */ + P_PIDFD => return refuse(guest, info, errno::EBADF), + _ => return refuse(guest, info, errno::EINVAL), + }; + park(guest, ChildWait { tid, which, options, out: info, rusage, waitid: true }) +} + +fn refuse(guest: &Guest, info: u64, e: i64) -> Answer { + if info != 0 && guest.write(info, &[0u8; SIGINFO_LEN]) < SIGINFO_LEN as i64 { + return Answer::value(errno::fail(errno::EFAULT)); + } + Answer::value(errno::fail(e)) +} diff --git a/userland/capsule_linux/src/linux/guest/siginfo.rs b/userland/capsule_linux/src/linux/guest/siginfo.rs index 936a11da3..15d66b982 100644 --- a/userland/capsule_linux/src/linux/guest/siginfo.rs +++ b/userland/capsule_linux/src/linux/guest/siginfo.rs @@ -21,6 +21,8 @@ pub const SI_USER: i32 = 0; pub const SI_TKILL: i32 = -6; +pub const CLD_EXITED: i32 = 1; +pub const CLD_KILLED: i32 = 2; pub const INFO_LEN: usize = 128; #[derive(Clone, Copy, Default)] diff --git a/userland/capsule_linux/src/linux/guest/sigstate.rs b/userland/capsule_linux/src/linux/guest/sigstate.rs index 2fcad0711..42ceee614 100644 --- a/userland/capsule_linux/src/linux/guest/sigstate.rs +++ b/userland/capsule_linux/src/linux/guest/sigstate.rs @@ -26,6 +26,7 @@ pub const SIGSEGV: u8 = 11; pub const SIGCHLD: u8 = 17; pub const SIGSTOP: u8 = 19; +pub const SA_NOCLDWAIT: u64 = 2; pub const SA_ONSTACK: u64 = 0x0800_0000; pub const SA_NODEFER: u64 = 0x4000_0000; pub const SA_RESETHAND: u64 = 0x8000_0000; diff --git a/userland/capsule_linux/src/linux/serve/family_reap.rs b/userland/capsule_linux/src/linux/serve/family_reap.rs index 3d626953a..a0c577a1f 100644 --- a/userland/capsule_linux/src/linux/serve/family_reap.rs +++ b/userland/capsule_linux/src/linux/serve/family_reap.rs @@ -14,25 +14,25 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Ending what has exited, and telling each parent: the ended child waits -//! until it is waited for, and a wait4 parked for it is answered. - -use nonos_libc::mk_foreign_reply; +//! Ending what has exited, and telling each parent: the ended child waits as +//! a zombie until it is waited for, and its exit signal, SIGCHLD unless clone +//! named another, is raised at the parent with CLD_EXITED or CLD_KILLED. A +//! parent that ignores SIGCHLD, or asked for SA_NOCLDWAIT, has its children +//! reaped as they end, as Linux does. use super::family::Family; -use super::pid_ns::PidNs; -use super::pid_out::value_out; -use crate::linux::abi::nr; -use crate::linux::call::reap_one; +use crate::linux::guest::siginfo::{SigInfo, CLD_EXITED, CLD_KILLED}; +use crate::linux::guest::sigstate::{SA_NOCLDWAIT, SIGCHLD}; use crate::linux::guest::Guest; impl Family { - /// End every process that asked to, deliver the signals that follow, - /// and go again while that ends anything more. + /// End every process that asked to, answer the waits and signals that + /// follows, and go again while that ends anything more. pub fn reap(&mut self) { self.settle_pipes(); loop { let ended = self.end_exited(); + self.settle_child_waits(); self.settle_signals(); if !ended && !self.guests.iter().any(|g| g.exited.is_some()) { return; @@ -41,14 +41,23 @@ impl Family { } } -/// Tell `p` that `gone` ended with `status`: kept until it is waited for, and -/// a parked wait4 answered with its pid in the guest's numbering. -pub fn tell_parent(p: &mut Guest, gone: &Guest, status: i32, ns: &mut PidNs) { - p.ended.push((gone.pid, status)); - if let Some((want, status, tid)) = p.waiting { - if let Some(value) = reap_one(p, want, status) { - p.waiting = None; - let _ = mk_foreign_reply(tid, value_out(ns, nr::WAIT4, value)); - } +/// Tell `p` that `gone` ended with `status`: kept as a zombie or reaped at +/// once, and its exit signal raised with CLD_EXITED or CLD_KILLED. +pub fn tell_parent(p: &mut Guest, gone: &Guest, status: i32) { + let sig = gone.signals.exit_signal; + let chld = p.signals.action(SIGCHLD as usize).unwrap_or_default(); + if sig == SIGCHLD && (chld.ignores() || chld.flags & SA_NOCLDWAIT != 0) { + p.children.retain(|c| *c != gone.pid); + } else { + p.ended.push((gone.pid, status)); + p.signals.kid_groups.push((gone.pid, gone.pgid)); + } + if sig != 0 { + let (code, value) = match status & 0x7f { + 0 => (CLD_EXITED, (status >> 8) & 0xff), + s => (CLD_KILLED, s), + }; + let info = SigInfo { value: value as u64, ..SigInfo::from(sig, code, gone.pid) }; + let _ = p.signals.raise(0, info); } } diff --git a/userland/capsule_linux/src/linux/serve/family_reap_end.rs b/userland/capsule_linux/src/linux/serve/family_reap_end.rs index ecdc224db..e7dbeea23 100644 --- a/userland/capsule_linux/src/linux/serve/family_reap_end.rs +++ b/userland/capsule_linux/src/linux/serve/family_reap_end.rs @@ -60,7 +60,7 @@ impl Family { if let Some(t) = gone.signals.vfork { answer(p, t, u64::from(shown)); } - tell_parent(p, &gone, status, &mut self.ns); + tell_parent(p, &gone, status); } any } diff --git a/userland/capsule_linux/src/linux/serve/family_wait.rs b/userland/capsule_linux/src/linux/serve/family_wait.rs index b62e5b3bc..812885a08 100644 --- a/userland/capsule_linux/src/linux/serve/family_wait.rs +++ b/userland/capsule_linux/src/linux/serve/family_wait.rs @@ -14,11 +14,14 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Answering a thread the family held parked: with the value its call -//! returns, or by entering the handler of a signal it now takes. +//! Answering wait4 and waitid. A child that has ended and fits is reported, +//! and reaped unless WNOWAIT; with none ended, WNOHANG answers 0, and a caller +//! with no child that could ever fit gets ECHILD. The family answers, since a +//! wait by group needs every child's group, which only it can see. use nonos_libc::mk_foreign_reply; +use super::family::Family; use crate::linux::guest::Guest; /// Reply to a parked thread, or enter the handler of a signal it now takes. @@ -27,3 +30,16 @@ pub fn answer(g: &mut Guest, tid: u32, value: u64) { let _ = mk_foreign_reply(tid, value); } } + +impl Family { + pub(super) fn settle_child_waits(&mut self) { + for i in 0..self.guests.len() { + for w in core::mem::take(&mut self.guests[i].signals.childwaits) { + match self.try_wait(i, &w) { + Some(v) => answer(&mut self.guests[i], w.tid, v), + None => self.guests[i].signals.childwaits.push(w), + } + } + } + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_wait_report.rs b/userland/capsule_linux/src/linux/serve/family_wait_report.rs new file mode 100644 index 000000000..2d8599184 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_wait_report.rs @@ -0,0 +1,65 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What a finished wait writes back: waitid's siginfo or wait4's status word, +//! and struct rusage. The kernel reports no CPU time for a guest, so rusage is +//! written as all zeros; Go always passes one, so it is still written, and the +//! log says once per family run that its zeros are not a measurement. + +use core::sync::atomic::{AtomicBool, Ordering}; + +use crate::linux::abi::errno; +use crate::linux::guest::siginfo::{SigInfo, CLD_EXITED, CLD_KILLED}; +use crate::linux::guest::sigstate::SIGCHLD; +use crate::linux::guest::sigwaits::ChildWait; +use crate::linux::guest::Guest; + +/// struct rusage: no CPU time is reported for a guest, so all of it is zero. +const RUSAGE_LEN: usize = 144; + +/// Set once the zero rusage has been said in the log. +static RUSAGE_SAID: AtomicBool = AtomicBool::new(false); + +/// Write what the caller asked for and give the value its call returns. +pub fn report(p: &Guest, w: &ChildWait, child: Option<(u32, i32)>, value: u64) -> u64 { + let wrote = match (w.waitid, child) { + (true, Some((pid, status))) => { + let (code, st) = match status & 0x7f { + 0 => (CLD_EXITED, (status >> 8) & 0xff), + s => (CLD_KILLED, s), + }; + let info = SigInfo { value: st as u64, ..SigInfo::from(SIGCHLD, code, pid) }; + w.out == 0 || p.write(w.out, &info.bytes()) == 128 + } + (true, None) => w.out == 0 || p.write(w.out, &[0u8; 128]) == 128, + (false, Some((_, status))) => w.out == 0 || p.write(w.out, &status.to_le_bytes()) == 4, + (false, None) => true, + }; + let usage = child.is_none() || w.rusage == 0 || zero_rusage(p, w.rusage); + match wrote && usage { + true => value, + false => errno::fail(errno::EFAULT), + } +} + +/// Write an all-zero rusage at `at`, saying so the first time in a family run. +fn zero_rusage(p: &Guest, at: u64) -> bool { + if !RUSAGE_SAID.swap(true, Ordering::Relaxed) { + let line = b"[LINUX] rusage: no CPU time is reported for a guest, written as zero\n"; + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + } + p.write(at, &[0u8; RUSAGE_LEN]) > 0 +} diff --git a/userland/capsule_linux/src/linux/serve/family_wait_try.rs b/userland/capsule_linux/src/linux/serve/family_wait_try.rs new file mode 100644 index 000000000..073caf39c --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_wait_try.rs @@ -0,0 +1,66 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! One parked wait4 or waitid tried against its caller's children: the first +//! ended child that fits is reported, or the call answers at once with none, +//! or it stays parked until a child ends. + +use super::family::Family; +use super::family_wait_report::report; +use crate::linux::abi::errno; +use crate::linux::call::{WALL, WCLONE, WEXITED, WNOHANG, WNOWAIT}; +use crate::linux::guest::sigwaits::{ChildWait, Which}; + +impl Family { + pub(super) fn try_wait(&mut self, i: usize, w: &ChildWait) -> Option { + let group_of = |pid: u32| { + let live = self.guests.iter().find(|g| g.pid == pid).map(|g| g.pgid); + live.or_else(|| { + self.guests[i].signals.kid_groups.iter().find(|k| k.0 == pid).map(|k| k.1) + }) + }; + let p = &self.guests[i]; + let fits = |pid: u32| { + let clone = p.signals.clone_kids.contains(&pid); + let kind = w.options & WALL != 0 || (w.options & WCLONE != 0) == clone; + kind && match w.which { + Which::Any => true, + Which::Pid(want) => pid == want, + Which::Group(g) => group_of(pid) == Some(g), + } + }; + let exits = w.options & WEXITED != 0; + let done = p.ended.iter().position(|(pid, _)| exits && fits(*pid)); + let any = p.children.iter().any(|c| fits(*c)); + let p = &mut self.guests[i]; + let Some(at) = done else { + return match (any, w.options & WNOHANG != 0) { + (false, _) => Some(report(p, w, None, errno::fail(errno::ECHILD))), + (true, true) => Some(report(p, w, None, 0)), + (true, false) => None, + }; + }; + let (pid, status) = p.ended[at]; + if w.options & WNOWAIT == 0 { + p.ended.remove(at); + p.children.retain(|c| *c != pid); + p.signals.clone_kids.retain(|c| *c != pid); + p.signals.kid_groups.retain(|k| k.0 != pid); + } + let shown = u64::from(crate::linux::serve::guest_pid(pid)); + Some(report(p, w, Some((pid, status)), if w.waitid { 0 } else { shown })) + } +} diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index 6bd4175c0..de7384585 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -30,6 +30,8 @@ mod family_signal; mod family_signal_route; mod family_sleep; mod family_wait; +mod family_wait_report; +mod family_wait_try; mod loop_impl; mod pid_map; mod pid_ns; diff --git a/userland/capsule_linux/src/linux/serve/pid_map.rs b/userland/capsule_linux/src/linux/serve/pid_map.rs index 2a10005c4..2328f29f7 100644 --- a/userland/capsule_linux/src/linux/serve/pid_map.rs +++ b/userland/capsule_linux/src/linux/serve/pid_map.rs @@ -41,6 +41,8 @@ pub fn frame_in(ns: &PidNs, frame: &ForeignFrame) -> Option { fn args_in(ns: &PidNs, call: u64, a: &mut [u64; 6]) -> Result<(), u64> { let (slots, missing): (&[usize], i64) = match call { nr::WAIT4 => (&[0], errno::ECHILD), + /* waitid names a pid or a group only for P_PID and P_PGID. */ + ns::WAITID if matches!(a[0], 1 | 2) && a[1] != 0 => (&[1], errno::ECHILD), ns::TGKILL | ns::RT_TGSIGQUEUEINFO => (&[0, 1], errno::ESRCH), ns::RT_SIGQUEUEINFO => (&[0], errno::ESRCH), np::KILL | np::TKILL | np::GETPGID | np::GETSID => (&[0], errno::ESRCH), diff --git a/userland/capsule_linux/src/linux/serve/route_life.rs b/userland/capsule_linux/src/linux/serve/route_life.rs index f21833b2a..cfd6dc95c 100644 --- a/userland/capsule_linux/src/linux/serve/route_life.rs +++ b/userland/capsule_linux/src/linux/serve/route_life.rs @@ -15,9 +15,9 @@ // along with this program. If not, see . //! The calls of process lifecycle and signals that can leave their caller -//! parked: a plain exit, a new process, and a signal sent where only the -//! family can say whether anyone received it. Asked first by `dispatch`, so -//! these are answered here whatever it holds. +//! parked: a plain exit, a new process, a wait for a child, and a signal sent +//! where only the family can say whether anyone received it. Asked first by +//! `dispatch`, so these are answered here whatever it holds. use nonos_libc::ForeignFrame; @@ -40,6 +40,8 @@ pub fn answer(guest: &mut Guest, frame: &ForeignFrame) -> Option { } nr::CLONE if a[0] & CLONE_THREAD == 0 => call::clone_process(guest, tid, a), nr::VFORK => call::vfork(guest, tid), + nr::WAIT4 => call::wait4_usage(guest, a[0], a[1], a[2], a[3], tid), + ns::WAITID => call::waitid(guest, tid, a), ns::KILL => call::kill_from(guest, tid, a[0], a[1]), ns::TKILL => call::tgkill_from(guest, tid, 0, a[0], a[1]), ns::TGKILL if (a[0] as i64) <= 0 => Answer::value(errno::fail(errno::EINVAL)), From 0c9b8fdc4b33e4d8085df08cf408db9cc3ec730d Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:03:09 +0000 Subject: [PATCH 09/26] linux: sigpipe raises SIGPIPE for a write to a pipe with no reader A write to a pipe no process can read answered EPIPE and nothing more. Linux raises SIGPIPE at the writing thread first, so a program that relies on the default to end it, as a shell pipeline does, kept writing into nothing, and a SIGPIPE handler never ran. sigpipe raises SIGPIPE at the writing thread, or at the process when given 0, with SI_USER and the writer's own pid, as Linux's pipe_write does. Caught, the handler runs over the EPIPE the write answers; ignored, only EPIPE is seen; at its default, the process ends. It is exported for the pipe write to call before it answers EPIPE; that write belongs to a change not made here, so sigpipe and SIGPIPE have no caller yet. --- userland/capsule_linux/src/linux/call/mod.rs | 2 +- userland/capsule_linux/src/linux/call/signal_send.rs | 12 ++++++++++-- userland/capsule_linux/src/linux/guest/sigstate.rs | 1 + 3 files changed, 12 insertions(+), 3 deletions(-) diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index 0c8824460..a101ad10d 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -78,7 +78,7 @@ pub use session::{getpgid, getsid, setpgid, setsid}; pub use signal::rt_sigaction; pub use signal_mask::rt_sigprocmask; pub use signal_queue::{rt_sigqueueinfo, rt_tgsigqueueinfo}; -pub use signal_send::{kill, kill_from, tgkill_from}; +pub use signal_send::{kill, kill_from, sigpipe, tgkill_from}; pub use signal_stack::sigaltstack; pub use sigreturn::rt_sigreturn; pub use sleep::{clock_nanosleep, nanosleep}; diff --git a/userland/capsule_linux/src/linux/call/signal_send.rs b/userland/capsule_linux/src/linux/call/signal_send.rs index 0b7107973..3b96aa524 100644 --- a/userland/capsule_linux/src/linux/call/signal_send.rs +++ b/userland/capsule_linux/src/linux/call/signal_send.rs @@ -14,7 +14,7 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `kill`, `tkill` and `tgkill`. A signal for the caller's own +//! `kill`, `tkill` and `tgkill`, and SIGPIPE. A signal for the caller's own //! process is queued here and taken as `serve::deliver` decides; one for //! another process of the family leaves through the outbox with the caller //! parked, and the family answers it once it knows whether anyone was there. @@ -23,7 +23,7 @@ use super::signal_post::post; use crate::linux::abi::errno; use crate::linux::guest::siginfo::{SigInfo, SI_TKILL, SI_USER}; -use crate::linux::guest::sigstate::NSIG; +use crate::linux::guest::sigstate::{NSIG, SIGPIPE}; use crate::linux::guest::sigwaits::Target; use crate::linux::guest::Guest; use crate::linux::serve::Answer; @@ -64,3 +64,11 @@ fn send(guest: &mut Guest, tid: u32, to: Target, signo: u64, code: i32) -> Answe let info = SigInfo::from(signo as u8, code, guest.pid); post(guest, tid, to, info) } + +/// A write to a pipe no process can read raises SIGPIPE at the writing +/// thread, as Linux's pipe_write does, before the write answers EPIPE: caught, +/// the handler runs over that EPIPE; ignored, only EPIPE is seen; at its +/// default, the process ends. `tid` 0 raises it at the process. +pub fn sigpipe(guest: &mut Guest, tid: u32) { + let _ = guest.signals.raise(tid, SigInfo::from(SIGPIPE, SI_USER, guest.pid)); +} diff --git a/userland/capsule_linux/src/linux/guest/sigstate.rs b/userland/capsule_linux/src/linux/guest/sigstate.rs index 42ceee614..fa80d80fa 100644 --- a/userland/capsule_linux/src/linux/guest/sigstate.rs +++ b/userland/capsule_linux/src/linux/guest/sigstate.rs @@ -23,6 +23,7 @@ pub const NSIG: usize = 64; pub const SIGKILL: u8 = 9; pub const SIGSEGV: u8 = 11; +pub const SIGPIPE: u8 = 13; pub const SIGCHLD: u8 = 17; pub const SIGSTOP: u8 = 19; From b27092ce9fadb68f1101fc542bf3f1f818b106da Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:03:19 +0000 Subject: [PATCH 10/26] linux: a caught signal ends the wait a thread is parked in A signal was only acted on when its thread next returned from a call. A thread parked in nanosleep, a futex wait, a pipe read or wait4 kept the signal until the wait ended by itself, so a handler for a signal sent to a thread in sleep(10) ran up to ten seconds late, and a signal whose default is fatal did not end a process whose threads were all parked. After every answer the family now looks at what each process has pending. An uncaught signal does its default at once, as Linux does when it is sent. A caught one is taken by a parked thread that does not block it: the thread leaves its wait, which is then never answered, and enters the handler over the parked call. The call answers EINTR, or under SA_RESTART runs again once the handler returns for the calls Linux restarts: reads and writes, the socket calls, wait4, waitid and an untimed futex wait. An interrupted relative nanosleep writes the time it had left. --- userland/capsule_linux/src/linux/guest/mod.rs | 1 + .../capsule_linux/src/linux/guest/sigpark.rs | 5 +- .../capsule_linux/src/linux/guest/sigstate.rs | 1 + .../capsule_linux/src/linux/guest/sigtake.rs | 5 ++ .../capsule_linux/src/linux/serve/deliver.rs | 6 +- .../src/linux/serve/deliver_interrupt.rs | 48 +++++++++++++++ .../src/linux/serve/deliver_rem.rs | 53 +++++++++++++++++ .../src/linux/serve/deliver_restart.rs | 51 ++++++++++++++++ .../src/linux/serve/deliver_wait.rs | 59 +++++++++++++++++++ .../src/linux/serve/family_signal.rs | 4 +- userland/capsule_linux/src/linux/serve/mod.rs | 4 ++ 11 files changed, 231 insertions(+), 6 deletions(-) create mode 100644 userland/capsule_linux/src/linux/serve/deliver_interrupt.rs create mode 100644 userland/capsule_linux/src/linux/serve/deliver_rem.rs create mode 100644 userland/capsule_linux/src/linux/serve/deliver_restart.rs create mode 100644 userland/capsule_linux/src/linux/serve/deliver_wait.rs diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index b26a52198..aa5cf7cdc 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -64,3 +64,4 @@ pub use layout::{ }; pub use mem::{page_down, page_up, span_within, MAX_SPAN, PAGE}; pub use region::Region; +pub use sigpark::Parked; diff --git a/userland/capsule_linux/src/linux/guest/sigpark.rs b/userland/capsule_linux/src/linux/guest/sigpark.rs index f9912efdd..6a4466180 100644 --- a/userland/capsule_linux/src/linux/guest/sigpark.rs +++ b/userland/capsule_linux/src/linux/guest/sigpark.rs @@ -14,8 +14,9 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Which wait a thread is parked in, and taking it out of every one, so a -//! thread that has gone leaves no wait behind to be answered. +//! Taking a thread out of the wait it is parked in, so a caught signal can end +//! that wait as Linux does: the wait is dropped here and never answered; the +//! handler's frame is the answer. use super::handle::Guest; diff --git a/userland/capsule_linux/src/linux/guest/sigstate.rs b/userland/capsule_linux/src/linux/guest/sigstate.rs index fa80d80fa..afec90e40 100644 --- a/userland/capsule_linux/src/linux/guest/sigstate.rs +++ b/userland/capsule_linux/src/linux/guest/sigstate.rs @@ -29,6 +29,7 @@ pub const SIGSTOP: u8 = 19; pub const SA_NOCLDWAIT: u64 = 2; pub const SA_ONSTACK: u64 = 0x0800_0000; +pub const SA_RESTART: u64 = 0x1000_0000; pub const SA_NODEFER: u64 = 0x4000_0000; pub const SA_RESETHAND: u64 = 0x8000_0000; diff --git a/userland/capsule_linux/src/linux/guest/sigtake.rs b/userland/capsule_linux/src/linux/guest/sigtake.rs index 8a30681a0..ab8116de6 100644 --- a/userland/capsule_linux/src/linux/guest/sigtake.rs +++ b/userland/capsule_linux/src/linux/guest/sigtake.rs @@ -39,4 +39,9 @@ impl Signals { pub fn discard(&mut self, signum: u8) { self.pending.retain(|(_, i)| i.signo != signum); } + + /// Each pending signal with the thread it is for, 0 for the process. + pub fn queued(&self) -> alloc::vec::Vec<(u32, u8)> { + self.pending.iter().map(|(t, i)| (*t, i.signo)).collect() + } } diff --git a/userland/capsule_linux/src/linux/serve/deliver.rs b/userland/capsule_linux/src/linux/serve/deliver.rs index 97800066a..ffbf471f6 100644 --- a/userland/capsule_linux/src/linux/serve/deliver.rs +++ b/userland/capsule_linux/src/linux/serve/deliver.rs @@ -17,9 +17,9 @@ //! A thread taking the signals it may take. A caught one enters its handler //! through Linux's rt_sigframe; an uncaught one does what its default says, //! and a default of ending the process ends all of it. A thread takes them -//! when it returns from a call, with the call's value in rax (maybe_deliver), -//! and when the kernel stops it running, on the registers it stopped with -//! (deliver_on). +//! when it returns from a call, with the call's value in rax (maybe_deliver); +//! when a signal ends the wait it is parked in (deliver_wait); and when the +//! kernel stops it running, on the registers it stopped with (deliver_on). use nonos_libc::{mk_foreign_context, ForeignRegs}; diff --git a/userland/capsule_linux/src/linux/serve/deliver_interrupt.rs b/userland/capsule_linux/src/linux/serve/deliver_interrupt.rs new file mode 100644 index 000000000..004b59736 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_interrupt.rs @@ -0,0 +1,48 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A caught signal ending a parked thread's wait: the handler is entered over +//! the parked call, whose answer is EINTR or a restart as deliver_restart +//! decides, and a relative sleep writes what it had left first. + +use nonos_libc::{mk_foreign_context, mk_foreign_reply, ForeignRegs}; + +use super::deliver_enter::enter; +use super::deliver_rem::write_rem; +use super::deliver_restart::rewind; +use crate::linux::guest::siginfo::SigInfo; +use crate::linux::guest::Guest; + +/// End `tid`'s wait with a handler entered over its parked call. +pub fn interrupt(guest: &mut Guest, tid: u32, info: SigInfo) { + let mut regs: ForeignRegs = [0; 18]; + if mk_foreign_context(tid, &mut regs) != 0 { + let _ = guest.signals.raise(tid, info); + return; + } + let Some(parked) = guest.leave_waits(tid) else { + let _ = guest.signals.raise(tid, info); + return; + }; + let act = guest.signals.action(info.signo as usize).unwrap_or_default(); + write_rem(guest, ®s, parked); + rewind(&mut regs, act.flags); + if !enter(guest, tid, ®s, info) { + /* Out of its wait with no handler entered: it must still be answered. */ + let _ = + mk_foreign_reply(tid, crate::linux::abi::errno::fail(crate::linux::abi::errno::EINTR)); + } +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_rem.rs b/userland/capsule_linux/src/linux/serve/deliver_rem.rs new file mode 100644 index 000000000..96e0d76af --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_rem.rs @@ -0,0 +1,53 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A relative sleep cut short by a handler writes the time it had left, as +//! Linux's nanosleep and relative clock_nanosleep do; an absolute one writes +//! nothing, as on Linux. + +use nonos_libc::ForeignRegs; + +use super::deliver_restart::{R10, RAX, RSI}; +use crate::linux::call::now_ms; +use crate::linux::guest::{Guest, Parked}; + +const CLOCK_MONOTONIC: u64 = 1; +const TIMER_ABSTIME: u64 = 1; +const NANOSLEEP: u64 = 35; +const CLOCK_NANOSLEEP: u64 = 230; + +/// A relative sleep cut short writes what was left of it where it was asked. +pub fn write_rem(guest: &Guest, regs: &ForeignRegs, parked: Parked) { + let Parked::Sleep(due) = parked else { + return; + }; + let rem = match regs[RAX] { + NANOSLEEP => regs[RSI], + CLOCK_NANOSLEEP if regs[RSI] & TIMER_ABSTIME == 0 => regs[R10], + _ => 0, + }; + let Some(now) = now_ms(CLOCK_MONOTONIC) else { + return; + }; + if rem == 0 { + return; + } + let left = due.saturating_sub(now); + let mut spec = [0u8; 16]; + spec[..8].copy_from_slice(&(left / 1000).to_le_bytes()); + spec[8..].copy_from_slice(&((left % 1000) * 1_000_000).to_le_bytes()); + let _ = guest.write(rem, &spec); +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_restart.rs b/userland/capsule_linux/src/linux/serve/deliver_restart.rs new file mode 100644 index 000000000..48b61266b --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_restart.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What an interrupted call answers, as Linux decides it. Reads and writes, +//! the socket calls, wait4, waitid and an untimed futex wait end with +//! ERESTARTSYS: under SA_RESTART the call runs again once the handler returns, +//! otherwise EINTR. Sleeps, timed waits, poll, select, epoll, pause and the +//! signal waits always answer EINTR, and a relative nanosleep writes the time +//! it had left (deliver_rem). + +use nonos_libc::ForeignRegs; + +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SA_RESTART; + +pub const R10: usize = 2; +pub const RSI: usize = 9; +pub const RAX: usize = 13; +const RIP: usize = 16; +/// The length of `syscall`, which a restart steps back over. +const SYSCALL_LEN: u64 = 2; + +/// read, write, readv, writev, accept, sendto, recvfrom, sendmsg, recvmsg, +/// wait4, waitid, accept4. +const RESTARTS: [u64; 12] = [0, 1, 19, 20, 43, 44, 45, 46, 47, 61, 247, 288]; +const FUTEX: u64 = 202; + +/// Set rax to what the handler returns into: the call again, or EINTR. The +/// parked frame's rax is still the call's number. +pub fn rewind(regs: &mut ForeignRegs, flags: u64) { + let nr = regs[RAX]; + let untimed_futex = nr == FUTEX && regs[R10] == 0; + if flags & SA_RESTART != 0 && (RESTARTS.contains(&nr) || untimed_futex) { + regs[RIP] = regs[RIP].wrapping_sub(SYSCALL_LEN); + return; + } + regs[RAX] = errno::fail(errno::EINTR); +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_wait.rs b/userland/capsule_linux/src/linux/serve/deliver_wait.rs new file mode 100644 index 000000000..b04af8d4e --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_wait.rs @@ -0,0 +1,59 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Signals that reach a process whose threads are parked, not returning. +//! Run after every answer: an uncaught signal's default acts at once, as +//! Linux's does when it is sent; and a caught signal ends the wait of a +//! thread that does not block it, with EINTR, or restarts the call under +//! SA_RESTART where Linux restarts it. + +use super::deliver_interrupt::interrupt; +use super::deliver_say::stop_unserved; +use crate::linux::call::killed; +use crate::linux::guest::sigdefault::{default_of, Default}; +use crate::linux::guest::sigstate::bit; +use crate::linux::guest::Guest; + +pub fn settle(guest: &mut Guest) { + for (tid, signo) in guest.signals.queued() { + if guest.exited.is_some() { + return; + } + let act = guest.signals.action(signo as usize).unwrap_or_default(); + let takers: alloc::vec::Vec = match tid { + 0 => guest.live_threads(), + t => alloc::vec![t], + }; + let free = |g: &Guest, t: &u32| g.signals.blocked(*t) & bit(signo) == 0; + if !act.catches() { + if takers.iter().any(|t| free(guest, t)) { + let _ = guest.signals.take(takers[0], bit(signo)); + match (act.ignores(), default_of(signo)) { + (false, Default::Terminate) => killed(guest, signo), + (false, Default::Stop) => stop_unserved(signo), + _ => {} + } + } + continue; + } + let parked = takers.iter().find(|t| free(guest, t) && guest.parked(**t).is_some()); + if let Some(&t) = parked { + if let Some(info) = guest.signals.take(t, bit(signo)) { + interrupt(guest, t, info); + } + } + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_signal.rs b/userland/capsule_linux/src/linux/serve/family_signal.rs index 0b757afc7..170da9ad2 100644 --- a/userland/capsule_linux/src/linux/serve/family_signal.rs +++ b/userland/capsule_linux/src/linux/serve/family_signal.rs @@ -15,12 +15,14 @@ // along with this program. If not, see . //! Signals between processes of the family, settled after every answer: the -//! outbox is routed to every process each signal names. +//! outbox is routed, and each process's parked threads take what now reaches +//! them. use super::family::Family; impl Family { pub(super) fn settle_signals(&mut self) { self.route_outbox(); + self.guests.iter_mut().for_each(super::deliver_wait::settle); } } diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index de7384585..8becd14e0 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -19,8 +19,12 @@ mod answer; mod deliver; mod deliver_enter; +mod deliver_interrupt; +mod deliver_rem; +mod deliver_restart; mod deliver_say; mod deliver_stack; +mod deliver_wait; mod dispatch; mod family; mod family_pipes; From f2ce1fcdc4df261fe2f7b75347115cc2e0f15cd8 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:03:23 +0000 Subject: [PATCH 11/26] linux: alarm and setitimer arm ITIMER_REAL, which raises SIGALRM alarm, setitimer and getitimer were not served, so a program that bounds a wait with SIGALRM, or keeps time with a periodic ITIMER_REAL, never saw the signal. ITIMER_REAL is now kept per process on the family's monotonic clock, in milliseconds, the finest step it keeps. alarm arms it once and answers the whole seconds the last one had left, rounded as Linux rounds them; setitimer arms it once or with a period and hands back the old value; getitimer reads what is left. When it runs out the family raises SIGALRM at the process and moves a periodic timer past every period that ended. Exec keeps it and fork does not, as on Linux. ITIMER_VIRTUAL and ITIMER_PROF count CPU time, which the kernel does not report to a supervisor: arming one is refused by name and reading one reports it disarmed. The host proofs check how a periodic timer moves on and that a one-shot one stops. Signals::next_due gives the nearest deadline for the serve loop's wait to end by; that wait is computed outside this change, so it has no caller yet and a timer fires when the loop next wakes. --- userland/capsule_linux/src/linux/call/mod.rs | 4 + .../src/linux/call/signal_itv.rs | 50 +++++++++++++ .../src/linux/call/signal_real.rs | 38 ++++++++++ .../src/linux/call/signal_timer.rs | 73 +++++++++++++++++++ userland/capsule_linux/src/linux/guest/mod.rs | 1 + .../capsule_linux/src/linux/guest/siginfo.rs | 1 + .../src/linux/guest/sigqueue_ops.rs | 8 +- .../capsule_linux/src/linux/guest/sigstate.rs | 1 + .../src/linux/guest/sigtimer_rearm.rs | 34 +++++++++ .../src/linux/serve/family_signal.rs | 13 +++- .../src/linux/serve/family_signal_fire.rs | 29 ++++++++ userland/capsule_linux/src/linux/serve/mod.rs | 1 + .../src/linux/serve/table_sig.rs | 9 ++- userland/capsule_linux_proofs/src/calls.rs | 11 ++- userland/capsule_linux_proofs/src/lib.rs | 2 +- userland/capsule_linux_proofs/src/tests.rs | 1 + .../src/tests/sigtimer_tests.rs | 33 +++++++++ 17 files changed, 299 insertions(+), 10 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/signal_itv.rs create mode 100644 userland/capsule_linux/src/linux/call/signal_real.rs create mode 100644 userland/capsule_linux/src/linux/call/signal_timer.rs create mode 100644 userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs create mode 100644 userland/capsule_linux/src/linux/serve/family_signal_fire.rs create mode 100644 userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index a101ad10d..000e1525c 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -43,11 +43,14 @@ pub mod sigframe; pub mod sigframe_build; mod sigframe_read; mod signal; +mod signal_itv; mod signal_mask; mod signal_post; mod signal_queue; +mod signal_real; mod signal_send; mod signal_stack; +mod signal_timer; mod sigreturn; mod sleep; mod spawn; @@ -80,6 +83,7 @@ pub use signal_mask::rt_sigprocmask; pub use signal_queue::{rt_sigqueueinfo, rt_tgsigqueueinfo}; pub use signal_send::{kill, kill_from, sigpipe, tgkill_from}; pub use signal_stack::sigaltstack; +pub use signal_timer::{alarm, getitimer, setitimer}; pub use sigreturn::rt_sigreturn; pub use sleep::{clock_nanosleep, nanosleep}; pub use spawn::{clone, clone_process, execve, fork, vfork, wait4, wait4_usage, waitid}; diff --git a/userland/capsule_linux/src/linux/call/signal_itv.rs b/userland/capsule_linux/src/linux/call/signal_itv.rs new file mode 100644 index 000000000..edaad03b2 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_itv.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `struct itimerval` in and out: interval, then value, each seconds and +//! microseconds. Microseconds round up to the millisecond the clock keeps. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +const USEC: u64 = 1_000_000; + +/// (interval, value) in milliseconds, or the errno a bad one earns. +pub fn read_val(guest: &Guest, at: u64) -> Result<(u64, u64), u64> { + let raw = guest.read(at, 32).ok_or(errno::fail(errno::EFAULT))?; + let w = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + let ms = |s: u64, us: u64| match us < USEC && (s as i64) >= 0 { + true => Ok(s.saturating_mul(1000).saturating_add(us.div_ceil(1000))), + false => Err(errno::fail(errno::EINVAL)), + }; + Ok((ms(w(0), w(8))?, ms(w(16), w(24))?)) +} + +/// Write (interval, value) at `at`, if the caller gave somewhere to write. +pub fn put(guest: &Guest, at: u64, interval: u64, value: u64) -> u64 { + if at == 0 { + return errno::ok(0); + } + let mut b = [0u8; 32]; + for (i, ms) in [interval, value].iter().enumerate() { + b[i * 16..i * 16 + 8].copy_from_slice(&(ms / 1000).to_le_bytes()); + b[i * 16 + 8..i * 16 + 16].copy_from_slice(&((ms % 1000) * 1000).to_le_bytes()); + } + match guest.write(at, &b) < 32 { + true => errno::fail(errno::EFAULT), + false => errno::ok(0), + } +} diff --git a/userland/capsule_linux/src/linux/call/signal_real.rs b/userland/capsule_linux/src/linux/call/signal_real.rs new file mode 100644 index 000000000..9e878eece --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_real.rs @@ -0,0 +1,38 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! ITIMER_REAL on the family's monotonic clock: arming it, and what it has +//! left. alarm and setitimer both come here. + +use crate::linux::call::now_ms; +use crate::linux::guest::sigtimer::Itimer; +use crate::linux::guest::Guest; + +const CLOCK_MONOTONIC: u64 = 1; + +/// Arm ITIMER_REAL for `value` ms, repeating every `interval`; 0 disarms. +/// Answers what it had left and its old interval. +pub fn arm(guest: &mut Guest, value: u64, interval: u64) -> (u64, u64) { + let was = remaining(guest); + let now = now_ms(CLOCK_MONOTONIC).unwrap_or(0); + guest.signals.real = (value != 0).then(|| Itimer { due: now.saturating_add(value), interval }); + was +} + +pub fn remaining(guest: &Guest) -> (u64, u64) { + let now = now_ms(CLOCK_MONOTONIC).unwrap_or(0); + guest.signals.real.map_or((0, 0), |t| (t.due.saturating_sub(now).max(1), t.interval)) +} diff --git a/userland/capsule_linux/src/linux/call/signal_timer.rs b/userland/capsule_linux/src/linux/call/signal_timer.rs new file mode 100644 index 000000000..e4fa90156 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_timer.rs @@ -0,0 +1,73 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! alarm, setitimer and getitimer. ITIMER_REAL counts on the monotonic clock +//! and raises SIGALRM at the process when it runs out; exec keeps it, fork +//! does not. ITIMER_VIRTUAL and ITIMER_PROF count the process's CPU time, +//! which the kernel does not report to a supervisor: arming one is refused by +//! name, and reading one reports it disarmed, which it always is. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::signal_itv::{put, read_val}; +use super::signal_real::{arm, remaining}; + +const ITIMER_REAL: u64 = 0; +const ITIMER_PROF: u64 = 2; + +/// alarm(seconds): a one-shot ITIMER_REAL, answering the whole seconds the +/// last one had left, rounded as Linux rounds them. +pub fn alarm(guest: &mut Guest, secs: u64) -> u64 { + let (was, _) = arm(guest, secs.min(u64::from(u32::MAX)).saturating_mul(1000), 0); + let (whole, ms) = (was / 1000, was % 1000); + errno::ok(whole + u64::from((whole == 0 && ms > 0) || ms >= 500)) +} + +pub fn setitimer(guest: &mut Guest, which: u64, new: u64, old: u64) -> u64 { + if which > ITIMER_PROF { + return errno::fail(errno::EINVAL); + } + let (interval, value) = match new { + 0 => (0, 0), + at => match read_val(guest, at) { + Ok(v) => v, + Err(e) => return e, + }, + }; + if which != ITIMER_REAL { + if value == 0 { + return put(guest, old, 0, 0); + } + let _ = nonos_libc::mk_debug(UNSERVED.as_ptr(), UNSERVED.len()); + return errno::fail(errno::ENOSYS); + } + let (was, every) = arm(guest, value, interval); + put(guest, old, every, was) +} + +const UNSERVED: &[u8] = b"[LINUX] unserved setitimer: no CPU-time clock for VIRTUAL or PROF\n"; + +pub fn getitimer(guest: &mut Guest, which: u64, out: u64) -> u64 { + if which > ITIMER_PROF { + return errno::fail(errno::EINVAL); + } + let (left, every) = match which { + ITIMER_REAL => remaining(guest), + _ => (0, 0), + }; + put(guest, out, every, left) +} diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index aa5cf7cdc..9372ac736 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -38,6 +38,7 @@ mod sigtake; pub mod sigthread; mod sigthread_copy; pub mod sigtimer; +mod sigtimer_rearm; pub mod sigwaits; mod layout; mod links; diff --git a/userland/capsule_linux/src/linux/guest/siginfo.rs b/userland/capsule_linux/src/linux/guest/siginfo.rs index 15d66b982..e9861eb4c 100644 --- a/userland/capsule_linux/src/linux/guest/siginfo.rs +++ b/userland/capsule_linux/src/linux/guest/siginfo.rs @@ -20,6 +20,7 @@ //! kernel pid in one. pub const SI_USER: i32 = 0; +pub const SI_KERNEL: i32 = 0x80; pub const SI_TKILL: i32 = -6; pub const CLD_EXITED: i32 = 1; pub const CLD_KILLED: i32 = 2; diff --git a/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs index 9eec3ad8d..5347da912 100644 --- a/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs +++ b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs @@ -15,7 +15,8 @@ // along with this program. If not, see . //! Reading and recording a process's signal state: the disposition of each -//! signal, and what is pending for a thread. +//! signal, the nearest deadline its timers and waits have, and what is +//! pending for a thread. use super::sigqueue::Signals; use super::sigstate::{bit, SigAction, NSIG}; @@ -32,6 +33,11 @@ impl Signals { (1..=NSIG).contains(&signum).then(|| self.actions[signum - 1]) } + /// The nearest deadline of a timer, for the serve loop's wait to end by. + pub fn next_due(&self) -> Option { + self.real.map(|t| t.due) + } + /// Every signal pending for `tid` or for its process, as a mask. pub fn pending_for(&self, tid: u32) -> u64 { let mine = self.pending.iter().filter(|(t, _)| *t == tid || *t == 0); diff --git a/userland/capsule_linux/src/linux/guest/sigstate.rs b/userland/capsule_linux/src/linux/guest/sigstate.rs index afec90e40..a14a7e0b2 100644 --- a/userland/capsule_linux/src/linux/guest/sigstate.rs +++ b/userland/capsule_linux/src/linux/guest/sigstate.rs @@ -24,6 +24,7 @@ pub const NSIG: usize = 64; pub const SIGKILL: u8 = 9; pub const SIGSEGV: u8 = 11; pub const SIGPIPE: u8 = 13; +pub const SIGALRM: u8 = 14; pub const SIGCHLD: u8 = 17; pub const SIGSTOP: u8 = 19; diff --git a/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs b/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs new file mode 100644 index 000000000..8757af10a --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs @@ -0,0 +1,34 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! How ITIMER_REAL moves on once it fires: a periodic one past every period +//! that ended unseen, a one-shot one stops. Pure, so the arithmetic is checked +//! without a guest. + +use super::sigtimer::Itimer; + +impl Itimer { + /// Move past every period that has ended by `now`; false for a one-shot. + pub fn rearm(&mut self, now: u64) -> bool { + if self.interval == 0 { + return false; + } + while self.due <= now { + self.due = self.due.saturating_add(self.interval); + } + true + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_signal.rs b/userland/capsule_linux/src/linux/serve/family_signal.rs index 170da9ad2..2ff67d604 100644 --- a/userland/capsule_linux/src/linux/serve/family_signal.rs +++ b/userland/capsule_linux/src/linux/serve/family_signal.rs @@ -14,15 +14,22 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Signals between processes of the family, settled after every answer: the -//! outbox is routed, and each process's parked threads take what now reaches -//! them. +//! Signals between processes of the family, and the timers that raise them, +//! settled after every answer: the outbox is routed, due timers fire, and +//! each process's parked threads take what now reaches them. use super::family::Family; +use super::family_signal_fire::fire; +use crate::linux::call::now_ms; + +const CLOCK_MONOTONIC: u64 = 1; impl Family { pub(super) fn settle_signals(&mut self) { self.route_outbox(); + if let Some(now) = now_ms(CLOCK_MONOTONIC) { + self.guests.iter_mut().for_each(|g| fire(g, now)); + } self.guests.iter_mut().for_each(super::deliver_wait::settle); } } diff --git a/userland/capsule_linux/src/linux/serve/family_signal_fire.rs b/userland/capsule_linux/src/linux/serve/family_signal_fire.rs new file mode 100644 index 000000000..aade6a7c1 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/family_signal_fire.rs @@ -0,0 +1,29 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The timers of a process that are due: ITIMER_REAL raises SIGALRM. + +use crate::linux::guest::siginfo::{SigInfo, SI_KERNEL}; +use crate::linux::guest::sigstate::SIGALRM; +use crate::linux::guest::Guest; + +/// Every timer of `g` due by `now`, fired. +pub fn fire(g: &mut Guest, now: u64) { + if let Some(mut t) = g.signals.real.filter(|t| t.due <= now) { + let _ = g.signals.raise(0, SigInfo::from(SIGALRM, SI_KERNEL, 0)); + g.signals.real = t.rearm(now).then_some(t); + } +} diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index 8becd14e0..c1bf80640 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -31,6 +31,7 @@ mod family_pipes; mod family_reap; mod family_reap_end; mod family_signal; +mod family_signal_fire; mod family_signal_route; mod family_sleep; mod family_wait; diff --git a/userland/capsule_linux/src/linux/serve/table_sig.rs b/userland/capsule_linux/src/linux/serve/table_sig.rs index 8ee668761..bf69bb632 100644 --- a/userland/capsule_linux/src/linux/serve/table_sig.rs +++ b/userland/capsule_linux/src/linux/serve/table_sig.rs @@ -14,10 +14,10 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Signal dispositions, masks and stacks: the calls of that family that -//! answer at once. +//! Signal dispositions, masks and stacks, and the timers that end in a +//! signal: the calls of that family that answer at once. -use crate::linux::abi::nr; +use crate::linux::abi::{nr, nr_sig as ns}; use crate::linux::call; use crate::linux::guest::Guest; @@ -26,6 +26,9 @@ pub fn sig_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option nr::RT_SIGACTION => call::rt_sigaction(guest, a[0], a[1], a[2], a[3]), nr::RT_SIGPROCMASK => call::rt_sigprocmask(guest, tid, a[0], a[1], a[2], a[3]), nr::SIGALTSTACK => call::sigaltstack(guest, tid, a[0], a[1]), + ns::ALARM => call::alarm(guest, a[0]), + ns::SETITIMER => call::setitimer(guest, a[0], a[1], a[2]), + ns::GETITIMER => call::getitimer(guest, a[0], a[1]), _ => return None, }) } diff --git a/userland/capsule_linux_proofs/src/calls.rs b/userland/capsule_linux_proofs/src/calls.rs index 5eb542041..7fb933908 100644 --- a/userland/capsule_linux_proofs/src/calls.rs +++ b/userland/capsule_linux_proofs/src/calls.rs @@ -15,8 +15,9 @@ // along with this program. If not, see . //! The capsule's pure call code, mounted from its tree as it ships: the -//! signal frame built and read back, and the reading of exec's shebang line. -//! None of it names the capsule's crate, so it runs here without a guest. +//! signal frame built and read back, the arithmetic of the timers that end in +//! a signal, and the reading of exec's shebang line. None of it names the +//! capsule's crate, so it runs here without a guest. #[path = "../../capsule_linux/src/linux/call/sigframe.rs"] pub mod sigframe; @@ -27,5 +28,11 @@ pub mod sigframe_build; #[path = "../../capsule_linux/src/linux/call/sigframe_read.rs"] pub mod sigframe_read; +#[path = "../../capsule_linux/src/linux/guest/sigtimer.rs"] +pub mod sigtimer; + +#[path = "../../capsule_linux/src/linux/guest/sigtimer_rearm.rs"] +pub mod sigtimer_rearm; + #[path = "../../capsule_linux/src/linux/call/spawn/exec_shebang.rs"] pub mod exec_shebang; diff --git a/userland/capsule_linux_proofs/src/lib.rs b/userland/capsule_linux_proofs/src/lib.rs index 5aef0ee56..0bbaa500e 100644 --- a/userland/capsule_linux_proofs/src/lib.rs +++ b/userland/capsule_linux_proofs/src/lib.rs @@ -63,7 +63,7 @@ pub mod server; pub mod route; pub mod calls; -pub use calls::{exec_shebang, sigframe, sigframe_build, sigframe_read}; +pub use calls::{exec_shebang, sigframe, sigframe_build, sigframe_read, sigtimer}; #[cfg(test)] pub mod image; diff --git a/userland/capsule_linux_proofs/src/tests.rs b/userland/capsule_linux_proofs/src/tests.rs index 498353d57..c28d4a409 100644 --- a/userland/capsule_linux_proofs/src/tests.rs +++ b/userland/capsule_linux_proofs/src/tests.rs @@ -43,6 +43,7 @@ mod route_tests; mod sigframe_layout_tests; mod sigframe_mutation_tests; mod sigframe_tests; +mod sigtimer_tests; mod service; mod stack_words_tests; mod stat_tests; diff --git a/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs b/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs new file mode 100644 index 000000000..ed4e0c52d --- /dev/null +++ b/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs @@ -0,0 +1,33 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The arithmetic of ITIMER_REAL: a periodic timer moves past every period +//! that ended while it was not looked at, and a one-shot one stops. + +use crate::sigtimer::Itimer; + +#[test] +fn a_periodic_itimer_moves_past_every_period_that_ended() { + let mut t = Itimer { due: 1000, interval: 200 }; + assert!(t.rearm(1450)); + assert_eq!(t.due, 1600); +} + +#[test] +fn a_one_shot_itimer_does_not_fire_again() { + let mut t = Itimer { due: 1000, interval: 0 }; + assert!(!t.rearm(1000)); +} From 6e6f70515f2beedd5d34c8f7a89e8f75a52045af Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:03:30 +0000 Subject: [PATCH 12/26] linux: pause, sigsuspend, sigtimedwait and sigpending are served pause, rt_sigsuspend, rt_sigtimedwait and rt_sigpending were not served, so a program could not wait for a signal, take one without a handler, or see what it held blocked. pause parks the thread until a handler runs and answers EINTR through it. rt_sigsuspend does the same under the mask it is given, and the handler's return puts the old mask back, since the frame carries it. rt_sigtimedwait takes a pending signal of its set at once, or parks until one arrives, answered with its number and siginfo, or until its timeout runs out, answered EAGAIN. rt_sigpending names what is pending and blocked for the calling thread. The sigtimedwait deadlines join the timers in next_due. The three waits are routed from route_life, which dispatch does not ask yet, so they have no caller until it does. --- userland/capsule_linux/src/linux/call/mod.rs | 4 + .../src/linux/call/signal_timedwait.rs | 75 +++++++++++++++++++ .../src/linux/call/signal_wait.rs | 60 +++++++++++++++ .../src/linux/guest/sigqueue_ops.rs | 6 +- .../src/linux/serve/deliver_sigwait.rs | 39 ++++++++++ .../src/linux/serve/deliver_wait.rs | 10 ++- .../src/linux/serve/family_signal_fire.rs | 19 ++++- userland/capsule_linux/src/linux/serve/mod.rs | 1 + .../src/linux/serve/route_life.rs | 9 ++- .../src/linux/serve/table_sig.rs | 1 + 10 files changed, 213 insertions(+), 11 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/signal_timedwait.rs create mode 100644 userland/capsule_linux/src/linux/call/signal_wait.rs create mode 100644 userland/capsule_linux/src/linux/serve/deliver_sigwait.rs diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index 000e1525c..0ed1d7050 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -50,7 +50,9 @@ mod signal_queue; mod signal_real; mod signal_send; mod signal_stack; +mod signal_timedwait; mod signal_timer; +mod signal_wait; mod sigreturn; mod sleep; mod spawn; @@ -84,6 +86,8 @@ pub use signal_queue::{rt_sigqueueinfo, rt_tgsigqueueinfo}; pub use signal_send::{kill, kill_from, sigpipe, tgkill_from}; pub use signal_stack::sigaltstack; pub use signal_timer::{alarm, getitimer, setitimer}; +pub use signal_timedwait::rt_sigtimedwait; +pub use signal_wait::{pause, rt_sigpending, rt_sigsuspend}; pub use sigreturn::rt_sigreturn; pub use sleep::{clock_nanosleep, nanosleep}; pub use spawn::{clone, clone_process, execve, fork, vfork, wait4, wait4_usage, waitid}; diff --git a/userland/capsule_linux/src/linux/call/signal_timedwait.rs b/userland/capsule_linux/src/linux/call/signal_timedwait.rs new file mode 100644 index 000000000..0f58217e9 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_timedwait.rs @@ -0,0 +1,75 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! rt_sigtimedwait takes a pending signal of its set without running a +//! handler, or parks until one comes or its time runs out with EAGAIN. + +use super::signal_wait::{read_set, SIGSET_LEN}; +use crate::linux::abi::errno; +use crate::linux::call::now_ms; +use crate::linux::guest::sigwaits::SigWait; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +const CLOCK_MONOTONIC: u64 = 1; +const NSEC: u64 = 1_000_000_000; + +pub fn rt_sigtimedwait( + guest: &mut Guest, + tid: u32, + set: u64, + info: u64, + ts: u64, + size: u64, +) -> Answer { + if size != SIGSET_LEN { + return Answer::value(errno::fail(errno::EINVAL)); + } + let Some(want) = read_set(guest, set) else { + return Answer::value(errno::fail(errno::EFAULT)); + }; + let due = match ts { + 0 => None, + at => match wait_ms(guest, at) { + Ok(ms) => now_ms(CLOCK_MONOTONIC).map(|now| now.saturating_add(ms)), + Err(e) => return Answer::value(e), + }, + }; + /* A signal of the set already waiting is taken now, whatever the timeout. */ + if let Some(got) = guest.signals.take(tid, want) { + let bytes = got.bytes(); + if info != 0 && guest.write(info, &bytes) < bytes.len() as i64 { + return Answer::value(errno::fail(errno::EFAULT)); + } + return Answer::value(errno::ok(u64::from(got.signo))); + } + if due.is_some_and(|d| now_ms(CLOCK_MONOTONIC).is_some_and(|now| d <= now)) { + return Answer::value(errno::fail(errno::EAGAIN)); + } + guest.signals.sigwaits.push(SigWait { tid, set: want, info, due }); + Answer::Park +} + +/// A relative timespec, in whole milliseconds rounded up. +fn wait_ms(guest: &Guest, at: u64) -> Result { + let raw = guest.read(at, 16).ok_or(errno::fail(errno::EFAULT))?; + let secs = u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8])); + let nanos = u64::from_le_bytes(raw[8..].try_into().unwrap_or([0; 8])); + if nanos >= NSEC || secs > i64::MAX as u64 { + return Err(errno::fail(errno::EINVAL)); + } + Ok(secs.saturating_mul(1000).saturating_add(nanos.div_ceil(1_000_000))) +} diff --git a/userland/capsule_linux/src/linux/call/signal_wait.rs b/userland/capsule_linux/src/linux/call/signal_wait.rs new file mode 100644 index 000000000..d90dfbadb --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_wait.rs @@ -0,0 +1,60 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Waiting for a signal. pause and rt_sigsuspend park until a handler runs, +//! and answer EINTR through it; sigsuspend waits under the mask it was given +//! and the handler's return puts the old one back. rt_sigpending names what +//! waits; rt_sigtimedwait is in signal_timedwait. + +use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::SigWait; +use crate::linux::guest::Guest; +use crate::linux::serve::Answer; + +pub const SIGSET_LEN: u64 = 8; + +pub fn pause(guest: &mut Guest, tid: u32) -> Answer { + guest.signals.sigwaits.push(SigWait { tid, set: 0, info: 0, due: None }); + Answer::Park +} + +pub fn rt_sigsuspend(guest: &mut Guest, tid: u32, set: u64, size: u64) -> Answer { + if size != SIGSET_LEN { + return Answer::value(errno::fail(errno::EINVAL)); + } + let Some(mask) = read_set(guest, set) else { + return Answer::value(errno::fail(errno::EFAULT)); + }; + let old = guest.signals.blocked(tid); + guest.signals.set_blocked(tid, mask); + guest.signals.thread(tid).saved = Some(old); + pause(guest, tid) +} + +pub fn rt_sigpending(guest: &mut Guest, tid: u32, out: u64, size: u64) -> u64 { + if size > SIGSET_LEN { + return errno::fail(errno::EINVAL); + } + let held = guest.signals.pending_for(tid) & guest.signals.blocked(tid); + match guest.write(out, &held.to_le_bytes()[..size as usize]) < size as i64 { + true => errno::fail(errno::EFAULT), + false => errno::ok(0), + } +} + +pub fn read_set(guest: &Guest, at: u64) -> Option { + guest.read(at, 8).map(|raw| u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8]))) +} diff --git a/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs index 5347da912..3ec8711f2 100644 --- a/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs +++ b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs @@ -33,9 +33,11 @@ impl Signals { (1..=NSIG).contains(&signum).then(|| self.actions[signum - 1]) } - /// The nearest deadline of a timer, for the serve loop's wait to end by. + /// The nearest deadline of a timer or a sigtimedwait, for the serve loop's + /// wait to end by. pub fn next_due(&self) -> Option { - self.real.map(|t| t.due) + let waits = self.sigwaits.iter().filter_map(|w| w.due); + self.real.map(|t| t.due).into_iter().chain(waits).min() } /// Every signal pending for `tid` or for its process, as a mask. diff --git a/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs b/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs new file mode 100644 index 000000000..ed8aa69fa --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs @@ -0,0 +1,39 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A thread in sigtimedwait takes a pending signal of its set without any +//! handler running, and its call answers with that signal's number, the +//! siginfo written where it asked. + +use nonos_libc::mk_foreign_reply; + +use crate::linux::guest::Guest; + +/// A thread in sigtimedwait takes a pending signal of its set, and its call +/// answers with that signal's number and siginfo. +pub fn taken_by_sigtimedwait(guest: &mut Guest) { + for w in guest.signals.sigwaits.clone().into_iter().filter(|w| w.set != 0) { + let Some(info) = guest.signals.take(w.tid, w.set) else { + continue; + }; + guest.signals.sigwaits.retain(|x| x.tid != w.tid); + let value = match w.info != 0 && guest.write(w.info, &info.bytes()) < 128 { + true => crate::linux::abi::errno::fail(crate::linux::abi::errno::EFAULT), + false => u64::from(info.signo), + }; + let _ = mk_foreign_reply(w.tid, value); + } +} diff --git a/userland/capsule_linux/src/linux/serve/deliver_wait.rs b/userland/capsule_linux/src/linux/serve/deliver_wait.rs index b04af8d4e..3cd8d2bbc 100644 --- a/userland/capsule_linux/src/linux/serve/deliver_wait.rs +++ b/userland/capsule_linux/src/linux/serve/deliver_wait.rs @@ -15,19 +15,21 @@ // along with this program. If not, see . //! Signals that reach a process whose threads are parked, not returning. -//! Run after every answer: an uncaught signal's default acts at once, as -//! Linux's does when it is sent; and a caught signal ends the wait of a -//! thread that does not block it, with EINTR, or restarts the call under -//! SA_RESTART where Linux restarts it. +//! Run after every answer: a sigtimedwait takes a signal in its set; an +//! uncaught signal's default acts at once, as Linux's does when it is sent; +//! and a caught signal ends the wait of a thread that does not block it, with +//! EINTR, or restarts the call under SA_RESTART where Linux restarts it. use super::deliver_interrupt::interrupt; use super::deliver_say::stop_unserved; +use super::deliver_sigwait::taken_by_sigtimedwait; use crate::linux::call::killed; use crate::linux::guest::sigdefault::{default_of, Default}; use crate::linux::guest::sigstate::bit; use crate::linux::guest::Guest; pub fn settle(guest: &mut Guest) { + taken_by_sigtimedwait(guest); for (tid, signo) in guest.signals.queued() { if guest.exited.is_some() { return; diff --git a/userland/capsule_linux/src/linux/serve/family_signal_fire.rs b/userland/capsule_linux/src/linux/serve/family_signal_fire.rs index aade6a7c1..00c825e86 100644 --- a/userland/capsule_linux/src/linux/serve/family_signal_fire.rs +++ b/userland/capsule_linux/src/linux/serve/family_signal_fire.rs @@ -14,16 +14,31 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! The timers of a process that are due: ITIMER_REAL raises SIGALRM. +//! The timers of a process that are due: ITIMER_REAL raises SIGALRM, and a +//! sigtimedwait whose time is up answers EAGAIN. +use super::family_wait::answer; +use crate::linux::abi::errno; use crate::linux::guest::siginfo::{SigInfo, SI_KERNEL}; use crate::linux::guest::sigstate::SIGALRM; use crate::linux::guest::Guest; -/// Every timer of `g` due by `now`, fired. +/// Every timer of `g` due by `now`, fired, and every sigtimedwait whose time +/// is up answered EAGAIN. pub fn fire(g: &mut Guest, now: u64) { if let Some(mut t) = g.signals.real.filter(|t| t.due <= now) { let _ = g.signals.raise(0, SigInfo::from(SIGALRM, SI_KERNEL, 0)); g.signals.real = t.rearm(now).then_some(t); } + let late: alloc::vec::Vec = g + .signals + .sigwaits + .iter() + .filter(|w| w.due.is_some_and(|d| d <= now)) + .map(|w| w.tid) + .collect(); + for tid in late { + g.signals.sigwaits.retain(|w| w.tid != tid); + answer(g, tid, errno::fail(errno::EAGAIN)); + } } diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index c1bf80640..684898bec 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -23,6 +23,7 @@ mod deliver_interrupt; mod deliver_rem; mod deliver_restart; mod deliver_say; +mod deliver_sigwait; mod deliver_stack; mod deliver_wait; mod dispatch; diff --git a/userland/capsule_linux/src/linux/serve/route_life.rs b/userland/capsule_linux/src/linux/serve/route_life.rs index cfd6dc95c..59bb3972e 100644 --- a/userland/capsule_linux/src/linux/serve/route_life.rs +++ b/userland/capsule_linux/src/linux/serve/route_life.rs @@ -15,9 +15,9 @@ // along with this program. If not, see . //! The calls of process lifecycle and signals that can leave their caller -//! parked: a plain exit, a new process, a wait for a child, and a signal sent -//! where only the family can say whether anyone received it. Asked first by -//! `dispatch`, so these are answered here whatever it holds. +//! parked: a plain exit, a new process, a wait for a child or a signal, and a +//! signal sent where only the family can say whether anyone received it. +//! Asked first by `dispatch`, so these are answered here whatever it holds. use nonos_libc::ForeignFrame; @@ -48,6 +48,9 @@ pub fn answer(guest: &mut Guest, frame: &ForeignFrame) -> Option { ns::TGKILL => call::tgkill_from(guest, tid, a[0], a[1], a[2]), ns::RT_SIGQUEUEINFO => call::rt_sigqueueinfo(guest, tid, a[0], a[1], a[2]), ns::RT_TGSIGQUEUEINFO => call::rt_tgsigqueueinfo(guest, tid, a[0], a[1], a[2], a[3]), + ns::PAUSE => call::pause(guest, tid), + ns::RT_SIGSUSPEND => call::rt_sigsuspend(guest, tid, a[0], a[1]), + ns::RT_SIGTIMEDWAIT => call::rt_sigtimedwait(guest, tid, a[0], a[1], a[2], a[3]), _ => return None, }) } diff --git a/userland/capsule_linux/src/linux/serve/table_sig.rs b/userland/capsule_linux/src/linux/serve/table_sig.rs index bf69bb632..2ae4f9c7f 100644 --- a/userland/capsule_linux/src/linux/serve/table_sig.rs +++ b/userland/capsule_linux/src/linux/serve/table_sig.rs @@ -26,6 +26,7 @@ pub fn sig_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option nr::RT_SIGACTION => call::rt_sigaction(guest, a[0], a[1], a[2], a[3]), nr::RT_SIGPROCMASK => call::rt_sigprocmask(guest, tid, a[0], a[1], a[2], a[3]), nr::SIGALTSTACK => call::sigaltstack(guest, tid, a[0], a[1]), + ns::RT_SIGPENDING => call::rt_sigpending(guest, tid, a[0], a[1]), ns::ALARM => call::alarm(guest, a[0]), ns::SETITIMER => call::setitimer(guest, a[0], a[1], a[2]), ns::GETITIMER => call::getitimer(guest, a[0], a[1]), From 949b45224fb8db02d8c66219e948daa2ca5673d1 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:03:35 +0000 Subject: [PATCH 13/26] linux: posix timers are served, with overruns counted as linux does timer_create, timer_settime, timer_gettime, timer_getoverrun and timer_delete were not served, so a program using POSIX timers failed at the first call. A process can now make timers on the realtime, monotonic, boottime, alarm and TAI clocks, numbered from 0 as Linux numbers them. A timer raises its signal at the process, or at the one thread SIGEV_THREAD_ID names, with SI_TIMER, its id and sigev_value; a NULL sigevent means SIGALRM with the id as its value, and SIGEV_NONE raises nothing. timer_settime arms it relative or absolute, once or with a period, and hands back the old setting. An expiry while the timer's signal still waits counts as an overrun, carried by the signal when it is taken and reported by timer_getoverrun. Deleting a timer drops its queued signal, and a timer whose signal was discarded may queue again. Exec drops every timer and fork gives the child none. Timers on a CPU-time clock are refused by name, since the kernel does not report a guest's CPU time to its supervisor. The host proofs check the overruns a late timer counts. --- userland/capsule_linux/src/linux/call/mod.rs | 8 +++ .../src/linux/call/signal_its.rs | 51 ++++++++++++++ .../src/linux/call/timer_create.rs | 68 +++++++++++++++++++ .../capsule_linux/src/linux/call/timer_ops.rs | 59 ++++++++++++++++ .../capsule_linux/src/linux/call/timer_set.rs | 56 +++++++++++++++ .../src/linux/call/timer_sigev.rs | 54 +++++++++++++++ .../capsule_linux/src/linux/guest/siginfo.rs | 1 + .../capsule_linux/src/linux/guest/siglive.rs | 1 + .../src/linux/guest/sigqueue_ops.rs | 3 +- .../capsule_linux/src/linux/guest/sigtake.rs | 31 +++++++-- .../src/linux/guest/sigtimer_rearm.rs | 27 ++++++-- .../src/linux/serve/family_signal_fire.rs | 25 ++++++- .../src/linux/serve/table_sig.rs | 5 ++ .../src/tests/sigtimer_tests.rs | 46 ++++++++++++- 14 files changed, 420 insertions(+), 15 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/signal_its.rs create mode 100644 userland/capsule_linux/src/linux/call/timer_create.rs create mode 100644 userland/capsule_linux/src/linux/call/timer_ops.rs create mode 100644 userland/capsule_linux/src/linux/call/timer_set.rs create mode 100644 userland/capsule_linux/src/linux/call/timer_sigev.rs diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index 0ed1d7050..272844132 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -43,6 +43,7 @@ pub mod sigframe; pub mod sigframe_build; mod sigframe_read; mod signal; +mod signal_its; mod signal_itv; mod signal_mask; mod signal_post; @@ -57,6 +58,10 @@ mod sigreturn; mod sleep; mod spawn; mod thread; +mod timer_create; +mod timer_ops; +mod timer_set; +mod timer_sigev; mod timeops; mod umask; mod uname; @@ -95,6 +100,9 @@ pub use spawn::{WALL, WCLONE, WEXITED, WNOHANG, WNOWAIT}; pub use clock::{clock_getres, clock_gettime, now_ms}; pub use epoch::{family_ms, mark_start}; pub use thread::{arch_prctl, getrandom}; +pub use timer_create::timer_create; +pub use timer_ops::{timer_delete, timer_getoverrun, timer_gettime}; +pub use timer_set::timer_settime; pub use timeops::{gettimeofday, time}; pub use umask::{umask, DEFAULT_UMASK}; pub use uname::uname; diff --git a/userland/capsule_linux/src/linux/call/signal_its.rs b/userland/capsule_linux/src/linux/call/signal_its.rs new file mode 100644 index 000000000..59500b6c2 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signal_its.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `struct itimerspec` in and out: interval, then value, each seconds and +//! nanoseconds. Nanoseconds round up to the millisecond the clock keeps. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +const NSEC: u64 = 1_000_000_000; +const MS: u64 = 1_000_000; + +/// (interval, value) in milliseconds, or the errno a bad one earns. +pub fn read_spec(guest: &Guest, at: u64) -> Result<(u64, u64), u64> { + let raw = guest.read(at, 32).ok_or(errno::fail(errno::EFAULT))?; + let w = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + let ms = |s: u64, ns: u64| match ns < NSEC && (s as i64) >= 0 { + true => Ok(s.saturating_mul(1000).saturating_add(ns.div_ceil(MS))), + false => Err(errno::fail(errno::EINVAL)), + }; + Ok((ms(w(0), w(8))?, ms(w(16), w(24))?)) +} + +/// Write (interval, value) at `at`, if the caller gave somewhere to write. +pub fn write_spec(guest: &Guest, at: u64, interval: u64, value: u64) -> u64 { + if at == 0 { + return errno::ok(0); + } + let mut b = [0u8; 32]; + for (i, ms) in [interval, value].iter().enumerate() { + b[i * 16..i * 16 + 8].copy_from_slice(&(ms / 1000).to_le_bytes()); + b[i * 16 + 8..i * 16 + 16].copy_from_slice(&((ms % 1000) * MS).to_le_bytes()); + } + match guest.write(at, &b) < 32 { + true => errno::fail(errno::EFAULT), + false => errno::ok(0), + } +} diff --git a/userland/capsule_linux/src/linux/call/timer_create.rs b/userland/capsule_linux/src/linux/call/timer_create.rs new file mode 100644 index 000000000..e045d7f43 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/timer_create.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! timer_create: a POSIX timer, disarmed. Ids count from 0 per process, as +//! Linux's do. Its signal goes to the process, or to the one thread +//! SIGEV_THREAD_ID names; a NULL sigevent means SIGALRM with the id as its +//! value. Timers on a CPU-time clock are refused by name: the kernel does not +//! report a guest's CPU time to its supervisor. + +use super::timer_sigev::read_sigevent; +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::SIGALRM; +use crate::linux::guest::sigtimer::PosixTimer; +use crate::linux::guest::Guest; + +const TIMERS_MAX: usize = 1024; + +pub fn timer_create(guest: &mut Guest, clock: u64, sevp: u64, out: u64) -> u64 { + match clock { + 0 | 1 | 7..=9 | 11 => {} + 2 | 3 => { + let line = b"[LINUX] unserved timer_create: no CPU-time clock\n"; + let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + return errno::fail(errno::ENOTSUP); + } + 4..=6 => return errno::fail(errno::ENOTSUP), + _ => return errno::fail(errno::EINVAL), + } + let id = (0..).find(|i| !guest.signals.timers.iter().any(|t| t.id == *i)).unwrap_or(0); + let mut t = PosixTimer { + id, + clock, + signo: SIGALRM, + tid: 0, + value: id as u64, + due: None, + interval: 0, + overrun: 0, + last_overrun: 0, + queued: false, + }; + if sevp != 0 { + if let Err(e) = read_sigevent(guest, sevp, &mut t) { + return e; + } + } + if guest.signals.timers.len() >= TIMERS_MAX { + return errno::fail(errno::EAGAIN); + } + if guest.write(out, &id.to_le_bytes()) < 4 { + return errno::fail(errno::EFAULT); + } + guest.signals.timers.push(t); + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/timer_ops.rs b/userland/capsule_linux/src/linux/call/timer_ops.rs new file mode 100644 index 000000000..7d6eb947b --- /dev/null +++ b/userland/capsule_linux/src/linux/call/timer_ops.rs @@ -0,0 +1,59 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! timer_gettime, timer_getoverrun and timer_delete; timer_settime is in +//! timer_set. Expiry raises the timer's signal with SI_TIMER, its id and +//! sigev_value; an expiry while that signal still waits counts as an overrun, +//! reported by the signal and by timer_getoverrun. + +use crate::linux::abi::errno; +use crate::linux::call::now_ms; +use crate::linux::guest::Guest; + +use super::signal_its::write_spec; + +const CLOCK_MONOTONIC: u64 = 1; + +pub fn timer_gettime(guest: &mut Guest, id: u64, out: u64) -> u64 { + let Some(at) = find(guest, id) else { + return errno::fail(errno::EINVAL); + }; + let t = guest.signals.timers[at]; + let now = now_ms(CLOCK_MONOTONIC).unwrap_or(0); + let left = t.due.map_or(0, |d| d.saturating_sub(now).max(1)); + write_spec(guest, out, t.interval, left) +} + +pub fn timer_getoverrun(guest: &mut Guest, id: u64) -> u64 { + match find(guest, id) { + Some(at) => errno::ok(guest.signals.timers[at].last_overrun as u64), + None => errno::fail(errno::EINVAL), + } +} + +pub fn timer_delete(guest: &mut Guest, id: u64) -> u64 { + let Some(at) = find(guest, id) else { + return errno::fail(errno::EINVAL); + }; + let t = guest.signals.timers.remove(at); + guest.signals.drop_timer_signal(t.id); + errno::ok(0) +} + +pub fn find(guest: &Guest, id: u64) -> Option { + let id = i32::try_from(id).ok()?; + guest.signals.timers.iter().position(|t| t.id == id) +} diff --git a/userland/capsule_linux/src/linux/call/timer_set.rs b/userland/capsule_linux/src/linux/call/timer_set.rs new file mode 100644 index 000000000..83041f6e4 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/timer_set.rs @@ -0,0 +1,56 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! timer_settime: arm a timer for a relative time, or an absolute one on its +//! own clock, repeating every interval; a zero value disarms it. The old +//! setting is written first when the caller asked for it. + +use super::signal_its::read_spec; +use super::timer_ops::{find, timer_gettime}; +use crate::linux::abi::errno; +use crate::linux::call::now_ms; +use crate::linux::guest::Guest; + +const TIMER_ABSTIME: u64 = 1; +const CLOCK_MONOTONIC: u64 = 1; + +pub fn timer_settime(guest: &mut Guest, id: u64, flags: u64, new: u64, old: u64) -> u64 { + let Some(at) = find(guest, id) else { + return errno::fail(errno::EINVAL); + }; + if new == 0 { + return errno::fail(errno::EINVAL); + } + let (interval, value) = match read_spec(guest, new) { + Ok(v) => v, + Err(e) => return e, + }; + let rc = timer_gettime(guest, id, old); + let t = &mut guest.signals.timers[at]; + let now = now_ms(CLOCK_MONOTONIC).unwrap_or(0); + let wait = match flags & TIMER_ABSTIME { + 0 => value, + _ => value.saturating_sub(now_ms(t.clock).unwrap_or(0)), + }; + t.due = (value != 0).then(|| now.saturating_add(wait)); + t.interval = if value == 0 { 0 } else { interval }; + t.overrun = 0; + if old != 0 { + rc + } else { + errno::ok(0) + } +} diff --git a/userland/capsule_linux/src/linux/call/timer_sigev.rs b/userland/capsule_linux/src/linux/call/timer_sigev.rs new file mode 100644 index 000000000..8be193c15 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/timer_sigev.rs @@ -0,0 +1,54 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The sigevent timer_create reads: which signal a timer raises and with what +//! value, SIGEV_NONE for none, and SIGEV_THREAD_ID for one thread of the +//! process rather than the whole of it. + +use crate::linux::abi::errno; +use crate::linux::guest::sigstate::NSIG; +use crate::linux::guest::sigtimer::PosixTimer; +use crate::linux::guest::Guest; + +const SIGEV_SIGNAL: u32 = 0; +const SIGEV_NONE: u32 = 1; +const SIGEV_THREAD: u32 = 2; +const SIGEV_THREAD_ID: u32 = 4; + +/// Read the sigevent at `sevp` into `t`, or the errno a bad one earns. +pub fn read_sigevent(guest: &Guest, sevp: u64, t: &mut PosixTimer) -> Result<(), u64> { + let Some(raw) = guest.read(sevp, 20) else { + return Err(errno::fail(errno::EFAULT)); + }; + let word = |i: usize| u32::from_le_bytes(raw[i..i + 4].try_into().unwrap_or([0; 4])); + let (signo, notify) = (word(8), word(12)); + t.value = u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8])); + t.signo = if notify == SIGEV_NONE { 0 } else { signo as u8 }; + let bad_signo = notify != SIGEV_NONE && (signo == 0 || signo as usize > NSIG); + match notify { + _ if bad_signo => return Err(errno::fail(errno::EINVAL)), + SIGEV_SIGNAL | SIGEV_NONE | SIGEV_THREAD => {} + n if n == SIGEV_SIGNAL | SIGEV_THREAD_ID => { + let tid = crate::linux::serve::kernel_pid(word(16)); + match tid.filter(|k| guest.live_threads().contains(k)) { + Some(k) => t.tid = k, + None => return Err(errno::fail(errno::EINVAL)), + } + } + _ => return Err(errno::fail(errno::EINVAL)), + } + Ok(()) +} diff --git a/userland/capsule_linux/src/linux/guest/siginfo.rs b/userland/capsule_linux/src/linux/guest/siginfo.rs index e9861eb4c..da0a7118b 100644 --- a/userland/capsule_linux/src/linux/guest/siginfo.rs +++ b/userland/capsule_linux/src/linux/guest/siginfo.rs @@ -21,6 +21,7 @@ pub const SI_USER: i32 = 0; pub const SI_KERNEL: i32 = 0x80; +pub const SI_TIMER: i32 = -2; pub const SI_TKILL: i32 = -6; pub const CLD_EXITED: i32 = 1; pub const CLD_KILLED: i32 = 2; diff --git a/userland/capsule_linux/src/linux/guest/siglive.rs b/userland/capsule_linux/src/linux/guest/siglive.rs index bfe811c33..ef6627502 100644 --- a/userland/capsule_linux/src/linux/guest/siglive.rs +++ b/userland/capsule_linux/src/linux/guest/siglive.rs @@ -36,5 +36,6 @@ impl Guest { let _ = self.leave_waits(tid); self.signals.threads.retain(|t| t.tid != tid); self.signals.pending.retain(|(t, _)| *t != tid); + self.signals.timers_requeue(); } } diff --git a/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs index 3ec8711f2..d475cf10b 100644 --- a/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs +++ b/userland/capsule_linux/src/linux/guest/sigqueue_ops.rs @@ -36,8 +36,9 @@ impl Signals { /// The nearest deadline of a timer or a sigtimedwait, for the serve loop's /// wait to end by. pub fn next_due(&self) -> Option { + let timers = self.timers.iter().filter_map(|t| t.due); let waits = self.sigwaits.iter().filter_map(|w| w.due); - self.real.map(|t| t.due).into_iter().chain(waits).min() + self.real.map(|t| t.due).into_iter().chain(timers).chain(waits).min() } /// Every signal pending for `tid` or for its process, as a mask. diff --git a/userland/capsule_linux/src/linux/guest/sigtake.rs b/userland/capsule_linux/src/linux/guest/sigtake.rs index ab8116de6..9ee8588c7 100644 --- a/userland/capsule_linux/src/linux/guest/sigtake.rs +++ b/userland/capsule_linux/src/linux/guest/sigtake.rs @@ -15,8 +15,8 @@ // along with this program. If not, see . //! Taking a signal, in Linux's order: the lowest-numbered first, the thread's -//! own before the process's; and dropping what a newly ignored signal leaves -//! pending. +//! own before the process's; and dropping what a deleted timer or a newly +//! ignored signal leaves pending. use super::siginfo::SigInfo; use super::sigqueue::Signals; @@ -24,7 +24,8 @@ use super::sigstate::bit; impl Signals { /// Take the next signal `tid` may take: one of `allow`, its own before the - /// process's, lowest first. + /// process's, lowest first. A timer's signal carries the overruns counted + /// while it waited, and its timer is free to queue again. pub fn take(&mut self, tid: u32, allow: u64) -> Option { let lowest = |want: u32| { let each = self.pending.iter().enumerate(); @@ -32,12 +33,34 @@ impl Signals { fit.min_by_key(|(_, (_, i))| i.signo).map(|(at, _)| at) }; let at = lowest(tid).or_else(|| lowest(0))?; - Some(self.pending.remove(at).1) + let mut info = self.pending.remove(at).1; + if let Some((id, _)) = info.timer { + if let Some(t) = self.timers.iter_mut().find(|t| t.id == id) { + info.timer = Some((id, t.overrun)); + t.last_overrun = t.overrun; + t.overrun = 0; + t.queued = false; + } + } + Some(info) + } + + /// Drop the signal a deleted timer left queued. + pub fn drop_timer_signal(&mut self, id: i32) { + self.pending.retain(|(_, i)| i.timer.is_none_or(|(t, _)| t != id)); } /// Drop every pending `signum`, as setting SIG_IGN on it does. pub fn discard(&mut self, signum: u8) { self.pending.retain(|(_, i)| i.signo != signum); + self.timers_requeue(); + } + + /// A timer whose queued signal was dropped may queue again. + pub fn timers_requeue(&mut self) { + for t in self.timers.iter_mut().filter(|t| t.queued) { + t.queued = self.pending.iter().any(|(_, i)| i.timer.is_some_and(|(id, _)| id == t.id)); + } } /// Each pending signal with the thread it is for, 0 for the process. diff --git a/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs b/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs index 8757af10a..d1ac5df70 100644 --- a/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs +++ b/userland/capsule_linux/src/linux/guest/sigtimer_rearm.rs @@ -14,11 +14,12 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! How ITIMER_REAL moves on once it fires: a periodic one past every period -//! that ended unseen, a one-shot one stops. Pure, so the arithmetic is checked -//! without a guest. +//! How a timer that ends in a signal moves on once it fires: a periodic one +//! past every period that ended unseen, counted as overruns for a POSIX +//! timer; a one-shot one stops. Pure, so the arithmetic is checked without a +//! guest. -use super::sigtimer::Itimer; +use super::sigtimer::{Itimer, PosixTimer}; impl Itimer { /// Move past every period that has ended by `now`; false for a one-shot. @@ -32,3 +33,21 @@ impl Itimer { true } } + +impl PosixTimer { + /// Fire at `now`: the periods that passed unseen are overruns, as Linux + /// counts them. False when the timer does not fire again. + pub fn rearm(&mut self, now: u64) -> bool { + let Some(due) = self.due else { + return false; + }; + if self.interval == 0 { + self.due = None; + return false; + } + let missed = now.saturating_sub(due) / self.interval; + self.overrun = self.overrun.saturating_add(missed.min(i32::MAX as u64) as i32); + self.due = Some(due + (missed + 1) * self.interval); + true + } +} diff --git a/userland/capsule_linux/src/linux/serve/family_signal_fire.rs b/userland/capsule_linux/src/linux/serve/family_signal_fire.rs index 00c825e86..a91dad558 100644 --- a/userland/capsule_linux/src/linux/serve/family_signal_fire.rs +++ b/userland/capsule_linux/src/linux/serve/family_signal_fire.rs @@ -14,12 +14,13 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! The timers of a process that are due: ITIMER_REAL raises SIGALRM, and a -//! sigtimedwait whose time is up answers EAGAIN. +//! The timers of a process that are due: ITIMER_REAL raises SIGALRM, a POSIX +//! timer raises its own signal or counts an overrun while that still waits, +//! and a sigtimedwait whose time is up answers EAGAIN. use super::family_wait::answer; use crate::linux::abi::errno; -use crate::linux::guest::siginfo::{SigInfo, SI_KERNEL}; +use crate::linux::guest::siginfo::{SigInfo, SI_KERNEL, SI_TIMER}; use crate::linux::guest::sigstate::SIGALRM; use crate::linux::guest::Guest; @@ -30,6 +31,24 @@ pub fn fire(g: &mut Guest, now: u64) { let _ = g.signals.raise(0, SigInfo::from(SIGALRM, SI_KERNEL, 0)); g.signals.real = t.rearm(now).then_some(t); } + for i in 0..g.signals.timers.len() { + let mut t = g.signals.timers[i]; + if t.due.is_none_or(|d| d > now) { + continue; + } + if t.queued { + t.overrun = t.overrun.saturating_add(1); + } else if t.signo != 0 { + let info = SigInfo { + timer: Some((t.id, 0)), + value: t.value, + ..SigInfo::from(t.signo, SI_TIMER, 0) + }; + t.queued = g.signals.raise(t.tid, info); + } + let _ = t.rearm(now); + g.signals.timers[i] = t; + } let late: alloc::vec::Vec = g .signals .sigwaits diff --git a/userland/capsule_linux/src/linux/serve/table_sig.rs b/userland/capsule_linux/src/linux/serve/table_sig.rs index 2ae4f9c7f..640c966ad 100644 --- a/userland/capsule_linux/src/linux/serve/table_sig.rs +++ b/userland/capsule_linux/src/linux/serve/table_sig.rs @@ -30,6 +30,11 @@ pub fn sig_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option ns::ALARM => call::alarm(guest, a[0]), ns::SETITIMER => call::setitimer(guest, a[0], a[1], a[2]), ns::GETITIMER => call::getitimer(guest, a[0], a[1]), + ns::TIMER_CREATE => call::timer_create(guest, a[0], a[1], a[2]), + ns::TIMER_SETTIME => call::timer_settime(guest, a[0], a[1], a[2], a[3]), + ns::TIMER_GETTIME => call::timer_gettime(guest, a[0], a[1]), + ns::TIMER_GETOVERRUN => call::timer_getoverrun(guest, a[0]), + ns::TIMER_DELETE => call::timer_delete(guest, a[0]), _ => return None, }) } diff --git a/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs b/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs index ed4e0c52d..4183ff3a3 100644 --- a/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs +++ b/userland/capsule_linux_proofs/src/tests/sigtimer_tests.rs @@ -14,10 +14,27 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! The arithmetic of ITIMER_REAL: a periodic timer moves past every period -//! that ended while it was not looked at, and a one-shot one stops. +//! The arithmetic of the timers that end in a signal: a periodic timer moves +//! past every period that ended while it was not looked at, a one-shot one +//! stops, and a POSIX timer counts the periods it missed as overruns, as +//! Linux's timer_getoverrun reports them. -use crate::sigtimer::Itimer; +use crate::sigtimer::{Itimer, PosixTimer}; + +fn posix(due: u64, interval: u64) -> PosixTimer { + PosixTimer { + id: 0, + clock: 1, + signo: 34, + tid: 0, + value: 0, + due: Some(due), + interval, + overrun: 0, + last_overrun: 0, + queued: false, + } +} #[test] fn a_periodic_itimer_moves_past_every_period_that_ended() { @@ -31,3 +48,26 @@ fn a_one_shot_itimer_does_not_fire_again() { let mut t = Itimer { due: 1000, interval: 0 }; assert!(!t.rearm(1000)); } + +#[test] +fn a_posix_timer_fired_late_counts_the_periods_it_missed() { + let mut t = posix(1000, 100); + assert!(t.rearm(1350)); + assert_eq!(t.overrun, 3); + assert_eq!(t.due, Some(1400)); +} + +#[test] +fn a_posix_timer_fired_on_time_counts_nothing() { + let mut t = posix(1000, 100); + assert!(t.rearm(1000)); + assert_eq!(t.overrun, 0); + assert_eq!(t.due, Some(1100)); +} + +#[test] +fn a_one_shot_posix_timer_disarms() { + let mut t = posix(1000, 0); + assert!(!t.rearm(1200)); + assert_eq!(t.due, None); +} From 478e293f9f3f2693849873bef162f239ce30a6a2 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:03:48 +0000 Subject: [PATCH 14/26] foreign: MkForeignFork takes an optional stack for the child MkForeignFork always started the child on its parent's stack pointer. A supervisor serving a clone that makes a process on a stack of its own, as musl's posix_spawn does, had no way to ask for that stack, so it could only refuse the call. MkForeignFork now takes the child's stack pointer as its second argument. Zero keeps the parent's, as every caller passes today, and anything else is where the child starts; a stack outside the user half is refused with EINVAL. The parked frame is read where it was before, only without the one-line wrapper around it. --- src/process/foreign/fork.rs | 31 +++++++++++++-------- src/syscall/microkernel/dispatch/process.rs | 2 +- 2 files changed, 21 insertions(+), 12 deletions(-) diff --git a/src/process/foreign/fork.rs b/src/process/foreign/fork.rs index feed314d9..b5ec85e6d 100644 --- a/src/process/foreign/fork.rs +++ b/src/process/foreign/fork.rs @@ -20,10 +20,18 @@ use super::peer_guard::pid_arg; use crate::process::core::ProcessState; use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_NOENT, ERRNO_PERM}; +/// The last address of the user half. +const USER_VA_MAX: u64 = 0x0000_7FFF_FFFF_FFFF; + /// `MkForeignFork`: a second process holding the first one's register state, /// with zero in its return register so the two can tell each other apart, -/// which is the whole of fork's contract to the program. -pub fn sys_foreign_fork(pid: u64) -> i64 { +/// which is the whole of fork's contract to the program. A non-zero `rsp` is +/// the stack the child starts on instead of its parent's, as a clone that +/// names a stack gives it; it must lie in the user half. +pub fn sys_foreign_fork(pid: u64, rsp: u64) -> i64 { + if rsp > USER_VA_MAX { + return ERRNO_INVAL; + } let Some(caller) = crate::process::current_pid() else { return ERRNO_INVAL; }; @@ -34,8 +42,8 @@ pub fn sys_foreign_fork(pid: u64) -> i64 { if super::registry::supervisor_of(parent) != Some(caller) { return ERRNO_PERM; } - let Some(state) = saved_state(parent) else { - // A guest that is not parked inside a syscall has no frame to copy. + let Some(state) = super::trap_frame::parked_frame(parent) else { + /* A guest that is not parked inside a syscall has no frame to copy. */ return ERRNO_NOENT; }; let child = match super::spawn::empty_guest(caller, b"fork") { @@ -44,9 +52,14 @@ pub fn sys_foreign_fork(pid: u64) -> i64 { }; let mut frame = state; frame.rax = 0; - // The thread pointer is a register the frame does not carry, so the child - // takes its forking thread's, read from that thread's PCB. Without this a - // fork from a thread that set its own FS would give the child a zero one. + if rsp != 0 { + frame.rsp = rsp; + } + /* + * The thread pointer is a register the frame does not carry, so the child + * takes its forking thread's, read from that thread's PCB. Without this a + * fork from a thread that set its own FS would give the child a zero one. + */ let parent_tls = crate::process::with_process(parent, |pcb| pcb.get_tls_base()).unwrap_or(0); crate::process::with_process(child, |pcb| { *pcb.saved_user_context.lock() = Some(frame); @@ -57,7 +70,3 @@ pub fn sys_foreign_fork(pid: u64) -> i64 { }); child as i64 } - -fn saved_state(pid: u32) -> Option { - super::trap_frame::parked_frame(pid) -} diff --git a/src/syscall/microkernel/dispatch/process.rs b/src/syscall/microkernel/dispatch/process.rs index dddf6a4a9..bc9223609 100644 --- a/src/syscall/microkernel/dispatch/process.rs +++ b/src/syscall/microkernel/dispatch/process.rs @@ -97,7 +97,7 @@ pub(super) fn handle(nr: u64, a: Args) -> Option { SYS_PEER_PROTECT => sys_peer_protect(a.a0, a.a1, a.a2, a.a3), SYS_FOREIGN_THREAD => sys_foreign_thread(a.a0, a.a1, a.a2, a.a3, a.a4), SYS_PEER_TLS => sys_peer_tls(a.a0, a.a1), - SYS_FOREIGN_FORK => sys_foreign_fork(a.a0), + SYS_FOREIGN_FORK => sys_foreign_fork(a.a0, a.a1), SYS_PEER_UNMAP => sys_peer_unmap(a.a0, a.a1, a.a2), SYS_FOREIGN_EXEC => sys_foreign_exec(a.a0, a.a1, a.a2), SYS_LOCAL_SIGN => sys_local_sign(a.a0, a.a1, a.a2, a.a3, a.a4), From 5f35d3b5b6a2fced0149fe9e8796de73bfe03d66 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:04:16 +0000 Subject: [PATCH 15/26] libc: mk_foreign_fork_at forks a guest onto a stack it names nonos_libc could only ask MkForeignFork for a child on its parent's stack, so a supervisor had no way to use the stack argument the kernel now takes. mk_foreign_fork_at passes a stack pointer for the child, and mk_foreign_fork is that with zero, as before. Both live in foreign_fork, next to the rest of the foreign calls, and are exported as the other foreign calls are. --- userland/libc/src/foreign.rs | 9 +-------- userland/libc/src/foreign_fork.rs | 32 +++++++++++++++++++++++++++++++ userland/libc/src/lib.rs | 6 ++++-- 3 files changed, 37 insertions(+), 10 deletions(-) create mode 100644 userland/libc/src/foreign_fork.rs diff --git a/userland/libc/src/foreign.rs b/userland/libc/src/foreign.rs index a50956d75..4bb9c833f 100644 --- a/userland/libc/src/foreign.rs +++ b/userland/libc/src/foreign.rs @@ -18,8 +18,7 @@ //! kernel refuses on its behalf. use crate::syscall::{ - call_raw, N_MK_FOREIGN_EXEC, N_MK_FOREIGN_FORK, N_MK_FOREIGN_REPLY, N_MK_FOREIGN_SPAWN, - N_MK_FOREIGN_START, + call_raw, N_MK_FOREIGN_EXEC, N_MK_FOREIGN_REPLY, N_MK_FOREIGN_SPAWN, N_MK_FOREIGN_START, N_MK_FOREIGN_THREAD, N_MK_FOREIGN_WAIT, }; @@ -42,12 +41,6 @@ pub fn mk_foreign_resume(pid: u32) -> i64 { call_raw(N_MK_FOREIGN_START, [pid as u64, 0, 0, 0, 0, 0]) } -/// A second process holding a guest's register state, with zero in its return -/// register. -pub fn mk_foreign_fork(pid: u32) -> i64 { - call_raw(N_MK_FOREIGN_FORK, [pid as u64, 0, 0, 0, 0, 0]) -} - /// Replace the program a parked guest is running. pub fn mk_foreign_exec(pid: u32, entry: u64, rsp: u64) -> i64 { call_raw(N_MK_FOREIGN_EXEC, [pid as u64, entry, rsp, 0, 0, 0]) diff --git a/userland/libc/src/foreign_fork.rs b/userland/libc/src/foreign_fork.rs new file mode 100644 index 000000000..93942203f --- /dev/null +++ b/userland/libc/src/foreign_fork.rs @@ -0,0 +1,32 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Forking a guest: a second process holding its register state, with zero +//! in its return register, on its parent's stack or one the caller names. + +use crate::syscall::{call_raw, N_MK_FOREIGN_FORK}; + +/// A second process holding a guest's register state, with zero in its return +/// register. +pub fn mk_foreign_fork(pid: u32) -> i64 { + mk_foreign_fork_at(pid, 0) +} + +/// A fork whose child starts on `rsp` rather than its parent's stack pointer, +/// as a clone that names a stack asks; zero keeps the parent's. +pub fn mk_foreign_fork_at(pid: u32, rsp: u64) -> i64 { + call_raw(N_MK_FOREIGN_FORK, [pid as u64, rsp, 0, 0, 0, 0]) +} diff --git a/userland/libc/src/lib.rs b/userland/libc/src/lib.rs index a1176472b..535b61f9a 100644 --- a/userland/libc/src/lib.rs +++ b/userland/libc/src/lib.rs @@ -27,6 +27,7 @@ pub mod capsule_verify; pub mod crypto; pub mod debug; pub mod foreign; +pub mod foreign_fork; pub mod foreign_frame; pub mod foreign_signal; pub mod graphics; @@ -82,9 +83,10 @@ pub use consent::{ mk_local_restore, }; pub use foreign::{ - mk_foreign_exec, mk_foreign_fork, mk_foreign_reply, mk_foreign_resume, mk_foreign_spawn, - mk_foreign_start, mk_foreign_thread, mk_foreign_wait, + mk_foreign_exec, mk_foreign_reply, mk_foreign_resume, mk_foreign_spawn, mk_foreign_start, + mk_foreign_thread, mk_foreign_wait, }; +pub use foreign_fork::{mk_foreign_fork, mk_foreign_fork_at}; pub use foreign_frame::{ForeignFrame, FOREIGN_NR_DIED}; pub use foreign_signal::{ mk_foreign_context, mk_foreign_signal, ForeignRegs, SIGNAL_DELIVER, SIGNAL_RETURN, From ac08871b1b432d2cd50526d26ab0c86c7afa2d5e Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:04:44 +0000 Subject: [PATCH 16/26] linux: a process clone that names a stack starts the child on it musl's posix_spawn, system and popen clone a vfork child onto a stack of its own, not the parent's. That clone was refused with ENOSYS, so each of them failed and no C program could start another through them. A clone that makes a process now hands its stack to the fork, and the child starts on it with the rest of its registers copied from the parent, as Linux starts it. With no stack named, the child runs on its parent's, as before. --- userland/capsule_linux/src/linux/call/spawn/fork.rs | 2 +- .../capsule_linux/src/linux/call/spawn/fork_child.rs | 6 ++++-- userland/capsule_linux/src/linux/call/spawn/vfork.rs | 12 +++++++----- .../src/linux/call/spawn/vfork_clone.rs | 10 ++++------ 4 files changed, 16 insertions(+), 14 deletions(-) diff --git a/userland/capsule_linux/src/linux/call/spawn/fork.rs b/userland/capsule_linux/src/linux/call/spawn/fork.rs index 720da83a6..37c2c624b 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork.rs @@ -24,7 +24,7 @@ use crate::linux::serve::Answer; use super::fork_child::fork_child; pub fn fork(guest: &mut Guest, caller: u32) -> Answer { - match fork_child(guest, caller, SIGCHLD, |_| {}) { + match fork_child(guest, caller, 0, SIGCHLD, |_| {}) { Ok(child) => Answer::value(errno::ok(child as u64)), Err(e) => Answer::value(e), } diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_child.rs b/userland/capsule_linux/src/linux/call/spawn/fork_child.rs index 062bd431e..85d9212c1 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork_child.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork_child.rs @@ -17,7 +17,7 @@ //! The copy every new process starts as: fork's, vfork's and a clone that //! makes a process all come here. -use nonos_libc::{mk_foreign_fork, mk_foreign_resume}; +use nonos_libc::{mk_foreign_fork_at, mk_foreign_resume}; use super::fork_copy::copy_spans; use crate::linux::abi::errno; @@ -27,14 +27,16 @@ use crate::linux::guest::Guest; /// A new process copied from this one, running, and adopted by the family /// once this answer is given. It raises `exit_signal` at its parent when it /// ends. Its signal state is the forking thread's, as Linux's fork gives it. +/// It starts on `stack` when that is not zero, as a clone naming a stack asks. /// `prep` runs on the child before it runs at all. pub(super) fn fork_child( guest: &mut Guest, caller: u32, + stack: u64, exit_signal: u8, prep: impl FnOnce(&mut Guest), ) -> Result { - let child = mk_foreign_fork(caller); + let child = mk_foreign_fork_at(caller, stack); if child < 0 { return Err(errno::fail(errno::ENOMEM)); } diff --git a/userland/capsule_linux/src/linux/call/spawn/vfork.rs b/userland/capsule_linux/src/linux/call/spawn/vfork.rs index 6144e076e..61291606f 100644 --- a/userland/capsule_linux/src/linux/call/spawn/vfork.rs +++ b/userland/capsule_linux/src/linux/call/spawn/vfork.rs @@ -18,9 +18,10 @@ //! rather than a thread (vfork_clone). The child is a copy, as fork's is: //! vfork's child shares its parent's memory on Linux, but it may only exec or //! exit, and the parent sleeps until it does, so what either sees is the -//! same. That is what Go's os/exec asks for with clone(CLONE_VFORK|CLONE_VM). -//! The parent parks here and the family answers it with the child's pid when -//! the child's exec succeeds or the child ends. +//! same. That is what Go's os/exec asks for with clone(CLONE_VFORK|CLONE_VM), +//! and musl's posix_spawn with a stack of the child's own, which the kernel's +//! fork starts it on. The parent parks here and the family answers it with +//! the child's pid when the child's exec succeeds or the child ends. use crate::linux::abi::errno; use crate::linux::guest::sigstate::SIGCHLD; @@ -30,7 +31,7 @@ use crate::linux::serve::Answer; use super::fork_child::fork_child; pub fn vfork(guest: &mut Guest, caller: u32) -> Answer { - start(guest, caller, SIGCHLD, true, |_| {}) + start(guest, caller, 0, SIGCHLD, true, |_| {}) } /// Fork a child and answer with its pid, or park the caller until the child @@ -38,11 +39,12 @@ pub fn vfork(guest: &mut Guest, caller: u32) -> Answer { pub(super) fn start( guest: &mut Guest, caller: u32, + stack: u64, signal: u8, parks: bool, prep: impl FnOnce(&mut Guest), ) -> Answer { - let child = match fork_child(guest, caller, signal, prep) { + let child = match fork_child(guest, caller, stack, signal, prep) { Ok(c) => c, Err(e) => return Answer::value(e), }; diff --git a/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs b/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs index c2a3d892a..4bac79300 100644 --- a/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs +++ b/userland/capsule_linux/src/linux/call/spawn/vfork_clone.rs @@ -15,9 +15,9 @@ // along with this program. If not, see . //! `clone` when it makes a process rather than a thread: a copy, as fork's -//! is, on its parent's stack, that raises the signal it named when it ends, -//! and with CLONE_VFORK parks its parent as vfork does. The tid words it -//! names are written as Linux writes them. +//! is, that starts on the stack the caller named, raises the signal it named +//! when it ends, and with CLONE_VFORK parks its parent as vfork does. The tid +//! words it names are written as Linux writes them. use super::vfork::start; use super::vfork_flags::{CLONE_CHILD_CLEARTID, CLONE_CHILD_SETTID, CLONE_PARENT_SETTID}; @@ -34,8 +34,6 @@ pub fn clone_process(guest: &mut Guest, caller: u32, a: [u64; 6]) -> Answer { Some("clone: flags beyond a copied process") } else if flags & CLONE_VM != 0 && flags & CLONE_VFORK == 0 { Some("clone: a process sharing its parent's memory") - } else if stack != 0 { - Some("clone: a new process on a stack of its own") } else { None }; @@ -58,7 +56,7 @@ pub fn clone_process(guest: &mut Guest, caller: u32, a: [u64; 6]) -> Answer { child.clear_tids.push((child.pid, a[3])); } }; - let answer = start(guest, caller, signal as u8, flags & CLONE_VFORK != 0, prep); + let answer = start(guest, caller, stack, signal as u8, flags & CLONE_VFORK != 0, prep); if flags & CLONE_PARENT_SETTID != 0 { /* Only a child made by this call is in `forked`. */ if let Some(child) = guest.forked.last().map(|c| c.pid) { From 3f26b6d114cae3f4fae5861ff00fc2a8ba9d1a6c Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 23:04:49 +0000 Subject: [PATCH 17/26] linux: five guests check process lifecycle and signals against linux Nothing in the guest set exercised a leader's plain exit, a program starting another, a parent told about its children, SIGPIPE, or a signal arriving while a thread waits, so each of the changes before this one could only be judged by reading it. Five guests now check them against what Linux does, each printing a line per part. leaderexit ends its leader with a plain exit and expects its worker to run on and set the status to 42. goexec starts programs from Go through syscall.ForkExec and os/exec, which use clone(CLONE_VFORK), and reads each child's status. sigchld catches SIGCHLD and checks its siginfo, then waitid and wait4 with their options and ECHILD, then starts programs through posix_spawn, system and popen. sigpipe checks a write to a widowed pipe caught, ignored and at its default. alarm checks that SIGALRM ends a sleep early, then setitimer, getitimer, pause, sigsuspend with sigpending, sigqueue into sigtimedwait, a timed out sigtimedwait and a POSIX timer, and reads its own ucontext at Linux's offsets. LifeGuests.mk, which Guests.mk includes, builds the C ones with musl and adds all five to the guest set. --- userland/linux_guests/Guests.mk | 3 + userland/linux_guests/LifeGuests.mk | 20 ++++++ userland/linux_guests/c/alarm.c | 75 +++++++++++++++++++++++ userland/linux_guests/c/alarm.h | 16 +++++ userland/linux_guests/c/alarm_timer.c | 40 ++++++++++++ userland/linux_guests/c/alarm_wait.c | 70 +++++++++++++++++++++ userland/linux_guests/c/leaderexit.c | 36 +++++++++++ userland/linux_guests/c/sigchld.c | 72 ++++++++++++++++++++++ userland/linux_guests/c/sigchld.h | 10 +++ userland/linux_guests/c/sigchld_spawn.c | 39 ++++++++++++ userland/linux_guests/c/sigchld_wait.c | 39 ++++++++++++ userland/linux_guests/c/sigpipe.c | 73 ++++++++++++++++++++++ userland/linux_guests/go/exec/forkexec.go | 39 ++++++++++++ userland/linux_guests/go/exec/go.mod | 3 + userland/linux_guests/go/exec/main.go | 72 ++++++++++++++++++++++ 15 files changed, 607 insertions(+) create mode 100644 userland/linux_guests/LifeGuests.mk create mode 100644 userland/linux_guests/c/alarm.c create mode 100644 userland/linux_guests/c/alarm.h create mode 100644 userland/linux_guests/c/alarm_timer.c create mode 100644 userland/linux_guests/c/alarm_wait.c create mode 100644 userland/linux_guests/c/leaderexit.c create mode 100644 userland/linux_guests/c/sigchld.c create mode 100644 userland/linux_guests/c/sigchld.h create mode 100644 userland/linux_guests/c/sigchld_spawn.c create mode 100644 userland/linux_guests/c/sigchld_wait.c create mode 100644 userland/linux_guests/c/sigpipe.c create mode 100644 userland/linux_guests/go/exec/forkexec.go create mode 100644 userland/linux_guests/go/exec/go.mod create mode 100644 userland/linux_guests/go/exec/main.go diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index 16a9232a4..e6bc56423 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -107,6 +107,9 @@ $(LINUX_GUESTS_C)/cthreads: $(LINUX_GUESTS_DIR)/c/cthreads.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cthreads,4974,4975,$(LINUX_GUESTS_C)/cthreads)) +# Process lifecycle and signals, each against Linux; see LifeGuests.mk. +include $(LINUX_GUESTS_DIR)/LifeGuests.mk + # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the # normal image, which does not set NONOS_LINUX_GUESTS, still ships them. diff --git a/userland/linux_guests/LifeGuests.mk b/userland/linux_guests/LifeGuests.mk new file mode 100644 index 000000000..828aff912 --- /dev/null +++ b/userland/linux_guests/LifeGuests.mk @@ -0,0 +1,20 @@ +# Guests for process lifecycle and signals. Included by Guests.mk. +# +# Process lifecycle and signals, each against Linux: a leader's plain exit +# that leaves its worker running, os/exec from Go through clone(CLONE_VFORK), +# SIGCHLD with waitid and wait4, SIGPIPE on a widowed pipe, and SIGALRM ending +# a sleep, with the timer and signal-wait calls. +$(LINUX_GUESTS_C)/leaderexit: $(LINUX_GUESTS_DIR)/c/leaderexit.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,leaderexit,4990,4991,$(LINUX_GUESTS_C)/leaderexit)) +$(GO_OUT)/exec: $(LINUX_GUESTS_DIR)/go/exec/forkexec.go +$(eval $(call LINUX_GUEST,goexec,4992,4993,$(GO_OUT)/exec)) +$(LINUX_GUESTS_C)/sigchld: $(addprefix $(LINUX_GUESTS_DIR)/c/,sigchld.c sigchld_wait.c sigchld_spawn.c sigchld.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $(filter %.c,$^) +$(eval $(call LINUX_GUEST,sigchld,4994,4995,$(LINUX_GUESTS_C)/sigchld)) +$(LINUX_GUESTS_C)/sigpipe: $(LINUX_GUESTS_DIR)/c/sigpipe.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,sigpipe,4996,4997,$(LINUX_GUESTS_C)/sigpipe)) +$(LINUX_GUESTS_C)/alarm: $(addprefix $(LINUX_GUESTS_DIR)/c/,alarm.c alarm_wait.c alarm_timer.c alarm.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $(filter %.c,$^) +$(eval $(call LINUX_GUEST,alarm,4998,4999,$(LINUX_GUESTS_C)/alarm)) diff --git a/userland/linux_guests/c/alarm.c b/userland/linux_guests/c/alarm.c new file mode 100644 index 000000000..dbd02f030 --- /dev/null +++ b/userland/linux_guests/c/alarm.c @@ -0,0 +1,75 @@ +/* + * Signals that arrive while a thread waits, and the calls that wait for them. + * alarm(1) then sleep(10): on Linux SIGALRM ends the sleep early with its + * handler run, well under two seconds in. Then setitimer and getitimer with + * pause; the signal waits and POSIX timers follow in alarm_wait.c. Every part + * prints its line. + */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include "alarm.h" + +volatile sig_atomic_t alarms; +static int failed, parts; + +static void on_alrm(int sig) { (void)sig; alarms++; } + +void part(int ok, const char *what, long n) { + printf("[C] alarm %s: %s (%ld)\n", ok ? "ok" : "FAIL", what, n); + fflush(stdout); + failed += !ok; + parts++; +} + +long ms_since(struct timespec *t0) { + struct timespec t; + clock_gettime(CLOCK_MONOTONIC, &t); + return (t.tv_sec - t0->tv_sec) * 1000 + (t.tv_nsec - t0->tv_nsec) / 1000000; +} + +void catch(int sig, void (*fn)(int)) { + struct sigaction sa; + memset(&sa, 0, sizeof sa); + sa.sa_handler = fn; + sigaction(sig, &sa, 0); +} + +int verdict(void) { + printf("[C] alarm %s: %d of %d parts held\n", failed ? "FAIL" : "PASS", parts - failed, parts); + fflush(stdout); + return failed != 0; +} + +int main(void) { + struct timespec t0; + catch(SIGALRM, on_alrm); + clock_gettime(CLOCK_MONOTONIC, &t0); + alarm(1); + unsigned left = sleep(10); + long ms = ms_since(&t0); + part(alarms == 1 && ms < 2000 && left >= 8, "alarm(1) ended sleep(10) early, ms", ms); + struct itimerval it = {{0, 200000}, {0, 200000}}, now; + alarms = 0; + clock_gettime(CLOCK_MONOTONIC, &t0); + setitimer(ITIMER_REAL, &it, 0); + /* Bounded: where pause fails at once instead of waiting, this still ends. */ + for (int i = 0; alarms < 3 && i < 50; i++) { + pause(); + } + ms = ms_since(&t0); + getitimer(ITIMER_REAL, &now); + part(ms >= 550 && ms < 1500 && now.it_interval.tv_usec == 200000 && + now.it_value.tv_usec <= 200000, + "setitimer 200 ms interval: three SIGALRMs through pause, ms", ms); + memset(&it, 0, sizeof it); + setitimer(ITIMER_REAL, &it, &now); + getitimer(ITIMER_REAL, &now); + part(now.it_value.tv_sec == 0 && now.it_value.tv_usec == 0, "a disarmed timer reads zero", 0); + part(alarm(0) == 0, "alarm(0) with nothing armed answers 0", 0); + signal_waits(); + posix_timer(); + return verdict(); +} diff --git a/userland/linux_guests/c/alarm.h b/userland/linux_guests/c/alarm.h new file mode 100644 index 000000000..36e7d7ade --- /dev/null +++ b/userland/linux_guests/c/alarm.h @@ -0,0 +1,16 @@ +/* + * What the alarm guest's two halves share: its SIGALRM and SIGUSR1 counts, + * what the SIGUSR1 handler read from its ucontext, and the part line every + * check prints. + */ +#include +#include + +extern volatile sig_atomic_t alarms, usr1, usr1_rip_in_text, usr1_saved_blocked; + +void part(int ok, const char *what, long n); +long ms_since(struct timespec *t0); +void catch(int sig, void (*fn)(int)); +void signal_waits(void); +void posix_timer(void); +int verdict(void); diff --git a/userland/linux_guests/c/alarm_timer.c b/userland/linux_guests/c/alarm_timer.c new file mode 100644 index 000000000..13b5954dd --- /dev/null +++ b/userland/linux_guests/c/alarm_timer.c @@ -0,0 +1,40 @@ +/* + * A POSIX timer, checked against Linux: timer_create on the monotonic clock + * with SIGEV_SIGNAL, armed for 100 ms with a 100 ms period, gives three + * SI_TIMER signals carrying its value through sigtimedwait; timer_gettime + * reads its period back, timer_getoverrun answers, timer_delete ends it. + */ +#define _GNU_SOURCE +#include + +#include "alarm.h" + +void posix_timer(void) { + sigset_t rt; + sigemptyset(&rt); + sigaddset(&rt, SIGRTMIN); + sigprocmask(SIG_BLOCK, &rt, 0); + struct sigevent ev; + memset(&ev, 0, sizeof ev); + ev.sigev_notify = SIGEV_SIGNAL; + ev.sigev_signo = SIGRTMIN; + ev.sigev_value.sival_int = 77; + timer_t id; + int rc = timer_create(CLOCK_MONOTONIC, &ev, &id); + struct itimerspec ts = {{0, 100000000}, {0, 100000000}}, got; + timer_settime(id, 0, &ts, 0); + siginfo_t si; + struct timespec wait = {1, 0}; + int fired = 0; + for (int i = 0; i < 3; i++) { + if (sigtimedwait(&rt, &si, &wait) == SIGRTMIN && si.si_code == SI_TIMER && + si.si_value.sival_int == 77) { + fired++; + } + } + timer_gettime(id, &got); + int over = timer_getoverrun(id); + part(rc == 0 && fired == 3 && got.it_interval.tv_nsec == 100000000 && over >= 0, + "timer_create 100 ms: three SI_TIMER signals, value 77", fired); + part(timer_delete(id) == 0, "timer_delete", 0); +} diff --git a/userland/linux_guests/c/alarm_wait.c b/userland/linux_guests/c/alarm_wait.c new file mode 100644 index 000000000..087339ac9 --- /dev/null +++ b/userland/linux_guests/c/alarm_wait.c @@ -0,0 +1,70 @@ +/* + * The signal waits, checked against Linux: a blocked SIGUSR1 shows in + * sigpending, sigsuspend runs its handler and answers EINTR with the mask put + * back, sigqueue hands a value to sigtimedwait, and a sigtimedwait with + * nothing pending times out. The SIGUSR1 handler reads its own ucontext: the + * interrupted rip must lie in this program's text and the saved mask must be + * the one sigsuspend replaced, which holds only at Linux's offsets. + */ +#define _GNU_SOURCE +#include +#include +#include +#include + +#include "alarm.h" + +extern char __executable_start[], etext[]; +volatile sig_atomic_t usr1, usr1_rip_in_text, usr1_saved_blocked; + +static void on_usr1(int sig, siginfo_t *info, void *ctx) { + (void)sig; + (void)info; + ucontext_t *uc = ctx; + unsigned long rip = (unsigned long)uc->uc_mcontext.gregs[REG_RIP]; + usr1_rip_in_text = rip >= (unsigned long)__executable_start && rip < (unsigned long)etext; + usr1_saved_blocked = sigismember(&uc->uc_sigmask, SIGUSR1); + usr1++; +} + +void signal_waits(void) { + struct sigaction sa; + memset(&sa, 0, sizeof sa); + sa.sa_sigaction = on_usr1; + sa.sa_flags = SA_SIGINFO; + sigaction(SIGUSR1, &sa, 0); + sigset_t block, none, pending, after, two; + sigemptyset(&block); + sigaddset(&block, SIGUSR1); + sigprocmask(SIG_BLOCK, &block, 0); + raise(SIGUSR1); + sigpending(&pending); + part(usr1 == 0 && sigismember(&pending, SIGUSR1), "a blocked SIGUSR1 waits, pending", 0); + sigemptyset(&none); + errno = 0; + int rc = sigsuspend(&none), e = errno; + sigprocmask(SIG_BLOCK, 0, &after); + part(rc == -1 && e == EINTR && usr1 == 1 && sigismember(&after, SIGUSR1), + "sigsuspend ran the handler, EINTR, mask restored", usr1); + part(usr1_rip_in_text && usr1_saved_blocked, + "the handler's ucontext: rip in the program's text, uc_sigmask the saved mask", 0); + + sigemptyset(&two); + sigaddset(&two, SIGUSR2); + sigprocmask(SIG_BLOCK, &two, 0); + union sigval v = {.sival_int = 1234}; + sigqueue(getpid(), SIGUSR2, v); + siginfo_t si; + struct timespec wait = {1, 0}, brief = {0, 100000000}, t0; + rc = sigtimedwait(&two, &si, &wait); + part(rc == SIGUSR2 && si.si_code == SI_QUEUE && si.si_value.sival_int == 1234 && + si.si_pid == getpid(), + "sigqueue into sigtimedwait: SI_QUEUE, value", si.si_value.sival_int); + clock_gettime(CLOCK_MONOTONIC, &t0); + errno = 0; + rc = sigtimedwait(&two, &si, &brief); + e = errno; + long ms = ms_since(&t0); + part(rc == -1 && e == EAGAIN && ms >= 90 && ms < 1000, "sigtimedwait timed out: EAGAIN, ms", + ms); +} diff --git a/userland/linux_guests/c/leaderexit.c b/userland/linux_guests/c/leaderexit.c new file mode 100644 index 000000000..c59e8ad33 --- /dev/null +++ b/userland/linux_guests/c/leaderexit.c @@ -0,0 +1,36 @@ +/* + * The leader ends with a plain exit, not exit_group. On Linux that ends only + * the leader: the process lives on in its other threads, and its status is + * the exit code of the last thread to end. The worker prints its line a second + * after the leader has gone and ends with 42, so a log with the PASS line and + * a status of 42 is Linux's behaviour; a process that ended at the leader's + * exit never prints the line and reports 0. + */ +#include +#include +#include +#include +#include + +static void *work(void *arg) { + (void)arg; + struct timespec t = {1, 0}; + nanosleep(&t, 0); + printf("[C] leaderexit PASS: worker ran 1 s after the leader's exit\n"); + fflush(stdout); + syscall(SYS_exit, 42); + return 0; +} + +int main(void) { + pthread_t t; + if (pthread_create(&t, 0, work, 0) != 0) { + printf("[C] leaderexit FAIL: create\n"); + fflush(stdout); + return 1; + } + printf("[C] leaderexit: leader exits, worker runs on\n"); + fflush(stdout); + syscall(SYS_exit, 0); + return 1; +} diff --git a/userland/linux_guests/c/sigchld.c b/userland/linux_guests/c/sigchld.c new file mode 100644 index 000000000..ed98cae08 --- /dev/null +++ b/userland/linux_guests/c/sigchld.c @@ -0,0 +1,72 @@ +/* + * A parent told about its children as Linux tells it. A child's exit raises + * SIGCHLD at a parent that catches it, with the child's pid, CLD_EXITED and + * its code in the siginfo; waitid(P_ALL, WEXITED) then reports the same + * child. The wait4 and waitid checks follow in sigchld_wait.c and musl's ways + * to start a program in sigchld_spawn.c. Every part runs and prints its line; + * the last line is PASS only if all of them held. + */ +#include +#include +#include +#include +#include +#include +#include "sigchld.h" + +static volatile sig_atomic_t got, got_code, got_status; +static volatile pid_t got_pid; +static int failed, parts; + +static void on_chld(int sig, siginfo_t *info, void *uc) { + (void)uc; + got = sig; + got_pid = info->si_pid; + got_code = info->si_code; + got_status = info->si_status; +} + +void part(int ok, const char *what) { + printf("[C] sigchld %s: %s\n", ok ? "ok" : "FAIL", what); + fflush(stdout); + failed += !ok; + parts++; +} + +pid_t child_exiting(int code, unsigned delay_ms) { + pid_t pid = fork(); + if (pid == 0) { + struct timespec t = {delay_ms / 1000, (delay_ms % 1000) * 1000000L}; + nanosleep(&t, 0); + _exit(code); + } + return pid; +} + +int main(void) { + struct sigaction sa; + memset(&sa, 0, sizeof sa); + sa.sa_sigaction = on_chld; + sa.sa_flags = SA_SIGINFO | SA_RESTART; + sigaction(SIGCHLD, &sa, 0); + pid_t a = child_exiting(7, 0); + /* Sleep in short steps until the handler has run or three seconds pass. */ + for (int i = 0; i < 300 && !got; i++) { + struct timespec t = {0, 10 * 1000 * 1000}; + nanosleep(&t, 0); + } + part(got == SIGCHLD && got_pid == a && got_code == CLD_EXITED && got_status == 7, + "SIGCHLD caught with the child's pid, CLD_EXITED and status 7"); + siginfo_t si; + memset(&si, 0, sizeof si); + int rc = waitid(P_ALL, 0, &si, WEXITED); + part(rc == 0 && si.si_pid == a && si.si_signo == SIGCHLD && si.si_code == CLD_EXITED && + si.si_status == 7, + "waitid(P_ALL, WEXITED) reports the child, CLD_EXITED, status 7"); + waits(); + spawns(); + printf("[C] sigchld %s: %d of %d parts held\n", failed ? "FAIL" : "PASS", parts - failed, + parts); + fflush(stdout); + return failed != 0; +} diff --git a/userland/linux_guests/c/sigchld.h b/userland/linux_guests/c/sigchld.h new file mode 100644 index 000000000..3c2a8b72a --- /dev/null +++ b/userland/linux_guests/c/sigchld.h @@ -0,0 +1,10 @@ +/* + * What the sigchld guest's files share: the part line every check prints, + * a child that exits after a delay, and the checks each file holds. + */ +#include + +void part(int ok, const char *what); +pid_t child_exiting(int code, unsigned delay_ms); +void waits(void); +void spawns(void); diff --git a/userland/linux_guests/c/sigchld_spawn.c b/userland/linux_guests/c/sigchld_spawn.c new file mode 100644 index 000000000..24b070be0 --- /dev/null +++ b/userland/linux_guests/c/sigchld_spawn.c @@ -0,0 +1,39 @@ +/* + * musl's own ways to start a program, which clone a vfork child onto a stack + * of its own: posix_spawn, system through the shell, and popen reading the + * child's output through a pipe. Then, with every child reaped, wait4 and + * waitid answer ECHILD. + */ +#include +#include +#include +#include +#include +#include +#include +#include "sigchld.h" + +extern char **environ; + +void spawns(void) { + pid_t d = -1; + int st = -1; + char *argv[] = {"/bin/cthreads", 0}; + int rc = posix_spawn(&d, "/bin/cthreads", 0, 0, argv, environ); + part(rc == 0 && waitpid(d, &st, 0) == d && WIFEXITED(st) && WEXITSTATUS(st) == 0, + "posix_spawn runs cthreads, status 0"); + st = system("/bin/leaderexit"); + part(WIFEXITED(st) && WEXITSTATUS(st) == 42, "system runs leaderexit through sh, status 42"); + char line[128] = {0}; + FILE *f = popen("/bin/cthreads", "r"); + int got_line = f && fgets(line, sizeof line, f) != 0; + int closed = f ? pclose(f) : -1; + part(got_line && strstr(line, "cthreads PASS") && WIFEXITED(closed) && WEXITSTATUS(closed) == 0, + "popen reads cthreads' PASS line through a pipe"); + siginfo_t si; + errno = 0; + part(wait4(-1, &st, 0, 0) == -1 && errno == ECHILD, "wait4 with no child left: ECHILD"); + errno = 0; + part(waitid(P_ALL, 0, &si, WEXITED) == -1 && errno == ECHILD, + "waitid with no child left: ECHILD"); +} diff --git a/userland/linux_guests/c/sigchld_wait.c b/userland/linux_guests/c/sigchld_wait.c new file mode 100644 index 000000000..93ec344d6 --- /dev/null +++ b/userland/linux_guests/c/sigchld_wait.c @@ -0,0 +1,39 @@ +/* + * wait4 and waitid against Linux: WNOHANG on a running child, waitid WNOWAIT + * looking without reaping, WUNTRACED, the status word of an exit and of a + * death by signal, and EINVAL for an option Linux does not know. + */ +#include +#include +#include +#include +#include +#include "sigchld.h" + +void waits(void) { + siginfo_t si; + pid_t b = child_exiting(3, 300); + int st = -1; + part(wait4(b, &st, WNOHANG, 0) == 0, "wait4 WNOHANG on a running child answers 0"); + memset(&si, 0, sizeof si); + si.si_pid = 1; + int rc = waitid(P_PID, b, &si, WEXITED | WNOHANG); + part(rc == 0 && si.si_pid == 0, "waitid WNOHANG on a running child answers 0, si_pid 0"); + memset(&si, 0, sizeof si); + rc = waitid(P_PID, b, &si, WEXITED | WNOWAIT); + part(rc == 0 && si.si_pid == b && si.si_status == 3, "waitid WNOWAIT reports and leaves it"); + rc = wait4(b, &st, WUNTRACED, 0); + part(rc == b && WIFEXITED(st) && WEXITSTATUS(st) == 3, + "wait4 WUNTRACED then reaps it: WIFEXITED, status 3"); + errno = 0; + part(wait4(-1, &st, 0x100, 0) == -1 && errno == EINVAL, "wait4 with an unknown option: EINVAL"); + pid_t c = fork(); + if (c == 0) { + signal(SIGTERM, SIG_DFL); + kill(getpid(), SIGTERM); + for (;;) pause(); + } + rc = wait4(c, &st, 0, 0); + part(rc == c && WIFSIGNALED(st) && WTERMSIG(st) == SIGTERM && !WIFEXITED(st), + "a child ended by SIGTERM: WIFSIGNALED, WTERMSIG 15"); +} diff --git a/userland/linux_guests/c/sigpipe.c b/userland/linux_guests/c/sigpipe.c new file mode 100644 index 000000000..020dc3238 --- /dev/null +++ b/userland/linux_guests/c/sigpipe.c @@ -0,0 +1,73 @@ +/* + * A write to a pipe no one can read raises SIGPIPE on the writing thread, as + * Linux does. Caught, the handler runs and the write returns EPIPE; ignored, + * the write returns EPIPE and nothing else happens; left at its default, it + * ends the process, which a parent sees as a death by signal 13. + */ +#include +#include +#include +#include +#include +#include + +static volatile sig_atomic_t caught; +static int failed, parts; + +static void on_pipe(int sig) { caught = sig; } + +static void part(int ok, const char *what) { + printf("[C] sigpipe %s: %s\n", ok ? "ok" : "FAIL", what); + fflush(stdout); + failed += !ok; + parts++; +} + +/* A pipe whose read end is already closed; the write end is returned. */ +static int widowed(void) { + int p[2]; + if (pipe(p) != 0) { + return -1; + } + close(p[0]); + return p[1]; +} + +int main(void) { + struct sigaction sa; + memset(&sa, 0, sizeof sa); + sa.sa_handler = on_pipe; + sigaction(SIGPIPE, &sa, 0); + int w = widowed(); + errno = 0; + ssize_t n = write(w, "x", 1); + int e = errno; + part(caught == SIGPIPE && n == -1 && e == EPIPE, "caught: the handler ran, write gave EPIPE"); + close(w); + + signal(SIGPIPE, SIG_IGN); + caught = 0; + w = widowed(); + errno = 0; + n = write(w, "x", 1); + e = errno; + part(caught == 0 && n == -1 && e == EPIPE, "ignored: write gave EPIPE, no handler"); + close(w); + + pid_t c = fork(); + if (c == 0) { + signal(SIGPIPE, SIG_DFL); + int cw = widowed(); + write(cw, "x", 1); + _exit(0); + } + int st = 0; + pid_t got = waitpid(c, &st, 0); + part(got == c && WIFSIGNALED(st) && WTERMSIG(st) == SIGPIPE, + "default: the writer died of signal 13"); + + printf("[C] sigpipe %s: %d of %d parts held\n", failed ? "FAIL" : "PASS", parts - failed, + parts); + fflush(stdout); + return failed != 0; +} diff --git a/userland/linux_guests/go/exec/forkexec.go b/userland/linux_guests/go/exec/forkexec.go new file mode 100644 index 000000000..91da4afe9 --- /dev/null +++ b/userland/linux_guests/go/exec/forkexec.go @@ -0,0 +1,39 @@ +/* + * The syscall.ForkExec half of goexec: Go's own clone(CLONE_VFORK|CLONE_VM| + * SIGCHLD) and execve, with the child's exec error sent back through a pipe, + * then Wait4 for each child's status. Nothing here needs the runtime's poller. + */ +package main + +import ( + "errors" + "fmt" + "syscall" +) + +/* run starts path with syscall.ForkExec and waits for it with Wait4. */ +func run(path string) (syscall.WaitStatus, error) { + attr := &syscall.ProcAttr{Files: []uintptr{0, 1, 2}} + pid, err := syscall.ForkExec(path, []string{path}, attr) + if err != nil { + return 0, err + } + var ws syscall.WaitStatus + got, err := syscall.Wait4(pid, &ws, 0, nil) + if err == nil && got != pid { + err = fmt.Errorf("wait4 answered %d for child %d", got, pid) + } + return ws, err +} + +func forkExecParts() { + ws, err := run("/bin/cthreads") + part(err == nil && ws.Exited() && ws.ExitStatus() == 0, + fmt.Sprintf("ForkExec cthreads: status %d, err %v", ws.ExitStatus(), err)) + ws, err = run("/bin/leaderexit") + part(err == nil && ws.Exited() && ws.ExitStatus() == 42, + fmt.Sprintf("ForkExec leaderexit: status %d, err %v", ws.ExitStatus(), err)) + _, err = run("/bin/no-such-program") + part(errors.Is(err, syscall.ENOENT), fmt.Sprintf("ForkExec a missing program: %v", err)) + fmt.Printf("[GO] goexec: ForkExec parts done, %d of %d held\n", parts-failed, parts) +} diff --git a/userland/linux_guests/go/exec/go.mod b/userland/linux_guests/go/exec/go.mod new file mode 100644 index 000000000..04b7c7215 --- /dev/null +++ b/userland/linux_guests/go/exec/go.mod @@ -0,0 +1,3 @@ +module nonos/guest/exec + +go 1.24 diff --git a/userland/linux_guests/go/exec/main.go b/userland/linux_guests/go/exec/main.go new file mode 100644 index 000000000..b14b4b059 --- /dev/null +++ b/userland/linux_guests/go/exec/main.go @@ -0,0 +1,72 @@ +/* + * Starting programs from Go. syscall.ForkExec is Go's own clone(CLONE_VFORK| + * CLONE_VM|SIGCHLD) and execve, with the child's exec error sent back through + * a pipe; Wait4 then reads each child's status. The children are the pthreads + * guest, a program that ends with status 42, and a path that does not exist. + * Then os/exec does the same through cmd.Output, which also needs the + * runtime's poller (eventfd, epoll) for its pipes; those parts run last. + * Stdin, and for Run the output too, is this program's own: a nil one makes + * os/exec open /dev/null, which the personality does not serve. + */ +package main + +import ( + "errors" + "fmt" + "io/fs" + "os" + "os/exec" + "strings" + "syscall" +) + +var failed, parts int + +func part(ok bool, what string) { + mark := "ok" + if !ok { + mark = "FAIL" + failed++ + } + parts++ + fmt.Printf("[GO] goexec %s: %s\n", mark, what) +} + +func own(c *exec.Cmd) *exec.Cmd { + c.Stdin, c.Stdout, c.Stderr = os.Stdin, os.Stdout, os.Stderr + return c +} + +func main() { + forkExecParts() + + cmd := exec.Command("/bin/cthreads") + cmd.Stdin = os.Stdin + out, err := cmd.Output() + line := strings.TrimSpace(string(out)) + code := -1 + if cmd.ProcessState != nil { + code = cmd.ProcessState.ExitCode() + } + part(err == nil && code == 0 && strings.Contains(line, "cthreads PASS"), + fmt.Sprintf("os/exec cthreads Output: status %d, err %v, output %q", code, err, line)) + + err = own(exec.Command("/bin/leaderexit")).Run() + var ee *exec.ExitError + part(errors.As(err, &ee) && ee.ExitCode() == 42, + fmt.Sprintf("os/exec leaderexit's status came back: %v", err)) + + err = own(exec.Command("/bin/no-such-program")).Run() + var pe *fs.PathError + part(errors.As(err, &pe) && pe.Op == "fork/exec" && errors.Is(pe.Err, syscall.ENOENT), + fmt.Sprintf("os/exec a missing program: %v", err)) + + verdict := "PASS" + if failed != 0 { + verdict = "FAIL" + } + fmt.Printf("[GO] goexec %s: %d of %d parts held\n", verdict, parts-failed, parts) + if failed != 0 { + os.Exit(1) + } +} From 75d14d10a8f995be21946f869011807d4bcff98c Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 00:16:54 +0000 Subject: [PATCH 18/26] linux: answer lifecycle and signal calls before dispatch The calls that can leave a thread parked in a new way, a leader's plain exit, a process clone, vfork, waitid, kill across the family, pause, sigsuspend, sigtimedwait and sigqueueinfo, were served in route_life, but nothing asked it: dispatch answered them the old way, or named them unserved. The family now asks route_life for each trap first, which counts the call, and dispatch answers everything else. wait4 now parks in the signal state's list, so the guest's single waiting slot is gone. --- userland/capsule_linux/src/linux/guest/handle.rs | 2 -- .../capsule_linux/src/linux/guest/handle_new.rs | 1 - userland/capsule_linux/src/linux/serve/family.rs | 2 +- .../capsule_linux/src/linux/serve/route_life.rs | 15 +++++++++++++-- 4 files changed, 14 insertions(+), 6 deletions(-) diff --git a/userland/capsule_linux/src/linux/guest/handle.rs b/userland/capsule_linux/src/linux/guest/handle.rs index 1f12ca784..a5d0c8bca 100644 --- a/userland/capsule_linux/src/linux/guest/handle.rs +++ b/userland/capsule_linux/src/linux/guest/handle.rs @@ -78,8 +78,6 @@ pub struct Guest { pub forked: Vec, /// Children that have ended, with their exit codes, until waited for. pub ended: Vec<(u32, i32)>, - /// A parked wait4: the pid it wants, where the status goes, the caller. - pub waiting: Option<(u64, u64, u32)>, /// Threads parked in a sleep: the monotonic deadline, and who. pub sleepers: Vec<(u64, u32)>, /// Calls parked until a descriptor they wait on is ready. diff --git a/userland/capsule_linux/src/linux/guest/handle_new.rs b/userland/capsule_linux/src/linux/guest/handle_new.rs index 0d8b798db..b290a2c50 100644 --- a/userland/capsule_linux/src/linux/guest/handle_new.rs +++ b/userland/capsule_linux/src/linux/guest/handle_new.rs @@ -57,7 +57,6 @@ impl Guest { umask: crate::linux::call::DEFAULT_UMASK, forked: Vec::new(), ended: Vec::new(), - waiting: None, sleepers: Vec::new(), blocked: Vec::new(), links: Default::default(), diff --git a/userland/capsule_linux/src/linux/serve/family.rs b/userland/capsule_linux/src/linux/serve/family.rs index 49ea32f48..132862ce0 100644 --- a/userland/capsule_linux/src/linux/serve/family.rs +++ b/userland/capsule_linux/src/linux/serve/family.rs @@ -26,7 +26,7 @@ use core::mem; use nonos_libc::{mk_foreign_reply, ForeignFrame, FOREIGN_NR_DIED}; use super::answer::Answer; -use super::dispatch::answer; +use super::route_life::answer; use super::pid_map::frame_in; use super::pid_ns::PidNs; use super::pid_out::value_out; diff --git a/userland/capsule_linux/src/linux/serve/route_life.rs b/userland/capsule_linux/src/linux/serve/route_life.rs index 59bb3972e..f24ed46c2 100644 --- a/userland/capsule_linux/src/linux/serve/route_life.rs +++ b/userland/capsule_linux/src/linux/serve/route_life.rs @@ -17,7 +17,7 @@ //! The calls of process lifecycle and signals that can leave their caller //! parked: a plain exit, a new process, a wait for a child or a signal, and a //! signal sent where only the family can say whether anyone received it. -//! Asked first by `dispatch`, so these are answered here whatever it holds. +//! Asked before `dispatch`, so these are answered here whatever it holds. use nonos_libc::ForeignFrame; @@ -30,7 +30,18 @@ use crate::linux::guest::Guest; /// clone's CLONE_THREAD: without it, clone makes a process. const CLONE_THREAD: u64 = 0x10000; -pub fn answer(guest: &mut Guest, frame: &ForeignFrame) -> Option { +/// One trap answered: here when it is one of these calls, else by dispatch. +pub fn answer(guest: &mut Guest, frame: &ForeignFrame) -> Answer { + match first(guest, frame) { + Some(got) => { + super::tally::call(); + got + } + None => super::dispatch::answer(guest, frame), + } +} + +fn first(guest: &mut Guest, frame: &ForeignFrame) -> Option { let (a, tid) = (frame.args(), frame.pid); Some(match frame.nr { nr::EXIT => call::exit_one(guest, tid, a[0]), From 240ee685e88a9233d219f1772dd3f68fa41f0cd7 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 00:17:02 +0000 Subject: [PATCH 19/26] linux: a new thread starts with its creator's signal mask A thread clone made started with every signal unblocked, whatever its creator had blocked. Go and musl block every signal around clone so a new thread cannot take one before its runtime is ready; here it could, on its first call. The new thread now takes its creator's mask, as Linux's clone gives it. --- userland/capsule_linux/src/linux/call/spawn/clone.rs | 1 + 1 file changed, 1 insertion(+) diff --git a/userland/capsule_linux/src/linux/call/spawn/clone.rs b/userland/capsule_linux/src/linux/call/spawn/clone.rs index 96d316879..284914e46 100644 --- a/userland/capsule_linux/src/linux/call/spawn/clone.rs +++ b/userland/capsule_linux/src/linux/call/spawn/clone.rs @@ -62,6 +62,7 @@ pub fn clone(guest: &mut Guest, frame: &ForeignFrame) -> Answer { } let tid = tid as u32; guest.threads.push(tid); + guest.signals.born(frame.pid, tid); /* its creator's mask, as clone gives */ // Linux writes the new tid where the caller asked, and ignores a word it // cannot write; musl keeps the parent's copy as the thread's own tid. if flags & CLONE_PARENT_SETTID != 0 { From 244566980ee6693a7c63a58339e058f710474372 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 00:17:03 +0000 Subject: [PATCH 20/26] linux: end the serve loop's wait at the next timer or signal wait The serve loop waited up to 250 ms unless a sleeper was due sooner, so SIGALRM, a POSIX timer's signal or a sigtimedwait timeout could come up to 250 ms late. The nearest of those deadlines now ends the wait too. --- userland/capsule_linux/src/linux/serve/family_sleep.rs | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/userland/capsule_linux/src/linux/serve/family_sleep.rs b/userland/capsule_linux/src/linux/serve/family_sleep.rs index 2f0e73985..8dfa0dbbc 100644 --- a/userland/capsule_linux/src/linux/serve/family_sleep.rs +++ b/userland/capsule_linux/src/linux/serve/family_sleep.rs @@ -47,8 +47,10 @@ impl Family { let sleeper = self .guests .iter() - .flat_map(|g| g.sleepers.iter().chain(g.futex_until.iter())) - .map(|&(d, _)| d.saturating_sub(now)) + .flat_map(|g| g.sleepers.iter().chain(g.futex_until.iter()).map(|&(d, _)| d)) + /* A timer or a sigtimedwait is due too. */ + .chain(self.guests.iter().filter_map(|g| g.signals.next_due())) + .map(|d| d.saturating_sub(now)) .min(); [sleeper, self.next_wait_ms(now)].into_iter().flatten().min() } From 2c6af30983683b418cf4ef4415854c8980a73c94 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 00:17:04 +0000 Subject: [PATCH 21/26] linux: a thread's death by signal keeps Linux's wait status A process whose thread died on a signal was given 128 plus the signal as its status, a shell's convention, which a parent reading it with WIFSIGNALED saw as an exit with a code of 139 or 137. It is now the signal's number, as Linux reports it. --- userland/capsule_linux/src/linux/serve/family.rs | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/userland/capsule_linux/src/linux/serve/family.rs b/userland/capsule_linux/src/linux/serve/family.rs index 132862ce0..a7283406b 100644 --- a/userland/capsule_linux/src/linux/serve/family.rs +++ b/userland/capsule_linux/src/linux/serve/family.rs @@ -79,15 +79,13 @@ impl Family { /// A guest thread ended on a signal. On Linux that ends the thread group, /// so the guest exits; reap then kills its other threads and answers any - /// waiter. The status carries the signal in the shell's 128+signo form. + /// waiter. The status is Linux's wait status: the signal's number. fn thread_died(&mut self, pid: u32, code: i32) { let Some(g) = self.guests.iter_mut().find(|g| g.owns(pid)) else { return; }; g.threads.retain(|t| *t != pid); - if g.exited.is_none() { - g.exited = Some(128 + signo_of(code)); - } + crate::linux::call::killed(g, signo_of(code) as u8); let line = alloc::format!("[LINUX] guest thread {pid} ended on a signal; ending the process\n"); crate::linux::start::say(line.as_bytes()); From 1b1e6aee918297f0e1e0daa17a40e8de8f4aac23 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 00:17:14 +0000 Subject: [PATCH 22/26] linux: raise SIGPIPE at a thread whose write answers EPIPE A write to a pipe no process can read answered EPIPE and nothing more, so a program relying on SIGPIPE's default to end it, as a shell pipeline's writer does, kept going, and a SIGPIPE handler never ran. A write, writev, or a send without MSG_NOSIGNAL that answers EPIPE now raises SIGPIPE at the writing thread before its answer, whether it is answered at once or after waiting, so the thread takes it on its way out as on Linux: caught, the handler runs over the EPIPE; ignored, only EPIPE is seen; at its default, the process ends. --- .../src/linux/serve/deliver_pipe.rs | 46 +++++++++++++++++++ .../capsule_linux/src/linux/serve/family.rs | 1 + .../src/linux/serve/family_waits.rs | 1 + userland/capsule_linux/src/linux/serve/mod.rs | 1 + 4 files changed, 49 insertions(+) create mode 100644 userland/capsule_linux/src/linux/serve/deliver_pipe.rs diff --git a/userland/capsule_linux/src/linux/serve/deliver_pipe.rs b/userland/capsule_linux/src/linux/serve/deliver_pipe.rs new file mode 100644 index 000000000..636d2e27a --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/deliver_pipe.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! SIGPIPE at the thread whose write found no reader. Linux raises it in the +//! write itself; here the write answers EPIPE and the answer is where the +//! signal is raised, before the reply, so the thread takes it on its way out +//! as Linux's does. A send with MSG_NOSIGNAL raises nothing. + +use crate::linux::abi::errno; +use crate::linux::call::sigpipe; +use crate::linux::guest::Guest; + +const WRITE: u64 = 1; +const WRITEV: u64 = 20; +const SENDTO: u64 = 44; +const SENDMSG: u64 = 46; +const MSG_NOSIGNAL: u64 = 0x4000; + +/// Raise SIGPIPE at `tid` when call `nr` with arguments `a` answered EPIPE. +pub fn broken_pipe(g: &mut Guest, tid: u32, nr: u64, a: [u64; 6], value: u64) { + if value != errno::fail(errno::EPIPE) { + return; + } + let quiet = match nr { + WRITE | WRITEV => false, + SENDTO => a[3] & MSG_NOSIGNAL != 0, + SENDMSG => a[2] & MSG_NOSIGNAL != 0, + _ => return, + }; + if !quiet { + sigpipe(g, tid); + } +} diff --git a/userland/capsule_linux/src/linux/serve/family.rs b/userland/capsule_linux/src/linux/serve/family.rs index a7283406b..92cb042e0 100644 --- a/userland/capsule_linux/src/linux/serve/family.rs +++ b/userland/capsule_linux/src/linux/serve/family.rs @@ -69,6 +69,7 @@ impl Family { let born = mem::take(&mut g.forked); if let Answer::Reply(value) = got { // A caught signal for this thread is delivered in place of the reply. + super::deliver_pipe::broken_pipe(g, frame.pid, frame.nr, frame.args(), value); let out = value_out(&mut self.ns, frame.nr, value); if !super::deliver::maybe_deliver(g, frame.pid, out) { let _ = mk_foreign_reply(frame.pid, out); diff --git a/userland/capsule_linux/src/linux/serve/family_waits.rs b/userland/capsule_linux/src/linux/serve/family_waits.rs index 735f19a44..6983f8be0 100644 --- a/userland/capsule_linux/src/linux/serve/family_waits.rs +++ b/userland/capsule_linux/src/linux/serve/family_waits.rs @@ -55,6 +55,7 @@ impl Family { } }; // A caught signal for this thread is delivered in place of the reply. + super::deliver_pipe::broken_pipe(g, wait.tid, wait.nr, wait.args, value); if !super::deliver::maybe_deliver(g, wait.tid, value) { let _ = mk_foreign_reply(wait.tid, value); } diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index 94c76e20f..917245ecf 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -19,6 +19,7 @@ mod answer; mod deliver; mod deliver_enter; +mod deliver_pipe; mod deliver_interrupt; mod deliver_rem; mod deliver_restart; From 3117b517f483e742a993f62acf93883d5082ee80 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 00:17:15 +0000 Subject: [PATCH 23/26] linux: serve /dev/null, /dev/zero, /dev/full and the random devices No path under /dev existed, so Go's os/exec, which opens /dev/null for every stream a command leaves nil, failed with ENOENT before it forked, and so did any program writing to /dev/null. The five character devices every Linux program may assume are now descriptors this capsule answers itself: null reads end of file and takes every write, zero reads zeros, full reads zeros and refuses writes with ENOSPC, random and urandom read the kernel's random bytes. fstat, stat and statx report them as character devices with Linux's major and minor numbers, lseek answers 0, and epoll refuses null, zero and full with EPERM as Linux's does. --- userland/capsule_linux/src/linux/call/io.rs | 2 + userland/capsule_linux/src/linux/file/dev.rs | 69 +++++++++++++++++++ .../capsule_linux/src/linux/file/dev_io.rs | 64 +++++++++++++++++ .../capsule_linux/src/linux/file/dev_stat.rs | 50 ++++++++++++++ .../capsule_linux/src/linux/file/epoll.rs | 4 +- .../capsule_linux/src/linux/file/meta/stat.rs | 15 ++-- .../src/linux/file/meta/statx.rs | 8 ++- userland/capsule_linux/src/linux/file/mod.rs | 4 ++ userland/capsule_linux/src/linux/file/open.rs | 3 +- userland/capsule_linux/src/linux/file/seek.rs | 5 +- .../capsule_linux/src/linux/guest/fd_kind.rs | 2 + 11 files changed, 218 insertions(+), 8 deletions(-) create mode 100644 userland/capsule_linux/src/linux/file/dev.rs create mode 100644 userland/capsule_linux/src/linux/file/dev_io.rs create mode 100644 userland/capsule_linux/src/linux/file/dev_stat.rs diff --git a/userland/capsule_linux/src/linux/call/io.rs b/userland/capsule_linux/src/linux/call/io.rs index f5145519b..ead9c31fa 100644 --- a/userland/capsule_linux/src/linux/call/io.rs +++ b/userland/capsule_linux/src/linux/call/io.rs @@ -37,6 +37,7 @@ pub fn write(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { Some(Kind::Unix) => crate::linux::unix::send(guest, fd, buf, len), Some(Kind::Pipe) => super::pipe_write(guest, fd, buf, len), Some(Kind::Event) => file::event_write(guest, fd, buf, len), + Some(Kind::Device) => file::dev_write(guest, fd, len), Some(Kind::Resolver) => net::dns::query(guest, fd, buf, len, LOOPBACK_53), Some(Kind::Dir) => errno::fail(errno::EISDIR), _ => errno::fail(errno::EBADF), @@ -52,6 +53,7 @@ pub fn read(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { Some(Kind::Unix) => crate::linux::unix::recv(guest, fd, buf, len), Some(Kind::Pipe) => super::pipe_read(guest, fd, buf, len), Some(Kind::Event) => file::event_read(guest, fd, buf, len), + Some(Kind::Device) => file::dev_read(guest, fd, buf, len), Some(Kind::Resolver) => net::dns::answer_out(guest, fd, buf, len).0, Some(Kind::Dir) => errno::fail(errno::EISDIR), _ => errno::fail(errno::EBADF), diff --git a/userland/capsule_linux/src/linux/file/dev.rs b/userland/capsule_linux/src/linux/file/dev.rs new file mode 100644 index 000000000..f47420bd1 --- /dev/null +++ b/userland/capsule_linux/src/linux/file/dev.rs @@ -0,0 +1,69 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The character devices every Linux program may assume: /dev/null, /dev/zero, +//! /dev/full, /dev/random and /dev/urandom. They are not files in the store; +//! each is a descriptor this capsule answers itself, with the major and minor +//! numbers Linux gives it, so fstat and stat say character device as Linux's. + +use crate::linux::abi::errno; +use crate::linux::guest::{Fd, Guest, Kind}; + +use super::flags::wants_write; +use super::slot::install; + +/// Path, major, minor. The handle of an open device is its index here. +const DEVICES: [(&[u8], u64, u64); 5] = [ + (b"/dev/null", 1, 3), + (b"/dev/zero", 1, 5), + (b"/dev/full", 1, 7), + (b"/dev/random", 1, 8), + (b"/dev/urandom", 1, 9), +]; +pub const NULL: u32 = 0; +pub const ZERO: u32 = 1; +pub const FULL: u32 = 2; + +/// The device a guest-visible path names, if it names one. +pub fn device_of(full: &[u8]) -> Option { + DEVICES.iter().position(|(p, _, _)| *p == full).map(|i| i as u32) +} + +/// Open the device `full` names; the caller has checked that it names one. +pub fn open_path(guest: &mut Guest, full: &[u8], flags: u64) -> u64 { + let Some(dev) = device_of(full) else { + return errno::fail(errno::ENOENT); + }; + let mut fd = Fd::empty(Kind::Device); + fd.handle = dev; + fd.path = full.to_vec(); + fd.writable = wants_write(flags); + match install(guest, fd) { + Some(n) => errno::ok(n), + None => errno::fail(errno::EMFILE), + } +} + +/// The major and minor numbers of an open device. +pub fn numbers(dev: u32) -> Option<(u64, u64)> { + DEVICES.get(dev as usize).map(|&(_, major, minor)| (major, minor)) +} + +/// Whether epoll may watch it: null, zero and full have no poll on Linux and +/// are refused with EPERM; the random devices can be waited on. +pub fn polls(dev: u32) -> bool { + dev > FULL +} diff --git a/userland/capsule_linux/src/linux/file/dev_io.rs b/userland/capsule_linux/src/linux/file/dev_io.rs new file mode 100644 index 000000000..17847c6a1 --- /dev/null +++ b/userland/capsule_linux/src/linux/file/dev_io.rs @@ -0,0 +1,64 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Reading and writing the character devices, as Linux's drivers answer: +//! /dev/null reads end of file and takes every write, /dev/zero reads zeros +//! and takes every write, /dev/full reads zeros and refuses every write with +//! ENOSPC, and the two random devices read the kernel's random bytes and take +//! writes without keeping them, as an unprivileged write to them does. + +use alloc::vec; + +use crate::linux::abi::errno; +use crate::linux::guest::{Guest, MAX_SPAN}; + +use super::dev::{FULL, NULL, ZERO}; + +/// Linux's random devices hand at most this much to one read. +const RANDOM_MAX: u64 = 32 << 20; + +fn device(guest: &Guest, fd: u64) -> u32 { + guest.fds.get(fd as usize).map_or(u32::MAX, |f| f.handle) +} + +pub fn read(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { + let dev = device(guest, fd); + if dev == NULL { + return errno::ok(0); + } + /* One step at a time, so a large read never holds a large buffer. */ + let want = if dev == ZERO || dev == FULL { len } else { len.min(RANDOM_MAX) }; + let mut done = 0; + while done < want { + let take = (want - done).min(MAX_SPAN) as usize; + let mut bytes = vec![0u8; take]; + if dev != ZERO && dev != FULL && nonos_libc::crypto_random(bytes.as_mut_ptr(), take) < 0 { + break; + } + if guest.write(buf + done, &bytes) < take as i64 { + return if done == 0 { errno::fail(errno::EFAULT) } else { errno::ok(done) }; + } + done += take as u64; + } + errno::ok(done) +} + +pub fn write(guest: &mut Guest, fd: u64, len: u64) -> u64 { + match device(guest, fd) { + FULL => errno::fail(errno::ENOSPC), + _ => errno::ok(len), + } +} diff --git a/userland/capsule_linux/src/linux/file/dev_stat.rs b/userland/capsule_linux/src/linux/file/dev_stat.rs new file mode 100644 index 000000000..f64043710 --- /dev/null +++ b/userland/capsule_linux/src/linux/file/dev_stat.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What stat, fstat and statx say about a character device: S_IFCHR with +//! read and write for everyone, as Linux's devtmpfs makes them, and the +//! device's major and minor numbers. + +use super::dev::numbers; + +const MODE: u32 = 0o020666; +/// st_mode and st_rdev in a `struct stat`. +const STAT_MODE: usize = 24; +const STAT_RDEV: usize = 40; +/// stx_mode, stx_rdev_major and stx_rdev_minor in a `struct statx`. +const STATX_MODE: usize = 28; +const STATX_RDEV: usize = 128; + +/// A `struct stat` made for a file, turned into the device's. st_rdev is +/// Linux's encoding of the two numbers. +pub fn as_device(stat: &mut [u8], dev: u32) { + let Some((major, minor)) = numbers(dev) else { + return; + }; + let rdev = (minor & 0xff) | ((major & 0xfff) << 8) | ((minor & !0xff) << 12); + stat[STAT_MODE..STAT_MODE + 4].copy_from_slice(&MODE.to_le_bytes()); + stat[STAT_RDEV..STAT_RDEV + 8].copy_from_slice(&rdev.to_le_bytes()); +} + +/// The same for a `struct statx`, which keeps the two numbers apart. +pub fn statx_device(buf: &mut [u8], dev: u32) { + let Some((major, minor)) = numbers(dev) else { + return; + }; + buf[STATX_MODE..STATX_MODE + 2].copy_from_slice(&(MODE as u16).to_le_bytes()); + buf[STATX_RDEV..STATX_RDEV + 4].copy_from_slice(&(major as u32).to_le_bytes()); + buf[STATX_RDEV + 4..STATX_RDEV + 8].copy_from_slice(&(minor as u32).to_le_bytes()); +} diff --git a/userland/capsule_linux/src/linux/file/epoll.rs b/userland/capsule_linux/src/linux/file/epoll.rs index 51435cfbd..8c56745bf 100644 --- a/userland/capsule_linux/src/linux/file/epoll.rs +++ b/userland/capsule_linux/src/linux/file/epoll.rs @@ -54,7 +54,9 @@ pub fn epoll_ctl(guest: &mut Guest, ep: u64, op: u64, fd: u64, event: u64) -> u6 if !open(ep) || !open(fd) { return errno::fail(errno::EBADF); } - if guest.fds.get(fd as usize).is_some_and(|f| matches!(f.kind, Kind::File | Kind::Dir)) { + let unpollable = |f: &Fd| f.kind == Kind::Device && !super::dev::polls(f.handle); + let refused = |f: &Fd| matches!(f.kind, Kind::File | Kind::Dir) || unpollable(f); + if guest.fds.get(fd as usize).is_some_and(refused) { return errno::fail(errno::EPERM); } let Some(list) = guest.fds.get_mut(ep as usize).filter(|f| f.kind == Kind::Epoll) else { diff --git a/userland/capsule_linux/src/linux/file/meta/stat.rs b/userland/capsule_linux/src/linux/file/meta/stat.rs index 90e0ee284..f8f032ac2 100644 --- a/userland/capsule_linux/src/linux/file/meta/stat.rs +++ b/userland/capsule_linux/src/linux/file/meta/stat.rs @@ -19,6 +19,7 @@ use crate::linux::abi::errno; use crate::linux::guest::{Guest, Kind}; +use super::super::dev::device_of; use super::super::flags::AT_FDCWD; use super::super::{path, resolve, store}; use super::statbuf::{build, inode, STAT_LEN}; @@ -27,6 +28,7 @@ use super::statbuf::{build, inode, STAT_LEN}; /// absent. `full` is guest-visible and is confined here. pub fn look(full: &[u8]) -> Option<(u64, bool)> { match store::stat_full(&resolve::key(full)) { + _ if device_of(full).is_some() => Some((0, false)), Ok((size, is_dir, _, _)) => Some((size, is_dir)), Err(_) => None, } @@ -43,7 +45,8 @@ pub fn fstat(guest: &mut Guest, fd: u64, out: u64) -> u64 { _ => (0, false), }; let ino = inode(&entry.path); - write_out(guest, out, size, is_dir, ino) + let dev = (entry.kind == Kind::Device).then_some(entry.handle); + write_out(guest, out, size, is_dir, ino, dev) } pub fn newfstatat(guest: &mut Guest, dirfd: u64, path_ptr: u64, out: u64) -> u64 { @@ -55,13 +58,17 @@ pub fn newfstatat(guest: &mut Guest, dirfd: u64, path_ptr: u64, out: u64) -> u64 } let full = guest.links.follow(resolve::visible(&guest.cwd, &name), true); match look(&full) { - Some((size, is_dir)) => write_out(guest, out, size, is_dir, inode(&full)), + Some((size, is_dir)) => write_out(guest, out, size, is_dir, inode(&full), device_of(&full)), None => errno::fail(errno::ENOENT), } } -fn write_out(guest: &Guest, out: u64, size: u64, is_dir: bool, ino: u64) -> u64 { - if guest.write(out, &build(size, is_dir, ino)) < STAT_LEN as i64 { +fn write_out(guest: &Guest, out: u64, size: u64, is_dir: bool, ino: u64, dev: Option) -> u64 { + let mut stat = build(size, is_dir, ino); + if let Some(d) = dev { + super::super::dev_stat::as_device(&mut stat, d); + } + if guest.write(out, &stat) < STAT_LEN as i64 { return errno::fail(errno::EFAULT); } errno::ok(0) diff --git a/userland/capsule_linux/src/linux/file/meta/statx.rs b/userland/capsule_linux/src/linux/file/meta/statx.rs index d1375b89e..0ba7c4558 100644 --- a/userland/capsule_linux/src/linux/file/meta/statx.rs +++ b/userland/capsule_linux/src/linux/file/meta/statx.rs @@ -41,7 +41,10 @@ pub fn statx(guest: &Guest, dirfd: u64, path: u64, out: u64) -> u64 { let Some(at) = resolve_at(guest, dirfd, path) else { return errno::fail(errno::EFAULT); }; - let Ok((size, is_dir, _, readonly)) = store::stat_full(&key(&at)) else { + let dev = super::super::dev::device_of(&at); + let Some((size, is_dir, _, readonly)) = + store::stat_full(&key(&at)).ok().or(dev.map(|_| (0, false, 0, false))) + else { return errno::fail(errno::ENOENT); }; let mode = if is_dir { S_IFDIR } else { S_IFREG } | if readonly { 0o555 } else { 0o755 }; @@ -53,6 +56,9 @@ pub fn statx(guest: &Guest, dirfd: u64, path: u64, out: u64) -> u64 { buf[32..40].copy_from_slice(&inode(&at).to_le_bytes()); // stx_ino buf[40..48].copy_from_slice(&size.to_le_bytes()); // stx_size buf[48..56].copy_from_slice(&size.div_ceil(512).to_le_bytes()); // stx_blocks + if let Some(d) = dev { + super::super::dev_stat::statx_device(&mut buf, d); + } match guest.write(out, &buf) { n if n < 0 => errno::fail(errno::EFAULT), _ => errno::ok(0), diff --git a/userland/capsule_linux/src/linux/file/mod.rs b/userland/capsule_linux/src/linux/file/mod.rs index 51bf8ea96..c9db777f0 100644 --- a/userland/capsule_linux/src/linux/file/mod.rs +++ b/userland/capsule_linux/src/linux/file/mod.rs @@ -20,6 +20,9 @@ mod at; mod clamp; pub(super) mod close; mod cstr; +mod dev; +mod dev_io; +mod dev_stat; mod dir; mod dir_children; mod dirent; @@ -59,6 +62,7 @@ mod write; pub use close::close; pub use cstr::read_cstr; +pub use dev_io::{read as dev_read, write as dev_write}; pub use dirents::getdents64; pub use dirops::{mkdirat, rmdir, unlinkat}; pub use epoll::{epoll_create, epoll_ctl}; diff --git a/userland/capsule_linux/src/linux/file/open.rs b/userland/capsule_linux/src/linux/file/open.rs index a36d29a0b..5ef6051a9 100644 --- a/userland/capsule_linux/src/linux/file/open.rs +++ b/userland/capsule_linux/src/linux/file/open.rs @@ -22,7 +22,7 @@ use crate::linux::abi::errno; use crate::linux::guest::{Guest, Kind}; use super::flags::{wants_write, AT_FDCWD, O_CLOEXEC, O_CREAT, O_DIRECTORY}; -use super::{dir, path, regular, resolve, store}; +use super::{dev, dir, path, regular, resolve, store}; pub fn openat(guest: &mut Guest, dirfd: u64, path_ptr: u64, flags: u64) -> u64 { let Some(name) = path::read_path(guest, path_ptr) else { @@ -34,6 +34,7 @@ pub fn openat(guest: &mut Guest, dirfd: u64, path_ptr: u64, flags: u64) -> u64 { }; let full = guest.links.follow(resolve::visible(&base, &name), true); let got = match store::stat(&resolve::key(&full)).ok() { + _ if dev::device_of(&full).is_some() => dev::open_path(guest, &full, flags), Some((_, true)) => dir::open(guest, full), Some((_, false)) if flags & O_DIRECTORY != 0 => errno::fail(errno::ENOTDIR), Some((size, false)) => regular::open(guest, full, size, flags), diff --git a/userland/capsule_linux/src/linux/file/seek.rs b/userland/capsule_linux/src/linux/file/seek.rs index 140e5610d..66002939f 100644 --- a/userland/capsule_linux/src/linux/file/seek.rs +++ b/userland/capsule_linux/src/linux/file/seek.rs @@ -14,7 +14,6 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - //! `lseek`. The position is this capsule's, not the server's: a read takes //! a window at an offset, so the descriptor's offset is the whole of it. @@ -29,6 +28,10 @@ pub fn lseek(guest: &mut Guest, fd: u64, offset: u64, whence: u64) -> u64 { let Some(entry) = guest.fds.get_mut(fd as usize) else { return errno::fail(errno::EBADF); }; + /* A character device has no position to move, and Linux answers 0. */ + if entry.kind == Kind::Device { + return errno::ok(0); + } if entry.kind != Kind::File { // A pipe or a console has no position, which Linux calls ESPIPE. return errno::fail(errno::ESPIPE); diff --git a/userland/capsule_linux/src/linux/guest/fd_kind.rs b/userland/capsule_linux/src/linux/guest/fd_kind.rs index 6888e6dd0..dafe25921 100644 --- a/userland/capsule_linux/src/linux/guest/fd_kind.rs +++ b/userland/capsule_linux/src/linux/guest/fd_kind.rs @@ -44,4 +44,6 @@ pub enum Kind { Resolver, /// An eventfd: a counter one thread adds to and another takes from. Event, + /// A character device this capsule answers: /dev/null and its kin. + Device, } From 9b551ebf64f7c0398e07b46ad760ed56bb57f738 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 00:17:28 +0000 Subject: [PATCH 24/26] linux: serve signalfd4 and signalfd signalfd4 was not served, so a program that reads its signals from a descriptor instead of running handlers, as event loops built on epoll do, could not start. A signalfd now takes a mask; a read by a thread takes the pending signals of that mask for the thread or its process as signalfd_siginfo records, laid out as Linux's, answers EAGAIN on a non-blocking descriptor when none is pending, and otherwise parks until one comes. poll and epoll see it readable once a signal of its mask waits. A second signalfd4 on it changes the mask, and a fork gives the child a copy. --- .../capsule_linux/src/linux/abi/nr_sig.rs | 2 + userland/capsule_linux/src/linux/call/io.rs | 1 + userland/capsule_linux/src/linux/call/mod.rs | 7 ++ .../src/linux/call/signal_timedwait.rs | 2 +- .../src/linux/call/signal_wait.rs | 2 +- .../capsule_linux/src/linux/call/signalfd.rs | 60 ++++++++++++++++ .../src/linux/call/signalfd_info.rs | 55 +++++++++++++++ .../src/linux/call/signalfd_read.rs | 47 +++++++++++++ .../src/linux/call/signalfd_take.rs | 69 +++++++++++++++++++ .../capsule_linux/src/linux/guest/fd_kind.rs | 2 + .../capsule_linux/src/linux/guest/sigqueue.rs | 2 + .../src/linux/guest/sigqueue_new.rs | 1 + .../src/linux/guest/sigthread_copy.rs | 4 +- .../capsule_linux/src/linux/guest/sigwaits.rs | 2 + userland/capsule_linux/src/linux/net/poll.rs | 1 + .../src/linux/serve/deliver_sigwait.rs | 9 +++ .../src/linux/serve/route_life.rs | 7 ++ 17 files changed, 270 insertions(+), 3 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/signalfd.rs create mode 100644 userland/capsule_linux/src/linux/call/signalfd_info.rs create mode 100644 userland/capsule_linux/src/linux/call/signalfd_read.rs create mode 100644 userland/capsule_linux/src/linux/call/signalfd_take.rs diff --git a/userland/capsule_linux/src/linux/abi/nr_sig.rs b/userland/capsule_linux/src/linux/abi/nr_sig.rs index 9a5f9e65a..6f5df6545 100644 --- a/userland/capsule_linux/src/linux/abi/nr_sig.rs +++ b/userland/capsule_linux/src/linux/abi/nr_sig.rs @@ -35,3 +35,5 @@ pub const TIMER_DELETE: u64 = 226; pub const TGKILL: u64 = 234; pub const WAITID: u64 = 247; pub const RT_TGSIGQUEUEINFO: u64 = 297; +pub const SIGNALFD: u64 = 282; +pub const SIGNALFD4: u64 = 289; diff --git a/userland/capsule_linux/src/linux/call/io.rs b/userland/capsule_linux/src/linux/call/io.rs index ead9c31fa..973662bf0 100644 --- a/userland/capsule_linux/src/linux/call/io.rs +++ b/userland/capsule_linux/src/linux/call/io.rs @@ -54,6 +54,7 @@ pub fn read(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { Some(Kind::Pipe) => super::pipe_read(guest, fd, buf, len), Some(Kind::Event) => file::event_read(guest, fd, buf, len), Some(Kind::Device) => file::dev_read(guest, fd, buf, len), + Some(Kind::Signal) => super::signalfd_now(guest, fd, buf, len), Some(Kind::Resolver) => net::dns::answer_out(guest, fd, buf, len).0, Some(Kind::Dir) => errno::fail(errno::EISDIR), _ => errno::fail(errno::EBADF), diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index 43d07c240..53010e330 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -48,6 +48,10 @@ pub mod sigframe_build; mod sigframe_read; mod signal; mod signal_its; +mod signalfd; +mod signalfd_info; +mod signalfd_read; +mod signalfd_take; mod signal_itv; mod signal_mask; mod signal_post; @@ -97,6 +101,9 @@ pub use session::{getpgid, getsid, setpgid, setsid}; pub use signal::rt_sigaction; pub use signal_mask::rt_sigprocmask; pub use signal_queue::{rt_sigqueueinfo, rt_tgsigqueueinfo}; +pub use signalfd::signalfd4; +pub use signalfd_read::signalfd_read; +pub use signalfd_take::{signalfd_bits, signalfd_now, signalfd_take}; pub use signal_send::{kill, kill_from, sigpipe, tgkill_from}; pub use signal_stack::sigaltstack; pub use signal_timer::{alarm, getitimer, setitimer}; diff --git a/userland/capsule_linux/src/linux/call/signal_timedwait.rs b/userland/capsule_linux/src/linux/call/signal_timedwait.rs index 0f58217e9..43236207a 100644 --- a/userland/capsule_linux/src/linux/call/signal_timedwait.rs +++ b/userland/capsule_linux/src/linux/call/signal_timedwait.rs @@ -59,7 +59,7 @@ pub fn rt_sigtimedwait( if due.is_some_and(|d| now_ms(CLOCK_MONOTONIC).is_some_and(|now| d <= now)) { return Answer::value(errno::fail(errno::EAGAIN)); } - guest.signals.sigwaits.push(SigWait { tid, set: want, info, due }); + guest.signals.sigwaits.push(SigWait { tid, set: want, info, due, records: 0 }); Answer::Park } diff --git a/userland/capsule_linux/src/linux/call/signal_wait.rs b/userland/capsule_linux/src/linux/call/signal_wait.rs index d90dfbadb..412bfb21f 100644 --- a/userland/capsule_linux/src/linux/call/signal_wait.rs +++ b/userland/capsule_linux/src/linux/call/signal_wait.rs @@ -27,7 +27,7 @@ use crate::linux::serve::Answer; pub const SIGSET_LEN: u64 = 8; pub fn pause(guest: &mut Guest, tid: u32) -> Answer { - guest.signals.sigwaits.push(SigWait { tid, set: 0, info: 0, due: None }); + guest.signals.sigwaits.push(SigWait { tid, set: 0, info: 0, due: None, records: 0 }); Answer::Park } diff --git a/userland/capsule_linux/src/linux/call/signalfd.rs b/userland/capsule_linux/src/linux/call/signalfd.rs new file mode 100644 index 000000000..6d72f8c22 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signalfd.rs @@ -0,0 +1,60 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `signalfd4` and `signalfd`: a descriptor that reads the signals of a mask +//! pending for the reading thread or its process, as `signalfd_siginfo` +//! records, instead of their handlers running. A read with none pending +//! answers EAGAIN on a non-blocking descriptor and otherwise parks the +//! thread until one comes, as Linux's does. + +use crate::linux::abi::errno; +use crate::linux::file::install; +use crate::linux::guest::sigstate::blockable; +use crate::linux::guest::{Fd, Guest, Kind}; + +const SFD_NONBLOCK: u64 = 0o4000; +const SFD_CLOEXEC: u64 = 0o2000000; +const SIGSET_LEN: u64 = 8; + +pub fn signalfd4(guest: &mut Guest, fd: u64, mask: u64, size: u64, flags: u64) -> u64 { + if size != SIGSET_LEN || flags & !(SFD_NONBLOCK | SFD_CLOEXEC) != 0 { + return errno::fail(errno::EINVAL); + } + let Some(raw) = guest.read(mask, 8) else { + return errno::fail(errno::EFAULT); + }; + let set = blockable(u64::from_le_bytes(raw[..8].try_into().unwrap_or([0; 8]))); + if fd as i64 != -1 { + /* An existing signalfd takes the new mask; any other descriptor is refused. */ + return match guest.fds.get(fd as usize).filter(|f| f.kind == Kind::Signal) { + Some(f) => { + let at = f.handle as usize; + guest.signals.sigfds[at] = set; + errno::ok(fd) + } + None => errno::fail(errno::EINVAL), + }; + } + guest.signals.sigfds.push(set); + let mut new = Fd::empty(Kind::Signal); + new.handle = (guest.signals.sigfds.len() - 1) as u32; + new.nonblock = flags & SFD_NONBLOCK != 0; + new.cloexec = flags & SFD_CLOEXEC != 0; + match install(guest, new) { + Some(n) => errno::ok(n), + None => errno::fail(errno::EMFILE), + } +} diff --git a/userland/capsule_linux/src/linux/call/signalfd_info.rs b/userland/capsule_linux/src/linux/call/signalfd_info.rs new file mode 100644 index 000000000..3c56a9c7e --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signalfd_info.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A signal as a signalfd hands it over: the 128-byte `signalfd_siginfo`, +//! laid out as Linux's, with the fields the siginfo carries moved to their +//! own places in it. + +use crate::linux::guest::siginfo::{SigInfo, CLD_EXITED, CLD_KILLED}; +use crate::linux::guest::sigstate::SIGCHLD; + +pub const SFD_INFO_LEN: usize = 128; + +impl SigInfo { + pub fn fd_bytes(&self) -> [u8; SFD_INFO_LEN] { + let mut b = [0u8; SFD_INFO_LEN]; + let mut put32 = |at: usize, v: u32| b[at..at + 4].copy_from_slice(&v.to_le_bytes()); + put32(0, u32::from(self.signo)); + put32(8, self.code as u32); + let pid = match self.pid { + 0 => 0, + k => crate::linux::serve::guest_pid(k), + }; + let child = self.signo == SIGCHLD && matches!(self.code, CLD_EXITED | CLD_KILLED); + match self.timer { + /* ssi_tid is the timer's id, and ssi_overrun its overruns. */ + Some((id, overrun)) => { + put32(24, id as u32); + put32(32, overrun as u32); + } + None => put32(12, pid), + } + if child { + put32(40, self.value as u32); + } else { + put32(44, self.value as u32); + } + if !child { + b[48..56].copy_from_slice(&self.value.to_le_bytes()); + } + b + } +} diff --git a/userland/capsule_linux/src/linux/call/signalfd_read.rs b/userland/capsule_linux/src/linux/call/signalfd_read.rs new file mode 100644 index 000000000..77657847b --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signalfd_read.rs @@ -0,0 +1,47 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A read of a signalfd by one thread: the signals of its mask pending for +//! that thread or its process, EAGAIN on a non-blocking descriptor when there +//! are none, or a wait until one comes. + +use super::signalfd_info::SFD_INFO_LEN; +use crate::linux::abi::errno; +use crate::linux::guest::sigwaits::SigWait; +use crate::linux::guest::{Guest, Kind}; +use crate::linux::serve::Answer; + +use super::signalfd_take::signalfd_take; + +/// A read by thread `tid`: what is pending now, EAGAIN, or a parked wait. +pub fn signalfd_read(guest: &mut Guest, tid: u32, fd: u64, buf: u64, len: u64) -> Answer { + let Some(f) = guest.fds.get(fd as usize).filter(|f| f.kind == Kind::Signal) else { + return Answer::value(errno::fail(errno::EBADF)); + }; + let (set, nonblock) = (guest.signals.sigfds[f.handle as usize], f.nonblock); + let records = len / SFD_INFO_LEN as u64; + if records == 0 { + return Answer::value(errno::fail(errno::EINVAL)); + } + if let Some(n) = signalfd_take(guest, tid, set, buf, records) { + return Answer::value(n); + } + if nonblock { + return Answer::value(errno::fail(errno::EAGAIN)); + } + guest.signals.sigwaits.push(SigWait { tid, set, info: buf, due: None, records }); + Answer::Park +} diff --git a/userland/capsule_linux/src/linux/call/signalfd_take.rs b/userland/capsule_linux/src/linux/call/signalfd_take.rs new file mode 100644 index 000000000..2b8ca6aa2 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/signalfd_take.rs @@ -0,0 +1,69 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Taking signals for a signalfd read: as many pending signals of the mask as +//! records fit, each written where the next record goes. None when nothing is +//! pending, so the read can wait or answer EAGAIN. + +use super::signalfd_info::SFD_INFO_LEN; +use crate::linux::abi::errno; +use crate::linux::guest::{Guest, Kind}; + +pub fn signalfd_take(guest: &mut Guest, tid: u32, set: u64, buf: u64, records: u64) -> Option { + let mut n = 0; + while n < records { + let Some(info) = guest.signals.take(tid, set) else { + break; + }; + let at = buf + n * SFD_INFO_LEN as u64; + if guest.write(at, &info.fd_bytes()) < SFD_INFO_LEN as i64 { + return Some(if n == 0 { errno::fail(errno::EFAULT) } else { n * SFD_INFO_LEN as u64 }); + } + n += 1; + } + (n > 0).then_some(n * SFD_INFO_LEN as u64) +} + +/// poll's POLLIN when a signal of the mask waits for the process or its +/// first thread; a read by another thread may find its own too. +pub fn signalfd_bits(guest: &Guest, fd: u64) -> u16 { + const POLLIN: u16 = 0x001; + let Some(f) = guest.fds.get(fd as usize).filter(|f| f.kind == Kind::Signal) else { + return 0; + }; + let set = guest.signals.sigfds.get(f.handle as usize).copied().unwrap_or(0); + let who = guest.live_threads().first().copied().unwrap_or(guest.pid); + if guest.signals.pending_for(who) & set != 0 { + POLLIN + } else { + 0 + } +} + +/// A read that reaches a signalfd another way than read(2), readv: what is +/// pending for the process or its first thread now, else EAGAIN. +pub fn signalfd_now(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { + let Some(f) = guest.fds.get(fd as usize).filter(|f| f.kind == Kind::Signal) else { + return errno::fail(errno::EBADF); + }; + let set = guest.signals.sigfds.get(f.handle as usize).copied().unwrap_or(0); + let who = guest.live_threads().first().copied().unwrap_or(guest.pid); + let records = len / SFD_INFO_LEN as u64; + match records { + 0 => errno::fail(errno::EINVAL), + n => signalfd_take(guest, who, set, buf, n).unwrap_or(errno::fail(errno::EAGAIN)), + } +} diff --git a/userland/capsule_linux/src/linux/guest/fd_kind.rs b/userland/capsule_linux/src/linux/guest/fd_kind.rs index dafe25921..4519659a5 100644 --- a/userland/capsule_linux/src/linux/guest/fd_kind.rs +++ b/userland/capsule_linux/src/linux/guest/fd_kind.rs @@ -46,4 +46,6 @@ pub enum Kind { Event, /// A character device this capsule answers: /dev/null and its kin. Device, + /// A signalfd: signals of a mask, read as records. + Signal, } diff --git a/userland/capsule_linux/src/linux/guest/sigqueue.rs b/userland/capsule_linux/src/linux/guest/sigqueue.rs index 56db59eb6..ca46189d0 100644 --- a/userland/capsule_linux/src/linux/guest/sigqueue.rs +++ b/userland/capsule_linux/src/linux/guest/sigqueue.rs @@ -55,4 +55,6 @@ pub struct Signals { pub clone_kids: Vec, /// The process group each ended child was in, for a wait by group. pub kid_groups: Vec<(u32, u32)>, + /// The mask of each signalfd, named by its descriptor's handle. + pub sigfds: Vec, } diff --git a/userland/capsule_linux/src/linux/guest/sigqueue_new.rs b/userland/capsule_linux/src/linux/guest/sigqueue_new.rs index 183333960..a1afd5ee6 100644 --- a/userland/capsule_linux/src/linux/guest/sigqueue_new.rs +++ b/userland/capsule_linux/src/linux/guest/sigqueue_new.rs @@ -38,6 +38,7 @@ impl Default for Signals { exit_signal: super::sigstate::SIGCHLD, clone_kids: Vec::new(), kid_groups: Vec::new(), + sigfds: Vec::new(), } } } diff --git a/userland/capsule_linux/src/linux/guest/sigthread_copy.rs b/userland/capsule_linux/src/linux/guest/sigthread_copy.rs index 0b1977b7d..439074569 100644 --- a/userland/capsule_linux/src/linux/guest/sigthread_copy.rs +++ b/userland/capsule_linux/src/linux/guest/sigthread_copy.rs @@ -25,7 +25,9 @@ impl Signals { /// A forked child: the dispositions, and the forking thread's mask and /// alternate stack for its one thread. Nothing pending, no timers. pub fn forked(&self, caller: u32, child: u32) -> Signals { - let mut s = Signals { actions: self.actions, ..Signals::default() }; + /* A signalfd is a descriptor, and the child holds a copy of each. */ + let sigfds = self.sigfds.clone(); + let mut s = Signals { actions: self.actions, sigfds, ..Signals::default() }; let mine = self.threads.iter().find(|t| t.tid == caller).copied(); let mut t = mine.unwrap_or(ThreadSig::new(child, 0)); t.tid = child; diff --git a/userland/capsule_linux/src/linux/guest/sigwaits.rs b/userland/capsule_linux/src/linux/guest/sigwaits.rs index 6dde91516..b07e0d3c1 100644 --- a/userland/capsule_linux/src/linux/guest/sigwaits.rs +++ b/userland/capsule_linux/src/linux/guest/sigwaits.rs @@ -30,6 +30,8 @@ pub struct SigWait { pub info: u64, /// When sigtimedwait gives up with EAGAIN. pub due: Option, + /// For a signalfd read, how many records fit where `info` points; 0 else. + pub records: u64, } /// Which children a wait4 or waitid asks about. diff --git a/userland/capsule_linux/src/linux/net/poll.rs b/userland/capsule_linux/src/linux/net/poll.rs index 55b30b9a4..08db2b49f 100644 --- a/userland/capsule_linux/src/linux/net/poll.rs +++ b/userland/capsule_linux/src/linux/net/poll.rs @@ -38,6 +38,7 @@ pub fn ready(guest: &Guest, fd: u64) -> u16 { Some(Kind::Timer) => crate::linux::file::timer_bits(guest, fd), Some(Kind::Pipe) => crate::linux::call::pipe_bits(guest, fd), Some(Kind::Event) => crate::linux::file::event_bits(guest, fd), + Some(Kind::Signal) => crate::linux::call::signalfd_bits(guest, fd), Some(Kind::Resolver) => resolver_bits(guest, fd), Some(_) => POLLIN | POLLOUT, } diff --git a/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs b/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs index ed8aa69fa..8c840cf16 100644 --- a/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs +++ b/userland/capsule_linux/src/linux/serve/deliver_sigwait.rs @@ -26,6 +26,15 @@ use crate::linux::guest::Guest; /// answers with that signal's number and siginfo. pub fn taken_by_sigtimedwait(guest: &mut Guest) { for w in guest.signals.sigwaits.clone().into_iter().filter(|w| w.set != 0) { + if w.records != 0 { + if let Some(n) = + crate::linux::call::signalfd_take(guest, w.tid, w.set, w.info, w.records) + { + guest.signals.sigwaits.retain(|x| x.tid != w.tid); + let _ = mk_foreign_reply(w.tid, n); + } + continue; + } let Some(info) = guest.signals.take(w.tid, w.set) else { continue; }; diff --git a/userland/capsule_linux/src/linux/serve/route_life.rs b/userland/capsule_linux/src/linux/serve/route_life.rs index f24ed46c2..0a2b06b1e 100644 --- a/userland/capsule_linux/src/linux/serve/route_life.rs +++ b/userland/capsule_linux/src/linux/serve/route_life.rs @@ -60,8 +60,15 @@ fn first(guest: &mut Guest, frame: &ForeignFrame) -> Option { ns::RT_SIGQUEUEINFO => call::rt_sigqueueinfo(guest, tid, a[0], a[1], a[2]), ns::RT_TGSIGQUEUEINFO => call::rt_tgsigqueueinfo(guest, tid, a[0], a[1], a[2], a[3]), ns::PAUSE => call::pause(guest, tid), + ns::SIGNALFD4 => Answer::value(call::signalfd4(guest, a[0], a[1], a[2], a[3])), + ns::SIGNALFD => Answer::value(call::signalfd4(guest, a[0], a[1], a[2], 0)), + nr::READ if signalfd(guest, a[0]) => call::signalfd_read(guest, tid, a[0], a[1], a[2]), ns::RT_SIGSUSPEND => call::rt_sigsuspend(guest, tid, a[0], a[1]), ns::RT_SIGTIMEDWAIT => call::rt_sigtimedwait(guest, tid, a[0], a[1], a[2], a[3]), _ => return None, }) } + +fn signalfd(guest: &Guest, fd: u64) -> bool { + guest.fds.get(fd as usize).is_some_and(|f| f.kind == crate::linux::guest::Kind::Signal) +} From 408e336f7c83655a69ddcded7207e2087ca656d1 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 00:17:28 +0000 Subject: [PATCH 25/26] mk: pack only the Linux guests LINUX_GUEST_SET names The Linux-guest test store packs every guest, and vfs loads a store of 16 MiB at most. With the waiting guests and the lifecycle guests together it holds 19.8 MB, and nonos-store-pack refuses it, so no guest boots. LINUX_GUEST_SET, when given, names the guests to pack; left empty, every guest is packed as before. Each guest is still built, signed and enrolled. --- userland/linux_guests/Guests.mk | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index fc3f3b9f9..9499ab0b1 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -54,10 +54,12 @@ nonos-mk-check-linux-guest-$(1)-keys: \ $(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_ed25519.pub \ $(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_mldsa65.pub LINUX_GUEST_STORE_DEPS += $$(linux-guest-$(1)_ARTIFACTS) $$(linux-guest-$(1)_ATTESTATION) -LINUX_GUEST_STORE_ENTRIES += --entry /linux$(or $(5),/bin/$(1))=$$(linux-guest-$(1)_BIN) \ +LINUX_GUEST_ENTRIES_$(1) := --entry /linux$(or $(5),/bin/$(1))=$$(linux-guest-$(1)_BIN) \ --entry /linux$(or $(5),/bin/$(1)).nonos_id_cert.bin=$$(linux-guest-$(1)_CERT) \ --entry /linux$(or $(5),/bin/$(1)).manifest.bin=$$(linux-guest-$(1)_MANIFEST) \ --entry /linux$(or $(5),/bin/$(1)).zk_trailer.bin=$$(linux-guest-$(1)_ATTESTATION) +# Every guest, or only those LINUX_GUEST_SET names: vfs loads 16 MiB at most. +LINUX_GUEST_STORE_ENTRIES += $$(if $$(filter $(1),$$(or $$(LINUX_GUEST_SET),$(1))),$$(LINUX_GUEST_ENTRIES_$(1))) endef $(eval $(call LINUX_GUEST,suite,4950,4951)) From 15067eb1252ea2256da50e44cc3bbaafdfed56b4 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 00:17:28 +0000 Subject: [PATCH 26/26] linux: goexec checks the devices and os/exec, alarm checks signalfd goexec now leaves each os/exec command's streams nil, as most Go programs do, so cmd.Output and Run open /dev/null, and checks the devices directly: /dev/null stats as a character device and reads end of file, /dev/zero reads zeros, /dev/full refuses a write with ENOSPC. alarm now checks a signalfd: EAGAIN with nothing pending, readable in poll once SIGUSR1 waits, the record's signal and pid, and a sigqueue value through a blocking read. --- userland/linux_guests/LifeGuests.mk | 4 +-- userland/linux_guests/c/alarm.c | 6 ++-- userland/linux_guests/c/alarm.h | 1 + userland/linux_guests/c/alarm_sigfd.c | 44 ++++++++++++++++++++++++ userland/linux_guests/go/exec/devices.go | 33 ++++++++++++++++++ userland/linux_guests/go/exec/main.go | 16 +++------ 6 files changed, 88 insertions(+), 16 deletions(-) create mode 100644 userland/linux_guests/c/alarm_sigfd.c create mode 100644 userland/linux_guests/go/exec/devices.go diff --git a/userland/linux_guests/LifeGuests.mk b/userland/linux_guests/LifeGuests.mk index 828aff912..69917cf01 100644 --- a/userland/linux_guests/LifeGuests.mk +++ b/userland/linux_guests/LifeGuests.mk @@ -7,7 +7,7 @@ $(LINUX_GUESTS_C)/leaderexit: $(LINUX_GUESTS_DIR)/c/leaderexit.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,leaderexit,4990,4991,$(LINUX_GUESTS_C)/leaderexit)) -$(GO_OUT)/exec: $(LINUX_GUESTS_DIR)/go/exec/forkexec.go +$(GO_OUT)/exec: $(addprefix $(LINUX_GUESTS_DIR)/go/exec/,forkexec.go devices.go) $(eval $(call LINUX_GUEST,goexec,4992,4993,$(GO_OUT)/exec)) $(LINUX_GUESTS_C)/sigchld: $(addprefix $(LINUX_GUESTS_DIR)/c/,sigchld.c sigchld_wait.c sigchld_spawn.c sigchld.h) @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $(filter %.c,$^) @@ -15,6 +15,6 @@ $(eval $(call LINUX_GUEST,sigchld,4994,4995,$(LINUX_GUESTS_C)/sigchld)) $(LINUX_GUESTS_C)/sigpipe: $(LINUX_GUESTS_DIR)/c/sigpipe.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,sigpipe,4996,4997,$(LINUX_GUESTS_C)/sigpipe)) -$(LINUX_GUESTS_C)/alarm: $(addprefix $(LINUX_GUESTS_DIR)/c/,alarm.c alarm_wait.c alarm_timer.c alarm.h) +$(LINUX_GUESTS_C)/alarm: $(addprefix $(LINUX_GUESTS_DIR)/c/,alarm.c alarm_wait.c alarm_timer.c alarm_sigfd.c alarm.h) @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $(filter %.c,$^) $(eval $(call LINUX_GUEST,alarm,4998,4999,$(LINUX_GUESTS_C)/alarm)) diff --git a/userland/linux_guests/c/alarm.c b/userland/linux_guests/c/alarm.c index dbd02f030..665491878 100644 --- a/userland/linux_guests/c/alarm.c +++ b/userland/linux_guests/c/alarm.c @@ -2,8 +2,8 @@ * Signals that arrive while a thread waits, and the calls that wait for them. * alarm(1) then sleep(10): on Linux SIGALRM ends the sleep early with its * handler run, well under two seconds in. Then setitimer and getitimer with - * pause; the signal waits and POSIX timers follow in alarm_wait.c. Every part - * prints its line. + * pause; the signal waits, POSIX timers and signalfd follow in their own + * files. Every part prints its line. */ #define _GNU_SOURCE #include @@ -11,7 +11,6 @@ #include #include #include "alarm.h" - volatile sig_atomic_t alarms; static int failed, parts; @@ -71,5 +70,6 @@ int main(void) { part(alarm(0) == 0, "alarm(0) with nothing armed answers 0", 0); signal_waits(); posix_timer(); + signal_fd(); return verdict(); } diff --git a/userland/linux_guests/c/alarm.h b/userland/linux_guests/c/alarm.h index 36e7d7ade..75fd94886 100644 --- a/userland/linux_guests/c/alarm.h +++ b/userland/linux_guests/c/alarm.h @@ -13,4 +13,5 @@ long ms_since(struct timespec *t0); void catch(int sig, void (*fn)(int)); void signal_waits(void); void posix_timer(void); +void signal_fd(void); int verdict(void); diff --git a/userland/linux_guests/c/alarm_sigfd.c b/userland/linux_guests/c/alarm_sigfd.c new file mode 100644 index 000000000..15a6f78a1 --- /dev/null +++ b/userland/linux_guests/c/alarm_sigfd.c @@ -0,0 +1,44 @@ +/* + * signalfd, checked against Linux: a blocked SIGUSR1 raised at the process + * reads back from a signalfd as a record with its number and the sender's + * pid, a non-blocking signalfd with nothing pending answers EAGAIN, poll + * reports it readable once a signal waits, and a sigqueue value arrives in + * ssi_int through a blocking read. + */ +#define _GNU_SOURCE +#include +#include +#include +#include + +#include "alarm.h" + +void signal_fd(void) { + sigset_t set; + sigemptyset(&set); + sigaddset(&set, SIGUSR1); + sigprocmask(SIG_BLOCK, &set, 0); + int fd = signalfd(-1, &set, SFD_NONBLOCK | SFD_CLOEXEC); + struct signalfd_siginfo si; + errno = 0; + ssize_t n = read(fd, &si, sizeof si); + part(fd >= 0 && n == -1 && errno == EAGAIN, "signalfd with nothing pending: EAGAIN", fd); + struct pollfd p = {fd, POLLIN, 0}; + int before = poll(&p, 1, 0); + kill(getpid(), SIGUSR1); + int after = poll(&p, 1, 0); + part(before == 0 && after == 1 && (p.revents & POLLIN), "poll: readable once SIGUSR1 waits", + after); + n = read(fd, &si, sizeof si); + part(n == sizeof si && si.ssi_signo == SIGUSR1 && si.ssi_pid == (unsigned)getpid(), + "signalfd reads SIGUSR1 with the sender's pid", si.ssi_signo); + close(fd); + sigaddset(&set, SIGUSR2); + int blocking = signalfd(-1, &set, 0); + union sigval v = {.sival_int = 4321}; + sigqueue(getpid(), SIGUSR2, v); + n = read(blocking, &si, sizeof si); + part(n == sizeof si && si.ssi_signo == SIGUSR2 && si.ssi_int == 4321, + "a blocking signalfd read takes sigqueue's value", si.ssi_int); + close(blocking); +} diff --git a/userland/linux_guests/go/exec/devices.go b/userland/linux_guests/go/exec/devices.go new file mode 100644 index 000000000..45c1bcf0e --- /dev/null +++ b/userland/linux_guests/go/exec/devices.go @@ -0,0 +1,33 @@ +/* + * The character devices, as Go's os package reaches them: /dev/null stats as + * a character device and reads end of file, /dev/zero reads zeros, and a + * write to /dev/full fails with ENOSPC. + */ +package main + +import ( + "errors" + "os" + "syscall" +) + +func deviceParts() { + st, err := os.Stat("/dev/null") + part(err == nil && st.Mode()&os.ModeCharDevice != 0, "/dev/null stats as a character device") + null, err := os.ReadFile("/dev/null") + part(err == nil && len(null) == 0, "/dev/null reads end of file") + z, err := os.Open("/dev/zero") + buf := make([]byte, 4096) + n := 0 + if err == nil { + n, err = z.Read(buf) + z.Close() + } + zeros := n == len(buf) + for _, b := range buf { + zeros = zeros && b == 0 + } + part(err == nil && zeros, "/dev/zero reads 4096 zeros") + err = os.WriteFile("/dev/full", []byte("x"), 0) + part(errors.Is(err, syscall.ENOSPC), "a write to /dev/full fails with ENOSPC") +} diff --git a/userland/linux_guests/go/exec/main.go b/userland/linux_guests/go/exec/main.go index b14b4b059..915d42c4e 100644 --- a/userland/linux_guests/go/exec/main.go +++ b/userland/linux_guests/go/exec/main.go @@ -4,9 +4,8 @@ * a pipe; Wait4 then reads each child's status. The children are the pthreads * guest, a program that ends with status 42, and a path that does not exist. * Then os/exec does the same through cmd.Output, which also needs the - * runtime's poller (eventfd, epoll) for its pipes; those parts run last. - * Stdin, and for Run the output too, is this program's own: a nil one makes - * os/exec open /dev/null, which the personality does not serve. + * runtime's poller (eventfd, epoll) for its pipes, and opens /dev/null for + * every stream left nil; those parts run last, after the devices' own checks. */ package main @@ -32,16 +31,11 @@ func part(ok bool, what string) { fmt.Printf("[GO] goexec %s: %s\n", mark, what) } -func own(c *exec.Cmd) *exec.Cmd { - c.Stdin, c.Stdout, c.Stderr = os.Stdin, os.Stdout, os.Stderr - return c -} - func main() { forkExecParts() + deviceParts() cmd := exec.Command("/bin/cthreads") - cmd.Stdin = os.Stdin out, err := cmd.Output() line := strings.TrimSpace(string(out)) code := -1 @@ -51,12 +45,12 @@ func main() { part(err == nil && code == 0 && strings.Contains(line, "cthreads PASS"), fmt.Sprintf("os/exec cthreads Output: status %d, err %v, output %q", code, err, line)) - err = own(exec.Command("/bin/leaderexit")).Run() + err = exec.Command("/bin/leaderexit").Run() var ee *exec.ExitError part(errors.As(err, &ee) && ee.ExitCode() == 42, fmt.Sprintf("os/exec leaderexit's status came back: %v", err)) - err = own(exec.Command("/bin/no-such-program")).Run() + err = exec.Command("/bin/no-such-program").Run() var pe *fs.PathError part(errors.As(err, &pe) && pe.Op == "fork/exec" && errors.Is(pe.Err, syscall.ENOENT), fmt.Sprintf("os/exec a missing program: %v", err))