From 4a50be4f3d0e68218db9b4af802977a602bf535e Mon Sep 17 00:00:00 2001 From: senseix21 Date: Fri, 2 Oct 2026 00:06:49 +0600 Subject: [PATCH 1/2] memory: fill no page for a foreign guest, and let PROT_NONE mean it The kernel demand-filled any user page a thread touched first, foreign guests included. A Linux guest's PROT_NONE reservation read as zeros, and a pthread's guard page, which musl leaves as the unopened bottom of a PROT_NONE stack reservation, took a fresh page and guarded nothing. Every page a guest is meant to have is already mapped by its supervisor with MkPeerMap, so a not-present fault in a foreign guest is now refused: the fault path ends the thread and the supervisor is told, as before. A peer protection of PROT_NONE now maps the page present without the user bit, so the frame keeps its bytes for a later protection that allows access while every guest access faults. perms_of moves to peer_protect beside the bit it reads, and peer_map takes it from there. Ported from #586 onto this branch; without it this branch lands the lane with the hole still open. --- src/memory/paging/manager/faults/demand.rs | 25 +++------ .../paging/manager/faults/demand_refuse.rs | 51 +++++++++++++++++++ src/memory/paging/manager/faults/mod.rs | 1 + src/process/foreign/peer_guard.rs | 6 +++ src/process/foreign/peer_map.rs | 15 +----- src/process/foreign/peer_protect.rs | 24 ++++++++- 6 files changed, 89 insertions(+), 33 deletions(-) create mode 100644 src/memory/paging/manager/faults/demand_refuse.rs diff --git a/src/memory/paging/manager/faults/demand.rs b/src/memory/paging/manager/faults/demand.rs index 8dc741b0ff..3a6e64f77e 100644 --- a/src/memory/paging/manager/faults/demand.rs +++ b/src/memory/paging/manager/faults/demand.rs @@ -29,27 +29,16 @@ impl PagingManager { virtual_addr: VirtAddr, stats: &PagingStatistics, ) -> PagingResult<()> { - // Only user-space addresses may be demand-backed. A not-present fault - // in the kernel half is never a legitimate lazy mapping; backing it - // silently would hand a capsule kernel-range memory. Surface it as an - // unhandled fault so the fault path kills the offender (user) or traps - // the real kernel bug, instead of papering over it. - if !layout::in_user_space(virtual_addr.as_u64()) { - return Err(PagingError::UnhandledPageFault); - } - - // Never demand-back the null page. A fault in the lowest page is a null - // or near-null dereference; backing it would silently satisfy the bug - // instead of trapping it. Leave the page unmapped as a guard so the - // fault path kills the offending capsule. - if virtual_addr.as_u64() < PAGE_SIZE_4K as u64 { + let pid = crate::process::current_pid().unwrap_or(0); + if super::demand_refuse::refused(virtual_addr.as_u64(), pid) { return Err(PagingError::UnhandledPageFault); } - // Charge the page against the faulting process's demand budget. A - // runaway capsule is refused here and killed by the fault path instead - // of exhausting physical memory. - let pid = crate::process::current_pid().unwrap_or(0); + /* + * Charge the page against the faulting process's demand budget. A + * runaway capsule is refused here and killed by the fault path instead + * of exhausting physical memory. + */ if !super::demand_cap::charge(pid) { return Err(PagingError::UnhandledPageFault); } diff --git a/src/memory/paging/manager/faults/demand_refuse.rs b/src/memory/paging/manager/faults/demand_refuse.rs new file mode 100644 index 0000000000..0c92672cce --- /dev/null +++ b/src/memory/paging/manager/faults/demand_refuse.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The pages the kernel never fills on a fault. + +use crate::memory::layout; +use crate::memory::paging::constants::PAGE_SIZE_4K; + +/// True when a not-present fault at `addr` in `pid` must not be filled. +pub(super) fn refused(addr: u64, pid: u32) -> bool { + /* + * Only user-space addresses may be demand-backed. A not-present fault + * in the kernel half is never a legitimate lazy mapping; backing it + * silently would hand a capsule kernel-range memory. Surface it as an + * unhandled fault so the fault path kills the offender (user) or traps + * the real kernel bug, instead of papering over it. + */ + if !layout::in_user_space(addr) { + return true; + } + /* + * Never demand-back the null page. A fault in the lowest page is a null + * or near-null dereference; backing it would silently satisfy the bug + * instead of trapping it. Leave the page unmapped as a guard so the + * fault path kills the offending capsule. + */ + if addr < PAGE_SIZE_4K as u64 { + return true; + } + /* + * A foreign guest's pages are exactly the ones its supervisor mapped for + * it. Filling any other page would hand the guest memory nobody gave it: + * a PROT_NONE reservation, a guard page, a hole. So the fault is refused, + * the fault path ends the thread, and its supervisor is told and decides + * what that means for the guest. + */ + crate::process::foreign::is_foreign(pid) +} diff --git a/src/memory/paging/manager/faults/mod.rs b/src/memory/paging/manager/faults/mod.rs index de2ee927fc..87fa92be28 100644 --- a/src/memory/paging/manager/faults/mod.rs +++ b/src/memory/paging/manager/faults/mod.rs @@ -17,4 +17,5 @@ mod cow; mod demand; mod demand_cap; +mod demand_refuse; mod handler; diff --git a/src/process/foreign/peer_guard.rs b/src/process/foreign/peer_guard.rs index 4fc9aa275a..019bf46472 100644 --- a/src/process/foreign/peer_guard.rs +++ b/src/process/foreign/peer_guard.rs @@ -38,6 +38,12 @@ pub(super) fn in_user_half(addr: u64, len: u64) -> bool { pub const PROT_WRITE: u64 = 1 << 0; pub const PROT_EXEC: u64 = 1 << 1; +/* + * No access from the guest at all. The page stays present with the user bit + * clear, so every guest access faults and the frame keeps its bytes for a + * later protection that allows access, as Linux keeps them. + */ +pub(super) const PROT_NONE: u64 = 1 << 2; /// The pid a syscall argument names. Refused rather than truncated: `as u32` diff --git a/src/process/foreign/peer_map.rs b/src/process/foreign/peer_map.rs index 7fc09c5a08..3ff4fac039 100644 --- a/src/process/foreign/peer_map.rs +++ b/src/process/foreign/peer_map.rs @@ -18,26 +18,15 @@ use crate::memory::addr::VirtAddr; use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid}; -use crate::memory::paging::types::PagePermissions; use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_NOMEM}; -use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_WRITE}; +use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE}; +use super::peer_protect::perms_of; fn span_ok(addr: u64, len: u64) -> bool { len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len) } -pub(super) fn perms_of(prot: u64) -> PagePermissions { - let mut perms = PagePermissions::READ | PagePermissions::USER; - if prot & PROT_WRITE != 0 { - perms = perms | PagePermissions::WRITE; - } - if prot & PROT_EXEC != 0 { - perms = perms | PagePermissions::EXECUTE; - } - perms -} - /// `MkPeerMap`: map `[addr, addr + len)` in a guest the caller supervises. pub fn sys_peer_map(pid: u64, addr: u64, len: u64, prot: u64) -> i64 { let Some(caller) = crate::process::current_pid() else { diff --git a/src/process/foreign/peer_protect.rs b/src/process/foreign/peer_protect.rs index 79d74e1ce5..cfde6b0b75 100644 --- a/src/process/foreign/peer_protect.rs +++ b/src/process/foreign/peer_protect.rs @@ -19,10 +19,12 @@ use crate::memory::addr::VirtAddr; use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid}; +use crate::memory::paging::types::PagePermissions; use crate::syscall::microkernel::errnos::{ERRNO_FAULT, ERRNO_INVAL}; -use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE}; -use super::peer_map::perms_of; +use super::peer_guard::{ + in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_NONE, PROT_WRITE, +}; fn span_ok(addr: u64, len: u64) -> bool { len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len) @@ -53,3 +55,21 @@ pub fn sys_peer_protect(pid: u64, addr: u64, len: u64, prot: u64) -> i64 { } 0 } + +pub(super) fn perms_of(prot: u64) -> PagePermissions { + /* + * Not USER: present for the kernel, which copies it at fork and frees it + * at teardown, and absent for every access the guest makes. + */ + if prot & PROT_NONE != 0 { + return PagePermissions::READ; + } + let mut perms = PagePermissions::READ | PagePermissions::USER; + if prot & PROT_WRITE != 0 { + perms = perms | PagePermissions::WRITE; + } + if prot & PROT_EXEC != 0 { + perms = perms | PagePermissions::EXECUTE; + } + perms +} From f5d969af8334dad38c460aebd1ec3944d56c5f98 Mon Sep 17 00:00:00 2001 From: senseix21 Date: Fri, 2 Oct 2026 06:01:35 +0600 Subject: [PATCH 2/2] linux: a span the guest closed stays closed, through mprotect and fork PROT_NONE had no way to reach the kernel. protect_span sent only write and exec bits, so mprotect(PROT_NONE) on a backed span asked for prot 0 and the kernel mapped it READ|USER: the guest kept reading what it had just given up. A PROT_NONE mmap was worse, reserving the span and leaving the kernel to demand-fill it on first touch, which is what let a musl guard page take a real frame. A region now records access alongside write and exec, and peer_prot turns the three into the bits a peer call carries, PEER_PROT_NONE included. protect_span sends that and records it with set_prot, so the region list fork maps a child from no longer holds the old protection. fork_copy maps each piece with the protection its span has now, so a span the parent closed comes back closed in the child. A reservation keeps nothing mapped, as before, but is marked without access: a touch before a commit faults, which is what Linux does. Ported from #586 onto this branch, alongside the kernel half. --- .../capsule_linux/src/linux/call/mem/prot.rs | 6 +-- .../src/linux/call/mem/prot_span.rs | 23 +++++----- .../src/linux/call/spawn/fork_copy.rs | 27 ++++------- .../src/linux/guest/mem_reserve.rs | 5 +- userland/capsule_linux/src/linux/guest/mod.rs | 3 +- .../capsule_linux/src/linux/guest/region.rs | 34 ++++++++++++-- .../src/linux/guest/region_prot.rs | 46 +++++++++++++++++++ userland/libc/src/peer.rs | 3 ++ 8 files changed, 108 insertions(+), 39 deletions(-) create mode 100644 userland/capsule_linux/src/linux/guest/region_prot.rs diff --git a/userland/capsule_linux/src/linux/call/mem/prot.rs b/userland/capsule_linux/src/linux/call/mem/prot.rs index 9d060a23ef..187c01c8ac 100644 --- a/userland/capsule_linux/src/linux/call/mem/prot.rs +++ b/userland/capsule_linux/src/linux/call/mem/prot.rs @@ -24,7 +24,7 @@ use super::prot_span::protect_span; pub const PROT_WRITE: u64 = 2; pub const PROT_EXEC: u64 = 4; /// PROT_READ, PROT_WRITE and PROT_EXEC together: any access at all. -const PROT_ANY: u64 = 7; +pub(super) const PROT_ANY: u64 = 7; /// A request for both at once. pub fn wx_refused(prot: u64) -> bool { @@ -79,8 +79,8 @@ pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 { return errno::fail(errno::ENOMEM); } } - // Every page is present now; this sets `prot` on all of them, - // including any the guest touched while the span was reserved. + // Every page is present now; this sets `prot` on all of them and + // records it, so a later fork gives the child no more than this. if protect_span(guest, at, piece, prot) < 0 { return errno::fail(errno::EACCES); } diff --git a/userland/capsule_linux/src/linux/call/mem/prot_span.rs b/userland/capsule_linux/src/linux/call/mem/prot_span.rs index aac1b7043a..b0f74b86b7 100644 --- a/userland/capsule_linux/src/linux/call/mem/prot_span.rs +++ b/userland/capsule_linux/src/linux/call/mem/prot_span.rs @@ -16,21 +16,20 @@ //! Reprotecting a span, a peer call at a time. -use nonos_libc::peer::{mk_peer_protect, PEER_PROT_EXEC, PEER_PROT_WRITE}; +use nonos_libc::peer::mk_peer_protect; -use crate::linux::guest::{Guest, MAX_SPAN}; +use crate::linux::guest::{peer_prot, Guest, MAX_SPAN}; -use super::prot::{PROT_EXEC, PROT_WRITE}; +use super::prot::{PROT_ANY, PROT_EXEC, PROT_WRITE}; -/// Set the protection of a span already mapped in the guest. -pub fn protect_span(guest: &Guest, addr: u64, span: u64, prot: u64) -> i64 { - let mut bits = 0; - if prot & PROT_WRITE != 0 { - bits |= PEER_PROT_WRITE; - } - if prot & PROT_EXEC != 0 { - bits |= PEER_PROT_EXEC; - } +/// Set the protection of a span already mapped in the guest, and record it on +/// the spans it covers. PROT_NONE keeps the pages and their bytes, reachable +/// by nothing the guest does. +pub fn protect_span(guest: &mut Guest, addr: u64, span: u64, prot: u64) -> i64 { + let (write, exec) = (prot & PROT_WRITE != 0, prot & PROT_EXEC != 0); + let access = prot & PROT_ANY != 0; + let bits = peer_prot(write, exec, access); + guest.set_prot(addr, span, write, exec, access); let mut done = 0; while done < span { let take = (span - done).min(MAX_SPAN); diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs index 2150dd41cd..a4d2ea2cc6 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs @@ -16,27 +16,31 @@ //! Copying a parent's spans into the child it just made. -use crate::linux::guest::{Guest, Region, MAX_SPAN}; -use nonos_libc::peer::{mk_peer_map, mk_peer_write, PEER_PROT_EXEC, PEER_PROT_WRITE}; +use crate::linux::guest::{Guest, MAX_SPAN}; +use nonos_libc::peer::{mk_peer_map, mk_peer_write}; /// Every span, mapped into the child and then filled from the parent. pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool { let spans = guest.regions.clone(); for span in spans { - // An unbacked reservation has no frames to copy; the child reserves it - // the same way, and its own first access faults a page in. + // An unbacked reservation has no frames to copy; the child holds the + // same reservation, and a touch there faults in the child as here. if !span.backed { continue; } /* * A piece at a time: the kernel takes at most MAX_SPAN a call, and a * region past it, a megabyte of static buffer for one, failed the - * whole fork. + * whole fork. Each piece gets the protection the span has now, + * PROT_NONE included: the kernel copies into a page whatever its + * protection, so the bytes still go in, and the child can do no more + * with them than the parent can. */ let mut done = 0; while done < span.len { let (at, take) = (span.at + done, (span.len - done).min(MAX_SPAN)); - if mk_peer_map(child, at, take, prot_of(&span)) < 0 || !copy_one(guest, child, at, take) + if mk_peer_map(child, at, take, span.peer_prot()) < 0 + || !copy_one(guest, child, at, take) { return false; } @@ -46,17 +50,6 @@ pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool { true } -fn prot_of(span: &Region) -> u64 { - let mut prot = 0; - if span.write { - prot |= PEER_PROT_WRITE; - } - if span.exec { - prot |= PEER_PROT_EXEC; - } - prot -} - fn copy_one(guest: &Guest, child: u32, at: u64, len: u64) -> bool { let Some(bytes) = guest.read(at, len as usize) else { return false; diff --git a/userland/capsule_linux/src/linux/guest/mem_reserve.rs b/userland/capsule_linux/src/linux/guest/mem_reserve.rs index d1caef268d..390d8611bf 100644 --- a/userland/capsule_linux/src/linux/guest/mem_reserve.rs +++ b/userland/capsule_linux/src/linux/guest/mem_reserve.rs @@ -54,12 +54,13 @@ impl Guest { } /// Take `len` of address space at `addr` without backing it: a PROT_NONE - /// reservation. Bytes appear, zeroed, when the guest first touches them. + /// reservation. Nothing is mapped, and the kernel fills no page for a + /// guest on its own, so a touch before a commit faults as Linux faults. pub fn reserve(&mut self, addr: u64, len: u64) -> i64 { let Some((start, span)) = span_within(addr, len, USER_MAX) else { return -1; }; - self.regions.push(Region::new(start, span, true, false, false)); + self.regions.push(Region { access: false, ..Region::new(start, span, false, false, false) }); 0 } } diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index 8b99943fc5..c15a28abed 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -41,6 +41,7 @@ mod region; mod region_cut; mod region_find; mod region_mark; +mod region_prot; mod sigpending; pub mod sigqueue; pub mod sigstack; @@ -61,6 +62,6 @@ pub use layout::{ }; pub use links::Links; pub use mem::{page_down, page_up, span_within, MAX_SPAN, PAGE}; -pub use region::Region; +pub use region::{peer_prot, Region}; pub use timer::Timer; pub use watch::{Watch, EPOLLET, EPOLLONESHOT}; diff --git a/userland/capsule_linux/src/linux/guest/region.rs b/userland/capsule_linux/src/linux/guest/region.rs index 23299d9a0f..b90a2aa01d 100644 --- a/userland/capsule_linux/src/linux/guest/region.rs +++ b/userland/capsule_linux/src/linux/guest/region.rs @@ -16,18 +16,23 @@ //! One span of a guest's address space, as this capsule laid it down. +use nonos_libc::peer::{PEER_PROT_EXEC, PEER_PROT_NONE, PEER_PROT_WRITE}; + #[derive(Clone, Copy)] pub struct Region { pub at: u64, pub len: u64, pub write: bool, pub exec: bool, + /// False for PROT_NONE: the guest may not touch the span at all. A backed + /// span keeps its pages and their bytes, present to the kernel only. + pub access: bool, /// File bytes mapped without exec, so never proved: mprotect may not /// make them executable later. pub unproven: bool, - /// False for a PROT_NONE reservation: address space taken, no frames yet. - /// The kernel demand-fills a page on first access, so reserving a large - /// span and committing a little costs only what is touched; fork skips it. + /// False for a PROT_NONE reservation: address space taken, no frames. + /// The kernel fills no page for a guest on its own, so a touch of one is a + /// fault; a commit maps the part asked for and records it backed. pub backed: bool, /// Bytes Linux would give back after MADV_DONTNEED, not zero: a file's, /// an ELF segment's or a shared mapping's. This capsule cannot give them @@ -37,7 +42,28 @@ pub struct Region { impl Region { /// A span this capsule laid down, anonymous until a mark says otherwise. + /// Reachable by the guest; a PROT_NONE span is built with `access` false. pub fn new(at: u64, len: u64, write: bool, exec: bool, backed: bool) -> Self { - Self { at, len, write, exec, unproven: false, backed, kept: false } + Self { at, len, write, exec, access: true, unproven: false, backed, kept: false } + } + + /// The protection the kernel is asked to give this span's pages. + pub fn peer_prot(&self) -> u64 { + peer_prot(self.write, self.exec, self.access) + } +} + +/// Peer protection bits for an access, a write and an exec permission. +pub fn peer_prot(write: bool, exec: bool, access: bool) -> u64 { + if !access { + return PEER_PROT_NONE; + } + let mut prot = 0; + if write { + prot |= PEER_PROT_WRITE; + } + if exec { + prot |= PEER_PROT_EXEC; } + prot } diff --git a/userland/capsule_linux/src/linux/guest/region_prot.rs b/userland/capsule_linux/src/linux/guest/region_prot.rs new file mode 100644 index 0000000000..705ddf967b --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/region_prot.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Recording a protection change on the spans a guest holds. + +use alloc::vec::Vec; + +use super::handle::Guest; +use super::region::Region; +use super::region_cut::cut; + +impl Guest { + /// Every backed part of `[at, at + len)` now has this protection. The + /// list is what fork maps the child from, so a list that kept the old + /// protection would give the child access the parent gave up. + pub fn set_prot(&mut self, at: u64, len: u64, write: bool, exec: bool, access: bool) { + let end = at.saturating_add(len); + let changed: Vec = self + .regions + .iter() + .filter(|r| r.backed && r.at < end && at < r.at.saturating_add(r.len)) + .map(|r| { + let from = r.at.max(at); + let to = r.at.saturating_add(r.len).min(end); + Region { at: from, len: to - from, write, exec, access, ..*r } + }) + .collect(); + for piece in &changed { + self.regions = cut(&self.regions, piece.at, piece.len); + } + self.regions.extend(changed); + } +} diff --git a/userland/libc/src/peer.rs b/userland/libc/src/peer.rs index 988f4cff93..29bb9ed16d 100644 --- a/userland/libc/src/peer.rs +++ b/userland/libc/src/peer.rs @@ -25,6 +25,9 @@ use crate::syscall::{ /// Pages of a guest may be written, and may be executed. pub const PEER_PROT_WRITE: u64 = 1 << 0; pub const PEER_PROT_EXEC: u64 = 1 << 1; +/// Pages the guest may not touch at all, their bytes kept for a later +/// protection that opens them: PROT_NONE. +pub const PEER_PROT_NONE: u64 = 1 << 2; /// Back a span of a guest's address space with fresh zeroed frames. pub fn mk_peer_map(pid: u32, addr: u64, len: u64, prot: u64) -> i64 {