Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 7 additions & 18 deletions src/memory/paging/manager/faults/demand.rs
Original file line number Diff line number Diff line change
Expand Up @@ -29,27 +29,16 @@ impl PagingManager {
virtual_addr: VirtAddr,
stats: &PagingStatistics,
) -> PagingResult<()> {
// Only user-space addresses may be demand-backed. A not-present fault
// in the kernel half is never a legitimate lazy mapping; backing it
// silently would hand a capsule kernel-range memory. Surface it as an
// unhandled fault so the fault path kills the offender (user) or traps
// the real kernel bug, instead of papering over it.
if !layout::in_user_space(virtual_addr.as_u64()) {
return Err(PagingError::UnhandledPageFault);
}

// Never demand-back the null page. A fault in the lowest page is a null
// or near-null dereference; backing it would silently satisfy the bug
// instead of trapping it. Leave the page unmapped as a guard so the
// fault path kills the offending capsule.
if virtual_addr.as_u64() < PAGE_SIZE_4K as u64 {
let pid = crate::process::current_pid().unwrap_or(0);
if super::demand_refuse::refused(virtual_addr.as_u64(), pid) {
return Err(PagingError::UnhandledPageFault);
}

// Charge the page against the faulting process's demand budget. A
// runaway capsule is refused here and killed by the fault path instead
// of exhausting physical memory.
let pid = crate::process::current_pid().unwrap_or(0);
/*
* Charge the page against the faulting process's demand budget. A
* runaway capsule is refused here and killed by the fault path instead
* of exhausting physical memory.
*/
if !super::demand_cap::charge(pid) {
return Err(PagingError::UnhandledPageFault);
}
Expand Down
51 changes: 51 additions & 0 deletions src/memory/paging/manager/faults/demand_refuse.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
// NONOS Operating System
// Copyright (C) 2026 NONOS Contributors
//
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU Affero General Public License as published by
// the Free Software Foundation, either version 3 of the License, or
// (at your option) any later version.
//
// This program is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU Affero General Public License for more details.
//
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see <https://www.gnu.org/licenses/>.

//! The pages the kernel never fills on a fault.

use crate::memory::layout;
use crate::memory::paging::constants::PAGE_SIZE_4K;

/// True when a not-present fault at `addr` in `pid` must not be filled.
pub(super) fn refused(addr: u64, pid: u32) -> bool {
/*
* Only user-space addresses may be demand-backed. A not-present fault
* in the kernel half is never a legitimate lazy mapping; backing it
* silently would hand a capsule kernel-range memory. Surface it as an
* unhandled fault so the fault path kills the offender (user) or traps
* the real kernel bug, instead of papering over it.
*/
if !layout::in_user_space(addr) {
return true;
}
/*
* Never demand-back the null page. A fault in the lowest page is a null
* or near-null dereference; backing it would silently satisfy the bug
* instead of trapping it. Leave the page unmapped as a guard so the
* fault path kills the offending capsule.
*/
if addr < PAGE_SIZE_4K as u64 {
return true;
}
/*
* A foreign guest's pages are exactly the ones its supervisor mapped for
* it. Filling any other page would hand the guest memory nobody gave it:
* a PROT_NONE reservation, a guard page, a hole. So the fault is refused,
* the fault path ends the thread, and its supervisor is told and decides
* what that means for the guest.
*/
crate::process::foreign::is_foreign(pid)
}
1 change: 1 addition & 0 deletions src/memory/paging/manager/faults/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -17,4 +17,5 @@
mod cow;
mod demand;
mod demand_cap;
mod demand_refuse;
mod handler;
6 changes: 6 additions & 0 deletions src/process/foreign/peer_guard.rs
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,12 @@ pub(super) fn in_user_half(addr: u64, len: u64) -> bool {

pub const PROT_WRITE: u64 = 1 << 0;
pub const PROT_EXEC: u64 = 1 << 1;
/*
* No access from the guest at all. The page stays present with the user bit
* clear, so every guest access faults and the frame keeps its bytes for a
* later protection that allows access, as Linux keeps them.
*/
pub(super) const PROT_NONE: u64 = 1 << 2;


/// The pid a syscall argument names. Refused rather than truncated: `as u32`
Expand Down
15 changes: 2 additions & 13 deletions src/process/foreign/peer_map.rs
Original file line number Diff line number Diff line change
Expand Up @@ -18,26 +18,15 @@

use crate::memory::addr::VirtAddr;
use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid};
use crate::memory::paging::types::PagePermissions;
use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_NOMEM};

use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_WRITE};
use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE};
use super::peer_protect::perms_of;

fn span_ok(addr: u64, len: u64) -> bool {
len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len)
}

pub(super) fn perms_of(prot: u64) -> PagePermissions {
let mut perms = PagePermissions::READ | PagePermissions::USER;
if prot & PROT_WRITE != 0 {
perms = perms | PagePermissions::WRITE;
}
if prot & PROT_EXEC != 0 {
perms = perms | PagePermissions::EXECUTE;
}
perms
}

/// `MkPeerMap`: map `[addr, addr + len)` in a guest the caller supervises.
pub fn sys_peer_map(pid: u64, addr: u64, len: u64, prot: u64) -> i64 {
let Some(caller) = crate::process::current_pid() else {
Expand Down
24 changes: 22 additions & 2 deletions src/process/foreign/peer_protect.rs
Original file line number Diff line number Diff line change
Expand Up @@ -19,10 +19,12 @@

use crate::memory::addr::VirtAddr;
use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid};
use crate::memory::paging::types::PagePermissions;
use crate::syscall::microkernel::errnos::{ERRNO_FAULT, ERRNO_INVAL};

use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE};
use super::peer_map::perms_of;
use super::peer_guard::{
in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_NONE, PROT_WRITE,
};

fn span_ok(addr: u64, len: u64) -> bool {
len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len)
Expand Down Expand Up @@ -53,3 +55,21 @@ pub fn sys_peer_protect(pid: u64, addr: u64, len: u64, prot: u64) -> i64 {
}
0
}

pub(super) fn perms_of(prot: u64) -> PagePermissions {
/*
* Not USER: present for the kernel, which copies it at fork and frees it
* at teardown, and absent for every access the guest makes.
*/
if prot & PROT_NONE != 0 {
return PagePermissions::READ;
}
let mut perms = PagePermissions::READ | PagePermissions::USER;
if prot & PROT_WRITE != 0 {
perms = perms | PagePermissions::WRITE;
}
if prot & PROT_EXEC != 0 {
perms = perms | PagePermissions::EXECUTE;
}
perms
}
6 changes: 3 additions & 3 deletions userland/capsule_linux/src/linux/call/mem/prot.rs
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@ use super::prot_span::protect_span;
pub const PROT_WRITE: u64 = 2;
pub const PROT_EXEC: u64 = 4;
/// PROT_READ, PROT_WRITE and PROT_EXEC together: any access at all.
const PROT_ANY: u64 = 7;
pub(super) const PROT_ANY: u64 = 7;

/// A request for both at once.
pub fn wx_refused(prot: u64) -> bool {
Expand Down Expand Up @@ -79,8 +79,8 @@ pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 {
return errno::fail(errno::ENOMEM);
}
}
// Every page is present now; this sets `prot` on all of them,
// including any the guest touched while the span was reserved.
// Every page is present now; this sets `prot` on all of them and
// records it, so a later fork gives the child no more than this.
if protect_span(guest, at, piece, prot) < 0 {
return errno::fail(errno::EACCES);
}
Expand Down
23 changes: 11 additions & 12 deletions userland/capsule_linux/src/linux/call/mem/prot_span.rs
Original file line number Diff line number Diff line change
Expand Up @@ -16,21 +16,20 @@

//! Reprotecting a span, a peer call at a time.

use nonos_libc::peer::{mk_peer_protect, PEER_PROT_EXEC, PEER_PROT_WRITE};
use nonos_libc::peer::mk_peer_protect;

use crate::linux::guest::{Guest, MAX_SPAN};
use crate::linux::guest::{peer_prot, Guest, MAX_SPAN};

use super::prot::{PROT_EXEC, PROT_WRITE};
use super::prot::{PROT_ANY, PROT_EXEC, PROT_WRITE};

/// Set the protection of a span already mapped in the guest.
pub fn protect_span(guest: &Guest, addr: u64, span: u64, prot: u64) -> i64 {
let mut bits = 0;
if prot & PROT_WRITE != 0 {
bits |= PEER_PROT_WRITE;
}
if prot & PROT_EXEC != 0 {
bits |= PEER_PROT_EXEC;
}
/// Set the protection of a span already mapped in the guest, and record it on
/// the spans it covers. PROT_NONE keeps the pages and their bytes, reachable
/// by nothing the guest does.
pub fn protect_span(guest: &mut Guest, addr: u64, span: u64, prot: u64) -> i64 {
let (write, exec) = (prot & PROT_WRITE != 0, prot & PROT_EXEC != 0);
let access = prot & PROT_ANY != 0;
let bits = peer_prot(write, exec, access);
guest.set_prot(addr, span, write, exec, access);
let mut done = 0;
while done < span {
let take = (span - done).min(MAX_SPAN);
Expand Down
27 changes: 10 additions & 17 deletions userland/capsule_linux/src/linux/call/spawn/fork_copy.rs
Original file line number Diff line number Diff line change
Expand Up @@ -16,27 +16,31 @@

//! Copying a parent's spans into the child it just made.

use crate::linux::guest::{Guest, Region, MAX_SPAN};
use nonos_libc::peer::{mk_peer_map, mk_peer_write, PEER_PROT_EXEC, PEER_PROT_WRITE};
use crate::linux::guest::{Guest, MAX_SPAN};
use nonos_libc::peer::{mk_peer_map, mk_peer_write};

/// Every span, mapped into the child and then filled from the parent.
pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool {
let spans = guest.regions.clone();
for span in spans {
// An unbacked reservation has no frames to copy; the child reserves it
// the same way, and its own first access faults a page in.
// An unbacked reservation has no frames to copy; the child holds the
// same reservation, and a touch there faults in the child as here.
if !span.backed {
continue;
}
/*
* A piece at a time: the kernel takes at most MAX_SPAN a call, and a
* region past it, a megabyte of static buffer for one, failed the
* whole fork.
* whole fork. Each piece gets the protection the span has now,
* PROT_NONE included: the kernel copies into a page whatever its
* protection, so the bytes still go in, and the child can do no more
* with them than the parent can.
*/
let mut done = 0;
while done < span.len {
let (at, take) = (span.at + done, (span.len - done).min(MAX_SPAN));
if mk_peer_map(child, at, take, prot_of(&span)) < 0 || !copy_one(guest, child, at, take)
if mk_peer_map(child, at, take, span.peer_prot()) < 0
|| !copy_one(guest, child, at, take)
{
return false;
}
Expand All @@ -46,17 +50,6 @@ pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool {
true
}

fn prot_of(span: &Region) -> u64 {
let mut prot = 0;
if span.write {
prot |= PEER_PROT_WRITE;
}
if span.exec {
prot |= PEER_PROT_EXEC;
}
prot
}

fn copy_one(guest: &Guest, child: u32, at: u64, len: u64) -> bool {
let Some(bytes) = guest.read(at, len as usize) else {
return false;
Expand Down
5 changes: 3 additions & 2 deletions userland/capsule_linux/src/linux/guest/mem_reserve.rs
Original file line number Diff line number Diff line change
Expand Up @@ -54,12 +54,13 @@ impl Guest {
}

/// Take `len` of address space at `addr` without backing it: a PROT_NONE
/// reservation. Bytes appear, zeroed, when the guest first touches them.
/// reservation. Nothing is mapped, and the kernel fills no page for a
/// guest on its own, so a touch before a commit faults as Linux faults.
pub fn reserve(&mut self, addr: u64, len: u64) -> i64 {
let Some((start, span)) = span_within(addr, len, USER_MAX) else {
return -1;
};
self.regions.push(Region::new(start, span, true, false, false));
self.regions.push(Region { access: false, ..Region::new(start, span, false, false, false) });
0
}
}
3 changes: 2 additions & 1 deletion userland/capsule_linux/src/linux/guest/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,7 @@ mod region;
mod region_cut;
mod region_find;
mod region_mark;
mod region_prot;
mod sigpending;
pub mod sigqueue;
pub mod sigstack;
Expand All @@ -61,6 +62,6 @@ pub use layout::{
};
pub use links::Links;
pub use mem::{page_down, page_up, span_within, MAX_SPAN, PAGE};
pub use region::Region;
pub use region::{peer_prot, Region};
pub use timer::Timer;
pub use watch::{Watch, EPOLLET, EPOLLONESHOT};
34 changes: 30 additions & 4 deletions userland/capsule_linux/src/linux/guest/region.rs
Original file line number Diff line number Diff line change
Expand Up @@ -16,18 +16,23 @@

//! One span of a guest's address space, as this capsule laid it down.

use nonos_libc::peer::{PEER_PROT_EXEC, PEER_PROT_NONE, PEER_PROT_WRITE};

#[derive(Clone, Copy)]
pub struct Region {
pub at: u64,
pub len: u64,
pub write: bool,
pub exec: bool,
/// False for PROT_NONE: the guest may not touch the span at all. A backed
/// span keeps its pages and their bytes, present to the kernel only.
pub access: bool,
/// File bytes mapped without exec, so never proved: mprotect may not
/// make them executable later.
pub unproven: bool,
/// False for a PROT_NONE reservation: address space taken, no frames yet.
/// The kernel demand-fills a page on first access, so reserving a large
/// span and committing a little costs only what is touched; fork skips it.
/// False for a PROT_NONE reservation: address space taken, no frames.
/// The kernel fills no page for a guest on its own, so a touch of one is a
/// fault; a commit maps the part asked for and records it backed.
pub backed: bool,
/// Bytes Linux would give back after MADV_DONTNEED, not zero: a file's,
/// an ELF segment's or a shared mapping's. This capsule cannot give them
Expand All @@ -37,7 +42,28 @@ pub struct Region {

impl Region {
/// A span this capsule laid down, anonymous until a mark says otherwise.
/// Reachable by the guest; a PROT_NONE span is built with `access` false.
pub fn new(at: u64, len: u64, write: bool, exec: bool, backed: bool) -> Self {
Self { at, len, write, exec, unproven: false, backed, kept: false }
Self { at, len, write, exec, access: true, unproven: false, backed, kept: false }
}

/// The protection the kernel is asked to give this span's pages.
pub fn peer_prot(&self) -> u64 {
peer_prot(self.write, self.exec, self.access)
}
}

/// Peer protection bits for an access, a write and an exec permission.
pub fn peer_prot(write: bool, exec: bool, access: bool) -> u64 {
if !access {
return PEER_PROT_NONE;
}
let mut prot = 0;
if write {
prot |= PEER_PROT_WRITE;
}
if exec {
prot |= PEER_PROT_EXEC;
}
prot
}
Loading
Loading