Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 7 additions & 18 deletions src/memory/paging/manager/faults/demand.rs
Original file line number Diff line number Diff line change
Expand Up @@ -29,27 +29,16 @@ impl PagingManager {
virtual_addr: VirtAddr,
stats: &PagingStatistics,
) -> PagingResult<()> {
// Only user-space addresses may be demand-backed. A not-present fault
// in the kernel half is never a legitimate lazy mapping; backing it
// silently would hand a capsule kernel-range memory. Surface it as an
// unhandled fault so the fault path kills the offender (user) or traps
// the real kernel bug, instead of papering over it.
if !layout::in_user_space(virtual_addr.as_u64()) {
return Err(PagingError::UnhandledPageFault);
}

// Never demand-back the null page. A fault in the lowest page is a null
// or near-null dereference; backing it would silently satisfy the bug
// instead of trapping it. Leave the page unmapped as a guard so the
// fault path kills the offending capsule.
if virtual_addr.as_u64() < PAGE_SIZE_4K as u64 {
let pid = crate::process::current_pid().unwrap_or(0);
if super::demand_refuse::refused(virtual_addr.as_u64(), pid) {
return Err(PagingError::UnhandledPageFault);
}

// Charge the page against the faulting process's demand budget. A
// runaway capsule is refused here and killed by the fault path instead
// of exhausting physical memory.
let pid = crate::process::current_pid().unwrap_or(0);
/*
* Charge the page against the faulting process's demand budget. A
* runaway capsule is refused here and killed by the fault path instead
* of exhausting physical memory.
*/
if !super::demand_cap::charge(pid) {
return Err(PagingError::UnhandledPageFault);
}
Expand Down
51 changes: 51 additions & 0 deletions src/memory/paging/manager/faults/demand_refuse.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
// NONOS Operating System
// Copyright (C) 2026 NONOS Contributors
//
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU Affero General Public License as published by
// the Free Software Foundation, either version 3 of the License, or
// (at your option) any later version.
//
// This program is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU Affero General Public License for more details.
//
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see <https://www.gnu.org/licenses/>.

//! The pages the kernel never fills on a fault.

use crate::memory::layout;
use crate::memory::paging::constants::PAGE_SIZE_4K;

/// True when a not-present fault at `addr` in `pid` must not be filled.
pub(super) fn refused(addr: u64, pid: u32) -> bool {
/*
* Only user-space addresses may be demand-backed. A not-present fault
* in the kernel half is never a legitimate lazy mapping; backing it
* silently would hand a capsule kernel-range memory. Surface it as an
* unhandled fault so the fault path kills the offender (user) or traps
* the real kernel bug, instead of papering over it.
*/
if !layout::in_user_space(addr) {
return true;
}
/*
* Never demand-back the null page. A fault in the lowest page is a null
* or near-null dereference; backing it would silently satisfy the bug
* instead of trapping it. Leave the page unmapped as a guard so the
* fault path kills the offending capsule.
*/
if addr < PAGE_SIZE_4K as u64 {
return true;
}
/*
* A foreign guest's pages are exactly the ones its supervisor mapped for
* it. Filling any other page would hand the guest memory nobody gave it:
* a PROT_NONE reservation, a guard page, a hole. So the fault is refused,
* the fault path ends the thread, and its supervisor is told and decides
* what that means for the guest.
*/
crate::process::foreign::is_foreign(pid)
}
1 change: 1 addition & 0 deletions src/memory/paging/manager/faults/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -17,4 +17,5 @@
mod cow;
mod demand;
mod demand_cap;
mod demand_refuse;
mod handler;
14 changes: 11 additions & 3 deletions src/process/foreign/peer_guard.rs
Original file line number Diff line number Diff line change
Expand Up @@ -21,11 +21,13 @@ use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_PERM};

pub(super) const PAGE: u64 = 4096;

// One call maps or copies at most this much, so a guest image crosses in
// bounded pieces and no single call holds the processor.
/*
* One call maps or copies at most this much, so a guest image crosses in
* bounded pieces and no single call holds the processor.
*/
pub(super) const MAX_SPAN: u64 = 1 << 20;

// The first address of the kernel half.
/* The first address of the kernel half. */
pub(super) const USER_VA_END: u64 = 0x0000_8000_0000_0000;

/// True when `[addr, addr + len)` lies wholly in the guest's own half.
Expand All @@ -38,6 +40,12 @@ pub(super) fn in_user_half(addr: u64, len: u64) -> bool {

pub const PROT_WRITE: u64 = 1 << 0;
pub const PROT_EXEC: u64 = 1 << 1;
/*
* No access from the guest at all. The page stays present with the user bit
* clear, so every guest access faults and the frame keeps its bytes for a
* later protection that allows access, as Linux keeps them.
*/
pub(super) const PROT_NONE: u64 = 1 << 2;


/// The pid a syscall argument names. Refused rather than truncated: `as u32`
Expand Down
15 changes: 2 additions & 13 deletions src/process/foreign/peer_map.rs
Original file line number Diff line number Diff line change
Expand Up @@ -18,26 +18,15 @@

use crate::memory::addr::VirtAddr;
use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid};
use crate::memory::paging::types::PagePermissions;
use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_NOMEM};

use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_WRITE};
use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE};
use super::peer_protect::perms_of;

fn span_ok(addr: u64, len: u64) -> bool {
len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len)
}

pub(super) fn perms_of(prot: u64) -> PagePermissions {
let mut perms = PagePermissions::READ | PagePermissions::USER;
if prot & PROT_WRITE != 0 {
perms = perms | PagePermissions::WRITE;
}
if prot & PROT_EXEC != 0 {
perms = perms | PagePermissions::EXECUTE;
}
perms
}

/// `MkPeerMap`: map `[addr, addr + len)` in a guest the caller supervises.
pub fn sys_peer_map(pid: u64, addr: u64, len: u64, prot: u64) -> i64 {
let Some(caller) = crate::process::current_pid() else {
Expand Down
24 changes: 22 additions & 2 deletions src/process/foreign/peer_protect.rs
Original file line number Diff line number Diff line change
Expand Up @@ -19,10 +19,12 @@

use crate::memory::addr::VirtAddr;
use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid};
use crate::memory::paging::types::PagePermissions;
use crate::syscall::microkernel::errnos::{ERRNO_FAULT, ERRNO_INVAL};

use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE};
use super::peer_map::perms_of;
use super::peer_guard::{
in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_NONE, PROT_WRITE,
};

fn span_ok(addr: u64, len: u64) -> bool {
len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len)
Expand Down Expand Up @@ -53,3 +55,21 @@ pub fn sys_peer_protect(pid: u64, addr: u64, len: u64, prot: u64) -> i64 {
}
0
}

pub(super) fn perms_of(prot: u64) -> PagePermissions {
/*
* Not USER: present for the kernel, which copies it at fork and frees it
* at teardown, and absent for every access the guest makes.
*/
if prot & PROT_NONE != 0 {
return PagePermissions::READ;
}
let mut perms = PagePermissions::READ | PagePermissions::USER;
if prot & PROT_WRITE != 0 {
perms = perms | PagePermissions::WRITE;
}
if prot & PROT_EXEC != 0 {
perms = perms | PagePermissions::EXECUTE;
}
perms
}
1 change: 1 addition & 0 deletions userland/capsule_linux/src/linux/abi/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -23,4 +23,5 @@ pub mod name;
pub mod nr;
pub mod nr_path;
pub mod nr_high;
pub mod nr_mem;
pub mod nr_sched;
1 change: 1 addition & 0 deletions userland/capsule_linux/src/linux/abi/nr.rs
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@
//! Linux x86_64 syscall numbers, by family.

pub use super::nr_high::*;
pub use super::nr_mem::*;
pub use super::nr_sched::*;

pub const READ: u64 = 0;
Expand Down
24 changes: 24 additions & 0 deletions userland/capsule_linux/src/linux/abi/nr_mem.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
// NONOS Operating System
// Copyright (C) 2026 NONOS Contributors
//
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU Affero General Public License as published by
// the Free Software Foundation, either version 3 of the License, or
// (at your option) any later version.
//
// This program is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU Affero General Public License for more details.
//
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see <https://www.gnu.org/licenses/>.
//! Memory calls the Linux x86_64 table has that the first families lacked.

pub const MSYNC: u64 = 26;
pub const MINCORE: u64 = 27;
pub const MLOCK: u64 = 149;
pub const MUNLOCK: u64 = 150;
pub const MLOCKALL: u64 = 151;
pub const MUNLOCKALL: u64 = 152;
pub const MLOCK2: u64 = 325;
64 changes: 64 additions & 0 deletions userland/capsule_linux/src/linux/call/mem/lock.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,64 @@
// NONOS Operating System
// Copyright (C) 2026 NONOS Contributors
//
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU Affero General Public License as published by
// the Free Software Foundation, either version 3 of the License, or
// (at your option) any later version.
//
// This program is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU Affero General Public License for more details.
//
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see <https://www.gnu.org/licenses/>.
//! `mlock`, `mlock2`, `munlock`, `mlockall` and `munlockall`.
//!
//! Locking keeps pages resident. Every page a guest holds here is resident
//! from the moment it is mapped until it is unmapped, and none is ever paged
//! out, so every page is already as locked as Linux can make it. What is left
//! of each call is Linux's checking of its arguments, answered the same way:
//! the span must be held, and the flags known. RLIMIT_MEMLOCK is reported
//! unlimited, so no lock is refused for its size.

use crate::linux::abi::errno;
use crate::linux::guest::{page_down, page_up, Guest, PAGE};

const MLOCK_ONFAULT: u64 = 1;
const MCL_CURRENT: u64 = 1;
const MCL_FUTURE: u64 = 2;
const MCL_ONFAULT: u64 = 4;

/// `mlock` and `munlock` alike: the address is rounded down, the length up,
/// and a span with a page the guest does not hold is ENOMEM.
pub fn mlock(guest: &Guest, addr: u64, len: u64) -> u64 {
let start = page_down(addr);
let Some(end) = addr.checked_add(len).filter(|e| *e <= u64::MAX - PAGE) else {
return errno::fail(errno::EINVAL);
};
let end = page_up(end);
if end == start {
return errno::ok(0);
}
if guest.mapped_from(start) < end - start {
return errno::fail(errno::ENOMEM);
}
errno::ok(0)
}

pub fn mlock2(guest: &Guest, addr: u64, len: u64, flags: u64) -> u64 {
if flags & !MLOCK_ONFAULT != 0 {
return errno::fail(errno::EINVAL);
}
mlock(guest, addr, len)
}

/// Linux refuses no flags, unknown flags, and MCL_ONFAULT on its own.
pub fn mlockall(flags: u64) -> u64 {
let known = MCL_CURRENT | MCL_FUTURE | MCL_ONFAULT;
if flags == 0 || flags & !known != 0 || flags == MCL_ONFAULT {
return errno::fail(errno::EINVAL);
}
errno::ok(0)
}
53 changes: 23 additions & 30 deletions userland/capsule_linux/src/linux/call/mem/map.rs
Original file line number Diff line number Diff line change
Expand Up @@ -17,50 +17,43 @@
//! `mmap`: anonymous pages, or a private mapping of a file.

use crate::linux::abi::errno;
use crate::linux::guest::{span_within, Guest, MMAP_LIMIT, USER_MAX};
use crate::linux::guest::{Guest, PAGE};

use super::map_anon::{anonymous, memfd};
use super::map_file::file;
use super::map_place::place;
use super::map_req::MapReq;
use super::prot::wx_refused;

const MAP_SHARED: u64 = 0x01;
const MAP_ANONYMOUS: u64 = 0x20;
const MAP_FIXED: u64 = 0x10;
const MAP_FIXED_NOREPLACE: u64 = 0x10_0000;

pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 {
if req.len == 0 {
/*
* Linux takes a file offset on a page boundary, and an exact address too;
* only a hint is rounded.
*/
let exact = req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0;
if req.len == 0 || req.off % PAGE != 0 || (exact && req.addr % PAGE != 0) {
return errno::fail(errno::EINVAL);
}
if wx_refused(req.prot) {
return errno::fail(errno::EPERM);
}
// MAP_FIXED is the exact address or failure. Page zero is never in the
// plan, and landing elsewhere would hand back memory the guest did not
// ask for, so it is refused, as Linux refuses it below mmap_min_addr.
if req.flags & MAP_FIXED != 0 && req.addr == 0 {
/*
* MAP_FIXED and MAP_FIXED_NOREPLACE are the exact address or failure.
* Page zero is never in the plan, and landing elsewhere would hand back
* memory the guest did not ask for, so it is refused, as Linux refuses it
* below mmap_min_addr.
*/
if exact && req.addr == 0 {
return errno::fail(errno::EPERM);
}
// The ceiling differs by who chose the address.
let (at, limit) = match req.fixed() {
Some(addr) => (addr, USER_MAX),
None => (guest.mmap_next, MMAP_LIMIT),
let spot = match place(guest, &req) {
Ok(spot) => spot,
Err(e) => return errno::fail(e),
};
let Some((at, span)) = span_within(at, req.len, limit) else {
return errno::fail(errno::ENOMEM);
};
if req.flags & MAP_ANONYMOUS != 0 {
return anonymous(guest, &req, at, span);
}
if crate::linux::file::is_memfd(guest, req.fd) {
return memfd(guest, &req, at, span);
}
if req.flags & MAP_SHARED != 0 {
/*
* Sharing a file between processes needs frames that two address
* spaces both point at, which no peer call offers.
*/
return errno::fail(errno::ENOSYS);
let out = super::map_kind::map_at(guest, &req, spot.at, spot.span);
if spot.from_cursor && (out as i64) >= 0 {
guest.mmap_next = spot.at + spot.span;
}
file(guest, &req, at, span)
out
}
16 changes: 9 additions & 7 deletions userland/capsule_linux/src/linux/call/mem/map_anon.rs
Original file line number Diff line number Diff line change
Expand Up @@ -44,10 +44,15 @@ pub fn memfd(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
}

pub fn anonymous(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
// A PROT_NONE anonymous mapping is a reservation: the runtime that makes it
// (Go's, for one) commits a fraction of it later with a fixed RW mapping.
// Backing the whole span here would spend real frames on address space no
// one has touched, so reserve it and let the first access fault a page in.
if !req.make_room(guest, at, span) {
return errno::fail(errno::ENOMEM);
}
/*
* A PROT_NONE anonymous mapping is a reservation: the runtime that makes it
* (Go's, for one) commits a fraction of it later with a fixed RW mapping.
* Backing the whole span here would spend real frames on address space no
* one may touch, so reserve it; a commit maps the part that is opened.
*/
let backed = if req.prot == 0 {
guest.reserve(at, span)
} else {
Expand All @@ -56,8 +61,5 @@ pub fn anonymous(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
if backed < 0 {
return errno::fail(errno::ENOMEM);
}
if req.fixed().is_none() {
guest.mmap_next += span;
}
errno::ok(at)
}
Loading
Loading