diff --git a/src/memory/paging/manager/faults/demand.rs b/src/memory/paging/manager/faults/demand.rs
index 8dc741b0f..3a6e64f77 100644
--- a/src/memory/paging/manager/faults/demand.rs
+++ b/src/memory/paging/manager/faults/demand.rs
@@ -29,27 +29,16 @@ impl PagingManager {
virtual_addr: VirtAddr,
stats: &PagingStatistics,
) -> PagingResult<()> {
- // Only user-space addresses may be demand-backed. A not-present fault
- // in the kernel half is never a legitimate lazy mapping; backing it
- // silently would hand a capsule kernel-range memory. Surface it as an
- // unhandled fault so the fault path kills the offender (user) or traps
- // the real kernel bug, instead of papering over it.
- if !layout::in_user_space(virtual_addr.as_u64()) {
- return Err(PagingError::UnhandledPageFault);
- }
-
- // Never demand-back the null page. A fault in the lowest page is a null
- // or near-null dereference; backing it would silently satisfy the bug
- // instead of trapping it. Leave the page unmapped as a guard so the
- // fault path kills the offending capsule.
- if virtual_addr.as_u64() < PAGE_SIZE_4K as u64 {
+ let pid = crate::process::current_pid().unwrap_or(0);
+ if super::demand_refuse::refused(virtual_addr.as_u64(), pid) {
return Err(PagingError::UnhandledPageFault);
}
- // Charge the page against the faulting process's demand budget. A
- // runaway capsule is refused here and killed by the fault path instead
- // of exhausting physical memory.
- let pid = crate::process::current_pid().unwrap_or(0);
+ /*
+ * Charge the page against the faulting process's demand budget. A
+ * runaway capsule is refused here and killed by the fault path instead
+ * of exhausting physical memory.
+ */
if !super::demand_cap::charge(pid) {
return Err(PagingError::UnhandledPageFault);
}
diff --git a/src/memory/paging/manager/faults/demand_refuse.rs b/src/memory/paging/manager/faults/demand_refuse.rs
new file mode 100644
index 000000000..0c92672cc
--- /dev/null
+++ b/src/memory/paging/manager/faults/demand_refuse.rs
@@ -0,0 +1,51 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The pages the kernel never fills on a fault.
+
+use crate::memory::layout;
+use crate::memory::paging::constants::PAGE_SIZE_4K;
+
+/// True when a not-present fault at `addr` in `pid` must not be filled.
+pub(super) fn refused(addr: u64, pid: u32) -> bool {
+ /*
+ * Only user-space addresses may be demand-backed. A not-present fault
+ * in the kernel half is never a legitimate lazy mapping; backing it
+ * silently would hand a capsule kernel-range memory. Surface it as an
+ * unhandled fault so the fault path kills the offender (user) or traps
+ * the real kernel bug, instead of papering over it.
+ */
+ if !layout::in_user_space(addr) {
+ return true;
+ }
+ /*
+ * Never demand-back the null page. A fault in the lowest page is a null
+ * or near-null dereference; backing it would silently satisfy the bug
+ * instead of trapping it. Leave the page unmapped as a guard so the
+ * fault path kills the offending capsule.
+ */
+ if addr < PAGE_SIZE_4K as u64 {
+ return true;
+ }
+ /*
+ * A foreign guest's pages are exactly the ones its supervisor mapped for
+ * it. Filling any other page would hand the guest memory nobody gave it:
+ * a PROT_NONE reservation, a guard page, a hole. So the fault is refused,
+ * the fault path ends the thread, and its supervisor is told and decides
+ * what that means for the guest.
+ */
+ crate::process::foreign::is_foreign(pid)
+}
diff --git a/src/memory/paging/manager/faults/mod.rs b/src/memory/paging/manager/faults/mod.rs
index de2ee927f..87fa92be2 100644
--- a/src/memory/paging/manager/faults/mod.rs
+++ b/src/memory/paging/manager/faults/mod.rs
@@ -17,4 +17,5 @@
mod cow;
mod demand;
mod demand_cap;
+mod demand_refuse;
mod handler;
diff --git a/src/process/foreign/peer_guard.rs b/src/process/foreign/peer_guard.rs
index 4fc9aa275..596dd00c1 100644
--- a/src/process/foreign/peer_guard.rs
+++ b/src/process/foreign/peer_guard.rs
@@ -21,11 +21,13 @@ use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_PERM};
pub(super) const PAGE: u64 = 4096;
-// One call maps or copies at most this much, so a guest image crosses in
-// bounded pieces and no single call holds the processor.
+/*
+ * One call maps or copies at most this much, so a guest image crosses in
+ * bounded pieces and no single call holds the processor.
+ */
pub(super) const MAX_SPAN: u64 = 1 << 20;
-// The first address of the kernel half.
+/* The first address of the kernel half. */
pub(super) const USER_VA_END: u64 = 0x0000_8000_0000_0000;
/// True when `[addr, addr + len)` lies wholly in the guest's own half.
@@ -38,6 +40,12 @@ pub(super) fn in_user_half(addr: u64, len: u64) -> bool {
pub const PROT_WRITE: u64 = 1 << 0;
pub const PROT_EXEC: u64 = 1 << 1;
+/*
+ * No access from the guest at all. The page stays present with the user bit
+ * clear, so every guest access faults and the frame keeps its bytes for a
+ * later protection that allows access, as Linux keeps them.
+ */
+pub(super) const PROT_NONE: u64 = 1 << 2;
/// The pid a syscall argument names. Refused rather than truncated: `as u32`
diff --git a/src/process/foreign/peer_map.rs b/src/process/foreign/peer_map.rs
index 3740c267a..2a309a4a5 100644
--- a/src/process/foreign/peer_map.rs
+++ b/src/process/foreign/peer_map.rs
@@ -18,26 +18,15 @@
use crate::memory::addr::VirtAddr;
use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid};
-use crate::memory::paging::types::PagePermissions;
use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_NOMEM};
-use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_WRITE};
+use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE};
+use super::peer_protect::perms_of;
fn span_ok(addr: u64, len: u64) -> bool {
len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len)
}
-pub(super) fn perms_of(prot: u64) -> PagePermissions {
- let mut perms = PagePermissions::READ | PagePermissions::USER;
- if prot & PROT_WRITE != 0 {
- perms = perms | PagePermissions::WRITE;
- }
- if prot & PROT_EXEC != 0 {
- perms = perms | PagePermissions::EXECUTE;
- }
- perms
-}
-
/// `MkPeerMap`: map `[addr, addr + len)` in a guest the caller supervises.
pub fn sys_peer_map(pid: u64, addr: u64, len: u64, prot: u64) -> i64 {
let Some(caller) = crate::process::current_pid() else {
diff --git a/src/process/foreign/peer_protect.rs b/src/process/foreign/peer_protect.rs
index 79d74e1ce..cfde6b0b7 100644
--- a/src/process/foreign/peer_protect.rs
+++ b/src/process/foreign/peer_protect.rs
@@ -19,10 +19,12 @@
use crate::memory::addr::VirtAddr;
use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid};
+use crate::memory::paging::types::PagePermissions;
use crate::syscall::microkernel::errnos::{ERRNO_FAULT, ERRNO_INVAL};
-use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE};
-use super::peer_map::perms_of;
+use super::peer_guard::{
+ in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_NONE, PROT_WRITE,
+};
fn span_ok(addr: u64, len: u64) -> bool {
len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len)
@@ -53,3 +55,21 @@ pub fn sys_peer_protect(pid: u64, addr: u64, len: u64, prot: u64) -> i64 {
}
0
}
+
+pub(super) fn perms_of(prot: u64) -> PagePermissions {
+ /*
+ * Not USER: present for the kernel, which copies it at fork and frees it
+ * at teardown, and absent for every access the guest makes.
+ */
+ if prot & PROT_NONE != 0 {
+ return PagePermissions::READ;
+ }
+ let mut perms = PagePermissions::READ | PagePermissions::USER;
+ if prot & PROT_WRITE != 0 {
+ perms = perms | PagePermissions::WRITE;
+ }
+ if prot & PROT_EXEC != 0 {
+ perms = perms | PagePermissions::EXECUTE;
+ }
+ perms
+}
diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs
index 68e04090a..667abbfa1 100644
--- a/userland/capsule_linux/src/linux/abi/mod.rs
+++ b/userland/capsule_linux/src/linux/abi/mod.rs
@@ -23,4 +23,5 @@ pub mod name;
pub mod nr;
pub mod nr_path;
pub mod nr_high;
+pub mod nr_mem;
pub mod nr_sched;
diff --git a/userland/capsule_linux/src/linux/abi/nr.rs b/userland/capsule_linux/src/linux/abi/nr.rs
index 1e800243b..03bd5c153 100644
--- a/userland/capsule_linux/src/linux/abi/nr.rs
+++ b/userland/capsule_linux/src/linux/abi/nr.rs
@@ -18,6 +18,7 @@
//! Linux x86_64 syscall numbers, by family.
pub use super::nr_high::*;
+pub use super::nr_mem::*;
pub use super::nr_sched::*;
pub const READ: u64 = 0;
diff --git a/userland/capsule_linux/src/linux/abi/nr_mem.rs b/userland/capsule_linux/src/linux/abi/nr_mem.rs
new file mode 100644
index 000000000..a6d3d83d7
--- /dev/null
+++ b/userland/capsule_linux/src/linux/abi/nr_mem.rs
@@ -0,0 +1,24 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+//! Memory calls the Linux x86_64 table has that the first families lacked.
+
+pub const MSYNC: u64 = 26;
+pub const MINCORE: u64 = 27;
+pub const MLOCK: u64 = 149;
+pub const MUNLOCK: u64 = 150;
+pub const MLOCKALL: u64 = 151;
+pub const MUNLOCKALL: u64 = 152;
+pub const MLOCK2: u64 = 325;
diff --git a/userland/capsule_linux/src/linux/call/mem/lock.rs b/userland/capsule_linux/src/linux/call/mem/lock.rs
new file mode 100644
index 000000000..dfbaa21d7
--- /dev/null
+++ b/userland/capsule_linux/src/linux/call/mem/lock.rs
@@ -0,0 +1,64 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+//! `mlock`, `mlock2`, `munlock`, `mlockall` and `munlockall`.
+//!
+//! Locking keeps pages resident. Every page a guest holds here is resident
+//! from the moment it is mapped until it is unmapped, and none is ever paged
+//! out, so every page is already as locked as Linux can make it. What is left
+//! of each call is Linux's checking of its arguments, answered the same way:
+//! the span must be held, and the flags known. RLIMIT_MEMLOCK is reported
+//! unlimited, so no lock is refused for its size.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::{page_down, page_up, Guest, PAGE};
+
+const MLOCK_ONFAULT: u64 = 1;
+const MCL_CURRENT: u64 = 1;
+const MCL_FUTURE: u64 = 2;
+const MCL_ONFAULT: u64 = 4;
+
+/// `mlock` and `munlock` alike: the address is rounded down, the length up,
+/// and a span with a page the guest does not hold is ENOMEM.
+pub fn mlock(guest: &Guest, addr: u64, len: u64) -> u64 {
+ let start = page_down(addr);
+ let Some(end) = addr.checked_add(len).filter(|e| *e <= u64::MAX - PAGE) else {
+ return errno::fail(errno::EINVAL);
+ };
+ let end = page_up(end);
+ if end == start {
+ return errno::ok(0);
+ }
+ if guest.mapped_from(start) < end - start {
+ return errno::fail(errno::ENOMEM);
+ }
+ errno::ok(0)
+}
+
+pub fn mlock2(guest: &Guest, addr: u64, len: u64, flags: u64) -> u64 {
+ if flags & !MLOCK_ONFAULT != 0 {
+ return errno::fail(errno::EINVAL);
+ }
+ mlock(guest, addr, len)
+}
+
+/// Linux refuses no flags, unknown flags, and MCL_ONFAULT on its own.
+pub fn mlockall(flags: u64) -> u64 {
+ let known = MCL_CURRENT | MCL_FUTURE | MCL_ONFAULT;
+ if flags == 0 || flags & !known != 0 || flags == MCL_ONFAULT {
+ return errno::fail(errno::EINVAL);
+ }
+ errno::ok(0)
+}
diff --git a/userland/capsule_linux/src/linux/call/mem/map.rs b/userland/capsule_linux/src/linux/call/mem/map.rs
index 904072cdb..1baa5ce4b 100644
--- a/userland/capsule_linux/src/linux/call/mem/map.rs
+++ b/userland/capsule_linux/src/linux/call/mem/map.rs
@@ -17,50 +17,43 @@
//! `mmap`: anonymous pages, or a private mapping of a file.
use crate::linux::abi::errno;
-use crate::linux::guest::{span_within, Guest, MMAP_LIMIT, USER_MAX};
+use crate::linux::guest::{Guest, PAGE};
-use super::map_anon::{anonymous, memfd};
-use super::map_file::file;
+use super::map_place::place;
use super::map_req::MapReq;
use super::prot::wx_refused;
-const MAP_SHARED: u64 = 0x01;
-const MAP_ANONYMOUS: u64 = 0x20;
const MAP_FIXED: u64 = 0x10;
+const MAP_FIXED_NOREPLACE: u64 = 0x10_0000;
pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 {
- if req.len == 0 {
+ /*
+ * Linux takes a file offset on a page boundary, and an exact address too;
+ * only a hint is rounded.
+ */
+ let exact = req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0;
+ if req.len == 0 || req.off % PAGE != 0 || (exact && req.addr % PAGE != 0) {
return errno::fail(errno::EINVAL);
}
if wx_refused(req.prot) {
return errno::fail(errno::EPERM);
}
- // MAP_FIXED is the exact address or failure. Page zero is never in the
- // plan, and landing elsewhere would hand back memory the guest did not
- // ask for, so it is refused, as Linux refuses it below mmap_min_addr.
- if req.flags & MAP_FIXED != 0 && req.addr == 0 {
+ /*
+ * MAP_FIXED and MAP_FIXED_NOREPLACE are the exact address or failure.
+ * Page zero is never in the plan, and landing elsewhere would hand back
+ * memory the guest did not ask for, so it is refused, as Linux refuses it
+ * below mmap_min_addr.
+ */
+ if exact && req.addr == 0 {
return errno::fail(errno::EPERM);
}
- // The ceiling differs by who chose the address.
- let (at, limit) = match req.fixed() {
- Some(addr) => (addr, USER_MAX),
- None => (guest.mmap_next, MMAP_LIMIT),
+ let spot = match place(guest, &req) {
+ Ok(spot) => spot,
+ Err(e) => return errno::fail(e),
};
- let Some((at, span)) = span_within(at, req.len, limit) else {
- return errno::fail(errno::ENOMEM);
- };
- if req.flags & MAP_ANONYMOUS != 0 {
- return anonymous(guest, &req, at, span);
- }
- if crate::linux::file::is_memfd(guest, req.fd) {
- return memfd(guest, &req, at, span);
- }
- if req.flags & MAP_SHARED != 0 {
- /*
- * Sharing a file between processes needs frames that two address
- * spaces both point at, which no peer call offers.
- */
- return errno::fail(errno::ENOSYS);
+ let out = super::map_kind::map_at(guest, &req, spot.at, spot.span);
+ if spot.from_cursor && (out as i64) >= 0 {
+ guest.mmap_next = spot.at + spot.span;
}
- file(guest, &req, at, span)
+ out
}
diff --git a/userland/capsule_linux/src/linux/call/mem/map_anon.rs b/userland/capsule_linux/src/linux/call/mem/map_anon.rs
index 0158b7f26..d3c8ef994 100644
--- a/userland/capsule_linux/src/linux/call/mem/map_anon.rs
+++ b/userland/capsule_linux/src/linux/call/mem/map_anon.rs
@@ -44,10 +44,15 @@ pub fn memfd(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
}
pub fn anonymous(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
- // A PROT_NONE anonymous mapping is a reservation: the runtime that makes it
- // (Go's, for one) commits a fraction of it later with a fixed RW mapping.
- // Backing the whole span here would spend real frames on address space no
- // one has touched, so reserve it and let the first access fault a page in.
+ if !req.make_room(guest, at, span) {
+ return errno::fail(errno::ENOMEM);
+ }
+ /*
+ * A PROT_NONE anonymous mapping is a reservation: the runtime that makes it
+ * (Go's, for one) commits a fraction of it later with a fixed RW mapping.
+ * Backing the whole span here would spend real frames on address space no
+ * one may touch, so reserve it; a commit maps the part that is opened.
+ */
let backed = if req.prot == 0 {
guest.reserve(at, span)
} else {
@@ -56,8 +61,5 @@ pub fn anonymous(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
if backed < 0 {
return errno::fail(errno::ENOMEM);
}
- if req.fixed().is_none() {
- guest.mmap_next += span;
- }
errno::ok(at)
}
diff --git a/userland/capsule_linux/src/linux/call/mem/map_file.rs b/userland/capsule_linux/src/linux/call/mem/map_file.rs
index 2961e9123..894d39531 100644
--- a/userland/capsule_linux/src/linux/call/mem/map_file.rs
+++ b/userland/capsule_linux/src/linux/call/mem/map_file.rs
@@ -36,7 +36,7 @@ pub fn file(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
if let Some(None) = proved {
return errno::fail(errno::EPERM);
}
- if guest.map(at, span, true, false) < 0 {
+ if !req.make_room(guest, at, span) || guest.map(at, span, true, false) < 0 {
return errno::fail(errno::ENOMEM);
}
if let Some(Some(bytes)) = proved {
@@ -48,7 +48,7 @@ pub fn file(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
if fill_read(guest, req, at) < 0 {
return errno::fail(errno::EACCES);
}
- // Not proved, since nothing asked to run it: it stays that way.
+ /* Not proved, since nothing asked to run it: it stays that way. */
guest.mark_unproven(at, span);
finish(guest, req, at, span)
}
@@ -57,8 +57,5 @@ fn finish(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
if protect_span(guest, at, span, req.prot) < 0 {
return errno::fail(errno::EACCES);
}
- if req.fixed().is_none() {
- guest.mmap_next += span;
- }
errno::ok(at)
}
diff --git a/userland/capsule_linux/src/linux/call/mem/map_free.rs b/userland/capsule_linux/src/linux/call/mem/map_free.rs
new file mode 100644
index 000000000..ccb509375
--- /dev/null
+++ b/userland/capsule_linux/src/linux/call/mem/map_free.rs
@@ -0,0 +1,40 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Where the mapping cursor finds room.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::{page_up, span_within, Guest, MMAP_LIMIT};
+
+/// The first span of `len` at or above the mapping cursor that meets
+/// nothing the guest holds.
+pub fn free_span(guest: &Guest, len: u64) -> Result<(u64, u64), i64> {
+ let mut at = guest.mmap_next;
+ loop {
+ let (start, span) = span_within(at, len, MMAP_LIMIT).ok_or(errno::ENOMEM)?;
+ let end = start + span;
+ let past = guest
+ .regions
+ .iter()
+ .filter(|r| r.at < end && start < r.at.saturating_add(r.len))
+ .map(|r| r.at.saturating_add(r.len))
+ .max();
+ match past {
+ None => return Ok((start, span)),
+ Some(next) => at = page_up(next),
+ }
+ }
+}
diff --git a/userland/capsule_linux/src/linux/call/mem/map_kind.rs b/userland/capsule_linux/src/linux/call/mem/map_kind.rs
new file mode 100644
index 000000000..b0b45e11c
--- /dev/null
+++ b/userland/capsule_linux/src/linux/call/mem/map_kind.rs
@@ -0,0 +1,44 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Which kind of mapping an mmap makes once it has a place.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::map_anon::{anonymous, memfd};
+use super::map_file::file;
+use super::map_req::MapReq;
+
+const MAP_SHARED: u64 = 0x01;
+const MAP_ANONYMOUS: u64 = 0x20;
+
+pub(super) fn map_at(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 {
+ if req.flags & MAP_ANONYMOUS != 0 {
+ return anonymous(guest, req, at, span);
+ }
+ if crate::linux::file::is_memfd(guest, req.fd) {
+ return memfd(guest, req, at, span);
+ }
+ if req.flags & MAP_SHARED != 0 {
+ /*
+ * Sharing a file between processes needs frames that two address
+ * spaces both point at, which no peer call offers.
+ */
+ return errno::fail(errno::ENOSYS);
+ }
+ file(guest, req, at, span)
+}
diff --git a/userland/capsule_linux/src/linux/call/mem/map_place.rs b/userland/capsule_linux/src/linux/call/mem/map_place.rs
new file mode 100644
index 000000000..830447d2b
--- /dev/null
+++ b/userland/capsule_linux/src/linux/call/mem/map_place.rs
@@ -0,0 +1,58 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Where an mmap goes. MAP_FIXED is the exact address; MAP_FIXED_NOREPLACE
+//! is the exact address or EEXIST when something is there; any other address
+//! is a hint taken only when nothing is there. Otherwise the mapping cursor
+//! chooses, skipping every span the guest already holds, since a new mapping
+//! laid over an old one would hand back the old pages.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::{page_up, span_within, Guest, USER_MAX};
+
+use super::map_req::MapReq;
+
+const MAP_FIXED: u64 = 0x10;
+const MAP_FIXED_NOREPLACE: u64 = 0x10_0000;
+
+pub struct Place {
+ pub at: u64,
+ pub span: u64,
+ /// Chosen by the cursor, which then moves past it.
+ pub from_cursor: bool,
+}
+
+/// The span the mapping takes, or the errno refusing it.
+pub fn place(guest: &Guest, req: &MapReq) -> Result {
+ let exact = |(at, span)| Place { at, span, from_cursor: false };
+ if req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0 {
+ let (at, span) = span_within(req.addr, req.len, USER_MAX).ok_or(errno::ENOMEM)?;
+ if req.flags & MAP_FIXED == 0 && guest.overlaps(at, span) {
+ return Err(errno::EEXIST);
+ }
+ return Ok(exact((at, span)));
+ }
+ /* Linux rounds a hint up to a page. */
+ if req.addr != 0 {
+ if let Some(got) = span_within(page_up(req.addr), req.len, USER_MAX) {
+ if !guest.overlaps(got.0, got.1) {
+ return Ok(exact(got));
+ }
+ }
+ }
+ let (at, span) = super::map_free::free_span(guest, req.len)?;
+ Ok(Place { at, span, from_cursor: true })
+}
diff --git a/userland/capsule_linux/src/linux/call/mem/map_req.rs b/userland/capsule_linux/src/linux/call/mem/map_req.rs
index c426dcb11..7e278ccd2 100644
--- a/userland/capsule_linux/src/linux/call/mem/map_req.rs
+++ b/userland/capsule_linux/src/linux/call/mem/map_req.rs
@@ -17,6 +17,10 @@
//! What a guest asked `mmap` for, in one value.
+use crate::linux::guest::Guest;
+
+const MAP_FIXED: u64 = 0x10;
+
pub struct MapReq {
pub addr: u64,
pub len: u64,
@@ -31,12 +35,11 @@ impl MapReq {
MapReq { addr: a[0], len: a[1], prot: a[2], flags: a[3], fd: a[4], off: a[5] }
}
- /// The address the guest named, or nothing when it left the choice
- /// to this capsule, which is the case the mapping cursor advances on.
- pub fn fixed(&self) -> Option {
- match self.addr {
- 0 => None,
- addr => Some(crate::linux::guest::page_down(addr)),
- }
+ /// Make room for a MAP_FIXED mapping at `[at, at + span)`. Linux replaces
+ /// whatever was there: the old pages go and the new mapping starts from
+ /// zeroes with its own protection. Called just before the new pages go in,
+ /// so a mapping refused earlier leaves the old one where it was.
+ pub fn make_room(&self, guest: &mut Guest, at: u64, span: u64) -> bool {
+ self.flags & MAP_FIXED == 0 || guest.unmap(at, span) >= 0
}
}
diff --git a/userland/capsule_linux/src/linux/call/mem/memory.rs b/userland/capsule_linux/src/linux/call/mem/memory.rs
index 28a2cfb6a..227da0d5c 100644
--- a/userland/capsule_linux/src/linux/call/mem/memory.rs
+++ b/userland/capsule_linux/src/linux/call/mem/memory.rs
@@ -17,7 +17,7 @@
//! `brk` and `munmap`.
use crate::linux::abi::errno;
-use crate::linux::guest::{page_up, Guest, BRK_BASE, BRK_LIMIT};
+use crate::linux::guest::{page_up, Guest, BRK_BASE, BRK_LIMIT, PAGE};
/// `brk(0)` reports the break; any other value moves it and reports where
/// it landed, which is Linux's contract and not an error channel.
@@ -29,20 +29,29 @@ pub fn brk(guest: &mut Guest, want: u64) -> u64 {
if want == 0 || want < BRK_BASE || want > BRK_LIMIT {
return errno::ok(guest.brk);
}
- let top = page_up(want);
- if top > guest.brk {
- let len = top - guest.brk;
- if guest.map(guest.brk, len, true, false) < 0 {
+ /* Whole pages: the page the old break sits in is already held. */
+ let (old, top) = (page_up(guest.brk), page_up(want));
+ if top > old {
+ /* Linux refuses a break that would run into a mapping. */
+ if guest.overlaps(old, top - old) || guest.map(old, top - old, true, false) < 0 {
return errno::ok(guest.brk);
}
}
+ /*
+ * A lower break gives the pages above it back, so growing again reads
+ * zeroes, as on Linux.
+ */
+ if top < old && guest.unmap(top, old - top) < 0 {
+ return errno::ok(guest.brk);
+ }
guest.brk = want;
errno::ok(guest.brk)
}
/// The pages go back to the kernel and leave the guest's region list.
pub fn munmap(guest: &mut Guest, addr: u64, len: u64) -> u64 {
- if len == 0 {
+ /* Linux takes an address on a page boundary, and rounds only the length. */
+ if len == 0 || addr % PAGE != 0 {
return errno::fail(errno::EINVAL);
}
match guest.unmap(addr, len) {
diff --git a/userland/capsule_linux/src/linux/call/mem/mod.rs b/userland/capsule_linux/src/linux/call/mem/mod.rs
index 423f1a164..a8f37dd84 100644
--- a/userland/capsule_linux/src/linux/call/mem/mod.rs
+++ b/userland/capsule_linux/src/linux/call/mem/mod.rs
@@ -14,23 +14,34 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-//! The calls that shape a guest's address space: `mmap`, `munmap`, `brk` and
-//! `mprotect`.
+//! The calls that shape a guest's address space: `mmap`, `munmap`, `brk`,
+//! `mprotect` and `mremap`, and the ones that ask about it or lock it.
+mod lock;
mod map;
mod map_anon;
mod map_exec;
mod map_file;
mod map_fill;
+mod map_free;
+mod map_kind;
+mod map_place;
mod map_req;
mod memory;
mod prot;
mod prot_span;
+mod prot_walk;
mod remap;
mod remap_move;
+mod remap_one;
+mod resident;
+mod sync;
+pub use lock::{mlock, mlock2, mlockall};
pub use map::mmap;
pub use map_req::MapReq;
pub use memory::{brk, munmap};
pub use prot::mprotect;
pub use remap::mremap;
+pub use resident::mincore;
+pub use sync::msync;
diff --git a/userland/capsule_linux/src/linux/call/mem/prot.rs b/userland/capsule_linux/src/linux/call/mem/prot.rs
index 9d060a23e..ce3bebf23 100644
--- a/userland/capsule_linux/src/linux/call/mem/prot.rs
+++ b/userland/capsule_linux/src/linux/call/mem/prot.rs
@@ -17,14 +17,12 @@
//! `mprotect`, and the rule that makes it necessary.
use crate::linux::abi::errno;
-use crate::linux::guest::{span_within, Guest, USER_MAX};
-
-use super::prot_span::protect_span;
+use crate::linux::guest::{span_within, Guest, PAGE, USER_MAX};
pub const PROT_WRITE: u64 = 2;
pub const PROT_EXEC: u64 = 4;
/// PROT_READ, PROT_WRITE and PROT_EXEC together: any access at all.
-const PROT_ANY: u64 = 7;
+pub const PROT_ANY: u64 = 7;
/// A request for both at once.
pub fn wx_refused(prot: u64) -> bool {
@@ -32,6 +30,10 @@ pub fn wx_refused(prot: u64) -> bool {
}
pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 {
+ /* Linux takes an address on a page boundary, and rounds only the length. */
+ if addr % PAGE != 0 {
+ return errno::fail(errno::EINVAL);
+ }
if len == 0 {
return errno::ok(0);
}
@@ -54,37 +56,5 @@ pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 {
if prot & PROT_EXEC != 0 && guest.span_unproven(start, span) {
return errno::fail(errno::EPERM);
}
- let end = start + span;
- let mut at = start;
- while at < end {
- let Some(r) = guest.regions.iter().find(|r| r.at <= at && at < r.at + r.len).copied()
- else {
- // Linux refuses a span with no mapping in it at all.
- return errno::fail(errno::ENOMEM);
- };
- let upto = end.min(r.at + r.len);
- let piece = upto - at;
- if !r.backed {
- /*
- * A PROT_NONE reservation has no pages for the kernel to
- * reprotect. Asking for access commits it, which is how musl makes
- * a thread stack: reserve with PROT_NONE, then mprotect the part
- * it uses to read-write. PROT_NONE on it changes nothing.
- */
- if prot & PROT_ANY == 0 {
- at = upto;
- continue;
- }
- if guest.commit(at, piece, prot & PROT_WRITE != 0, prot & PROT_EXEC != 0) < 0 {
- return errno::fail(errno::ENOMEM);
- }
- }
- // Every page is present now; this sets `prot` on all of them,
- // including any the guest touched while the span was reserved.
- if protect_span(guest, at, piece, prot) < 0 {
- return errno::fail(errno::EACCES);
- }
- at = upto;
- }
- errno::ok(0)
+ super::prot_walk::walk(guest, start, span, prot)
}
diff --git a/userland/capsule_linux/src/linux/call/mem/prot_span.rs b/userland/capsule_linux/src/linux/call/mem/prot_span.rs
index aac1b7043..bbe49caf6 100644
--- a/userland/capsule_linux/src/linux/call/mem/prot_span.rs
+++ b/userland/capsule_linux/src/linux/call/mem/prot_span.rs
@@ -16,21 +16,18 @@
//! Reprotecting a span, a peer call at a time.
-use nonos_libc::peer::{mk_peer_protect, PEER_PROT_EXEC, PEER_PROT_WRITE};
+use nonos_libc::peer::mk_peer_protect;
-use crate::linux::guest::{Guest, MAX_SPAN};
+use crate::linux::guest::{peer_prot, Guest, MAX_SPAN};
-use super::prot::{PROT_EXEC, PROT_WRITE};
+use super::prot::{PROT_ANY, PROT_EXEC, PROT_WRITE};
-/// Set the protection of a span already mapped in the guest.
-pub fn protect_span(guest: &Guest, addr: u64, span: u64, prot: u64) -> i64 {
- let mut bits = 0;
- if prot & PROT_WRITE != 0 {
- bits |= PEER_PROT_WRITE;
- }
- if prot & PROT_EXEC != 0 {
- bits |= PEER_PROT_EXEC;
- }
+/// Set the protection of a span already mapped in the guest, and record it
+/// on the spans it covers.
+pub fn protect_span(guest: &mut Guest, addr: u64, span: u64, prot: u64) -> i64 {
+ let (write, exec, access) =
+ (prot & PROT_WRITE != 0, prot & PROT_EXEC != 0, prot & PROT_ANY != 0);
+ let bits = peer_prot(write, exec, access);
let mut done = 0;
while done < span {
let take = (span - done).min(MAX_SPAN);
@@ -40,5 +37,6 @@ pub fn protect_span(guest: &Guest, addr: u64, span: u64, prot: u64) -> i64 {
}
done += take;
}
+ guest.set_prot(addr, span, write, exec, access);
0
}
diff --git a/userland/capsule_linux/src/linux/call/mem/prot_walk.rs b/userland/capsule_linux/src/linux/call/mem/prot_walk.rs
new file mode 100644
index 000000000..ab11f99a5
--- /dev/null
+++ b/userland/capsule_linux/src/linux/call/mem/prot_walk.rs
@@ -0,0 +1,56 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Walking an mprotect span one mapping at a time.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::prot::{PROT_ANY, PROT_EXEC, PROT_WRITE};
+use super::prot_span::protect_span;
+
+/// Give `[start, start + span)` the protection `prot`, mapping by mapping.
+pub(super) fn walk(guest: &mut Guest, start: u64, span: u64, prot: u64) -> u64 {
+ let end = start + span;
+ let mut at = start;
+ while at < end {
+ let Some(r) = guest.regions.iter().find(|r| r.at <= at && at < r.at + r.len).copied()
+ else {
+ /* Linux refuses a span with no mapping in it at all. */
+ return errno::fail(errno::ENOMEM);
+ };
+ let upto = end.min(r.at + r.len);
+ let piece = upto - at;
+ if !r.backed {
+ /*
+ * A PROT_NONE reservation has no pages for the kernel to
+ * reprotect. Asking for access commits it, which is how musl makes
+ * a thread stack: reserve with PROT_NONE, then mprotect the part
+ * it uses to read-write. PROT_NONE on it changes nothing. The
+ * commit maps the piece with `prot` and records it.
+ */
+ if prot & PROT_ANY != 0
+ && guest.commit(at, piece, prot & PROT_WRITE != 0, prot & PROT_EXEC != 0) < 0
+ {
+ return errno::fail(errno::ENOMEM);
+ }
+ } else if protect_span(guest, at, piece, prot) < 0 {
+ return errno::fail(errno::EACCES);
+ }
+ at = upto;
+ }
+ errno::ok(0)
+}
diff --git a/userland/capsule_linux/src/linux/call/mem/remap.rs b/userland/capsule_linux/src/linux/call/mem/remap.rs
index add8ae3a4..85c41a2b0 100644
--- a/userland/capsule_linux/src/linux/call/mem/remap.rs
+++ b/userland/capsule_linux/src/linux/call/mem/remap.rs
@@ -18,23 +18,24 @@
//! or move with MREMAP_MAYMOVE. glibc's realloc of a large block is this.
use crate::linux::abi::errno;
-use crate::linux::guest::{page_up, span_within, Guest, MMAP_LIMIT, PAGE};
+use crate::linux::guest::{page_up, span_within, Guest, PAGE, USER_MAX};
const MAYMOVE: u64 = 1;
pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u64) -> u64 {
- // MREMAP_FIXED and DONTUNMAP choose the destination; neither is offered.
+ /* MREMAP_FIXED and DONTUNMAP choose the destination; neither is offered. */
if flags & !MAYMOVE != 0 || old % PAGE != 0 || old_len == 0 || new_len == 0 {
return errno::fail(errno::EINVAL);
}
let (old_len, new_len) = (page_up(old_len), page_up(new_len));
- if guest.mapped_from(old) < old_len {
- return errno::fail(errno::EFAULT);
- }
let Some(r) = guest.regions.iter().find(|r| r.at <= old && old < r.at + r.len).copied() else {
return errno::fail(errno::EFAULT);
};
- // Code was proved where it was mapped; a moved copy would not be.
+ /* Linux moves one mapping at a time: the old span must lie inside one. */
+ if !super::remap_one::one_mapping(guest, old, old_len, &r) {
+ return errno::fail(errno::EFAULT);
+ }
+ /* Code was proved where it was mapped; a moved copy would not be. */
if r.exec {
return errno::fail(errno::EPERM);
}
@@ -46,16 +47,15 @@ pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u6
}
let tail = old + old_len;
let grow = new_len - old_len;
- let free = !guest.regions.iter().any(|g| g.at < tail + grow && tail < g.at + g.len);
- if free
- && span_within(tail, grow, MMAP_LIMIT).is_some()
- && guest.map(tail, grow, r.write, false) >= 0
+ /* The grown part is the same mapping: its protection, its backing. */
+ if !guest.overlaps(tail, grow)
+ && span_within(tail, grow, USER_MAX).is_some()
+ && guest.map_like(tail, grow, &r) >= 0
{
- guest.mmap_next = guest.mmap_next.max(tail + grow);
return errno::ok(old);
}
if flags & MAYMOVE == 0 {
return errno::fail(errno::ENOMEM);
}
- super::remap_move::moved(guest, old, old_len, new_len, r.write)
+ super::remap_move::moved(guest, old, old_len, new_len, &r)
}
diff --git a/userland/capsule_linux/src/linux/call/mem/remap_move.rs b/userland/capsule_linux/src/linux/call/mem/remap_move.rs
index b8b3930d7..2b10dad7a 100644
--- a/userland/capsule_linux/src/linux/call/mem/remap_move.rs
+++ b/userland/capsule_linux/src/linux/call/mem/remap_move.rs
@@ -17,29 +17,26 @@
//! `mremap` when the block cannot grow where it is: moved to fresh pages.
use crate::linux::abi::errno;
-use crate::linux::guest::{span_within, Guest, MMAP_LIMIT};
+use crate::linux::guest::{Guest, Region};
-const PROT_READ: u64 = 1;
+use super::map_free::free_span;
-// A fresh span at the mapping cursor, the old bytes copied in, the old span gone.
-pub(super) fn moved(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, write: bool) -> u64 {
- let Some((at, span)) = span_within(guest.mmap_next, new_len, MMAP_LIMIT) else {
- return errno::fail(errno::ENOMEM);
- };
- let Some(bytes) = guest.read(old, old_len as usize) else {
- return errno::fail(errno::EFAULT);
+/// A fresh span at the mapping cursor held like the old one, with the old
+/// protection, backing and provenance; the old bytes copied in; the old span
+/// gone. A move that fails part way leaves nothing new behind.
+pub(super) fn moved(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, like: &Region) -> u64 {
+ let (at, span) = match free_span(guest, new_len) {
+ Ok(got) => got,
+ Err(e) => return errno::fail(e),
};
- // Writable while the bytes go in; the old protection after.
- if guest.map(at, span, true, false) < 0 {
+ if guest.map_like(at, span, like) < 0 {
return errno::fail(errno::ENOMEM);
}
- guest.mmap_next += span;
- if guest.write(at, &bytes) < bytes.len() as i64 {
+ if like.backed && !guest.copy_within(old, at, old_len) {
+ let _ = guest.unmap(at, span);
return errno::fail(errno::EFAULT);
}
- if !write {
- let _ = super::prot::mprotect(guest, at, span, PROT_READ);
- }
let _ = guest.unmap(old, old_len);
+ guest.mmap_next = at + span;
errno::ok(at)
}
diff --git a/userland/capsule_linux/src/linux/call/mem/remap_one.rs b/userland/capsule_linux/src/linux/call/mem/remap_one.rs
new file mode 100644
index 000000000..40d026315
--- /dev/null
+++ b/userland/capsule_linux/src/linux/call/mem/remap_one.rs
@@ -0,0 +1,38 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Whether an mremap span is one mapping, as Linux requires.
+
+use crate::linux::guest::{Guest, Region};
+
+/// Every page of `[at, at + len)` is held with the same protection, backing
+/// and provenance as `like`, which is what Linux keeps as one mapping.
+pub(super) fn one_mapping(guest: &Guest, at: u64, len: u64, like: &Region) -> bool {
+ let end = at + len;
+ let mut reach = at;
+ while reach < end {
+ let Some(r) = guest.regions.iter().find(|r| r.at <= reach && reach < r.at + r.len) else {
+ return false;
+ };
+ let same = (r.write, r.exec, r.access, r.backed, r.unproven)
+ == (like.write, like.exec, like.access, like.backed, like.unproven);
+ if !same {
+ return false;
+ }
+ reach = r.at + r.len;
+ }
+ true
+}
diff --git a/userland/capsule_linux/src/linux/call/mem/resident.rs b/userland/capsule_linux/src/linux/call/mem/resident.rs
new file mode 100644
index 000000000..dba74b088
--- /dev/null
+++ b/userland/capsule_linux/src/linux/call/mem/resident.rs
@@ -0,0 +1,54 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+//! `mincore`: which pages of a span are resident.
+//!
+//! A backed span is resident page for page, since the personality maps every
+//! page of it when it is committed and the kernel pages nothing out, whatever
+//! its protection now. A reservation has no page at all. So the region list
+//! answers exactly, one byte per page, 1 for resident.
+
+use alloc::vec::Vec;
+
+use crate::linux::abi::errno;
+use crate::linux::guest::{page_up, Guest, MAX_SPAN, PAGE, USER_MAX};
+
+pub fn mincore(guest: &Guest, addr: u64, len: u64, vec: u64) -> u64 {
+ if addr % PAGE != 0 {
+ return errno::fail(errno::EINVAL);
+ }
+ let Some(end) = addr.checked_add(len).filter(|e| *e <= USER_MAX) else {
+ return errno::fail(errno::ENOMEM);
+ };
+ let end = page_up(end);
+ let mut at = addr;
+ let mut out: Vec = Vec::new();
+ let mut written = 0u64;
+ while at < end {
+ let Some(r) = guest.regions.iter().find(|r| r.at <= at && at < r.at + r.len) else {
+ return errno::fail(errno::ENOMEM);
+ };
+ out.push(r.backed as u8);
+ at += PAGE;
+ if out.len() as u64 == MAX_SPAN || at >= end {
+ if guest.write(vec + written, &out) < out.len() as i64 {
+ return errno::fail(errno::EFAULT);
+ }
+ written += out.len() as u64;
+ out.clear();
+ }
+ }
+ errno::ok(0)
+}
diff --git a/userland/capsule_linux/src/linux/call/mem/sync.rs b/userland/capsule_linux/src/linux/call/mem/sync.rs
new file mode 100644
index 000000000..665ff0cfa
--- /dev/null
+++ b/userland/capsule_linux/src/linux/call/mem/sync.rs
@@ -0,0 +1,44 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+//! `msync`: write a shared file mapping back to its file.
+//!
+//! Every file mapping here is private, since MAP_SHARED of a file is refused,
+//! and a private mapping has nothing to write back: Linux answers msync on
+//! one with its argument checks alone, and so does this.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::{page_up, Guest, PAGE};
+
+const MS_ASYNC: u64 = 1;
+const MS_INVALIDATE: u64 = 2;
+const MS_SYNC: u64 = 4;
+
+pub fn msync(guest: &Guest, addr: u64, len: u64, flags: u64) -> u64 {
+ if flags & !(MS_ASYNC | MS_INVALIDATE | MS_SYNC) != 0 || addr % PAGE != 0 {
+ return errno::fail(errno::EINVAL);
+ }
+ if flags & MS_ASYNC != 0 && flags & MS_SYNC != 0 {
+ return errno::fail(errno::EINVAL);
+ }
+ let Some(end) = addr.checked_add(page_up(len)).filter(|_| len <= u64::MAX - PAGE) else {
+ return errno::fail(errno::ENOMEM);
+ };
+ /* Linux reports a span with a page nothing maps as ENOMEM. */
+ if end > addr && guest.mapped_from(addr) < end - addr {
+ return errno::fail(errno::ENOMEM);
+ }
+ errno::ok(0)
+}
diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs
index ab445d17c..6b506d6ee 100644
--- a/userland/capsule_linux/src/linux/call/mod.rs
+++ b/userland/capsule_linux/src/linux/call/mod.rs
@@ -33,7 +33,7 @@ mod limits;
mod limits_table;
mod glibc;
mod glibc_sched;
-mod mem;
+pub mod mem;
mod pipe;
mod pipe_dup;
mod pipe_end;
diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs
index 3cf86191a..e7b7b3a87 100644
--- a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs
+++ b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs
@@ -16,39 +16,42 @@
//! Copying a parent's spans into the child it just made.
-use crate::linux::guest::{Guest, Region};
-use nonos_libc::peer::{mk_peer_map, mk_peer_write, PEER_PROT_EXEC, PEER_PROT_WRITE};
+use crate::linux::guest::{Guest, MAX_SPAN};
+use nonos_libc::peer::{mk_peer_map, mk_peer_write};
/// Every span, mapped into the child and then filled from the parent.
pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool {
let spans = guest.regions.clone();
for span in spans {
- // An unbacked reservation has no frames to copy; the child reserves it
- // the same way, and its own first access faults a page in.
+ /*
+ * An unbacked reservation has no frames to copy; the child holds the
+ * same reservation, and a touch there faults in the child as here.
+ */
if !span.backed {
continue;
}
- if mk_peer_map(child, span.at, span.len, prot_of(&span)) < 0 {
- return false;
- }
- if !copy_one(guest, child, span.at, span.len) {
- return false;
+ /*
+ * The kernel maps and copies at most MAX_SPAN in one peer call, so a
+ * larger span, which a Go heap is, crosses in pieces. Each piece gets
+ * the protection the span has now, PROT_NONE included: the kernel
+ * copies into a page whatever its protection, so the bytes still go
+ * in, and the child can do no more with them than the parent can.
+ */
+ let mut done = 0;
+ while done < span.len {
+ let take = (span.len - done).min(MAX_SPAN);
+ if mk_peer_map(child, span.at + done, take, span.peer_prot()) < 0 {
+ return false;
+ }
+ if !copy_one(guest, child, span.at + done, take) {
+ return false;
+ }
+ done += take;
}
}
true
}
-fn prot_of(span: &Region) -> u64 {
- let mut prot = 0;
- if span.write {
- prot |= PEER_PROT_WRITE;
- }
- if span.exec {
- prot |= PEER_PROT_EXEC;
- }
- prot
-}
-
fn copy_one(guest: &Guest, child: u32, at: u64, len: u64) -> bool {
let Some(bytes) = guest.read(at, len as usize) else {
return false;
diff --git a/userland/capsule_linux/src/linux/guest/mem_like.rs b/userland/capsule_linux/src/linux/guest/mem_like.rs
new file mode 100644
index 000000000..aa51cd440
--- /dev/null
+++ b/userland/capsule_linux/src/linux/guest/mem_like.rs
@@ -0,0 +1,55 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A span that takes after another: what mremap makes when a mapping grows
+//! or moves, since Linux gives the new part the mapping's own protection.
+
+use super::handle::Guest;
+use super::mem::MAX_SPAN;
+use super::mem_span::map_span;
+use super::region::Region;
+
+impl Guest {
+ /// `[at, at + len)` held like `like`: the same protection, provenance and
+ /// backing. A reservation stays one, with no pages.
+ pub fn map_like(&mut self, at: u64, len: u64, like: &Region) -> i64 {
+ if like.backed {
+ let rc = map_span(self.pid, at, len, like.peer_prot());
+ if rc < 0 {
+ return rc;
+ }
+ }
+ self.regions.push(Region { at, len, ..*like });
+ 0
+ }
+
+ /// Copy `len` bytes from `from` to `to` inside the guest, a megabyte at a
+ /// time. The kernel copies whatever the pages' protection is.
+ pub fn copy_within(&self, from: u64, to: u64, len: u64) -> bool {
+ let mut done = 0;
+ while done < len {
+ let take = (len - done).min(MAX_SPAN);
+ let Some(bytes) = self.read(from + done, take as usize) else {
+ return false;
+ };
+ if self.write(to + done, &bytes) < bytes.len() as i64 {
+ return false;
+ }
+ done += take;
+ }
+ true
+ }
+}
diff --git a/userland/capsule_linux/src/linux/guest/mem_map.rs b/userland/capsule_linux/src/linux/guest/mem_map.rs
index 83404d6f4..9ae63e81f 100644
--- a/userland/capsule_linux/src/linux/guest/mem_map.rs
+++ b/userland/capsule_linux/src/linux/guest/mem_map.rs
@@ -16,36 +16,23 @@
//! Backing a span of a guest with pages.
-use nonos_libc::peer::{mk_peer_map, PEER_PROT_EXEC, PEER_PROT_WRITE};
-
use super::handle::Guest;
use super::layout::USER_MAX;
-use super::mem::{span_within, MAX_SPAN};
-use super::region::Region;
+use super::mem::span_within;
+use super::mem_span::map_span;
+use super::region::{peer_prot, Region};
use super::region_cut::cut;
impl Guest {
/// Pages covering `[addr, addr + len)`.
pub fn map(&mut self, addr: u64, len: u64, write: bool, exec: bool) -> i64 {
- // Bounded by the top of the guest's area, which is the stack.
+ /* Bounded by the top of the guest's area, which is the stack. */
let Some((start, span)) = span_within(addr, len, USER_MAX) else {
return -1;
};
- let mut prot = 0;
- if write {
- prot |= PEER_PROT_WRITE;
- }
- if exec {
- prot |= PEER_PROT_EXEC;
- }
- let mut done = 0;
- while done < span {
- let take = (span - done).min(MAX_SPAN);
- let rc = mk_peer_map(self.pid, start + done, take, prot);
- if rc < 0 {
- return rc;
- }
- done += take;
+ let rc = map_span(self.pid, start, span, peer_prot(write, exec, true));
+ if rc < 0 {
+ return rc;
}
/*
* Remembered because fork copies a guest by walking what its
@@ -56,6 +43,7 @@ impl Guest {
len: span,
write,
exec,
+ access: true,
unproven: false,
backed: true,
});
@@ -63,45 +51,24 @@ impl Guest {
}
/// Back `[at, at + len)` of a reservation with the given protection, the
- /// commit a fixed mmap makes. Pages the guest has not touched get zeroed
- /// frames; pages it has touched keep their contents, since peer_map skips
- /// a page that is already there. The span is then recorded as backed, in
- /// place of the reservation it came from, so fork copies it.
+ /// commit an mprotect that asks for access makes. The kernel fills no page
+ /// a guest touches on its own, so every page here is new and zeroed. The
+ /// span is then recorded as backed, in place of the reservation it came
+ /// from, so fork copies it.
pub fn commit(&mut self, at: u64, len: u64, write: bool, exec: bool) -> i64 {
- let mut prot = 0;
- if write {
- prot |= PEER_PROT_WRITE;
- }
- if exec {
- prot |= PEER_PROT_EXEC;
- }
- let mut done = 0;
- while done < len {
- let take = (len - done).min(MAX_SPAN);
- let rc = mk_peer_map(self.pid, at + done, take, prot);
- if rc < 0 {
- return rc;
- }
- done += take;
+ let rc = map_span(self.pid, at, len, peer_prot(write, exec, true));
+ if rc < 0 {
+ return rc;
}
self.regions = cut(&self.regions, at, len);
- self.regions.push(Region { at, len, write, exec, unproven: false, backed: true });
- 0
- }
-
- /// Take `len` of address space at `addr` without backing it: a PROT_NONE
- /// reservation. Bytes appear, zeroed, when the guest first touches them.
- pub fn reserve(&mut self, addr: u64, len: u64) -> i64 {
- let Some((start, span)) = span_within(addr, len, USER_MAX) else {
- return -1;
- };
self.regions.push(Region {
- at: start,
- len: span,
- write: true,
- exec: false,
+ at,
+ len,
+ write,
+ exec,
+ access: true,
unproven: false,
- backed: false,
+ backed: true,
});
0
}
diff --git a/userland/capsule_linux/src/linux/guest/mem_reserve.rs b/userland/capsule_linux/src/linux/guest/mem_reserve.rs
new file mode 100644
index 000000000..bd533af92
--- /dev/null
+++ b/userland/capsule_linux/src/linux/guest/mem_reserve.rs
@@ -0,0 +1,43 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Taking address space for a guest without backing it.
+
+use super::handle::Guest;
+use super::layout::USER_MAX;
+use super::mem::span_within;
+use super::region::Region;
+
+impl Guest {
+ /// Take `len` of address space at `addr` without backing it: a PROT_NONE
+ /// reservation. No page exists until a commit maps one; a touch before
+ /// that is a fault, as it is on Linux.
+ pub fn reserve(&mut self, addr: u64, len: u64) -> i64 {
+ let Some((start, span)) = span_within(addr, len, USER_MAX) else {
+ return -1;
+ };
+ self.regions.push(Region {
+ at: start,
+ len: span,
+ write: false,
+ exec: false,
+ access: false,
+ unproven: false,
+ backed: false,
+ });
+ 0
+ }
+}
diff --git a/userland/capsule_linux/src/linux/guest/mem_span.rs b/userland/capsule_linux/src/linux/guest/mem_span.rs
new file mode 100644
index 000000000..520876792
--- /dev/null
+++ b/userland/capsule_linux/src/linux/guest/mem_span.rs
@@ -0,0 +1,35 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Backing a span of a guest, a megabyte at a time.
+
+use nonos_libc::peer::mk_peer_map;
+
+use super::mem::MAX_SPAN;
+
+/// `MkPeerMap` over a span, a megabyte at a time.
+pub(super) fn map_span(pid: u32, at: u64, len: u64, prot: u64) -> i64 {
+ let mut done = 0;
+ while done < len {
+ let take = (len - done).min(MAX_SPAN);
+ let rc = mk_peer_map(pid, at + done, take, prot);
+ if rc < 0 {
+ return rc;
+ }
+ done += take;
+ }
+ 0
+}
diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs
index 514a26bdd..b0e86ebb1 100644
--- a/userland/capsule_linux/src/linux/guest/mod.rs
+++ b/userland/capsule_linux/src/linux/guest/mod.rs
@@ -36,12 +36,16 @@ mod links_list;
mod links_load;
mod mem;
mod mem_copy;
+mod mem_like;
mod mem_map;
+mod mem_reserve;
+mod mem_span;
mod mem_unmap;
mod region;
mod region_cut;
mod region_find;
mod region_mark;
+mod region_prot;
mod threads;
mod timer;
mod watch;
@@ -57,6 +61,6 @@ pub use layout::{
STACK_TOP, USER_MAX,
};
pub use mem::{page_down, page_up, span_within, MAX_SPAN, PAGE};
-pub use region::Region;
+pub use region::{peer_prot, Region};
pub use timer::Timer;
pub use watch::{Watch, EPOLLET, EPOLLONESHOT};
diff --git a/userland/capsule_linux/src/linux/guest/region.rs b/userland/capsule_linux/src/linux/guest/region.rs
index 8caf7b1a1..9ed68d6c9 100644
--- a/userland/capsule_linux/src/linux/guest/region.rs
+++ b/userland/capsule_linux/src/linux/guest/region.rs
@@ -16,17 +16,44 @@
//! One span of a guest's address space, as this capsule laid it down.
+use nonos_libc::peer::{PEER_PROT_EXEC, PEER_PROT_NONE, PEER_PROT_WRITE};
+
#[derive(Clone, Copy)]
pub struct Region {
pub at: u64,
pub len: u64,
pub write: bool,
pub exec: bool,
+ /// False for PROT_NONE: the guest may not touch the span at all. A backed
+ /// span keeps its pages and their bytes, present to the kernel only.
+ pub access: bool,
/// File bytes mapped without exec, so never proved: mprotect may not
/// make them executable later.
pub unproven: bool,
- /// False for a PROT_NONE reservation: address space taken, no frames yet.
- /// The kernel demand-fills a page on first access, so reserving a large
- /// span and committing a little costs only what is touched; fork skips it.
+ /// False for a PROT_NONE reservation: address space taken, no frames.
+ /// The kernel fills no page for a guest on its own, so a touch of one is a
+ /// fault; a commit maps the part asked for and records it backed.
pub backed: bool,
}
+
+impl Region {
+ /// The protection the kernel is asked to give this span's pages.
+ pub fn peer_prot(&self) -> u64 {
+ peer_prot(self.write, self.exec, self.access)
+ }
+}
+
+/// Peer protection bits for an access, a write and an exec permission.
+pub fn peer_prot(write: bool, exec: bool, access: bool) -> u64 {
+ if !access {
+ return PEER_PROT_NONE;
+ }
+ let mut prot = 0;
+ if write {
+ prot |= PEER_PROT_WRITE;
+ }
+ if exec {
+ prot |= PEER_PROT_EXEC;
+ }
+ prot
+}
diff --git a/userland/capsule_linux/src/linux/guest/region_find.rs b/userland/capsule_linux/src/linux/guest/region_find.rs
index aabfe4cd0..e557af1a7 100644
--- a/userland/capsule_linux/src/linux/guest/region_find.rs
+++ b/userland/capsule_linux/src/linux/guest/region_find.rs
@@ -36,4 +36,10 @@ impl Guest {
}
reach - addr
}
+
+ /// Whether any span the guest holds meets `[at, at + len)`.
+ pub fn overlaps(&self, at: u64, len: u64) -> bool {
+ let end = at.saturating_add(len);
+ self.regions.iter().any(|r| r.at < end && at < r.at.saturating_add(r.len))
+ }
}
diff --git a/userland/capsule_linux/src/linux/guest/region_prot.rs b/userland/capsule_linux/src/linux/guest/region_prot.rs
new file mode 100644
index 000000000..705ddf967
--- /dev/null
+++ b/userland/capsule_linux/src/linux/guest/region_prot.rs
@@ -0,0 +1,46 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Recording a protection change on the spans a guest holds.
+
+use alloc::vec::Vec;
+
+use super::handle::Guest;
+use super::region::Region;
+use super::region_cut::cut;
+
+impl Guest {
+ /// Every backed part of `[at, at + len)` now has this protection. The
+ /// list is what fork maps the child from, so a list that kept the old
+ /// protection would give the child access the parent gave up.
+ pub fn set_prot(&mut self, at: u64, len: u64, write: bool, exec: bool, access: bool) {
+ let end = at.saturating_add(len);
+ let changed: Vec = self
+ .regions
+ .iter()
+ .filter(|r| r.backed && r.at < end && at < r.at.saturating_add(r.len))
+ .map(|r| {
+ let from = r.at.max(at);
+ let to = r.at.saturating_add(r.len).min(end);
+ Region { at: from, len: to - from, write, exec, access, ..*r }
+ })
+ .collect();
+ for piece in &changed {
+ self.regions = cut(&self.regions, piece.at, piece.len);
+ }
+ self.regions.extend(changed);
+ }
+}
diff --git a/userland/capsule_linux/src/linux/serve/table_mem.rs b/userland/capsule_linux/src/linux/serve/table_mem.rs
index 4ca19272b..71e3cf67c 100644
--- a/userland/capsule_linux/src/linux/serve/table_mem.rs
+++ b/userland/capsule_linux/src/linux/serve/table_mem.rs
@@ -27,6 +27,12 @@ pub fn mem_ops(guest: &mut Guest, nr: u64, a: [u64; 6]) -> Option {
nr::MUNMAP => call::munmap(guest, a[0], a[1]),
nr::MPROTECT => call::mprotect(guest, a[0], a[1], a[2]),
nr::MREMAP => call::mremap(guest, a[0], a[1], a[2], a[3]),
+ nr::MSYNC => call::mem::msync(guest, a[0], a[1], a[2]),
+ nr::MINCORE => call::mem::mincore(guest, a[0], a[1], a[2]),
+ nr::MLOCK | nr::MUNLOCK => call::mem::mlock(guest, a[0], a[1]),
+ nr::MLOCK2 => call::mem::mlock2(guest, a[0], a[1], a[2]),
+ nr::MLOCKALL => call::mem::mlockall(a[0]),
+ nr::MUNLOCKALL => errno::ok(0),
// Advice, and this capsule takes none of it.
nr::MADVISE => errno::ok(0),
_ => return None,
diff --git a/userland/libc/src/peer.rs b/userland/libc/src/peer.rs
index 988f4cff9..29bb9ed16 100644
--- a/userland/libc/src/peer.rs
+++ b/userland/libc/src/peer.rs
@@ -25,6 +25,9 @@ use crate::syscall::{
/// Pages of a guest may be written, and may be executed.
pub const PEER_PROT_WRITE: u64 = 1 << 0;
pub const PEER_PROT_EXEC: u64 = 1 << 1;
+/// Pages the guest may not touch at all, their bytes kept for a later
+/// protection that opens them: PROT_NONE.
+pub const PEER_PROT_NONE: u64 = 1 << 2;
/// Back a span of a guest's address space with fresh zeroed frames.
pub fn mk_peer_map(pid: u32, addr: u64, len: u64, prot: u64) -> i64 {
diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk
index c42c8e873..3848f6f20 100644
--- a/userland/linux_guests/Guests.mk
+++ b/userland/linux_guests/Guests.mk
@@ -119,6 +119,32 @@ $(LINUX_GUESTS_C)/cwait: $(LINUX_GUESTS_DIR)/c/cwait.c
@mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $<
$(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait))
+# The memory proofs, one program whose first argument names the proof, so the
+# store carries one binary and one set of proofs for all five:
+# guardpage a pthread recursing into its guard page ends on SIGSEGV, 139
+# protnone PROT_NONE means no access, and bytes survive a close and reopen
+# protfork a fork after mprotect gives the child the protection set now
+# touchfork bytes written into a reservation survive a fork; MAP_FIXED
+# over a written page replaces it with zeroes
+# memcalls mmap placement, brk, alignment, mremap, and mlock, msync and
+# mincore, each against Linux's answer
+MEMPROOF_PARTS := guardpage protnone protfork touchfork memcalls escape
+# Files the proofs share, built as they are: no main of their own.
+MEMPROOF_SHARED := memproof_run memcalls_map memcalls_remap memcalls_lock escape_map escape_exec
+MEMPROOF_SRCS := $(foreach p,memproof $(MEMPROOF_PARTS) $(MEMPROOF_SHARED),\
+ $(LINUX_GUESTS_DIR)/c/$(p).c) $(LINUX_GUESTS_DIR)/c/memproof.h $(LINUX_GUESTS_DIR)/c/memcalls.h $(LINUX_GUESTS_DIR)/c/escape.h
+$(LINUX_GUESTS_C)/memproof: $(MEMPROOF_SRCS)
+ @mkdir -p $(@D)/memproof.o
+ @for p in $(MEMPROOF_PARTS); do \
+ musl-gcc -O2 -c -Dmain=$${p}_main -o $(@D)/memproof.o/$$p.o $(LINUX_GUESTS_DIR)/c/$$p.c || exit 1; \
+ done
+ @for p in $(MEMPROOF_SHARED); do \
+ musl-gcc -O2 -c -o $(@D)/memproof.o/$$p.o $(LINUX_GUESTS_DIR)/c/$$p.c || exit 1; \
+ done
+ @musl-gcc -O2 -static -o $@ $(LINUX_GUESTS_DIR)/c/memproof.c \
+ $(foreach p,$(MEMPROOF_PARTS) $(MEMPROOF_SHARED),$(@D)/memproof.o/$(p).o)
+$(eval $(call LINUX_GUEST,memproof,4980,4981,$(LINUX_GUESTS_C)/memproof))
+
# The Linux-guest test store is about guests, not the desktop's media and demo
# capsules. Drop both so the signed guest set fits the vfs load budget; the
# normal image, which does not set NONOS_LINUX_GUESTS, still ships them.
diff --git a/userland/linux_guests/c/escape.c b/userland/linux_guests/c/escape.c
new file mode 100644
index 000000000..4253fb8e0
--- /dev/null
+++ b/userland/linux_guests/c/escape.c
@@ -0,0 +1,31 @@
+/*
+ * A Linux guest trying to break out of its own address space through the
+ * syscall surface it is given: reach the kernel half, wrap a span, map
+ * write-and-execute, add execute to a file it never proved, call a number
+ * that is not served, and step past the end of what it holds. Each part
+ * passes when the machine refuses; one boot names any that got through.
+ */
+#include
+#include
+
+#include "escape.h"
+
+void held(const char *name, int ok, long got) {
+ char detail[48];
+ snprintf(detail, sizeof detail, "got %ld", got);
+ part_line("escape", name, ok, detail);
+}
+
+long erc(long v) {
+ return v == -1 ? -errno : v;
+}
+
+int main(void) {
+ reach_kernel();
+ wrap_span();
+ wx_map();
+ exec_escalate();
+ forged_call();
+ past_end();
+ return finish("escape");
+}
diff --git a/userland/linux_guests/c/escape.h b/userland/linux_guests/c/escape.h
new file mode 100644
index 000000000..355e7253d
--- /dev/null
+++ b/userland/linux_guests/c/escape.h
@@ -0,0 +1,19 @@
+/* What the escape attack files share. Each part passes when NONOS refuses. */
+#ifndef ESCAPE_H
+#define ESCAPE_H
+
+#include "memproof.h"
+
+/* Record an attack: ok means the machine held (refused it). */
+void held(const char *name, int ok, long got);
+/* -errno of a call (escape) that returned -1, or its value. */
+long erc(long v);
+
+void reach_kernel(void);
+void wrap_span(void);
+void wx_map(void);
+void exec_escalate(void);
+void forged_call(void);
+void past_end(void);
+
+#endif
diff --git a/userland/linux_guests/c/escape_exec.c b/userland/linux_guests/c/escape_exec.c
new file mode 100644
index 000000000..5fa27175a
--- /dev/null
+++ b/userland/linux_guests/c/escape_exec.c
@@ -0,0 +1,50 @@
+/* Attacks on what a guest may run and call: adding execute to a file it never
+ * proved, a syscall number that is not served, and a step past a mapping. */
+#define _GNU_SOURCE
+#include
+#include
+#include
+#include
+#include
+
+#include "escape.h"
+
+static volatile char *edge;
+
+void exec_escalate(void) {
+ /* Map this program read-only, then try to make it executable. It was
+ * mapped without execute, so it was never proved; NONOS refuses. */
+ int fd = open("/bin/memproof", O_RDONLY);
+ if (fd < 0) {
+ fd = open("/proc/self/exe", O_RDONLY);
+ }
+ if (fd < 0) {
+ held("exec added to an unproven file mapping is refused", 0, -errno);
+ return;
+ }
+ void *p = mmap(0, PG, PROT_READ, MAP_PRIVATE, fd, 0);
+ close(fd);
+ if (p == MAP_FAILED) {
+ held("exec added to an unproven file mapping is refused", 0, -errno);
+ return;
+ }
+ long r = erc(mprotect(p, PG, PROT_READ | PROT_EXEC));
+ held("exec added to an unproven file mapping is refused", r == -EPERM, r);
+}
+
+void forged_call(void) {
+ /* A syscall number the personality does not serve. */
+ long r = erc(syscall(0x462));
+ held("an unserved syscall number answers ENOSYS", r == -ENOSYS, r);
+}
+
+static void step_past(void) {
+ edge[PG] = 1;
+}
+
+void past_end(void) {
+ /* Two pages, second given back, so the page written is a certain hole. */
+ edge = mmap(0, 2 * PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ munmap((void *)(edge + PG), PG);
+ part("escape", "a write into an unmapped hole faults", step_past, 1);
+}
diff --git a/userland/linux_guests/c/escape_map.c b/userland/linux_guests/c/escape_map.c
new file mode 100644
index 000000000..0cecc8ea3
--- /dev/null
+++ b/userland/linux_guests/c/escape_map.c
@@ -0,0 +1,40 @@
+/* Attacks on where and how a guest may map: the kernel half, a wrapping
+ * span, and a page that is both writable and executable. */
+#include
+#include
+#include
+
+#include "escape.h"
+
+/* The first address of the kernel half, which no guest mapping may reach. */
+#define KERNEL_HALF 0x0000800000000000UL
+
+static int fixed_refused(uintptr_t at) {
+ void *p = mmap((void *)at, PG, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+ if (p == MAP_FAILED) {
+ return 1;
+ }
+ /* It answered with an address; it must at least not be the one asked. */
+ munmap(p, PG);
+ return (uintptr_t)p != at;
+}
+
+void reach_kernel(void) {
+ held("MAP_FIXED into the kernel half is refused", fixed_refused(KERNEL_HALF), 0);
+ held("MAP_FIXED one page below the kernel half is refused", fixed_refused(KERNEL_HALF - PG), 0);
+}
+
+void wrap_span(void) {
+ /* A length that wraps past the top of the address space. */
+ void *p = mmap(0, (size_t)-PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ long got = p == MAP_FAILED ? -errno : 0;
+ held("a mapping whose length wraps is refused", p == MAP_FAILED, got);
+}
+
+void wx_map(void) {
+ /* Write and execute at once: NONOS refuses it, Linux allows it. */
+ void *p = mmap(0, PG, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ long got = p == MAP_FAILED ? -errno : 0;
+ held("a write-and-execute mapping is refused", p == MAP_FAILED, got);
+}
diff --git a/userland/linux_guests/c/guardpage.c b/userland/linux_guests/c/guardpage.c
new file mode 100644
index 000000000..34e2ad1d8
--- /dev/null
+++ b/userland/linux_guests/c/guardpage.c
@@ -0,0 +1,66 @@
+/*
+ * A pthread recurses until it walks off the bottom of its stack. musl reserves
+ * each thread stack with PROT_NONE and opens all but the lowest part with
+ * mprotect, so the part it leaves closed is the guard. On Linux the first touch
+ * of the guard is SIGSEGV, which ends the whole process with status 139. The
+ * recursion stops by itself 64 KiB below the guard, so a guard that guards
+ * nothing prints the FAIL line with how far it ran; no PASS line exists, since
+ * the only correct outcome is that the process does not get to print one.
+ */
+#define _GNU_SOURCE
+#include
+#include
+#include
+#include
+#include
+
+#include "memproof.h"
+
+#define FRAME 1024
+#define PAST (64 * 1024)
+
+static uintptr_t lo;
+static uintptr_t deepest;
+
+static unsigned dive(unsigned depth) {
+ volatile char frame[FRAME];
+ uintptr_t here = (uintptr_t)frame;
+ frame[0] = (char)depth;
+ frame[FRAME - 1] = (char)depth;
+ deepest = here;
+ if (here < lo - PAST) {
+ return depth;
+ }
+ return dive(depth + 1) + frame[0] - frame[FRAME - 1];
+}
+
+static void *worker(void *arg) {
+ (void)arg;
+ pthread_attr_t a;
+ void *base;
+ size_t size, guard;
+ char line[160];
+ pthread_getattr_np(pthread_self(), &a);
+ pthread_attr_getstack(&a, &base, &size);
+ pthread_attr_getguardsize(&a, &guard);
+ lo = (uintptr_t)base;
+ snprintf(line, sizeof line, "[C] guardpage: stack %zu KiB, guard %zu KiB below 0x%lx; recursing\n",
+ size / 1024, guard / 1024, (unsigned long)lo);
+ say(line);
+ unsigned depth = dive(0);
+ snprintf(line, sizeof line,
+ "[C] guardpage FAIL: %u frames, ran %lu KiB below the stack with no fault\n", depth,
+ (unsigned long)((lo - deepest) / 1024));
+ say(line);
+ return 0;
+}
+
+int main(void) {
+ pthread_t t;
+ if (pthread_create(&t, 0, worker, 0) != 0) {
+ say("[C] guardpage FAIL: no thread\n");
+ return 1;
+ }
+ pthread_join(t, 0);
+ return 1;
+}
diff --git a/userland/linux_guests/c/memcalls.c b/userland/linux_guests/c/memcalls.c
new file mode 100644
index 000000000..437cd19dd
--- /dev/null
+++ b/userland/linux_guests/c/memcalls.c
@@ -0,0 +1,61 @@
+/*
+ * The memory calls answered as Linux answers them: where mmap puts a hint, the
+ * break giving pages back, unaligned addresses refused, mremap keeping a
+ * mapping's protection, and mlock, msync and mincore with their errnos. Every
+ * part runs and prints one line, so one boot names every part that fails.
+ * Parts that must fault run in a forked child.
+ */
+#include
+#include
+#include
+#include
+#include
+
+#include "memcalls.h"
+
+volatile char *mc_p;
+
+void check(const char *name, int ok, long got) {
+ char detail[48];
+ snprintf(detail, sizeof detail, "got %ld", got);
+ part_line("memcalls", name, ok, detail);
+}
+
+void faults(const char *name, void (*fn)(void)) {
+ pid_t c = fork();
+ if (c == 0) {
+ fn();
+ _exit(0);
+ }
+ int st = 0;
+ waitpid(c, &st, 0);
+ check(name, segv(st), st);
+}
+
+long rc(long v) {
+ return v == -1 ? -errno : v;
+}
+
+char *anon(long len, int prot) {
+ return mmap(0, len, prot, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+}
+
+void mc_write(void) {
+ mc_p[0] = 1;
+}
+
+void mc_read(void) {
+ (void)mc_p[0];
+}
+
+int main(void) {
+ placement();
+ breaks();
+ alignment();
+ remaps();
+ provenance();
+ locks();
+ syncs();
+ cores();
+ return finish("memcalls");
+}
diff --git a/userland/linux_guests/c/memcalls.h b/userland/linux_guests/c/memcalls.h
new file mode 100644
index 000000000..483923926
--- /dev/null
+++ b/userland/linux_guests/c/memcalls.h
@@ -0,0 +1,30 @@
+/* What the memcalls files share. */
+#ifndef MEMCALLS_H
+#define MEMCALLS_H
+
+#include "memproof.h"
+
+#define NOREPLACE 0x100000
+
+/* The page a part run in a child touches. */
+extern volatile char *mc_p;
+
+void check(const char *name, int ok, long got);
+/* A part run in a forked child that must die of SIGSEGV. */
+void faults(const char *name, void (*fn)(void));
+/* -errno of a call that returned -1, or its value. */
+long rc(long v);
+char *anon(long len, int prot);
+void mc_write(void);
+void mc_read(void);
+
+void placement(void);
+void breaks(void);
+void alignment(void);
+void remaps(void);
+void provenance(void);
+void locks(void);
+void syncs(void);
+void cores(void);
+
+#endif
diff --git a/userland/linux_guests/c/memcalls_lock.c b/userland/linux_guests/c/memcalls_lock.c
new file mode 100644
index 000000000..b8b74ce02
--- /dev/null
+++ b/userland/linux_guests/c/memcalls_lock.c
@@ -0,0 +1,55 @@
+/* memcalls: mlock, msync and mincore with Linux's errnos. */
+#define _GNU_SOURCE
+#include
+#include
+#include
+#include
+#include
+
+#include "memcalls.h"
+
+void locks(void) {
+ char *m = anon(3 * PG, PROT_READ | PROT_WRITE);
+ check("mlock of a mapping", rc(mlock(m, 3 * PG)) == 0, rc(mlock(m, 3 * PG)));
+ check("munlock of a mapping", rc(munlock(m, 3 * PG)) == 0, rc(munlock(m, 3 * PG)));
+ munmap(m + PG, PG);
+ check("mlock over a hole is ENOMEM", rc(mlock(m, 3 * PG)) == -ENOMEM, rc(mlock(m, 3 * PG)));
+ check("mlock2 with an unknown flag is EINVAL", rc(syscall(SYS_mlock2, m, PG, 2)) == -EINVAL,
+ rc(syscall(SYS_mlock2, m, PG, 2)));
+ check("mlock2 MLOCK_ONFAULT", rc(syscall(SYS_mlock2, m, PG, 1)) == 0,
+ rc(syscall(SYS_mlock2, m, PG, 1)));
+ check("mlockall(0) is EINVAL", rc(mlockall(0)) == -EINVAL, rc(mlockall(0)));
+ check("mlockall(MCL_ONFAULT) alone is EINVAL", rc(mlockall(MCL_ONFAULT)) == -EINVAL,
+ rc(mlockall(MCL_ONFAULT)));
+ check("mlockall(MCL_CURRENT)", rc(mlockall(MCL_CURRENT)) == 0, rc(mlockall(MCL_CURRENT)));
+ check("munlockall", rc(munlockall()) == 0, rc(munlockall()));
+}
+
+void syncs(void) {
+ char *m = anon(3 * PG, PROT_READ | PROT_WRITE);
+ check("msync MS_SYNC", rc(msync(m, 3 * PG, MS_SYNC)) == 0, rc(msync(m, 3 * PG, MS_SYNC)));
+ check("msync unaligned is EINVAL", rc(msync(m + 1, PG, MS_SYNC)) == -EINVAL,
+ rc(msync(m + 1, PG, MS_SYNC)));
+ check("msync MS_ASYNC|MS_SYNC is EINVAL", rc(msync(m, PG, MS_ASYNC | MS_SYNC)) == -EINVAL,
+ rc(msync(m, PG, MS_ASYNC | MS_SYNC)));
+ check("msync unknown flag is EINVAL", rc(msync(m, PG, 8)) == -EINVAL, rc(msync(m, PG, 8)));
+ munmap(m + PG, PG);
+ check("msync over a hole is ENOMEM", rc(msync(m, 3 * PG, MS_SYNC)) == -ENOMEM,
+ rc(msync(m, 3 * PG, MS_SYNC)));
+}
+
+void cores(void) {
+ char *r = anon(4 * PG, PROT_NONE);
+ mprotect(r + PG, PG, PROT_READ | PROT_WRITE);
+ r[PG] = 1;
+ unsigned char v[4] = { 9, 9, 9, 9 };
+ long got = rc(mincore(r, 4 * PG, v));
+ check("mincore of a reservation with one page opened is 0,1,0,0",
+ got == 0 && v[0] == 0 && v[1] == 1 && v[2] == 0 && v[3] == 0,
+ got ? got : v[0] | v[1] << 8 | v[2] << 16 | (long)v[3] << 24);
+ check("mincore unaligned is EINVAL", rc(mincore(r + 1, PG, v)) == -EINVAL,
+ rc(mincore(r + 1, PG, v)));
+ munmap(r + 2 * PG, PG);
+ check("mincore over a hole is ENOMEM", rc(mincore(r, 4 * PG, v)) == -ENOMEM,
+ rc(mincore(r, 4 * PG, v)));
+}
diff --git a/userland/linux_guests/c/memcalls_map.c b/userland/linux_guests/c/memcalls_map.c
new file mode 100644
index 000000000..2adb3be11
--- /dev/null
+++ b/userland/linux_guests/c/memcalls_map.c
@@ -0,0 +1,66 @@
+/* memcalls: where mmap puts a mapping, the break, and unaligned addresses. */
+#define _GNU_SOURCE
+#include
+#include
+#include
+#include
+#include
+
+#include "memcalls.h"
+
+void placement(void) {
+ char *m = anon(PG, PROT_READ | PROT_WRITE);
+ m[0] = 0x11;
+ char *n = mmap(m, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ check("hint on a mapping lands elsewhere, zeroed", n != m && n[0] == 0 && m[0] == 0x11,
+ (long)(n - m));
+ void *q = mmap(m, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0);
+ long e = q == MAP_FAILED ? -errno : 0;
+ check("MAP_FIXED_NOREPLACE on a mapping is EEXIST", e == -EEXIST && m[0] == 0x11, e);
+ /*
+ * A mapping placed just above the last one mmap chose: the next mmap that
+ * leaves the choice to the system must not land on it. On a system that
+ * already holds that address the probe cannot be placed, and says so.
+ */
+ char *a = anon(PG, PROT_READ | PROT_WRITE);
+ char *f = mmap(a + PG, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0);
+ if (f == MAP_FAILED) {
+ check("next mmap skips a mapping placed above the last (probe address taken)",
+ errno == EEXIST, -errno);
+ return;
+ }
+ f[0] = 0x44;
+ char *b = anon(PG, PROT_READ | PROT_WRITE);
+ check("next mmap skips a mapping placed above the last", b != f && b[0] == 0 && f[0] == 0x44,
+ (long)(b - f));
+}
+
+void breaks(void) {
+ long cur = syscall(SYS_brk, 0);
+ long up = syscall(SYS_brk, cur + 2 * PG);
+ ((volatile char *)cur)[PG] = 0x22;
+ syscall(SYS_brk, cur);
+ syscall(SYS_brk, cur + 2 * PG);
+ char b = ((volatile char *)cur)[PG];
+ check("brk down and up again reads zero", up == cur + 2 * PG && b == 0, b);
+ syscall(SYS_brk, cur);
+ long top = (cur + PG - 1) & ~(long)(PG - 1);
+ void *in = mmap((void *)(top + PG), PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0);
+ long got = syscall(SYS_brk, top + 4 * PG);
+ check("brk into a mapping is refused", in != MAP_FAILED && got == cur, got - cur);
+ munmap(in, PG);
+}
+
+void alignment(void) {
+ char *m = anon(2 * PG, PROT_READ | PROT_WRITE);
+ check("munmap unaligned is EINVAL", rc(munmap(m + 1, PG)) == -EINVAL, rc(munmap(m + 1, PG)));
+ /* musl's mprotect rounds the address down itself; the kernel's does not. */
+ long e = rc(syscall(SYS_mprotect, m + 1, PG, PROT_READ));
+ check("mprotect unaligned is EINVAL", e == -EINVAL, e);
+ void *q = mmap(m + 1, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+ e = q == MAP_FAILED ? -errno : 0;
+ check("MAP_FIXED unaligned is EINVAL", e == -EINVAL, e);
+ /* musl's mmap refuses an unaligned offset itself; the kernel's must too. */
+ e = rc(syscall(SYS_mmap, 0, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS, -1, 1));
+ check("mmap offset unaligned is EINVAL", e == -EINVAL, e);
+}
diff --git a/userland/linux_guests/c/memcalls_remap.c b/userland/linux_guests/c/memcalls_remap.c
new file mode 100644
index 000000000..add9b6d0e
--- /dev/null
+++ b/userland/linux_guests/c/memcalls_remap.c
@@ -0,0 +1,56 @@
+/* memcalls: mremap keeping a mapping's protection, backing and provenance. */
+#define _GNU_SOURCE
+#include
+#include
+#include
+#include
+#include
+
+#include "memcalls.h"
+
+void remaps(void) {
+ char *m = anon(PG, PROT_READ | PROT_WRITE);
+ m[0] = 0x33;
+ mprotect(m, PG, PROT_READ);
+ char *g = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE);
+ check("mremap of a read-only page keeps the byte", g != MAP_FAILED && g[0] == 0x33, g[0]);
+ mc_p = g + PG;
+ faults("mremap grown part of a read-only page faults on write", mc_write);
+ char *r = anon(2 * PG, PROT_NONE);
+ anon(PG, PROT_NONE);
+ char *q = mremap(r, 2 * PG, 4 * PG, MREMAP_MAYMOVE);
+ check("mremap of a reservation succeeds", q != MAP_FAILED, (long)(q == MAP_FAILED));
+ mc_p = q + 3 * PG;
+ faults("mremap grown reservation still faults on read", mc_read);
+ char *two = anon(2 * PG, PROT_READ | PROT_WRITE);
+ mprotect(two + PG, PG, PROT_READ);
+ void *x = mremap(two, 2 * PG, 3 * PG, MREMAP_MAYMOVE);
+ long e = x == MAP_FAILED ? -errno : 0;
+ check("mremap across two mappings is EFAULT", e == -EFAULT, e);
+}
+
+/*
+ * A file mapped without exec was never proved; where mprotect refuses to make
+ * it executable, it must refuse the copy mremap moved too. Host Linux allows
+ * both, and the part checks only that the two answers agree.
+ */
+void provenance(void) {
+ /* This program's own file: /bin/memproof in the store, itself on a host. */
+ int fd = open("/bin/memproof", O_RDONLY);
+ if (fd < 0) {
+ fd = open("/proc/self/exe", O_RDONLY);
+ }
+ char *m = mmap(0, PG, PROT_READ, MAP_PRIVATE, fd, 0);
+ if (m == MAP_FAILED) {
+ check("file mapping for the provenance part", 0, -errno);
+ return;
+ }
+ long first = rc(mprotect(m, PG, PROT_READ | PROT_EXEC));
+ mprotect(m, PG, PROT_READ);
+ mmap(m + PG, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0);
+ char *n = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE);
+ long moved = n == MAP_FAILED ? -1000 : rc(mprotect(n, PG, PROT_READ | PROT_EXEC));
+ check("moved unproven file bytes refused exec as before the move", n != m && moved == first,
+ moved);
+ close(fd);
+}
diff --git a/userland/linux_guests/c/memproof.c b/userland/linux_guests/c/memproof.c
new file mode 100644
index 000000000..2f154eccf
--- /dev/null
+++ b/userland/linux_guests/c/memproof.c
@@ -0,0 +1,39 @@
+/*
+ * The memory proofs as one program, so the test store carries one binary and
+ * one set of proofs for all of them instead of one per proof: the store has a
+ * fixed load budget and each proof set is most of a guest's size there. The
+ * first argument names the proof; each is its own file, built with its main
+ * renamed, and runs exactly as it would as a program of its own.
+ */
+#include
+#include
+
+int guardpage_main(void);
+int protnone_main(void);
+int protfork_main(void);
+int touchfork_main(void);
+int memcalls_main(void);
+int escape_main(void);
+
+static const struct {
+ const char *name;
+ int (*run)(void);
+} proofs[] = {
+ { "guardpage", guardpage_main },
+ { "protnone", protnone_main },
+ { "protfork", protfork_main },
+ { "touchfork", touchfork_main },
+ { "memcalls", memcalls_main },
+ { "escape", escape_main },
+};
+
+int main(int argc, char **argv) {
+ for (unsigned i = 0; argc > 1 && i < sizeof proofs / sizeof proofs[0]; i++) {
+ if (strcmp(argv[1], proofs[i].name) == 0) {
+ return proofs[i].run();
+ }
+ }
+ fputs("[C] memproof FAIL: name a proof: guardpage protnone protfork touchfork memcalls escape\n",
+ stdout);
+ return 2;
+}
diff --git a/userland/linux_guests/c/memproof.h b/userland/linux_guests/c/memproof.h
new file mode 100644
index 000000000..69dd07152
--- /dev/null
+++ b/userland/linux_guests/c/memproof.h
@@ -0,0 +1,27 @@
+/*
+ * What every memory proof shares: printing a line, running a part in a forked
+ * child, reading how the child ended, and counting parts for the last line.
+ */
+#ifndef MEMPROOF_H
+#define MEMPROOF_H
+
+#define PG 4096
+
+void say(const char *s);
+
+/*
+ * A SIGSEGV death. The personality reports a signal death as exit status
+ * 128+signo, which counts as the same SIGSEGV; the raw status is printed.
+ */
+int segv(int st);
+
+/* Count one part and print its line: "[C] : ok|FAIL ()". */
+void part_line(const char *proof, const char *name, int ok, const char *detail);
+
+/* Run `fn` in a forked child that must, or must not, die of SIGSEGV. */
+void part(const char *proof, const char *name, void (*fn)(void), int must_fault);
+
+/* Print the PASS or FAIL line and give the exit code. */
+int finish(const char *proof);
+
+#endif
diff --git a/userland/linux_guests/c/memproof_run.c b/userland/linux_guests/c/memproof_run.c
new file mode 100644
index 000000000..4d6a1100f
--- /dev/null
+++ b/userland/linux_guests/c/memproof_run.c
@@ -0,0 +1,53 @@
+/* The parts of memproof.h every proof shares. */
+#include
+#include
+#include
+#include
+#include
+
+#include "memproof.h"
+
+static int passed, failed;
+
+void say(const char *s) {
+ write(1, s, strlen(s));
+}
+
+int segv(int st) {
+ return (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) ||
+ (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV);
+}
+
+void part_line(const char *proof, const char *name, int ok, const char *detail) {
+ char line[200];
+ ok ? passed++ : failed++;
+ snprintf(line, sizeof line, "[C] %s %s: %s (%s)\n", proof, name, ok ? "ok" : "FAIL", detail);
+ say(line);
+}
+
+void part(const char *proof, const char *name, void (*fn)(void), int must_fault) {
+ char detail[64];
+ pid_t c = fork();
+ if (c < 0) {
+ part_line(proof, name, 0, "fork failed");
+ return;
+ }
+ if (c == 0) {
+ fn();
+ _exit(0);
+ }
+ int st = 0;
+ waitpid(c, &st, 0);
+ int clean = WIFEXITED(st) && WEXITSTATUS(st) == 0;
+ snprintf(detail, sizeof detail, "%s, status 0x%x", must_fault ? "must fault" : "must not fault",
+ st);
+ part_line(proof, name, must_fault ? segv(st) : clean, detail);
+}
+
+int finish(const char *proof) {
+ char line[160];
+ snprintf(line, sizeof line, "[C] %s %s: %d parts ok, %d failed\n", proof,
+ failed ? "FAIL" : "PASS", passed, failed);
+ say(line);
+ return failed ? 1 : 0;
+}
diff --git a/userland/linux_guests/c/protfork.c b/userland/linux_guests/c/protfork.c
new file mode 100644
index 000000000..5f4060322
--- /dev/null
+++ b/userland/linux_guests/c/protfork.c
@@ -0,0 +1,68 @@
+/*
+ * A fork gives the child the parent's mappings with the protection they have
+ * now, not the one they were made with. Each part changes a protection with
+ * mprotect, forks, and the child tries one access; the parent reads how the
+ * child ended.
+ */
+#include
+#include
+
+#include "memproof.h"
+
+static volatile char *p;
+
+static void write_it(void) {
+ p[0] = 1;
+ if (p[0] != 1) {
+ _exit(2);
+ }
+}
+static void read_it(void) {
+ if (p[0] != 0x5a) {
+ _exit(2);
+ }
+}
+
+static volatile char *fresh(int pages) {
+ volatile char *m =
+ mmap(0, pages * PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ for (int i = 0; i < pages; i++) {
+ m[i * PG] = 0x5a;
+ }
+ return m;
+}
+
+int main(void) {
+ const char *me = "protfork";
+ p = fresh(1);
+ mprotect((void *)p, PG, PROT_READ);
+ part(me, "RW to R, child writes", write_it, 1);
+ part(me, "RW to R, child reads the byte", read_it, 0);
+
+ p = fresh(1);
+ mprotect((void *)p, PG, PROT_NONE);
+ part(me, "RW to NONE, child reads", read_it, 1);
+
+ p = fresh(1);
+ mprotect((void *)p, PG, PROT_READ);
+ mprotect((void *)p, PG, PROT_READ | PROT_WRITE);
+ part(me, "RW to R to RW, child writes", write_it, 0);
+
+ volatile char *m = fresh(3);
+ mprotect((void *)(m + PG), PG, PROT_READ);
+ p = m;
+ part(me, "middle page R, child writes the first", write_it, 0);
+ p = m + PG;
+ part(me, "middle page R, child writes the middle", write_it, 1);
+ p = m + 2 * PG;
+ part(me, "middle page R, child writes the last", write_it, 0);
+ /* Larger than the kernel's 1 MiB per peer call, so fork copies it in pieces. */
+ long big = 2 * 1024 * 1024;
+ p = mmap(0, big, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ p += big - PG;
+ p[0] = 0x5a;
+ part(me, "2 MiB RW, child reads the last page", read_it, 0);
+ mprotect((void *)(p - big + PG), big, PROT_NONE);
+ part(me, "2 MiB RW to NONE, child reads the last page", read_it, 1);
+ return finish(me);
+}
diff --git a/userland/linux_guests/c/protnone.c b/userland/linux_guests/c/protnone.c
new file mode 100644
index 000000000..dfc9f5275
--- /dev/null
+++ b/userland/linux_guests/c/protnone.c
@@ -0,0 +1,47 @@
+/*
+ * PROT_NONE means no access. Each part runs in a forked child and the parent
+ * reads how the child ended: a part that must fault passes only when the child
+ * dies of SIGSEGV, a part that must not fault passes only when it exits 0.
+ * Every part runs, so one boot names every part that fails.
+ */
+#include
+#include
+
+#include "memproof.h"
+
+static volatile char *p;
+
+static void read_it(void) {
+ if (p[0] != 0x5a) {
+ _exit(2);
+ }
+}
+static void write_it(void) {
+ p[0] = 1;
+}
+static void read_below(void) {
+ (void)p[-1];
+}
+
+int main(void) {
+ const char *me = "protnone";
+ p = mmap(0, PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ part(me, "read of an mmap PROT_NONE page", read_it, 1);
+ part(me, "write to an mmap PROT_NONE page", write_it, 1);
+
+ p = mmap(0, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ p[0] = 0x5a;
+ mprotect((void *)p, PG, PROT_NONE);
+ part(me, "read after mprotect RW to PROT_NONE", read_it, 1);
+ mprotect((void *)p, PG, PROT_READ);
+ part(me, "read after PROT_NONE back to R keeps the byte", read_it, 0);
+ part(me, "write to a PROT_READ page", write_it, 1);
+
+ char *r = mmap(0, 3 * PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ mprotect(r + PG, PG, PROT_READ | PROT_WRITE);
+ p = (volatile char *)(r + PG);
+ p[0] = 0x5a;
+ part(me, "read of the opened page of a reservation", read_it, 0);
+ part(me, "read of the closed page below it", read_below, 1);
+ return finish(me);
+}
diff --git a/userland/linux_guests/c/touchfork.c b/userland/linux_guests/c/touchfork.c
new file mode 100644
index 000000000..447204f89
--- /dev/null
+++ b/userland/linux_guests/c/touchfork.c
@@ -0,0 +1,57 @@
+/*
+ * Bytes a guest wrote into a reservation survive a fork. A reservation is a
+ * PROT_NONE mapping; a program opens part of it with mprotect, writes, and may
+ * close it again. Linux keeps the bytes through all of that and gives the child
+ * a copy. Touching a page never opened is SIGSEGV, in the child as anywhere.
+ * MAP_FIXED over a mapping replaces it: the new pages read zero, as Linux says.
+ */
+#include
+#include
+
+#include "memproof.h"
+
+#define PAGES 16
+
+static volatile char *r;
+
+static void check_bytes(void) {
+ for (int i = 4; i < 8; i++) {
+ if (r[i * PG] != (char)(0x40 + i) || r[i * PG + PG - 1] != (char)(0x50 + i)) {
+ _exit(2);
+ }
+ }
+}
+static void open_then_check(void) {
+ mprotect((void *)(r + 4 * PG), 4 * PG, PROT_READ);
+ check_bytes();
+}
+static void touch_unopened(void) {
+ (void)r[0];
+}
+static void check_zero(void) {
+ if (r[0] != 0) {
+ _exit(2);
+ }
+}
+
+int main(void) {
+ const char *me = "touchfork";
+ int anon = MAP_PRIVATE | MAP_ANONYMOUS;
+ r = mmap(0, PAGES * PG, PROT_NONE, anon, -1, 0);
+ mprotect((void *)(r + 4 * PG), 4 * PG, PROT_READ | PROT_WRITE);
+ for (int i = 4; i < 8; i++) {
+ r[i * PG] = (char)(0x40 + i);
+ r[i * PG + PG - 1] = (char)(0x50 + i);
+ }
+ part(me, "opened part of a reservation, child reads the bytes", check_bytes, 0);
+ mprotect((void *)(r + 4 * PG), 4 * PG, PROT_NONE);
+ part(me, "closed again, child opens it and reads the bytes", open_then_check, 0);
+ part(me, "child touches a page never opened", touch_unopened, 1);
+
+ r = mmap(0, PG, PROT_READ | PROT_WRITE, anon, -1, 0);
+ r[0] = 0x77;
+ mmap((void *)r, PG, PROT_NONE, anon | MAP_FIXED, -1, 0);
+ mprotect((void *)r, PG, PROT_READ);
+ part(me, "MAP_FIXED PROT_NONE over written page, child reads zero", check_zero, 0);
+ return finish(me);
+}