From e0482b2ef29d0895139c9d92341bd090904df52c Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 17:00:18 +0000 Subject: [PATCH 01/13] memory: fill no page for a foreign guest on its own The kernel demand-filled any user page a thread touched first, foreign guests included. A Linux guest's PROT_NONE reservation read as zeros, and a pthread's guard page, which musl leaves as the unopened bottom of a PROT_NONE stack reservation, took a fresh page and guarded nothing: a recursion ran straight through it. Every page a Linux guest is meant to have is already mapped by its supervisor with MkPeerMap: brk, the initial stack, the ELF segments, file maps, anonymous maps and mprotect commits. Demand fill only ever gave a guest pages nobody had mapped for it. Now a not-present fault in a foreign guest is refused. The fault path ends the thread with -11, the kernel posts the death to the supervisor as before, and the personality ends the process with status 139. The page tables stay the one record of what a guest holds; nothing new is kept in the kernel. guardpage, a musl pthread recursing into its guard page, proves it: the process ends on SIGSEGV with status 139 and never prints the line it prints when it runs 64 KiB below the stack. --- src/memory/paging/manager/faults/demand.rs | 11 +++- userland/linux_guests/Guests.mk | 5 ++ userland/linux_guests/c/guardpage.c | 66 ++++++++++++++++++++++ 3 files changed, 81 insertions(+), 1 deletion(-) create mode 100644 userland/linux_guests/c/guardpage.c diff --git a/src/memory/paging/manager/faults/demand.rs b/src/memory/paging/manager/faults/demand.rs index 8dc741b0ff..3ace5e0c3a 100644 --- a/src/memory/paging/manager/faults/demand.rs +++ b/src/memory/paging/manager/faults/demand.rs @@ -46,10 +46,19 @@ impl PagingManager { return Err(PagingError::UnhandledPageFault); } + // A foreign guest's pages are exactly the ones its supervisor mapped + // for it. Filling any other page would hand the guest memory nobody + // gave it: a PROT_NONE reservation, a guard page, a hole. So the fault + // is refused, the fault path ends the thread, and its supervisor is + // told and decides what that means for the guest. + let pid = crate::process::current_pid().unwrap_or(0); + if crate::process::foreign::is_foreign(pid) { + return Err(PagingError::UnhandledPageFault); + } + // Charge the page against the faulting process's demand budget. A // runaway capsule is refused here and killed by the fault path instead // of exhausting physical memory. - let pid = crate::process::current_pid().unwrap_or(0); if !super::demand_cap::charge(pid) { return Err(PagingError::UnhandledPageFault); } diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index c42c8e8735..52547185e8 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -118,6 +118,11 @@ $(eval $(call LINUX_GUEST,cthreads,4974,4975,$(LINUX_GUESTS_C)/cthreads)) $(LINUX_GUESTS_C)/cwait: $(LINUX_GUESTS_DIR)/c/cwait.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait)) +# A pthread recursing into its guard page: the process must end on SIGSEGV +# with status 139, and the line it prints if it runs past the guard never shows. +$(LINUX_GUESTS_C)/guardpage: $(LINUX_GUESTS_DIR)/c/guardpage.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,guardpage,4980,4981,$(LINUX_GUESTS_C)/guardpage)) # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the diff --git a/userland/linux_guests/c/guardpage.c b/userland/linux_guests/c/guardpage.c new file mode 100644 index 0000000000..716737e2f8 --- /dev/null +++ b/userland/linux_guests/c/guardpage.c @@ -0,0 +1,66 @@ +// A pthread recurses until it walks off the bottom of its stack. musl reserves +// each thread stack with PROT_NONE and opens all but the lowest part with +// mprotect, so the part it leaves closed is the guard. On Linux the first touch +// of the guard is SIGSEGV, which ends the whole process with status 139. The +// recursion stops by itself 64 KiB below the guard, so a guard that guards +// nothing prints the FAIL line with how far it ran; no PASS line exists, since +// the only correct outcome is that the process does not get to print one. +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#define FRAME 1024 +#define PAST (64 * 1024) + +static uintptr_t lo; +static uintptr_t deepest; + +static void say(const char *s) { + write(1, s, strlen(s)); +} + +static unsigned dive(unsigned depth) { + volatile char frame[FRAME]; + uintptr_t here = (uintptr_t)frame; + frame[0] = (char)depth; + frame[FRAME - 1] = (char)depth; + deepest = here; + if (here < lo - PAST) { + return depth; + } + return dive(depth + 1) + frame[0] - frame[FRAME - 1]; +} + +static void *worker(void *arg) { + (void)arg; + pthread_attr_t a; + void *base; + size_t size, guard; + char line[160]; + pthread_getattr_np(pthread_self(), &a); + pthread_attr_getstack(&a, &base, &size); + pthread_attr_getguardsize(&a, &guard); + lo = (uintptr_t)base; + snprintf(line, sizeof line, "[C] guardpage: stack %zu KiB, guard %zu KiB below 0x%lx; recursing\n", + size / 1024, guard / 1024, (unsigned long)lo); + say(line); + unsigned depth = dive(0); + snprintf(line, sizeof line, + "[C] guardpage FAIL: %u frames, ran %lu KiB below the stack with no fault\n", depth, + (unsigned long)((lo - deepest) / 1024)); + say(line); + return 0; +} + +int main(void) { + pthread_t t; + if (pthread_create(&t, 0, worker, 0) != 0) { + say("[C] guardpage FAIL: no thread\n"); + return 1; + } + pthread_join(t, 0); + return 1; +} From 19515257d5ae2bac744856bba2c84ec1e1bae9eb Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 17:01:17 +0000 Subject: [PATCH 02/13] linux: make PROT_NONE mean no access on pages a guest has The peer protection bits had write and exec and nothing for no access: MkPeerMap and MkPeerProtect with neither bit gave a present, readable user page. So mprotect(PROT_NONE) on a mapped span, and a PROT_NONE file mapping, left the pages readable, and the region list had no way to say a backed span was closed. MkPeerMap and MkPeerProtect now take PEER_PROT_NONE. The page stays present with the user bit clear: every guest access faults, the kernel still copies it at fork and frees it at teardown, and the bytes are there again when a later mprotect opens it, as Linux keeps them. A region records whether it allows access at all, and one function turns a region's protection into peer bits for map, commit, mprotect and the fork copy. protnone proves it part by part in forked children. --- src/process/foreign/peer_guard.rs | 4 + src/process/foreign/peer_map.rs | 9 ++- .../capsule_linux/src/linux/call/mem/prot.rs | 18 ++--- .../src/linux/call/mem/prot_span.rs | 16 ++-- .../src/linux/call/spawn/fork_copy.rs | 26 +++--- .../capsule_linux/src/linux/guest/mem_map.rs | 79 ++++++++++--------- userland/capsule_linux/src/linux/guest/mod.rs | 2 +- .../capsule_linux/src/linux/guest/region.rs | 33 +++++++- userland/libc/src/peer.rs | 3 + userland/linux_guests/Guests.mk | 7 ++ userland/linux_guests/c/protnone.c | 79 +++++++++++++++++++ 11 files changed, 195 insertions(+), 81 deletions(-) create mode 100644 userland/linux_guests/c/protnone.c diff --git a/src/process/foreign/peer_guard.rs b/src/process/foreign/peer_guard.rs index 4fc9aa275a..976db9d820 100644 --- a/src/process/foreign/peer_guard.rs +++ b/src/process/foreign/peer_guard.rs @@ -38,6 +38,10 @@ pub(super) fn in_user_half(addr: u64, len: u64) -> bool { pub const PROT_WRITE: u64 = 1 << 0; pub const PROT_EXEC: u64 = 1 << 1; +// No access from the guest at all. The page stays present with the user bit +// clear, so every guest access faults and the frame keeps its bytes for a +// later protection that allows access, as Linux keeps them. +pub(super) const PROT_NONE: u64 = 1 << 2; /// The pid a syscall argument names. Refused rather than truncated: `as u32` diff --git a/src/process/foreign/peer_map.rs b/src/process/foreign/peer_map.rs index 3740c267a6..06a66fe85d 100644 --- a/src/process/foreign/peer_map.rs +++ b/src/process/foreign/peer_map.rs @@ -21,13 +21,20 @@ use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid}; use crate::memory::paging::types::PagePermissions; use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_NOMEM}; -use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_WRITE}; +use super::peer_guard::{ + in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_NONE, PROT_WRITE, +}; fn span_ok(addr: u64, len: u64) -> bool { len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len) } pub(super) fn perms_of(prot: u64) -> PagePermissions { + // Not USER: present for the kernel, which copies it at fork and frees it + // at teardown, and absent for every access the guest makes. + if prot & PROT_NONE != 0 { + return PagePermissions::READ; + } let mut perms = PagePermissions::READ | PagePermissions::USER; if prot & PROT_WRITE != 0 { perms = perms | PagePermissions::WRITE; diff --git a/userland/capsule_linux/src/linux/call/mem/prot.rs b/userland/capsule_linux/src/linux/call/mem/prot.rs index 9d060a23ef..c25271dbb9 100644 --- a/userland/capsule_linux/src/linux/call/mem/prot.rs +++ b/userland/capsule_linux/src/linux/call/mem/prot.rs @@ -24,7 +24,7 @@ use super::prot_span::protect_span; pub const PROT_WRITE: u64 = 2; pub const PROT_EXEC: u64 = 4; /// PROT_READ, PROT_WRITE and PROT_EXEC together: any access at all. -const PROT_ANY: u64 = 7; +pub const PROT_ANY: u64 = 7; /// A request for both at once. pub fn wx_refused(prot: u64) -> bool { @@ -69,19 +69,15 @@ pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 { * A PROT_NONE reservation has no pages for the kernel to * reprotect. Asking for access commits it, which is how musl makes * a thread stack: reserve with PROT_NONE, then mprotect the part - * it uses to read-write. PROT_NONE on it changes nothing. + * it uses to read-write. PROT_NONE on it changes nothing. The + * commit maps the piece with `prot` and records it. */ - if prot & PROT_ANY == 0 { - at = upto; - continue; - } - if guest.commit(at, piece, prot & PROT_WRITE != 0, prot & PROT_EXEC != 0) < 0 { + if prot & PROT_ANY != 0 + && guest.commit(at, piece, prot & PROT_WRITE != 0, prot & PROT_EXEC != 0) < 0 + { return errno::fail(errno::ENOMEM); } - } - // Every page is present now; this sets `prot` on all of them, - // including any the guest touched while the span was reserved. - if protect_span(guest, at, piece, prot) < 0 { + } else if protect_span(guest, at, piece, prot) < 0 { return errno::fail(errno::EACCES); } at = upto; diff --git a/userland/capsule_linux/src/linux/call/mem/prot_span.rs b/userland/capsule_linux/src/linux/call/mem/prot_span.rs index aac1b7043a..e7fba9460e 100644 --- a/userland/capsule_linux/src/linux/call/mem/prot_span.rs +++ b/userland/capsule_linux/src/linux/call/mem/prot_span.rs @@ -16,21 +16,17 @@ //! Reprotecting a span, a peer call at a time. -use nonos_libc::peer::{mk_peer_protect, PEER_PROT_EXEC, PEER_PROT_WRITE}; +use nonos_libc::peer::mk_peer_protect; -use crate::linux::guest::{Guest, MAX_SPAN}; +use crate::linux::guest::{peer_prot, Guest, MAX_SPAN}; -use super::prot::{PROT_EXEC, PROT_WRITE}; +use super::prot::{PROT_ANY, PROT_EXEC, PROT_WRITE}; /// Set the protection of a span already mapped in the guest. pub fn protect_span(guest: &Guest, addr: u64, span: u64, prot: u64) -> i64 { - let mut bits = 0; - if prot & PROT_WRITE != 0 { - bits |= PEER_PROT_WRITE; - } - if prot & PROT_EXEC != 0 { - bits |= PEER_PROT_EXEC; - } + let (write, exec, access) = + (prot & PROT_WRITE != 0, prot & PROT_EXEC != 0, prot & PROT_ANY != 0); + let bits = peer_prot(write, exec, access); let mut done = 0; while done < span { let take = (span - done).min(MAX_SPAN); diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs index 3cf86191a5..fa114ebf0e 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs @@ -16,19 +16,24 @@ //! Copying a parent's spans into the child it just made. -use crate::linux::guest::{Guest, Region}; -use nonos_libc::peer::{mk_peer_map, mk_peer_write, PEER_PROT_EXEC, PEER_PROT_WRITE}; +use crate::linux::guest::Guest; +use nonos_libc::peer::{mk_peer_map, mk_peer_write}; /// Every span, mapped into the child and then filled from the parent. pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool { let spans = guest.regions.clone(); for span in spans { - // An unbacked reservation has no frames to copy; the child reserves it - // the same way, and its own first access faults a page in. + // An unbacked reservation has no frames to copy; the child holds the + // same reservation, and a touch there faults in the child as here. if !span.backed { continue; } - if mk_peer_map(child, span.at, span.len, prot_of(&span)) < 0 { + /* + * The protection the span has now, PROT_NONE included: the kernel + * copies into a page whatever its protection, so the bytes still go + * in, and the child can do no more with them than the parent can. + */ + if mk_peer_map(child, span.at, span.len, span.peer_prot()) < 0 { return false; } if !copy_one(guest, child, span.at, span.len) { @@ -38,17 +43,6 @@ pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool { true } -fn prot_of(span: &Region) -> u64 { - let mut prot = 0; - if span.write { - prot |= PEER_PROT_WRITE; - } - if span.exec { - prot |= PEER_PROT_EXEC; - } - prot -} - fn copy_one(guest: &Guest, child: u32, at: u64, len: u64) -> bool { let Some(bytes) = guest.read(at, len as usize) else { return false; diff --git a/userland/capsule_linux/src/linux/guest/mem_map.rs b/userland/capsule_linux/src/linux/guest/mem_map.rs index 83404d6f44..6b1f484dda 100644 --- a/userland/capsule_linux/src/linux/guest/mem_map.rs +++ b/userland/capsule_linux/src/linux/guest/mem_map.rs @@ -16,12 +16,12 @@ //! Backing a span of a guest with pages. -use nonos_libc::peer::{mk_peer_map, PEER_PROT_EXEC, PEER_PROT_WRITE}; +use nonos_libc::peer::mk_peer_map; use super::handle::Guest; use super::layout::USER_MAX; use super::mem::{span_within, MAX_SPAN}; -use super::region::Region; +use super::region::{peer_prot, Region}; use super::region_cut::cut; impl Guest { @@ -31,21 +31,9 @@ impl Guest { let Some((start, span)) = span_within(addr, len, USER_MAX) else { return -1; }; - let mut prot = 0; - if write { - prot |= PEER_PROT_WRITE; - } - if exec { - prot |= PEER_PROT_EXEC; - } - let mut done = 0; - while done < span { - let take = (span - done).min(MAX_SPAN); - let rc = mk_peer_map(self.pid, start + done, take, prot); - if rc < 0 { - return rc; - } - done += take; + let rc = map_span(self.pid, start, span, peer_prot(write, exec, true)); + if rc < 0 { + return rc; } /* * Remembered because fork copies a guest by walking what its @@ -56,6 +44,7 @@ impl Guest { len: span, write, exec, + access: true, unproven: false, backed: true, }); @@ -63,34 +52,31 @@ impl Guest { } /// Back `[at, at + len)` of a reservation with the given protection, the - /// commit a fixed mmap makes. Pages the guest has not touched get zeroed - /// frames; pages it has touched keep their contents, since peer_map skips - /// a page that is already there. The span is then recorded as backed, in - /// place of the reservation it came from, so fork copies it. + /// commit an mprotect that asks for access makes. The kernel fills no page + /// a guest touches on its own, so every page here is new and zeroed. The + /// span is then recorded as backed, in place of the reservation it came + /// from, so fork copies it. pub fn commit(&mut self, at: u64, len: u64, write: bool, exec: bool) -> i64 { - let mut prot = 0; - if write { - prot |= PEER_PROT_WRITE; - } - if exec { - prot |= PEER_PROT_EXEC; - } - let mut done = 0; - while done < len { - let take = (len - done).min(MAX_SPAN); - let rc = mk_peer_map(self.pid, at + done, take, prot); - if rc < 0 { - return rc; - } - done += take; + let rc = map_span(self.pid, at, len, peer_prot(write, exec, true)); + if rc < 0 { + return rc; } self.regions = cut(&self.regions, at, len); - self.regions.push(Region { at, len, write, exec, unproven: false, backed: true }); + self.regions.push(Region { + at, + len, + write, + exec, + access: true, + unproven: false, + backed: true, + }); 0 } /// Take `len` of address space at `addr` without backing it: a PROT_NONE - /// reservation. Bytes appear, zeroed, when the guest first touches them. + /// reservation. No page exists until a commit maps one; a touch before + /// that is a fault, as it is on Linux. pub fn reserve(&mut self, addr: u64, len: u64) -> i64 { let Some((start, span)) = span_within(addr, len, USER_MAX) else { return -1; @@ -98,11 +84,26 @@ impl Guest { self.regions.push(Region { at: start, len: span, - write: true, + write: false, exec: false, + access: false, unproven: false, backed: false, }); 0 } } + +/// `MkPeerMap` over a span, a megabyte at a time. +pub(super) fn map_span(pid: u32, at: u64, len: u64, prot: u64) -> i64 { + let mut done = 0; + while done < len { + let take = (len - done).min(MAX_SPAN); + let rc = mk_peer_map(pid, at + done, take, prot); + if rc < 0 { + return rc; + } + done += take; + } + 0 +} diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index 514a26bdd4..83f958019d 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -57,6 +57,6 @@ pub use layout::{ STACK_TOP, USER_MAX, }; pub use mem::{page_down, page_up, span_within, MAX_SPAN, PAGE}; -pub use region::Region; +pub use region::{peer_prot, Region}; pub use timer::Timer; pub use watch::{Watch, EPOLLET, EPOLLONESHOT}; diff --git a/userland/capsule_linux/src/linux/guest/region.rs b/userland/capsule_linux/src/linux/guest/region.rs index 8caf7b1a1f..9ed68d6c90 100644 --- a/userland/capsule_linux/src/linux/guest/region.rs +++ b/userland/capsule_linux/src/linux/guest/region.rs @@ -16,17 +16,44 @@ //! One span of a guest's address space, as this capsule laid it down. +use nonos_libc::peer::{PEER_PROT_EXEC, PEER_PROT_NONE, PEER_PROT_WRITE}; + #[derive(Clone, Copy)] pub struct Region { pub at: u64, pub len: u64, pub write: bool, pub exec: bool, + /// False for PROT_NONE: the guest may not touch the span at all. A backed + /// span keeps its pages and their bytes, present to the kernel only. + pub access: bool, /// File bytes mapped without exec, so never proved: mprotect may not /// make them executable later. pub unproven: bool, - /// False for a PROT_NONE reservation: address space taken, no frames yet. - /// The kernel demand-fills a page on first access, so reserving a large - /// span and committing a little costs only what is touched; fork skips it. + /// False for a PROT_NONE reservation: address space taken, no frames. + /// The kernel fills no page for a guest on its own, so a touch of one is a + /// fault; a commit maps the part asked for and records it backed. pub backed: bool, } + +impl Region { + /// The protection the kernel is asked to give this span's pages. + pub fn peer_prot(&self) -> u64 { + peer_prot(self.write, self.exec, self.access) + } +} + +/// Peer protection bits for an access, a write and an exec permission. +pub fn peer_prot(write: bool, exec: bool, access: bool) -> u64 { + if !access { + return PEER_PROT_NONE; + } + let mut prot = 0; + if write { + prot |= PEER_PROT_WRITE; + } + if exec { + prot |= PEER_PROT_EXEC; + } + prot +} diff --git a/userland/libc/src/peer.rs b/userland/libc/src/peer.rs index 988f4cff93..29bb9ed16d 100644 --- a/userland/libc/src/peer.rs +++ b/userland/libc/src/peer.rs @@ -25,6 +25,9 @@ use crate::syscall::{ /// Pages of a guest may be written, and may be executed. pub const PEER_PROT_WRITE: u64 = 1 << 0; pub const PEER_PROT_EXEC: u64 = 1 << 1; +/// Pages the guest may not touch at all, their bytes kept for a later +/// protection that opens them: PROT_NONE. +pub const PEER_PROT_NONE: u64 = 1 << 2; /// Back a span of a guest's address space with fresh zeroed frames. pub fn mk_peer_map(pid: u32, addr: u64, len: u64, prot: u64) -> i64 { diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index 52547185e8..f53a5d5163 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -124,6 +124,13 @@ $(LINUX_GUESTS_C)/guardpage: $(LINUX_GUESTS_DIR)/c/guardpage.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,guardpage,4980,4981,$(LINUX_GUESTS_C)/guardpage)) +# PROT_NONE means no access: a read or write of a PROT_NONE mmap, of a page +# mprotect closed, and of the closed page below an opened one each fault, and +# bytes survive a close and reopen. +$(LINUX_GUESTS_C)/protnone: $(LINUX_GUESTS_DIR)/c/protnone.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,protnone,4982,4983,$(LINUX_GUESTS_C)/protnone)) + # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the # normal image, which does not set NONOS_LINUX_GUESTS, still ships them. diff --git a/userland/linux_guests/c/protnone.c b/userland/linux_guests/c/protnone.c new file mode 100644 index 0000000000..dac6a37fc4 --- /dev/null +++ b/userland/linux_guests/c/protnone.c @@ -0,0 +1,79 @@ +// PROT_NONE means no access. Each part runs in a forked child and the parent +// reads how the child ended: a part that must fault passes only when the child +// dies of SIGSEGV, a part that must not fault passes only when it exits 0. +// Every part runs, so one boot names every part that fails. The personality +// reports a signal death as exit status 128+signo, which is counted as the +// same SIGSEGV; the raw status is printed either way. +#include +#include +#include +#include +#include +#include + +#define PG 4096 + +static int failed, passed; +static volatile char *p; + +static void say(const char *s) { + write(1, s, strlen(s)); +} + +static void part(const char *name, void (*fn)(void), int must_fault) { + char line[160]; + pid_t c = fork(); + if (c == 0) { + fn(); + _exit(0); + } + int st = 0; + waitpid(c, &st, 0); + int segv = (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || + (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); + int clean = WIFEXITED(st) && WEXITSTATUS(st) == 0; + int ok = must_fault ? segv : clean; + ok ? passed++ : failed++; + snprintf(line, sizeof line, "[C] protnone %s: %s (%s, status 0x%x)\n", name, + ok ? "ok" : "FAIL", must_fault ? "must fault" : "must not fault", st); + say(line); +} + +static void read_it(void) { + if (p[0] != 0x5a) { + _exit(2); + } +} +static void write_it(void) { + p[0] = 1; +} +static void read_below(void) { + (void)p[-1]; +} + +int main(void) { + char line[160]; + p = mmap(0, PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + part("read of an mmap PROT_NONE page", read_it, 1); + part("write to an mmap PROT_NONE page", write_it, 1); + + p = mmap(0, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + p[0] = 0x5a; + mprotect((void *)p, PG, PROT_NONE); + part("read after mprotect RW to PROT_NONE", read_it, 1); + mprotect((void *)p, PG, PROT_READ); + part("read after PROT_NONE back to R keeps the byte", read_it, 0); + part("write to a PROT_READ page", write_it, 1); + + char *r = mmap(0, 3 * PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + mprotect(r + PG, PG, PROT_READ | PROT_WRITE); + p = (volatile char *)(r + PG); + p[0] = 0x5a; + part("read of the opened page of a reservation", read_it, 0); + part("read of the closed page below it", read_below, 1); + + snprintf(line, sizeof line, "[C] protnone %s: %d parts ok, %d failed\n", + failed ? "FAIL" : "PASS", passed, failed); + say(line); + return failed ? 1 : 0; +} From 187181d0b206447f539420270f910216d3cac3f1 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 17:01:36 +0000 Subject: [PATCH 03/13] linux: record the protection mprotect sets on a backed span mprotect changed the page protection of a backed span in the kernel but left the region list with the protection the span was mapped with. Fork maps the child from that list, so after mprotect RW to R, or RW to PROT_NONE, the child got the span read-write: a write the parent could not make succeeded in the child. Every protection change on a backed span goes through protect_span, and protect_span now records the new write, exec and access on exactly the part it changed, splitting the region around it and keeping whether the bytes were proved. Fork maps the child with what the parent has now. protfork proves it: RW to R, RW to PROT_NONE, RW to R to RW, and one page closed in the middle of three, each checked in a forked child. --- .../src/linux/call/mem/prot_span.rs | 6 +- userland/capsule_linux/src/linux/guest/mod.rs | 1 + .../src/linux/guest/region_prot.rs | 46 ++++++++++ userland/linux_guests/Guests.mk | 6 ++ userland/linux_guests/c/protfork.c | 91 +++++++++++++++++++ 5 files changed, 148 insertions(+), 2 deletions(-) create mode 100644 userland/capsule_linux/src/linux/guest/region_prot.rs create mode 100644 userland/linux_guests/c/protfork.c diff --git a/userland/capsule_linux/src/linux/call/mem/prot_span.rs b/userland/capsule_linux/src/linux/call/mem/prot_span.rs index e7fba9460e..bbe49caf61 100644 --- a/userland/capsule_linux/src/linux/call/mem/prot_span.rs +++ b/userland/capsule_linux/src/linux/call/mem/prot_span.rs @@ -22,8 +22,9 @@ use crate::linux::guest::{peer_prot, Guest, MAX_SPAN}; use super::prot::{PROT_ANY, PROT_EXEC, PROT_WRITE}; -/// Set the protection of a span already mapped in the guest. -pub fn protect_span(guest: &Guest, addr: u64, span: u64, prot: u64) -> i64 { +/// Set the protection of a span already mapped in the guest, and record it +/// on the spans it covers. +pub fn protect_span(guest: &mut Guest, addr: u64, span: u64, prot: u64) -> i64 { let (write, exec, access) = (prot & PROT_WRITE != 0, prot & PROT_EXEC != 0, prot & PROT_ANY != 0); let bits = peer_prot(write, exec, access); @@ -36,5 +37,6 @@ pub fn protect_span(guest: &Guest, addr: u64, span: u64, prot: u64) -> i64 { } done += take; } + guest.set_prot(addr, span, write, exec, access); 0 } diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index 83f958019d..2f5f7d0239 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -42,6 +42,7 @@ mod region; mod region_cut; mod region_find; mod region_mark; +mod region_prot; mod threads; mod timer; mod watch; diff --git a/userland/capsule_linux/src/linux/guest/region_prot.rs b/userland/capsule_linux/src/linux/guest/region_prot.rs new file mode 100644 index 0000000000..705ddf967b --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/region_prot.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Recording a protection change on the spans a guest holds. + +use alloc::vec::Vec; + +use super::handle::Guest; +use super::region::Region; +use super::region_cut::cut; + +impl Guest { + /// Every backed part of `[at, at + len)` now has this protection. The + /// list is what fork maps the child from, so a list that kept the old + /// protection would give the child access the parent gave up. + pub fn set_prot(&mut self, at: u64, len: u64, write: bool, exec: bool, access: bool) { + let end = at.saturating_add(len); + let changed: Vec = self + .regions + .iter() + .filter(|r| r.backed && r.at < end && at < r.at.saturating_add(r.len)) + .map(|r| { + let from = r.at.max(at); + let to = r.at.saturating_add(r.len).min(end); + Region { at: from, len: to - from, write, exec, access, ..*r } + }) + .collect(); + for piece in &changed { + self.regions = cut(&self.regions, piece.at, piece.len); + } + self.regions.extend(changed); + } +} diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index f53a5d5163..57e43ea315 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -131,6 +131,12 @@ $(LINUX_GUESTS_C)/protnone: $(LINUX_GUESTS_DIR)/c/protnone.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,protnone,4982,4983,$(LINUX_GUESTS_C)/protnone)) +# A fork after mprotect: the child gets the protection the parent has now, so +# a write to a page the parent made read-only faults in the child. +$(LINUX_GUESTS_C)/protfork: $(LINUX_GUESTS_DIR)/c/protfork.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,protfork,4984,4985,$(LINUX_GUESTS_C)/protfork)) + # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the # normal image, which does not set NONOS_LINUX_GUESTS, still ships them. diff --git a/userland/linux_guests/c/protfork.c b/userland/linux_guests/c/protfork.c new file mode 100644 index 0000000000..8b0463caa9 --- /dev/null +++ b/userland/linux_guests/c/protfork.c @@ -0,0 +1,91 @@ +// A fork gives the child the parent's mappings with the protection they have +// now, not the one they were made with. Each part changes a protection with +// mprotect, forks, and the child tries one access; the parent reads how the +// child ended. The personality reports a signal death as exit status +// 128+signo, which is counted as the same SIGSEGV; the raw status is printed. +#include +#include +#include +#include +#include +#include + +#define PG 4096 + +static int failed, passed; +static volatile char *p; + +static void say(const char *s) { + write(1, s, strlen(s)); +} + +static void part(const char *name, void (*fn)(void), int must_fault) { + char line[160]; + pid_t c = fork(); + if (c == 0) { + fn(); + _exit(0); + } + int st = 0; + waitpid(c, &st, 0); + int segv = (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || + (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); + int clean = WIFEXITED(st) && WEXITSTATUS(st) == 0; + int ok = must_fault ? segv : clean; + ok ? passed++ : failed++; + snprintf(line, sizeof line, "[C] protfork %s: %s (%s, status 0x%x)\n", name, + ok ? "ok" : "FAIL", must_fault ? "must fault" : "must not fault", st); + say(line); +} + +static void write_it(void) { + p[0] = 1; + if (p[0] != 1) { + _exit(2); + } +} +static void read_it(void) { + if (p[0] != 0x5a) { + _exit(2); + } +} + +static volatile char *fresh(int pages) { + volatile char *m = + mmap(0, pages * PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + for (int i = 0; i < pages; i++) { + m[i * PG] = 0x5a; + } + return m; +} + +int main(void) { + char line[160]; + p = fresh(1); + mprotect((void *)p, PG, PROT_READ); + part("RW to R, child writes", write_it, 1); + part("RW to R, child reads the byte", read_it, 0); + + p = fresh(1); + mprotect((void *)p, PG, PROT_NONE); + part("RW to NONE, child reads", read_it, 1); + + p = fresh(1); + mprotect((void *)p, PG, PROT_READ); + mprotect((void *)p, PG, PROT_READ | PROT_WRITE); + part("RW to R to RW, child writes", write_it, 0); + + volatile char *m = fresh(3); + mprotect((void *)(m + PG), PG, PROT_READ); + p = m; + part("middle page R, child writes the first", write_it, 0); + p = m + PG; + part("middle page R, child writes the middle", write_it, 1); + p = m + 2 * PG; + part("middle page R, child writes the last", write_it, 0); + + snprintf(line, sizeof line, "[C] protfork %s: %d parts ok, %d failed\n", + failed ? "FAIL" : "PASS", passed, failed); + say(line); + return failed ? 1 : 0; +} From 7232acff6396c8760d156ab58049c374d39e034b Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 17:02:47 +0000 Subject: [PATCH 04/13] linux: replace what a MAP_FIXED mapping lands on A MAP_FIXED mmap over pages the guest already had kept them: MkPeerMap skips a page that is present, so the old bytes and the old protection stayed, and the region list held the old span and the new one on top of each other. A program that wrote a page, mapped PROT_NONE over it with MAP_FIXED and opened it again read its old bytes, where Linux gives zeroes, and fork copied the doubled span twice. MAP_FIXED now unmaps the target span just before the new pages go in, as Linux replaces a mapping. It is done after every refusal the mapping can meet, so a refused mapping leaves the old one in place. touchfork proves it with the reservation cases around it: bytes written into an opened part of a reservation reach a forked child, still do after the part is closed and opened again in the child, a page never opened faults in the child, and MAP_FIXED PROT_NONE over a written page reads zero once opened. --- .../src/linux/call/mem/map_anon.rs | 5 +- .../src/linux/call/mem/map_file.rs | 2 +- .../src/linux/call/mem/map_req.rs | 12 +++ userland/linux_guests/Guests.mk | 7 ++ userland/linux_guests/c/touchfork.c | 85 +++++++++++++++++++ 5 files changed, 109 insertions(+), 2 deletions(-) create mode 100644 userland/linux_guests/c/touchfork.c diff --git a/userland/capsule_linux/src/linux/call/mem/map_anon.rs b/userland/capsule_linux/src/linux/call/mem/map_anon.rs index 0158b7f263..2dd8d11048 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_anon.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_anon.rs @@ -44,10 +44,13 @@ pub fn memfd(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { } pub fn anonymous(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { + if !req.make_room(guest, at, span) { + return errno::fail(errno::ENOMEM); + } // A PROT_NONE anonymous mapping is a reservation: the runtime that makes it // (Go's, for one) commits a fraction of it later with a fixed RW mapping. // Backing the whole span here would spend real frames on address space no - // one has touched, so reserve it and let the first access fault a page in. + // one may touch, so reserve it; a commit maps the part that is opened. let backed = if req.prot == 0 { guest.reserve(at, span) } else { diff --git a/userland/capsule_linux/src/linux/call/mem/map_file.rs b/userland/capsule_linux/src/linux/call/mem/map_file.rs index 2961e91233..6e086f854f 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_file.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_file.rs @@ -36,7 +36,7 @@ pub fn file(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if let Some(None) = proved { return errno::fail(errno::EPERM); } - if guest.map(at, span, true, false) < 0 { + if !req.make_room(guest, at, span) || guest.map(at, span, true, false) < 0 { return errno::fail(errno::ENOMEM); } if let Some(Some(bytes)) = proved { diff --git a/userland/capsule_linux/src/linux/call/mem/map_req.rs b/userland/capsule_linux/src/linux/call/mem/map_req.rs index c426dcb115..91c149766f 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_req.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_req.rs @@ -17,6 +17,10 @@ //! What a guest asked `mmap` for, in one value. +use crate::linux::guest::Guest; + +const MAP_FIXED: u64 = 0x10; + pub struct MapReq { pub addr: u64, pub len: u64, @@ -39,4 +43,12 @@ impl MapReq { addr => Some(crate::linux::guest::page_down(addr)), } } + + /// Make room for a MAP_FIXED mapping at `[at, at + span)`. Linux replaces + /// whatever was there: the old pages go and the new mapping starts from + /// zeroes with its own protection. Called just before the new pages go in, + /// so a mapping refused earlier leaves the old one where it was. + pub fn make_room(&self, guest: &mut Guest, at: u64, span: u64) -> bool { + self.flags & MAP_FIXED == 0 || guest.unmap(at, span) >= 0 + } } diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index 57e43ea315..133f4b2fba 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -137,6 +137,13 @@ $(LINUX_GUESTS_C)/protfork: $(LINUX_GUESTS_DIR)/c/protfork.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,protfork,4984,4985,$(LINUX_GUESTS_C)/protfork)) +# Bytes written into an opened part of a reservation survive closing it and a +# fork; a page never opened faults in the child; and MAP_FIXED over a written +# page replaces it with zeroes. +$(LINUX_GUESTS_C)/touchfork: $(LINUX_GUESTS_DIR)/c/touchfork.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,touchfork,4986,4987,$(LINUX_GUESTS_C)/touchfork)) + # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the # normal image, which does not set NONOS_LINUX_GUESTS, still ships them. diff --git a/userland/linux_guests/c/touchfork.c b/userland/linux_guests/c/touchfork.c new file mode 100644 index 0000000000..133e2b4e42 --- /dev/null +++ b/userland/linux_guests/c/touchfork.c @@ -0,0 +1,85 @@ +// Bytes a guest wrote into a reservation survive a fork. A reservation is a +// PROT_NONE mapping; a program opens part of it with mprotect, writes, and may +// close it again. Linux keeps the bytes through all of that and gives the child +// a copy. Touching a page never opened is SIGSEGV, in the child as anywhere. +// MAP_FIXED over a mapping replaces it: the new pages read zero, as Linux says. +#include +#include +#include +#include +#include +#include + +#define PG 4096 +#define PAGES 16 + +static int failed, passed; +static volatile char *r; + +static void say(const char *s) { + write(1, s, strlen(s)); +} + +static void part(const char *name, void (*fn)(void), int must_fault) { + char line[160]; + pid_t c = fork(); + if (c == 0) { + fn(); + _exit(0); + } + int st = 0; + waitpid(c, &st, 0); + int segv = (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || + (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); + int clean = WIFEXITED(st) && WEXITSTATUS(st) == 0; + int ok = must_fault ? segv : clean; + ok ? passed++ : failed++; + snprintf(line, sizeof line, "[C] touchfork %s: %s (%s, status 0x%x)\n", name, + ok ? "ok" : "FAIL", must_fault ? "must fault" : "must not fault", st); + say(line); +} + +static void check_bytes(void) { + for (int i = 4; i < 8; i++) { + if (r[i * PG] != (char)(0x40 + i) || r[i * PG + PG - 1] != (char)(0x50 + i)) { + _exit(2); + } + } +} +static void open_then_check(void) { + mprotect((void *)(r + 4 * PG), 4 * PG, PROT_READ); + check_bytes(); +} +static void touch_unopened(void) { + (void)r[0]; +} +static void check_zero(void) { + if (r[0] != 0) { + _exit(2); + } +} + +int main(void) { + char line[160]; + r = mmap(0, PAGES * PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + mprotect((void *)(r + 4 * PG), 4 * PG, PROT_READ | PROT_WRITE); + for (int i = 4; i < 8; i++) { + r[i * PG] = (char)(0x40 + i); + r[i * PG + PG - 1] = (char)(0x50 + i); + } + part("opened part of a reservation, child reads the bytes", check_bytes, 0); + mprotect((void *)(r + 4 * PG), 4 * PG, PROT_NONE); + part("closed again, child opens it and reads the bytes", open_then_check, 0); + part("child touches a page never opened", touch_unopened, 1); + + r = mmap(0, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + r[0] = 0x77; + mmap((void *)r, PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); + mprotect((void *)r, PG, PROT_READ); + part("MAP_FIXED PROT_NONE over written page, child reads zero", check_zero, 0); + + snprintf(line, sizeof line, "[C] touchfork %s: %d parts ok, %d failed\n", + failed ? "FAIL" : "PASS", passed, failed); + say(line); + return failed ? 1 : 0; +} From 259d2d962cc99a47fda8fda11620a5bcc9cc5a6c Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 17:03:58 +0000 Subject: [PATCH 05/13] linux: never lay a new mapping over one the guest holds mmap took any nonzero address as fixed, hint or not, and the mapping cursor took the next address above the last mapping it chose without looking at what the guest had mapped there itself. Either way a new mapping could land on an old one: MkPeerMap skips a present page, so the guest got the old bytes back as its new, zeroed mapping, and the region list held both. MAP_FIXED_NOREPLACE was taken as a hint and mapped over whatever was there. Placement now follows Linux. MAP_FIXED is the exact address; MAP_FIXED_NOREPLACE is the exact address or EEXIST when anything is there; any other address is a hint, rounded up to a page and used only when nothing is there. Otherwise the cursor chooses and skips every span the guest holds, and moves past a mapping only when it chose it. --- .../capsule_linux/src/linux/call/mem/map.rs | 37 ++++++---- .../src/linux/call/mem/map_anon.rs | 3 - .../src/linux/call/mem/map_file.rs | 3 - .../src/linux/call/mem/map_place.rs | 71 +++++++++++++++++++ .../src/linux/call/mem/map_req.rs | 9 --- .../capsule_linux/src/linux/call/mem/mod.rs | 1 + .../src/linux/guest/region_find.rs | 6 ++ 7 files changed, 100 insertions(+), 30 deletions(-) create mode 100644 userland/capsule_linux/src/linux/call/mem/map_place.rs diff --git a/userland/capsule_linux/src/linux/call/mem/map.rs b/userland/capsule_linux/src/linux/call/mem/map.rs index 904072cdbf..cd950cca24 100644 --- a/userland/capsule_linux/src/linux/call/mem/map.rs +++ b/userland/capsule_linux/src/linux/call/mem/map.rs @@ -17,16 +17,18 @@ //! `mmap`: anonymous pages, or a private mapping of a file. use crate::linux::abi::errno; -use crate::linux::guest::{span_within, Guest, MMAP_LIMIT, USER_MAX}; +use crate::linux::guest::Guest; use super::map_anon::{anonymous, memfd}; use super::map_file::file; +use super::map_place::place; use super::map_req::MapReq; use super::prot::wx_refused; const MAP_SHARED: u64 = 0x01; const MAP_ANONYMOUS: u64 = 0x20; const MAP_FIXED: u64 = 0x10; +const MAP_FIXED_NOREPLACE: u64 = 0x10_0000; pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 { if req.len == 0 { @@ -35,25 +37,30 @@ pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 { if wx_refused(req.prot) { return errno::fail(errno::EPERM); } - // MAP_FIXED is the exact address or failure. Page zero is never in the - // plan, and landing elsewhere would hand back memory the guest did not - // ask for, so it is refused, as Linux refuses it below mmap_min_addr. - if req.flags & MAP_FIXED != 0 && req.addr == 0 { + // MAP_FIXED and MAP_FIXED_NOREPLACE are the exact address or failure. + // Page zero is never in the plan, and landing elsewhere would hand back + // memory the guest did not ask for, so it is refused, as Linux refuses it + // below mmap_min_addr. + if req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0 && req.addr == 0 { return errno::fail(errno::EPERM); } - // The ceiling differs by who chose the address. - let (at, limit) = match req.fixed() { - Some(addr) => (addr, USER_MAX), - None => (guest.mmap_next, MMAP_LIMIT), - }; - let Some((at, span)) = span_within(at, req.len, limit) else { - return errno::fail(errno::ENOMEM); + let spot = match place(guest, &req) { + Ok(spot) => spot, + Err(e) => return errno::fail(e), }; + let out = map_at(guest, &req, spot.at, spot.span); + if spot.from_cursor && (out as i64) >= 0 { + guest.mmap_next = spot.at + spot.span; + } + out +} + +fn map_at(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if req.flags & MAP_ANONYMOUS != 0 { - return anonymous(guest, &req, at, span); + return anonymous(guest, req, at, span); } if crate::linux::file::is_memfd(guest, req.fd) { - return memfd(guest, &req, at, span); + return memfd(guest, req, at, span); } if req.flags & MAP_SHARED != 0 { /* @@ -62,5 +69,5 @@ pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 { */ return errno::fail(errno::ENOSYS); } - file(guest, &req, at, span) + file(guest, req, at, span) } diff --git a/userland/capsule_linux/src/linux/call/mem/map_anon.rs b/userland/capsule_linux/src/linux/call/mem/map_anon.rs index 2dd8d11048..bf3bc34480 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_anon.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_anon.rs @@ -59,8 +59,5 @@ pub fn anonymous(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if backed < 0 { return errno::fail(errno::ENOMEM); } - if req.fixed().is_none() { - guest.mmap_next += span; - } errno::ok(at) } diff --git a/userland/capsule_linux/src/linux/call/mem/map_file.rs b/userland/capsule_linux/src/linux/call/mem/map_file.rs index 6e086f854f..a6e3447942 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_file.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_file.rs @@ -57,8 +57,5 @@ fn finish(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if protect_span(guest, at, span, req.prot) < 0 { return errno::fail(errno::EACCES); } - if req.fixed().is_none() { - guest.mmap_next += span; - } errno::ok(at) } diff --git a/userland/capsule_linux/src/linux/call/mem/map_place.rs b/userland/capsule_linux/src/linux/call/mem/map_place.rs new file mode 100644 index 0000000000..103425fd4b --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/map_place.rs @@ -0,0 +1,71 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where an mmap goes. MAP_FIXED is the exact address; MAP_FIXED_NOREPLACE +//! is the exact address or EEXIST when something is there; any other address +//! is a hint taken only when nothing is there. Otherwise the mapping cursor +//! chooses, skipping every span the guest already holds, since a new mapping +//! laid over an old one would hand back the old pages. + +use crate::linux::abi::errno; +use crate::linux::guest::{page_up, span_within, Guest, MMAP_LIMIT, USER_MAX}; + +use super::map_req::MapReq; + +const MAP_FIXED: u64 = 0x10; +const MAP_FIXED_NOREPLACE: u64 = 0x10_0000; + +pub struct Place { + pub at: u64, + pub span: u64, + /// Chosen by the cursor, which then moves past it. + pub from_cursor: bool, +} + +/// The span the mapping takes, or the errno refusing it. +pub fn place(guest: &Guest, req: &MapReq) -> Result { + let exact = |(at, span)| Place { at, span, from_cursor: false }; + if req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0 { + let (at, span) = span_within(req.addr, req.len, USER_MAX).ok_or(errno::ENOMEM)?; + if req.flags & MAP_FIXED == 0 && guest.overlaps(at, span) { + return Err(errno::EEXIST); + } + return Ok(exact((at, span))); + } + // Linux rounds a hint up to a page. + if req.addr != 0 { + if let Some(got) = span_within(page_up(req.addr), req.len, USER_MAX) { + if !guest.overlaps(got.0, got.1) { + return Ok(exact(got)); + } + } + } + let mut at = guest.mmap_next; + loop { + let (start, span) = span_within(at, req.len, MMAP_LIMIT).ok_or(errno::ENOMEM)?; + let end = start + span; + let past = guest + .regions + .iter() + .filter(|r| r.at < end && start < r.at.saturating_add(r.len)) + .map(|r| r.at.saturating_add(r.len)) + .max(); + match past { + None => return Ok(Place { at: start, span, from_cursor: true }), + Some(next) => at = page_up(next), + } + } +} diff --git a/userland/capsule_linux/src/linux/call/mem/map_req.rs b/userland/capsule_linux/src/linux/call/mem/map_req.rs index 91c149766f..7e278ccd2a 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_req.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_req.rs @@ -35,15 +35,6 @@ impl MapReq { MapReq { addr: a[0], len: a[1], prot: a[2], flags: a[3], fd: a[4], off: a[5] } } - /// The address the guest named, or nothing when it left the choice - /// to this capsule, which is the case the mapping cursor advances on. - pub fn fixed(&self) -> Option { - match self.addr { - 0 => None, - addr => Some(crate::linux::guest::page_down(addr)), - } - } - /// Make room for a MAP_FIXED mapping at `[at, at + span)`. Linux replaces /// whatever was there: the old pages go and the new mapping starts from /// zeroes with its own protection. Called just before the new pages go in, diff --git a/userland/capsule_linux/src/linux/call/mem/mod.rs b/userland/capsule_linux/src/linux/call/mem/mod.rs index 423f1a164a..95f3daef11 100644 --- a/userland/capsule_linux/src/linux/call/mem/mod.rs +++ b/userland/capsule_linux/src/linux/call/mem/mod.rs @@ -22,6 +22,7 @@ mod map_anon; mod map_exec; mod map_file; mod map_fill; +mod map_place; mod map_req; mod memory; mod prot; diff --git a/userland/capsule_linux/src/linux/guest/region_find.rs b/userland/capsule_linux/src/linux/guest/region_find.rs index aabfe4cd05..e557af1a79 100644 --- a/userland/capsule_linux/src/linux/guest/region_find.rs +++ b/userland/capsule_linux/src/linux/guest/region_find.rs @@ -36,4 +36,10 @@ impl Guest { } reach - addr } + + /// Whether any span the guest holds meets `[at, at + len)`. + pub fn overlaps(&self, at: u64, len: u64) -> bool { + let end = at.saturating_add(len); + self.regions.iter().any(|r| r.at < end && at < r.at.saturating_add(r.len)) + } } From 137901e31ecffc31b6d119dd7bff810286ab600d Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 17:04:26 +0000 Subject: [PATCH 06/13] linux: move the break in whole pages, and give pages back when it drops brk mapped from the old break itself, which is rarely page aligned, so every call that grew the break pushed a second region for the page the break already sat in, and fork copied that page once per call. A lower break only moved the number: the pages above it stayed mapped, and growing again handed back the old bytes where Linux gives zeroes. And a break that grew into a mapping the guest had placed in the heap area with MAP_FIXED mapped over it. The break now grows from the page above the old one, is refused, as Linux refuses it, when that span meets a mapping, and a lower break unmaps the whole pages above it. --- .../capsule_linux/src/linux/call/mem/memory.rs | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/userland/capsule_linux/src/linux/call/mem/memory.rs b/userland/capsule_linux/src/linux/call/mem/memory.rs index 28a2cfb6a3..d39f823a6e 100644 --- a/userland/capsule_linux/src/linux/call/mem/memory.rs +++ b/userland/capsule_linux/src/linux/call/mem/memory.rs @@ -29,13 +29,19 @@ pub fn brk(guest: &mut Guest, want: u64) -> u64 { if want == 0 || want < BRK_BASE || want > BRK_LIMIT { return errno::ok(guest.brk); } - let top = page_up(want); - if top > guest.brk { - let len = top - guest.brk; - if guest.map(guest.brk, len, true, false) < 0 { + // Whole pages: the page the old break sits in is already held. + let (old, top) = (page_up(guest.brk), page_up(want)); + if top > old { + // Linux refuses a break that would run into a mapping. + if guest.overlaps(old, top - old) || guest.map(old, top - old, true, false) < 0 { return errno::ok(guest.brk); } } + // A lower break gives the pages above it back, so growing again reads + // zeroes, as on Linux. + if top < old && guest.unmap(top, old - top) < 0 { + return errno::ok(guest.brk); + } guest.brk = want; errno::ok(guest.brk) } From f60b8190b6ae01e0aa4494ba47678c2a644cbdee Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 17:05:40 +0000 Subject: [PATCH 07/13] linux: keep a mapping's protection and provenance through mremap mremap gave the part it grew read-write whatever the mapping was, so a PROT_NONE or read-only mapping grew a writable tail. A move read the old bytes and failed with EFAULT on a reservation, which has none; restored only read-only or read-write on the copy, so a PROT_NONE mapping came back readable; left the new span mapped when the copy failed; and dropped the mark that the bytes came from a file nothing proved, so the moved copy could then be made executable with mprotect, past the check that refuses it where the file was mapped. It also accepted an old span running across mappings with different protections. The grown part and the moved copy are now held like the mapping they came from: its protection, its backing and its provenance. A reservation grows or moves as a reservation with no bytes to copy. A move that fails unmaps what it made. An old span that is not one mapping is refused with EFAULT, as Linux refuses it. The move goes to a free span found the same way mmap finds one. --- .../src/linux/call/mem/map_place.rs | 11 +++- .../capsule_linux/src/linux/call/mem/remap.rs | 39 +++++++++---- .../src/linux/call/mem/remap_move.rs | 29 +++++----- .../capsule_linux/src/linux/guest/mem_like.rs | 55 +++++++++++++++++++ userland/capsule_linux/src/linux/guest/mod.rs | 1 + 5 files changed, 107 insertions(+), 28 deletions(-) create mode 100644 userland/capsule_linux/src/linux/guest/mem_like.rs diff --git a/userland/capsule_linux/src/linux/call/mem/map_place.rs b/userland/capsule_linux/src/linux/call/mem/map_place.rs index 103425fd4b..7131c02ba7 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_place.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_place.rs @@ -53,9 +53,16 @@ pub fn place(guest: &Guest, req: &MapReq) -> Result { } } } + let (at, span) = free_span(guest, req.len)?; + Ok(Place { at, span, from_cursor: true }) +} + +/// The first span of `len` at or above the mapping cursor that meets +/// nothing the guest holds. +pub fn free_span(guest: &Guest, len: u64) -> Result<(u64, u64), i64> { let mut at = guest.mmap_next; loop { - let (start, span) = span_within(at, req.len, MMAP_LIMIT).ok_or(errno::ENOMEM)?; + let (start, span) = span_within(at, len, MMAP_LIMIT).ok_or(errno::ENOMEM)?; let end = start + span; let past = guest .regions @@ -64,7 +71,7 @@ pub fn place(guest: &Guest, req: &MapReq) -> Result { .map(|r| r.at.saturating_add(r.len)) .max(); match past { - None => return Ok(Place { at: start, span, from_cursor: true }), + None => return Ok((start, span)), Some(next) => at = page_up(next), } } diff --git a/userland/capsule_linux/src/linux/call/mem/remap.rs b/userland/capsule_linux/src/linux/call/mem/remap.rs index add8ae3a4e..a046279c90 100644 --- a/userland/capsule_linux/src/linux/call/mem/remap.rs +++ b/userland/capsule_linux/src/linux/call/mem/remap.rs @@ -18,7 +18,7 @@ //! or move with MREMAP_MAYMOVE. glibc's realloc of a large block is this. use crate::linux::abi::errno; -use crate::linux::guest::{page_up, span_within, Guest, MMAP_LIMIT, PAGE}; +use crate::linux::guest::{page_up, span_within, Guest, Region, PAGE, USER_MAX}; const MAYMOVE: u64 = 1; @@ -28,12 +28,13 @@ pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u6 return errno::fail(errno::EINVAL); } let (old_len, new_len) = (page_up(old_len), page_up(new_len)); - if guest.mapped_from(old) < old_len { - return errno::fail(errno::EFAULT); - } let Some(r) = guest.regions.iter().find(|r| r.at <= old && old < r.at + r.len).copied() else { return errno::fail(errno::EFAULT); }; + // Linux moves one mapping at a time: the old span must lie inside one. + if !one_mapping(guest, old, old_len, &r) { + return errno::fail(errno::EFAULT); + } // Code was proved where it was mapped; a moved copy would not be. if r.exec { return errno::fail(errno::EPERM); @@ -46,16 +47,34 @@ pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u6 } let tail = old + old_len; let grow = new_len - old_len; - let free = !guest.regions.iter().any(|g| g.at < tail + grow && tail < g.at + g.len); - if free - && span_within(tail, grow, MMAP_LIMIT).is_some() - && guest.map(tail, grow, r.write, false) >= 0 + // The grown part is the same mapping: its protection, its backing. + if !guest.overlaps(tail, grow) + && span_within(tail, grow, USER_MAX).is_some() + && guest.map_like(tail, grow, &r) >= 0 { - guest.mmap_next = guest.mmap_next.max(tail + grow); return errno::ok(old); } if flags & MAYMOVE == 0 { return errno::fail(errno::ENOMEM); } - super::remap_move::moved(guest, old, old_len, new_len, r.write) + super::remap_move::moved(guest, old, old_len, new_len, &r) +} + +/// Every page of `[at, at + len)` is held with the same protection, backing +/// and provenance as `like`, which is what Linux keeps as one mapping. +fn one_mapping(guest: &Guest, at: u64, len: u64, like: &Region) -> bool { + let end = at + len; + let mut reach = at; + while reach < end { + let Some(r) = guest.regions.iter().find(|r| r.at <= reach && reach < r.at + r.len) else { + return false; + }; + let same = (r.write, r.exec, r.access, r.backed, r.unproven) + == (like.write, like.exec, like.access, like.backed, like.unproven); + if !same { + return false; + } + reach = r.at + r.len; + } + true } diff --git a/userland/capsule_linux/src/linux/call/mem/remap_move.rs b/userland/capsule_linux/src/linux/call/mem/remap_move.rs index b8b3930d7e..b1f7b7f6b4 100644 --- a/userland/capsule_linux/src/linux/call/mem/remap_move.rs +++ b/userland/capsule_linux/src/linux/call/mem/remap_move.rs @@ -17,29 +17,26 @@ //! `mremap` when the block cannot grow where it is: moved to fresh pages. use crate::linux::abi::errno; -use crate::linux::guest::{span_within, Guest, MMAP_LIMIT}; +use crate::linux::guest::{Guest, Region}; -const PROT_READ: u64 = 1; +use super::map_place::free_span; -// A fresh span at the mapping cursor, the old bytes copied in, the old span gone. -pub(super) fn moved(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, write: bool) -> u64 { - let Some((at, span)) = span_within(guest.mmap_next, new_len, MMAP_LIMIT) else { - return errno::fail(errno::ENOMEM); - }; - let Some(bytes) = guest.read(old, old_len as usize) else { - return errno::fail(errno::EFAULT); +/// A fresh span at the mapping cursor held like the old one, with the old +/// protection, backing and provenance; the old bytes copied in; the old span +/// gone. A move that fails part way leaves nothing new behind. +pub(super) fn moved(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, like: &Region) -> u64 { + let (at, span) = match free_span(guest, new_len) { + Ok(got) => got, + Err(e) => return errno::fail(e), }; - // Writable while the bytes go in; the old protection after. - if guest.map(at, span, true, false) < 0 { + if guest.map_like(at, span, like) < 0 { return errno::fail(errno::ENOMEM); } - guest.mmap_next += span; - if guest.write(at, &bytes) < bytes.len() as i64 { + if like.backed && !guest.copy_within(old, at, old_len) { + let _ = guest.unmap(at, span); return errno::fail(errno::EFAULT); } - if !write { - let _ = super::prot::mprotect(guest, at, span, PROT_READ); - } let _ = guest.unmap(old, old_len); + guest.mmap_next = at + span; errno::ok(at) } diff --git a/userland/capsule_linux/src/linux/guest/mem_like.rs b/userland/capsule_linux/src/linux/guest/mem_like.rs new file mode 100644 index 0000000000..6c6ae90150 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/mem_like.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A span that takes after another: what mremap makes when a mapping grows +//! or moves, since Linux gives the new part the mapping's own protection. + +use super::handle::Guest; +use super::mem::MAX_SPAN; +use super::mem_map::map_span; +use super::region::Region; + +impl Guest { + /// `[at, at + len)` held like `like`: the same protection, provenance and + /// backing. A reservation stays one, with no pages. + pub fn map_like(&mut self, at: u64, len: u64, like: &Region) -> i64 { + if like.backed { + let rc = map_span(self.pid, at, len, like.peer_prot()); + if rc < 0 { + return rc; + } + } + self.regions.push(Region { at, len, ..*like }); + 0 + } + + /// Copy `len` bytes from `from` to `to` inside the guest, a megabyte at a + /// time. The kernel copies whatever the pages' protection is. + pub fn copy_within(&self, from: u64, to: u64, len: u64) -> bool { + let mut done = 0; + while done < len { + let take = (len - done).min(MAX_SPAN); + let Some(bytes) = self.read(from + done, take as usize) else { + return false; + }; + if self.write(to + done, &bytes) < bytes.len() as i64 { + return false; + } + done += take; + } + true + } +} diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index 2f5f7d0239..4994278f5d 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -36,6 +36,7 @@ mod links_list; mod links_load; mod mem; mod mem_copy; +mod mem_like; mod mem_map; mod mem_unmap; mod region; From 737472033e187ebba1212c8163225adf2b9096e2 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 17:06:04 +0000 Subject: [PATCH 08/13] linux: refuse an unaligned address where Linux refuses it munmap and mprotect rounded an address that was not on a page boundary down to one, and so acted on bytes below the address the guest named. mmap did the same for a MAP_FIXED address, and took a file offset that was not on a page boundary. Linux answers all four with EINVAL. They now answer EINVAL. Only a length is rounded, and a hint address without MAP_FIXED, as on Linux. --- userland/capsule_linux/src/linux/call/mem/map.rs | 9 ++++++--- userland/capsule_linux/src/linux/call/mem/memory.rs | 5 +++-- userland/capsule_linux/src/linux/call/mem/prot.rs | 6 +++++- 3 files changed, 14 insertions(+), 6 deletions(-) diff --git a/userland/capsule_linux/src/linux/call/mem/map.rs b/userland/capsule_linux/src/linux/call/mem/map.rs index cd950cca24..8571e54762 100644 --- a/userland/capsule_linux/src/linux/call/mem/map.rs +++ b/userland/capsule_linux/src/linux/call/mem/map.rs @@ -17,7 +17,7 @@ //! `mmap`: anonymous pages, or a private mapping of a file. use crate::linux::abi::errno; -use crate::linux::guest::Guest; +use crate::linux::guest::{Guest, PAGE}; use super::map_anon::{anonymous, memfd}; use super::map_file::file; @@ -31,7 +31,10 @@ const MAP_FIXED: u64 = 0x10; const MAP_FIXED_NOREPLACE: u64 = 0x10_0000; pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 { - if req.len == 0 { + // Linux takes a file offset on a page boundary, and an exact address too; + // only a hint is rounded. + let exact = req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0; + if req.len == 0 || req.off % PAGE != 0 || (exact && req.addr % PAGE != 0) { return errno::fail(errno::EINVAL); } if wx_refused(req.prot) { @@ -41,7 +44,7 @@ pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 { // Page zero is never in the plan, and landing elsewhere would hand back // memory the guest did not ask for, so it is refused, as Linux refuses it // below mmap_min_addr. - if req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0 && req.addr == 0 { + if exact && req.addr == 0 { return errno::fail(errno::EPERM); } let spot = match place(guest, &req) { diff --git a/userland/capsule_linux/src/linux/call/mem/memory.rs b/userland/capsule_linux/src/linux/call/mem/memory.rs index d39f823a6e..fe1b4003b3 100644 --- a/userland/capsule_linux/src/linux/call/mem/memory.rs +++ b/userland/capsule_linux/src/linux/call/mem/memory.rs @@ -17,7 +17,7 @@ //! `brk` and `munmap`. use crate::linux::abi::errno; -use crate::linux::guest::{page_up, Guest, BRK_BASE, BRK_LIMIT}; +use crate::linux::guest::{page_up, Guest, BRK_BASE, BRK_LIMIT, PAGE}; /// `brk(0)` reports the break; any other value moves it and reports where /// it landed, which is Linux's contract and not an error channel. @@ -48,7 +48,8 @@ pub fn brk(guest: &mut Guest, want: u64) -> u64 { /// The pages go back to the kernel and leave the guest's region list. pub fn munmap(guest: &mut Guest, addr: u64, len: u64) -> u64 { - if len == 0 { + // Linux takes an address on a page boundary, and rounds only the length. + if len == 0 || addr % PAGE != 0 { return errno::fail(errno::EINVAL); } match guest.unmap(addr, len) { diff --git a/userland/capsule_linux/src/linux/call/mem/prot.rs b/userland/capsule_linux/src/linux/call/mem/prot.rs index c25271dbb9..4062b07461 100644 --- a/userland/capsule_linux/src/linux/call/mem/prot.rs +++ b/userland/capsule_linux/src/linux/call/mem/prot.rs @@ -17,7 +17,7 @@ //! `mprotect`, and the rule that makes it necessary. use crate::linux::abi::errno; -use crate::linux::guest::{span_within, Guest, USER_MAX}; +use crate::linux::guest::{span_within, Guest, PAGE, USER_MAX}; use super::prot_span::protect_span; @@ -32,6 +32,10 @@ pub fn wx_refused(prot: u64) -> bool { } pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 { + // Linux takes an address on a page boundary, and rounds only the length. + if addr % PAGE != 0 { + return errno::fail(errno::EINVAL); + } if len == 0 { return errno::ok(0); } From f87ffa02978230c3dbe7ba5f7d0cf01602929cce Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 17:09:04 +0000 Subject: [PATCH 09/13] linux: serve mlock, munlock, mlockall, munlockall, msync and mincore A guest calling mlock, mlock2, munlock, mlockall, munlockall, msync or mincore got an unserved line and ENOSYS, where Linux answers each. In this memory model every page a guest holds is resident from the commit that maps it until it is unmapped, and nothing is paged out. So the lock calls have nothing left to do but Linux's argument checks: the span must be held (ENOMEM), the flags known (EINVAL), and RLIMIT_MEMLOCK is already reported unlimited. Every file mapping is private, since MAP_SHARED of a file is refused, so msync has nothing to write back and answers its checks alone, ENOMEM for a span with a hole included. mincore answers from the region list, one byte per page: 1 for a backed page, whatever its protection, and 0 for a reservation page, which has no frame. memcalls proves these and the placement, brk, alignment and mremap changes before it, one line per part. --- userland/capsule_linux/src/linux/abi/mod.rs | 1 + userland/capsule_linux/src/linux/abi/nr.rs | 1 + .../capsule_linux/src/linux/abi/nr_mem.rs | 24 ++ .../capsule_linux/src/linux/call/mem/lock.rs | 64 +++++ .../capsule_linux/src/linux/call/mem/mod.rs | 10 +- .../src/linux/call/mem/resident.rs | 54 +++++ .../capsule_linux/src/linux/call/mem/sync.rs | 44 ++++ userland/capsule_linux/src/linux/call/mod.rs | 4 +- .../src/linux/serve/table_mem.rs | 6 + userland/linux_guests/Guests.mk | 7 + userland/linux_guests/c/memcalls.c | 224 ++++++++++++++++++ 11 files changed, 436 insertions(+), 3 deletions(-) create mode 100644 userland/capsule_linux/src/linux/abi/nr_mem.rs create mode 100644 userland/capsule_linux/src/linux/call/mem/lock.rs create mode 100644 userland/capsule_linux/src/linux/call/mem/resident.rs create mode 100644 userland/capsule_linux/src/linux/call/mem/sync.rs create mode 100644 userland/linux_guests/c/memcalls.c diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs index 68e04090aa..0e5daf6a1d 100644 --- a/userland/capsule_linux/src/linux/abi/mod.rs +++ b/userland/capsule_linux/src/linux/abi/mod.rs @@ -24,3 +24,4 @@ pub mod nr; pub mod nr_path; pub mod nr_high; pub mod nr_sched; +pub mod nr_mem; diff --git a/userland/capsule_linux/src/linux/abi/nr.rs b/userland/capsule_linux/src/linux/abi/nr.rs index 1e800243be..8db6360936 100644 --- a/userland/capsule_linux/src/linux/abi/nr.rs +++ b/userland/capsule_linux/src/linux/abi/nr.rs @@ -19,6 +19,7 @@ pub use super::nr_high::*; pub use super::nr_sched::*; +pub use super::nr_mem::*; pub const READ: u64 = 0; pub const WRITE: u64 = 1; diff --git a/userland/capsule_linux/src/linux/abi/nr_mem.rs b/userland/capsule_linux/src/linux/abi/nr_mem.rs new file mode 100644 index 0000000000..a6d3d83d79 --- /dev/null +++ b/userland/capsule_linux/src/linux/abi/nr_mem.rs @@ -0,0 +1,24 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . +//! Memory calls the Linux x86_64 table has that the first families lacked. + +pub const MSYNC: u64 = 26; +pub const MINCORE: u64 = 27; +pub const MLOCK: u64 = 149; +pub const MUNLOCK: u64 = 150; +pub const MLOCKALL: u64 = 151; +pub const MUNLOCKALL: u64 = 152; +pub const MLOCK2: u64 = 325; diff --git a/userland/capsule_linux/src/linux/call/mem/lock.rs b/userland/capsule_linux/src/linux/call/mem/lock.rs new file mode 100644 index 0000000000..dfbaa21d78 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/lock.rs @@ -0,0 +1,64 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . +//! `mlock`, `mlock2`, `munlock`, `mlockall` and `munlockall`. +//! +//! Locking keeps pages resident. Every page a guest holds here is resident +//! from the moment it is mapped until it is unmapped, and none is ever paged +//! out, so every page is already as locked as Linux can make it. What is left +//! of each call is Linux's checking of its arguments, answered the same way: +//! the span must be held, and the flags known. RLIMIT_MEMLOCK is reported +//! unlimited, so no lock is refused for its size. + +use crate::linux::abi::errno; +use crate::linux::guest::{page_down, page_up, Guest, PAGE}; + +const MLOCK_ONFAULT: u64 = 1; +const MCL_CURRENT: u64 = 1; +const MCL_FUTURE: u64 = 2; +const MCL_ONFAULT: u64 = 4; + +/// `mlock` and `munlock` alike: the address is rounded down, the length up, +/// and a span with a page the guest does not hold is ENOMEM. +pub fn mlock(guest: &Guest, addr: u64, len: u64) -> u64 { + let start = page_down(addr); + let Some(end) = addr.checked_add(len).filter(|e| *e <= u64::MAX - PAGE) else { + return errno::fail(errno::EINVAL); + }; + let end = page_up(end); + if end == start { + return errno::ok(0); + } + if guest.mapped_from(start) < end - start { + return errno::fail(errno::ENOMEM); + } + errno::ok(0) +} + +pub fn mlock2(guest: &Guest, addr: u64, len: u64, flags: u64) -> u64 { + if flags & !MLOCK_ONFAULT != 0 { + return errno::fail(errno::EINVAL); + } + mlock(guest, addr, len) +} + +/// Linux refuses no flags, unknown flags, and MCL_ONFAULT on its own. +pub fn mlockall(flags: u64) -> u64 { + let known = MCL_CURRENT | MCL_FUTURE | MCL_ONFAULT; + if flags == 0 || flags & !known != 0 || flags == MCL_ONFAULT { + return errno::fail(errno::EINVAL); + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/mem/mod.rs b/userland/capsule_linux/src/linux/call/mem/mod.rs index 95f3daef11..da5c5496ff 100644 --- a/userland/capsule_linux/src/linux/call/mem/mod.rs +++ b/userland/capsule_linux/src/linux/call/mem/mod.rs @@ -14,9 +14,10 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! The calls that shape a guest's address space: `mmap`, `munmap`, `brk` and -//! `mprotect`. +//! The calls that shape a guest's address space: `mmap`, `munmap`, `brk`, +//! `mprotect` and `mremap`, and the ones that ask about it or lock it. +mod lock; mod map; mod map_anon; mod map_exec; @@ -29,9 +30,14 @@ mod prot; mod prot_span; mod remap; mod remap_move; +mod resident; +mod sync; +pub use lock::{mlock, mlock2, mlockall}; pub use map::mmap; pub use map_req::MapReq; pub use memory::{brk, munmap}; pub use prot::mprotect; pub use remap::mremap; +pub use resident::mincore; +pub use sync::msync; diff --git a/userland/capsule_linux/src/linux/call/mem/resident.rs b/userland/capsule_linux/src/linux/call/mem/resident.rs new file mode 100644 index 0000000000..dba74b0887 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/resident.rs @@ -0,0 +1,54 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . +//! `mincore`: which pages of a span are resident. +//! +//! A backed span is resident page for page, since the personality maps every +//! page of it when it is committed and the kernel pages nothing out, whatever +//! its protection now. A reservation has no page at all. So the region list +//! answers exactly, one byte per page, 1 for resident. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::{page_up, Guest, MAX_SPAN, PAGE, USER_MAX}; + +pub fn mincore(guest: &Guest, addr: u64, len: u64, vec: u64) -> u64 { + if addr % PAGE != 0 { + return errno::fail(errno::EINVAL); + } + let Some(end) = addr.checked_add(len).filter(|e| *e <= USER_MAX) else { + return errno::fail(errno::ENOMEM); + }; + let end = page_up(end); + let mut at = addr; + let mut out: Vec = Vec::new(); + let mut written = 0u64; + while at < end { + let Some(r) = guest.regions.iter().find(|r| r.at <= at && at < r.at + r.len) else { + return errno::fail(errno::ENOMEM); + }; + out.push(r.backed as u8); + at += PAGE; + if out.len() as u64 == MAX_SPAN || at >= end { + if guest.write(vec + written, &out) < out.len() as i64 { + return errno::fail(errno::EFAULT); + } + written += out.len() as u64; + out.clear(); + } + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/mem/sync.rs b/userland/capsule_linux/src/linux/call/mem/sync.rs new file mode 100644 index 0000000000..64b3834648 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/sync.rs @@ -0,0 +1,44 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . +//! `msync`: write a shared file mapping back to its file. +//! +//! Every file mapping here is private, since MAP_SHARED of a file is refused, +//! and a private mapping has nothing to write back: Linux answers msync on +//! one with its argument checks alone, and so does this. + +use crate::linux::abi::errno; +use crate::linux::guest::{page_up, Guest, PAGE}; + +const MS_ASYNC: u64 = 1; +const MS_INVALIDATE: u64 = 2; +const MS_SYNC: u64 = 4; + +pub fn msync(guest: &Guest, addr: u64, len: u64, flags: u64) -> u64 { + if flags & !(MS_ASYNC | MS_INVALIDATE | MS_SYNC) != 0 || addr % PAGE != 0 { + return errno::fail(errno::EINVAL); + } + if flags & MS_ASYNC != 0 && flags & MS_SYNC != 0 { + return errno::fail(errno::EINVAL); + } + let Some(end) = addr.checked_add(page_up(len)).filter(|_| len <= u64::MAX - PAGE) else { + return errno::fail(errno::ENOMEM); + }; + // Linux reports a span with a page nothing maps as ENOMEM. + if end > addr && guest.mapped_from(addr) < end - addr { + return errno::fail(errno::ENOMEM); + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index ab445d17c0..08e27b2b85 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -65,7 +65,9 @@ pub use life::{exit, exit_thread, set_tid_address}; pub use limits::{getrlimit, prlimit64}; pub use glibc::prctl; pub use glibc_sched::{clone3, getcpu, membarrier, sched_getaffinity}; -pub use mem::{brk, mmap, mprotect, mremap, munmap, MapReq}; +pub use mem::{ + brk, mincore, mlock, mlock2, mlockall, mmap, mprotect, mremap, msync, munmap, MapReq, +}; pub use pipe::pipe2; pub use pipe_dup::{dup, dup2}; pub use pipe_io::write as pipe_write; diff --git a/userland/capsule_linux/src/linux/serve/table_mem.rs b/userland/capsule_linux/src/linux/serve/table_mem.rs index 4ca19272bf..cded74bc72 100644 --- a/userland/capsule_linux/src/linux/serve/table_mem.rs +++ b/userland/capsule_linux/src/linux/serve/table_mem.rs @@ -27,6 +27,12 @@ pub fn mem_ops(guest: &mut Guest, nr: u64, a: [u64; 6]) -> Option { nr::MUNMAP => call::munmap(guest, a[0], a[1]), nr::MPROTECT => call::mprotect(guest, a[0], a[1], a[2]), nr::MREMAP => call::mremap(guest, a[0], a[1], a[2], a[3]), + nr::MSYNC => call::msync(guest, a[0], a[1], a[2]), + nr::MINCORE => call::mincore(guest, a[0], a[1], a[2]), + nr::MLOCK | nr::MUNLOCK => call::mlock(guest, a[0], a[1]), + nr::MLOCK2 => call::mlock2(guest, a[0], a[1], a[2]), + nr::MLOCKALL => call::mlockall(a[0]), + nr::MUNLOCKALL => errno::ok(0), // Advice, and this capsule takes none of it. nr::MADVISE => errno::ok(0), _ => return None, diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index 133f4b2fba..bdc1730f34 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -144,6 +144,13 @@ $(LINUX_GUESTS_C)/touchfork: $(LINUX_GUESTS_DIR)/c/touchfork.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,touchfork,4986,4987,$(LINUX_GUESTS_C)/touchfork)) +# The memory calls against Linux's answers: mmap placement, brk giving pages +# back, unaligned addresses, mremap keeping protection and provenance, and +# mlock, mlock2, mlockall, munlockall, msync and mincore with their errnos. +$(LINUX_GUESTS_C)/memcalls: $(LINUX_GUESTS_DIR)/c/memcalls.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,memcalls,4988,4989,$(LINUX_GUESTS_C)/memcalls)) + # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the # normal image, which does not set NONOS_LINUX_GUESTS, still ships them. diff --git a/userland/linux_guests/c/memcalls.c b/userland/linux_guests/c/memcalls.c new file mode 100644 index 0000000000..b663e11416 --- /dev/null +++ b/userland/linux_guests/c/memcalls.c @@ -0,0 +1,224 @@ +// The memory calls answered as Linux answers them: where mmap puts a hint, the +// break giving pages back, unaligned addresses refused, mremap keeping a +// mapping's protection, and mlock, msync and mincore with their errnos. Every +// part runs and prints one line, so one boot names every part that fails. +// Parts that must fault run in a forked child; the personality reports a +// signal death as exit status 128+signo, counted as the same SIGSEGV. +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#define PG 4096 +#define NOREPLACE 0x100000 + +static int failed, passed; +static volatile char *p; + +static void say(const char *s) { + write(1, s, strlen(s)); +} + +static void check(const char *name, int ok, long got) { + char line[160]; + ok ? passed++ : failed++; + snprintf(line, sizeof line, "[C] memcalls %s: %s (got %ld)\n", name, ok ? "ok" : "FAIL", got); + say(line); +} + +// -errno of a call that returned -1, or its value. +static long rc(long v) { + return v == -1 ? -errno : v; +} + +static void faults(const char *name, void (*fn)(void)) { + pid_t c = fork(); + if (c == 0) { + fn(); + _exit(0); + } + int st = 0; + waitpid(c, &st, 0); + int segv = (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || + (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); + check(name, segv, st); +} + +static void write_p(void) { + p[0] = 1; +} +static void read_p(void) { + (void)p[0]; +} + +static char *anon(long len, int prot) { + return mmap(0, len, prot, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); +} + +static void placement(void) { + char *m = anon(PG, PROT_READ | PROT_WRITE); + m[0] = 0x11; + char *n = mmap(m, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + check("hint on a mapping lands elsewhere, zeroed", n != m && n[0] == 0 && m[0] == 0x11, + (long)(n - m)); + void *q = mmap(m, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + long e = q == MAP_FAILED ? -errno : 0; + check("MAP_FIXED_NOREPLACE on a mapping is EEXIST", e == -EEXIST && m[0] == 0x11, e); + // A mapping placed just above the last one mmap chose: the next mmap that + // leaves the choice to the system must not land on it. On a system that + // already holds that address the probe cannot be placed, and says so. + char *a = anon(PG, PROT_READ | PROT_WRITE); + char *f = mmap(a + PG, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + if (f == MAP_FAILED) { + check("next mmap skips a mapping placed above the last (probe address taken)", + errno == EEXIST, -errno); + return; + } + f[0] = 0x44; + char *b = anon(PG, PROT_READ | PROT_WRITE); + check("next mmap skips a mapping placed above the last", b != f && b[0] == 0 && f[0] == 0x44, + (long)(b - f)); +} + +static void breaks(void) { + long cur = syscall(SYS_brk, 0); + long up = syscall(SYS_brk, cur + 2 * PG); + ((volatile char *)cur)[PG] = 0x22; + syscall(SYS_brk, cur); + syscall(SYS_brk, cur + 2 * PG); + char b = ((volatile char *)cur)[PG]; + check("brk down and up again reads zero", up == cur + 2 * PG && b == 0, b); + syscall(SYS_brk, cur); + long top = (cur + PG - 1) & ~(long)(PG - 1); + void *in = mmap((void *)(top + PG), PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + long got = syscall(SYS_brk, top + 4 * PG); + check("brk into a mapping is refused", in != MAP_FAILED && got == cur, got - cur); + munmap(in, PG); +} + +static void alignment(void) { + char *m = anon(2 * PG, PROT_READ | PROT_WRITE); + check("munmap unaligned is EINVAL", rc(munmap(m + 1, PG)) == -EINVAL, rc(munmap(m + 1, PG))); + // musl's mprotect rounds the address down itself; the kernel's does not. + long e = rc(syscall(SYS_mprotect, m + 1, PG, PROT_READ)); + check("mprotect unaligned is EINVAL", e == -EINVAL, e); + void *q = mmap(m + 1, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); + e = q == MAP_FAILED ? -errno : 0; + check("MAP_FIXED unaligned is EINVAL", e == -EINVAL, e); + // musl's mmap refuses an unaligned offset itself; the kernel's must too. + e = rc(syscall(SYS_mmap, 0, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS, -1, 1)); + check("mmap offset unaligned is EINVAL", e == -EINVAL, e); +} + +static void remaps(void) { + char *m = anon(PG, PROT_READ | PROT_WRITE); + m[0] = 0x33; + mprotect(m, PG, PROT_READ); + char *g = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE); + check("mremap of a read-only page keeps the byte", g != MAP_FAILED && g[0] == 0x33, g[0]); + p = g + PG; + faults("mremap grown part of a read-only page faults on write", write_p); + char *r = anon(2 * PG, PROT_NONE); + anon(PG, PROT_NONE); + char *q = mremap(r, 2 * PG, 4 * PG, MREMAP_MAYMOVE); + check("mremap of a reservation succeeds", q != MAP_FAILED, (long)(q == MAP_FAILED)); + p = q + 3 * PG; + faults("mremap grown reservation still faults on read", read_p); + char *two = anon(2 * PG, PROT_READ | PROT_WRITE); + mprotect(two + PG, PG, PROT_READ); + void *x = mremap(two, 2 * PG, 3 * PG, MREMAP_MAYMOVE); + long e = x == MAP_FAILED ? -errno : 0; + check("mremap across two mappings is EFAULT", e == -EFAULT, e); +} + +// A file mapped without exec was never proved; where mprotect refuses to make +// it executable, it must refuse the copy mremap moved too. Host Linux allows +// both, and the part checks only that the two answers agree. +static void provenance(void) { + // This program's own file: /bin/memcalls in the store, itself on a host. + int fd = open("/bin/memcalls", O_RDONLY); + if (fd < 0) { + fd = open("/proc/self/exe", O_RDONLY); + } + char *m = mmap(0, PG, PROT_READ, MAP_PRIVATE, fd, 0); + if (m == MAP_FAILED) { + check("file mapping for the provenance part", 0, -errno); + return; + } + long first = rc(mprotect(m, PG, PROT_READ | PROT_EXEC)); + mprotect(m, PG, PROT_READ); + mmap(m + PG, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + char *n = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE); + long moved = n == MAP_FAILED ? -1000 : rc(mprotect(n, PG, PROT_READ | PROT_EXEC)); + check("moved unproven file bytes refused exec as before the move", n != m && moved == first, + moved); + close(fd); +} + +static void locks(void) { + char *m = anon(3 * PG, PROT_READ | PROT_WRITE); + check("mlock of a mapping", rc(mlock(m, 3 * PG)) == 0, rc(mlock(m, 3 * PG))); + check("munlock of a mapping", rc(munlock(m, 3 * PG)) == 0, rc(munlock(m, 3 * PG))); + munmap(m + PG, PG); + check("mlock over a hole is ENOMEM", rc(mlock(m, 3 * PG)) == -ENOMEM, rc(mlock(m, 3 * PG))); + check("mlock2 with an unknown flag is EINVAL", rc(syscall(SYS_mlock2, m, PG, 2)) == -EINVAL, + rc(syscall(SYS_mlock2, m, PG, 2))); + check("mlock2 MLOCK_ONFAULT", rc(syscall(SYS_mlock2, m, PG, 1)) == 0, + rc(syscall(SYS_mlock2, m, PG, 1))); + check("mlockall(0) is EINVAL", rc(mlockall(0)) == -EINVAL, rc(mlockall(0))); + check("mlockall(MCL_ONFAULT) alone is EINVAL", rc(mlockall(MCL_ONFAULT)) == -EINVAL, + rc(mlockall(MCL_ONFAULT))); + check("mlockall(MCL_CURRENT)", rc(mlockall(MCL_CURRENT)) == 0, rc(mlockall(MCL_CURRENT))); + check("munlockall", rc(munlockall()) == 0, rc(munlockall())); +} + +static void syncs(void) { + char *m = anon(3 * PG, PROT_READ | PROT_WRITE); + check("msync MS_SYNC", rc(msync(m, 3 * PG, MS_SYNC)) == 0, rc(msync(m, 3 * PG, MS_SYNC))); + check("msync unaligned is EINVAL", rc(msync(m + 1, PG, MS_SYNC)) == -EINVAL, + rc(msync(m + 1, PG, MS_SYNC))); + check("msync MS_ASYNC|MS_SYNC is EINVAL", rc(msync(m, PG, MS_ASYNC | MS_SYNC)) == -EINVAL, + rc(msync(m, PG, MS_ASYNC | MS_SYNC))); + check("msync unknown flag is EINVAL", rc(msync(m, PG, 8)) == -EINVAL, rc(msync(m, PG, 8))); + munmap(m + PG, PG); + check("msync over a hole is ENOMEM", rc(msync(m, 3 * PG, MS_SYNC)) == -ENOMEM, + rc(msync(m, 3 * PG, MS_SYNC))); +} + +static void cores(void) { + char *r = anon(4 * PG, PROT_NONE); + mprotect(r + PG, PG, PROT_READ | PROT_WRITE); + r[PG] = 1; + unsigned char v[4] = { 9, 9, 9, 9 }; + long got = rc(mincore(r, 4 * PG, v)); + check("mincore of a reservation with one page opened is 0,1,0,0", + got == 0 && v[0] == 0 && v[1] == 1 && v[2] == 0 && v[3] == 0, + got ? got : v[0] | v[1] << 8 | v[2] << 16 | (long)v[3] << 24); + check("mincore unaligned is EINVAL", rc(mincore(r + 1, PG, v)) == -EINVAL, + rc(mincore(r + 1, PG, v))); + munmap(r + 2 * PG, PG); + check("mincore over a hole is ENOMEM", rc(mincore(r, 4 * PG, v)) == -ENOMEM, + rc(mincore(r, 4 * PG, v))); +} + +int main(void) { + char line[160]; + placement(); + breaks(); + alignment(); + remaps(); + provenance(); + locks(); + syncs(); + cores(); + snprintf(line, sizeof line, "[C] memcalls %s: %d parts ok, %d failed\n", + failed ? "FAIL" : "PASS", passed, failed); + say(line); + return failed ? 1 : 0; +} From 242bcf5736189bc76716cde10295a606b0756995 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 09:07:21 +0000 Subject: [PATCH 10/13] linux: build the memory proofs as one guest, so the test store loads With the gopoll, gopreempt and cwait guests enrolled beside the five memory proof guests, the guest-test store no longer loaded: nonos-store-pack counted 18039927 payload bytes against the vfs budget of 16777216 and refused it, so no guest could boot. Each guest in the store is its binary plus its certificate, manifest and a 327 KiB proof trailer, and the five memory guests came to 1907576 bytes of it, 1683520 of them proofs. The five proofs are now one program, memproof, whose first argument names the proof: guardpage, protnone, protfork, touchfork or memcalls. Each proof keeps its own file and is built with its main renamed, so it runs exactly as it did as a program of its own. memcalls maps its own file for the provenance part, which is now /bin/memproof. The ids 4982 to 4989 are free again. --- userland/capsule_linux/src/linux/abi/mod.rs | 2 +- userland/capsule_linux/src/linux/abi/nr.rs | 2 +- userland/linux_guests/Guests.mk | 49 ++++++++------------- userland/linux_guests/c/memcalls.c | 4 +- userland/linux_guests/c/memproof.c | 35 +++++++++++++++ 5 files changed, 57 insertions(+), 35 deletions(-) create mode 100644 userland/linux_guests/c/memproof.c diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs index 0e5daf6a1d..667abbfa15 100644 --- a/userland/capsule_linux/src/linux/abi/mod.rs +++ b/userland/capsule_linux/src/linux/abi/mod.rs @@ -23,5 +23,5 @@ pub mod name; pub mod nr; pub mod nr_path; pub mod nr_high; -pub mod nr_sched; pub mod nr_mem; +pub mod nr_sched; diff --git a/userland/capsule_linux/src/linux/abi/nr.rs b/userland/capsule_linux/src/linux/abi/nr.rs index 8db6360936..03bd5c1532 100644 --- a/userland/capsule_linux/src/linux/abi/nr.rs +++ b/userland/capsule_linux/src/linux/abi/nr.rs @@ -18,8 +18,8 @@ //! Linux x86_64 syscall numbers, by family. pub use super::nr_high::*; -pub use super::nr_sched::*; pub use super::nr_mem::*; +pub use super::nr_sched::*; pub const READ: u64 = 0; pub const WRITE: u64 = 1; diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index bdc1730f34..0f48028daf 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -118,38 +118,25 @@ $(eval $(call LINUX_GUEST,cthreads,4974,4975,$(LINUX_GUESTS_C)/cthreads)) $(LINUX_GUESTS_C)/cwait: $(LINUX_GUESTS_DIR)/c/cwait.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait)) -# A pthread recursing into its guard page: the process must end on SIGSEGV -# with status 139, and the line it prints if it runs past the guard never shows. -$(LINUX_GUESTS_C)/guardpage: $(LINUX_GUESTS_DIR)/c/guardpage.c - @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< -$(eval $(call LINUX_GUEST,guardpage,4980,4981,$(LINUX_GUESTS_C)/guardpage)) - -# PROT_NONE means no access: a read or write of a PROT_NONE mmap, of a page -# mprotect closed, and of the closed page below an opened one each fault, and -# bytes survive a close and reopen. -$(LINUX_GUESTS_C)/protnone: $(LINUX_GUESTS_DIR)/c/protnone.c - @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< -$(eval $(call LINUX_GUEST,protnone,4982,4983,$(LINUX_GUESTS_C)/protnone)) - -# A fork after mprotect: the child gets the protection the parent has now, so -# a write to a page the parent made read-only faults in the child. -$(LINUX_GUESTS_C)/protfork: $(LINUX_GUESTS_DIR)/c/protfork.c - @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< -$(eval $(call LINUX_GUEST,protfork,4984,4985,$(LINUX_GUESTS_C)/protfork)) -# Bytes written into an opened part of a reservation survive closing it and a -# fork; a page never opened faults in the child; and MAP_FIXED over a written -# page replaces it with zeroes. -$(LINUX_GUESTS_C)/touchfork: $(LINUX_GUESTS_DIR)/c/touchfork.c - @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< -$(eval $(call LINUX_GUEST,touchfork,4986,4987,$(LINUX_GUESTS_C)/touchfork)) - -# The memory calls against Linux's answers: mmap placement, brk giving pages -# back, unaligned addresses, mremap keeping protection and provenance, and -# mlock, mlock2, mlockall, munlockall, msync and mincore with their errnos. -$(LINUX_GUESTS_C)/memcalls: $(LINUX_GUESTS_DIR)/c/memcalls.c - @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< -$(eval $(call LINUX_GUEST,memcalls,4988,4989,$(LINUX_GUESTS_C)/memcalls)) +# The memory proofs, one program whose first argument names the proof, so the +# store carries one binary and one set of proofs for all five: +# guardpage a pthread recursing into its guard page ends on SIGSEGV, 139 +# protnone PROT_NONE means no access, and bytes survive a close and reopen +# protfork a fork after mprotect gives the child the protection set now +# touchfork bytes written into a reservation survive a fork; MAP_FIXED +# over a written page replaces it with zeroes +# memcalls mmap placement, brk, alignment, mremap, and mlock, msync and +# mincore, each against Linux's answer +MEMPROOF_PARTS := guardpage protnone protfork touchfork memcalls +$(LINUX_GUESTS_C)/memproof: $(LINUX_GUESTS_DIR)/c/memproof.c \ + $(foreach p,$(MEMPROOF_PARTS),$(LINUX_GUESTS_DIR)/c/$(p).c) + @mkdir -p $(@D)/memproof.o + @for p in $(MEMPROOF_PARTS); do \ + musl-gcc -O2 -c -Dmain=$${p}_main -o $(@D)/memproof.o/$$p.o $(LINUX_GUESTS_DIR)/c/$$p.c || exit 1; \ + done + @musl-gcc -O2 -static -o $@ $< $(foreach p,$(MEMPROOF_PARTS),$(@D)/memproof.o/$(p).o) +$(eval $(call LINUX_GUEST,memproof,4980,4981,$(LINUX_GUESTS_C)/memproof)) # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the diff --git a/userland/linux_guests/c/memcalls.c b/userland/linux_guests/c/memcalls.c index b663e11416..d8fe1c6d73 100644 --- a/userland/linux_guests/c/memcalls.c +++ b/userland/linux_guests/c/memcalls.c @@ -141,8 +141,8 @@ static void remaps(void) { // it executable, it must refuse the copy mremap moved too. Host Linux allows // both, and the part checks only that the two answers agree. static void provenance(void) { - // This program's own file: /bin/memcalls in the store, itself on a host. - int fd = open("/bin/memcalls", O_RDONLY); + // This program's own file: /bin/memproof in the store, itself on a host. + int fd = open("/bin/memproof", O_RDONLY); if (fd < 0) { fd = open("/proc/self/exe", O_RDONLY); } diff --git a/userland/linux_guests/c/memproof.c b/userland/linux_guests/c/memproof.c new file mode 100644 index 0000000000..fc486b76a0 --- /dev/null +++ b/userland/linux_guests/c/memproof.c @@ -0,0 +1,35 @@ +// The memory proofs as one program, so the test store carries one binary and +// one set of proofs for all of them instead of one per proof: the store has a +// fixed load budget and each proof set is most of a guest's size there. The +// first argument names the proof; each is its own file, built with its main +// renamed, and runs exactly as it would as a program of its own. +#include +#include + +int guardpage_main(void); +int protnone_main(void); +int protfork_main(void); +int touchfork_main(void); +int memcalls_main(void); + +static const struct { + const char *name; + int (*run)(void); +} proofs[] = { + { "guardpage", guardpage_main }, + { "protnone", protnone_main }, + { "protfork", protfork_main }, + { "touchfork", touchfork_main }, + { "memcalls", memcalls_main }, +}; + +int main(int argc, char **argv) { + for (unsigned i = 0; argc > 1 && i < sizeof proofs / sizeof proofs[0]; i++) { + if (strcmp(argv[1], proofs[i].name) == 0) { + return proofs[i].run(); + } + } + fputs("[C] memproof FAIL: name a proof: guardpage protnone protfork touchfork memcalls\n", + stdout); + return 2; +} From a42cf1f43dea0daf1de0c7d348f71753b04902e4 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 09:07:21 +0000 Subject: [PATCH 11/13] linux: split the memory files to 75 lines and write comments as blocks Nine files this branch touches ran past 75 lines, from 76 for map.rs to 224 for memcalls.c, and the comments added in the kernel, personality and guest files were written with // where the rule is /* */. Each long file is split along a line it already had: the three refusals of a demand fill go to faults/demand_refuse.rs; perms_of moves into peer_protect.rs; reserve and map_span leave mem_map.rs; the mprotect walk, the mremap one-mapping check, the mmap kind and the cursor search each get a file. The memory proofs share one file for printing, running a part in a child and counting, and memcalls is split by call family. call/mod.rs is back to its own export line and makes mem public, so the lock, msync and mincore calls are named call::mem::... in the table. No behaviour changes. --- src/memory/paging/manager/faults/demand.rs | 32 +-- .../paging/manager/faults/demand_refuse.rs | 51 +++++ src/memory/paging/manager/faults/mod.rs | 1 + src/process/foreign/peer_guard.rs | 16 +- src/process/foreign/peer_map.rs | 22 +- src/process/foreign/peer_protect.rs | 24 +- .../capsule_linux/src/linux/call/mem/map.rs | 39 +--- .../src/linux/call/mem/map_anon.rs | 10 +- .../src/linux/call/mem/map_file.rs | 2 +- .../src/linux/call/mem/map_free.rs | 40 ++++ .../src/linux/call/mem/map_kind.rs | 44 ++++ .../src/linux/call/mem/map_place.rs | 26 +-- .../src/linux/call/mem/memory.rs | 12 +- .../capsule_linux/src/linux/call/mem/mod.rs | 4 + .../capsule_linux/src/linux/call/mem/prot.rs | 34 +-- .../src/linux/call/mem/prot_walk.rs | 56 +++++ .../capsule_linux/src/linux/call/mem/remap.rs | 31 +-- .../src/linux/call/mem/remap_move.rs | 2 +- .../src/linux/call/mem/remap_one.rs | 38 ++++ .../capsule_linux/src/linux/call/mem/sync.rs | 2 +- userland/capsule_linux/src/linux/call/mod.rs | 6 +- .../src/linux/call/spawn/fork_copy.rs | 6 +- .../capsule_linux/src/linux/guest/mem_like.rs | 2 +- .../capsule_linux/src/linux/guest/mem_map.rs | 40 +--- .../src/linux/guest/mem_reserve.rs | 43 ++++ .../capsule_linux/src/linux/guest/mem_span.rs | 35 +++ userland/capsule_linux/src/linux/guest/mod.rs | 2 + .../src/linux/serve/table_mem.rs | 10 +- userland/linux_guests/Guests.mk | 13 +- userland/linux_guests/c/guardpage.c | 22 +- userland/linux_guests/c/memcalls.c | 209 ++---------------- userland/linux_guests/c/memcalls.h | 30 +++ userland/linux_guests/c/memcalls_lock.c | 55 +++++ userland/linux_guests/c/memcalls_map.c | 66 ++++++ userland/linux_guests/c/memcalls_remap.c | 56 +++++ userland/linux_guests/c/memproof.c | 12 +- userland/linux_guests/c/memproof.h | 27 +++ userland/linux_guests/c/memproof_run.c | 49 ++++ userland/linux_guests/c/protfork.c | 63 ++---- userland/linux_guests/c/protnone.c | 64 ++---- userland/linux_guests/c/touchfork.c | 66 ++---- 41 files changed, 792 insertions(+), 570 deletions(-) create mode 100644 src/memory/paging/manager/faults/demand_refuse.rs create mode 100644 userland/capsule_linux/src/linux/call/mem/map_free.rs create mode 100644 userland/capsule_linux/src/linux/call/mem/map_kind.rs create mode 100644 userland/capsule_linux/src/linux/call/mem/prot_walk.rs create mode 100644 userland/capsule_linux/src/linux/call/mem/remap_one.rs create mode 100644 userland/capsule_linux/src/linux/guest/mem_reserve.rs create mode 100644 userland/capsule_linux/src/linux/guest/mem_span.rs create mode 100644 userland/linux_guests/c/memcalls.h create mode 100644 userland/linux_guests/c/memcalls_lock.c create mode 100644 userland/linux_guests/c/memcalls_map.c create mode 100644 userland/linux_guests/c/memcalls_remap.c create mode 100644 userland/linux_guests/c/memproof.h create mode 100644 userland/linux_guests/c/memproof_run.c diff --git a/src/memory/paging/manager/faults/demand.rs b/src/memory/paging/manager/faults/demand.rs index 3ace5e0c3a..3a6e64f77e 100644 --- a/src/memory/paging/manager/faults/demand.rs +++ b/src/memory/paging/manager/faults/demand.rs @@ -29,36 +29,16 @@ impl PagingManager { virtual_addr: VirtAddr, stats: &PagingStatistics, ) -> PagingResult<()> { - // Only user-space addresses may be demand-backed. A not-present fault - // in the kernel half is never a legitimate lazy mapping; backing it - // silently would hand a capsule kernel-range memory. Surface it as an - // unhandled fault so the fault path kills the offender (user) or traps - // the real kernel bug, instead of papering over it. - if !layout::in_user_space(virtual_addr.as_u64()) { - return Err(PagingError::UnhandledPageFault); - } - - // Never demand-back the null page. A fault in the lowest page is a null - // or near-null dereference; backing it would silently satisfy the bug - // instead of trapping it. Leave the page unmapped as a guard so the - // fault path kills the offending capsule. - if virtual_addr.as_u64() < PAGE_SIZE_4K as u64 { - return Err(PagingError::UnhandledPageFault); - } - - // A foreign guest's pages are exactly the ones its supervisor mapped - // for it. Filling any other page would hand the guest memory nobody - // gave it: a PROT_NONE reservation, a guard page, a hole. So the fault - // is refused, the fault path ends the thread, and its supervisor is - // told and decides what that means for the guest. let pid = crate::process::current_pid().unwrap_or(0); - if crate::process::foreign::is_foreign(pid) { + if super::demand_refuse::refused(virtual_addr.as_u64(), pid) { return Err(PagingError::UnhandledPageFault); } - // Charge the page against the faulting process's demand budget. A - // runaway capsule is refused here and killed by the fault path instead - // of exhausting physical memory. + /* + * Charge the page against the faulting process's demand budget. A + * runaway capsule is refused here and killed by the fault path instead + * of exhausting physical memory. + */ if !super::demand_cap::charge(pid) { return Err(PagingError::UnhandledPageFault); } diff --git a/src/memory/paging/manager/faults/demand_refuse.rs b/src/memory/paging/manager/faults/demand_refuse.rs new file mode 100644 index 0000000000..0c92672cce --- /dev/null +++ b/src/memory/paging/manager/faults/demand_refuse.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The pages the kernel never fills on a fault. + +use crate::memory::layout; +use crate::memory::paging::constants::PAGE_SIZE_4K; + +/// True when a not-present fault at `addr` in `pid` must not be filled. +pub(super) fn refused(addr: u64, pid: u32) -> bool { + /* + * Only user-space addresses may be demand-backed. A not-present fault + * in the kernel half is never a legitimate lazy mapping; backing it + * silently would hand a capsule kernel-range memory. Surface it as an + * unhandled fault so the fault path kills the offender (user) or traps + * the real kernel bug, instead of papering over it. + */ + if !layout::in_user_space(addr) { + return true; + } + /* + * Never demand-back the null page. A fault in the lowest page is a null + * or near-null dereference; backing it would silently satisfy the bug + * instead of trapping it. Leave the page unmapped as a guard so the + * fault path kills the offending capsule. + */ + if addr < PAGE_SIZE_4K as u64 { + return true; + } + /* + * A foreign guest's pages are exactly the ones its supervisor mapped for + * it. Filling any other page would hand the guest memory nobody gave it: + * a PROT_NONE reservation, a guard page, a hole. So the fault is refused, + * the fault path ends the thread, and its supervisor is told and decides + * what that means for the guest. + */ + crate::process::foreign::is_foreign(pid) +} diff --git a/src/memory/paging/manager/faults/mod.rs b/src/memory/paging/manager/faults/mod.rs index de2ee927fc..87fa92be28 100644 --- a/src/memory/paging/manager/faults/mod.rs +++ b/src/memory/paging/manager/faults/mod.rs @@ -17,4 +17,5 @@ mod cow; mod demand; mod demand_cap; +mod demand_refuse; mod handler; diff --git a/src/process/foreign/peer_guard.rs b/src/process/foreign/peer_guard.rs index 976db9d820..596dd00c1a 100644 --- a/src/process/foreign/peer_guard.rs +++ b/src/process/foreign/peer_guard.rs @@ -21,11 +21,13 @@ use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_PERM}; pub(super) const PAGE: u64 = 4096; -// One call maps or copies at most this much, so a guest image crosses in -// bounded pieces and no single call holds the processor. +/* + * One call maps or copies at most this much, so a guest image crosses in + * bounded pieces and no single call holds the processor. + */ pub(super) const MAX_SPAN: u64 = 1 << 20; -// The first address of the kernel half. +/* The first address of the kernel half. */ pub(super) const USER_VA_END: u64 = 0x0000_8000_0000_0000; /// True when `[addr, addr + len)` lies wholly in the guest's own half. @@ -38,9 +40,11 @@ pub(super) fn in_user_half(addr: u64, len: u64) -> bool { pub const PROT_WRITE: u64 = 1 << 0; pub const PROT_EXEC: u64 = 1 << 1; -// No access from the guest at all. The page stays present with the user bit -// clear, so every guest access faults and the frame keeps its bytes for a -// later protection that allows access, as Linux keeps them. +/* + * No access from the guest at all. The page stays present with the user bit + * clear, so every guest access faults and the frame keeps its bytes for a + * later protection that allows access, as Linux keeps them. + */ pub(super) const PROT_NONE: u64 = 1 << 2; diff --git a/src/process/foreign/peer_map.rs b/src/process/foreign/peer_map.rs index 06a66fe85d..2a309a4a55 100644 --- a/src/process/foreign/peer_map.rs +++ b/src/process/foreign/peer_map.rs @@ -18,33 +18,15 @@ use crate::memory::addr::VirtAddr; use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid}; -use crate::memory::paging::types::PagePermissions; use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_NOMEM}; -use super::peer_guard::{ - in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_NONE, PROT_WRITE, -}; +use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE}; +use super::peer_protect::perms_of; fn span_ok(addr: u64, len: u64) -> bool { len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len) } -pub(super) fn perms_of(prot: u64) -> PagePermissions { - // Not USER: present for the kernel, which copies it at fork and frees it - // at teardown, and absent for every access the guest makes. - if prot & PROT_NONE != 0 { - return PagePermissions::READ; - } - let mut perms = PagePermissions::READ | PagePermissions::USER; - if prot & PROT_WRITE != 0 { - perms = perms | PagePermissions::WRITE; - } - if prot & PROT_EXEC != 0 { - perms = perms | PagePermissions::EXECUTE; - } - perms -} - /// `MkPeerMap`: map `[addr, addr + len)` in a guest the caller supervises. pub fn sys_peer_map(pid: u64, addr: u64, len: u64, prot: u64) -> i64 { let Some(caller) = crate::process::current_pid() else { diff --git a/src/process/foreign/peer_protect.rs b/src/process/foreign/peer_protect.rs index 79d74e1ce5..cfde6b0b75 100644 --- a/src/process/foreign/peer_protect.rs +++ b/src/process/foreign/peer_protect.rs @@ -19,10 +19,12 @@ use crate::memory::addr::VirtAddr; use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid}; +use crate::memory::paging::types::PagePermissions; use crate::syscall::microkernel::errnos::{ERRNO_FAULT, ERRNO_INVAL}; -use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE}; -use super::peer_map::perms_of; +use super::peer_guard::{ + in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_NONE, PROT_WRITE, +}; fn span_ok(addr: u64, len: u64) -> bool { len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len) @@ -53,3 +55,21 @@ pub fn sys_peer_protect(pid: u64, addr: u64, len: u64, prot: u64) -> i64 { } 0 } + +pub(super) fn perms_of(prot: u64) -> PagePermissions { + /* + * Not USER: present for the kernel, which copies it at fork and frees it + * at teardown, and absent for every access the guest makes. + */ + if prot & PROT_NONE != 0 { + return PagePermissions::READ; + } + let mut perms = PagePermissions::READ | PagePermissions::USER; + if prot & PROT_WRITE != 0 { + perms = perms | PagePermissions::WRITE; + } + if prot & PROT_EXEC != 0 { + perms = perms | PagePermissions::EXECUTE; + } + perms +} diff --git a/userland/capsule_linux/src/linux/call/mem/map.rs b/userland/capsule_linux/src/linux/call/mem/map.rs index 8571e54762..1baa5ce4ba 100644 --- a/userland/capsule_linux/src/linux/call/mem/map.rs +++ b/userland/capsule_linux/src/linux/call/mem/map.rs @@ -19,20 +19,18 @@ use crate::linux::abi::errno; use crate::linux::guest::{Guest, PAGE}; -use super::map_anon::{anonymous, memfd}; -use super::map_file::file; use super::map_place::place; use super::map_req::MapReq; use super::prot::wx_refused; -const MAP_SHARED: u64 = 0x01; -const MAP_ANONYMOUS: u64 = 0x20; const MAP_FIXED: u64 = 0x10; const MAP_FIXED_NOREPLACE: u64 = 0x10_0000; pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 { - // Linux takes a file offset on a page boundary, and an exact address too; - // only a hint is rounded. + /* + * Linux takes a file offset on a page boundary, and an exact address too; + * only a hint is rounded. + */ let exact = req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0; if req.len == 0 || req.off % PAGE != 0 || (exact && req.addr % PAGE != 0) { return errno::fail(errno::EINVAL); @@ -40,10 +38,12 @@ pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 { if wx_refused(req.prot) { return errno::fail(errno::EPERM); } - // MAP_FIXED and MAP_FIXED_NOREPLACE are the exact address or failure. - // Page zero is never in the plan, and landing elsewhere would hand back - // memory the guest did not ask for, so it is refused, as Linux refuses it - // below mmap_min_addr. + /* + * MAP_FIXED and MAP_FIXED_NOREPLACE are the exact address or failure. + * Page zero is never in the plan, and landing elsewhere would hand back + * memory the guest did not ask for, so it is refused, as Linux refuses it + * below mmap_min_addr. + */ if exact && req.addr == 0 { return errno::fail(errno::EPERM); } @@ -51,26 +51,9 @@ pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 { Ok(spot) => spot, Err(e) => return errno::fail(e), }; - let out = map_at(guest, &req, spot.at, spot.span); + let out = super::map_kind::map_at(guest, &req, spot.at, spot.span); if spot.from_cursor && (out as i64) >= 0 { guest.mmap_next = spot.at + spot.span; } out } - -fn map_at(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { - if req.flags & MAP_ANONYMOUS != 0 { - return anonymous(guest, req, at, span); - } - if crate::linux::file::is_memfd(guest, req.fd) { - return memfd(guest, req, at, span); - } - if req.flags & MAP_SHARED != 0 { - /* - * Sharing a file between processes needs frames that two address - * spaces both point at, which no peer call offers. - */ - return errno::fail(errno::ENOSYS); - } - file(guest, req, at, span) -} diff --git a/userland/capsule_linux/src/linux/call/mem/map_anon.rs b/userland/capsule_linux/src/linux/call/mem/map_anon.rs index bf3bc34480..d3c8ef9946 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_anon.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_anon.rs @@ -47,10 +47,12 @@ pub fn anonymous(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if !req.make_room(guest, at, span) { return errno::fail(errno::ENOMEM); } - // A PROT_NONE anonymous mapping is a reservation: the runtime that makes it - // (Go's, for one) commits a fraction of it later with a fixed RW mapping. - // Backing the whole span here would spend real frames on address space no - // one may touch, so reserve it; a commit maps the part that is opened. + /* + * A PROT_NONE anonymous mapping is a reservation: the runtime that makes it + * (Go's, for one) commits a fraction of it later with a fixed RW mapping. + * Backing the whole span here would spend real frames on address space no + * one may touch, so reserve it; a commit maps the part that is opened. + */ let backed = if req.prot == 0 { guest.reserve(at, span) } else { diff --git a/userland/capsule_linux/src/linux/call/mem/map_file.rs b/userland/capsule_linux/src/linux/call/mem/map_file.rs index a6e3447942..894d395312 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_file.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_file.rs @@ -48,7 +48,7 @@ pub fn file(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if fill_read(guest, req, at) < 0 { return errno::fail(errno::EACCES); } - // Not proved, since nothing asked to run it: it stays that way. + /* Not proved, since nothing asked to run it: it stays that way. */ guest.mark_unproven(at, span); finish(guest, req, at, span) } diff --git a/userland/capsule_linux/src/linux/call/mem/map_free.rs b/userland/capsule_linux/src/linux/call/mem/map_free.rs new file mode 100644 index 0000000000..ccb509375e --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/map_free.rs @@ -0,0 +1,40 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where the mapping cursor finds room. + +use crate::linux::abi::errno; +use crate::linux::guest::{page_up, span_within, Guest, MMAP_LIMIT}; + +/// The first span of `len` at or above the mapping cursor that meets +/// nothing the guest holds. +pub fn free_span(guest: &Guest, len: u64) -> Result<(u64, u64), i64> { + let mut at = guest.mmap_next; + loop { + let (start, span) = span_within(at, len, MMAP_LIMIT).ok_or(errno::ENOMEM)?; + let end = start + span; + let past = guest + .regions + .iter() + .filter(|r| r.at < end && start < r.at.saturating_add(r.len)) + .map(|r| r.at.saturating_add(r.len)) + .max(); + match past { + None => return Ok((start, span)), + Some(next) => at = page_up(next), + } + } +} diff --git a/userland/capsule_linux/src/linux/call/mem/map_kind.rs b/userland/capsule_linux/src/linux/call/mem/map_kind.rs new file mode 100644 index 0000000000..b0b45e11c2 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/map_kind.rs @@ -0,0 +1,44 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which kind of mapping an mmap makes once it has a place. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::map_anon::{anonymous, memfd}; +use super::map_file::file; +use super::map_req::MapReq; + +const MAP_SHARED: u64 = 0x01; +const MAP_ANONYMOUS: u64 = 0x20; + +pub(super) fn map_at(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { + if req.flags & MAP_ANONYMOUS != 0 { + return anonymous(guest, req, at, span); + } + if crate::linux::file::is_memfd(guest, req.fd) { + return memfd(guest, req, at, span); + } + if req.flags & MAP_SHARED != 0 { + /* + * Sharing a file between processes needs frames that two address + * spaces both point at, which no peer call offers. + */ + return errno::fail(errno::ENOSYS); + } + file(guest, req, at, span) +} diff --git a/userland/capsule_linux/src/linux/call/mem/map_place.rs b/userland/capsule_linux/src/linux/call/mem/map_place.rs index 7131c02ba7..830447d2b9 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_place.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_place.rs @@ -21,7 +21,7 @@ //! laid over an old one would hand back the old pages. use crate::linux::abi::errno; -use crate::linux::guest::{page_up, span_within, Guest, MMAP_LIMIT, USER_MAX}; +use crate::linux::guest::{page_up, span_within, Guest, USER_MAX}; use super::map_req::MapReq; @@ -45,7 +45,7 @@ pub fn place(guest: &Guest, req: &MapReq) -> Result { } return Ok(exact((at, span))); } - // Linux rounds a hint up to a page. + /* Linux rounds a hint up to a page. */ if req.addr != 0 { if let Some(got) = span_within(page_up(req.addr), req.len, USER_MAX) { if !guest.overlaps(got.0, got.1) { @@ -53,26 +53,6 @@ pub fn place(guest: &Guest, req: &MapReq) -> Result { } } } - let (at, span) = free_span(guest, req.len)?; + let (at, span) = super::map_free::free_span(guest, req.len)?; Ok(Place { at, span, from_cursor: true }) } - -/// The first span of `len` at or above the mapping cursor that meets -/// nothing the guest holds. -pub fn free_span(guest: &Guest, len: u64) -> Result<(u64, u64), i64> { - let mut at = guest.mmap_next; - loop { - let (start, span) = span_within(at, len, MMAP_LIMIT).ok_or(errno::ENOMEM)?; - let end = start + span; - let past = guest - .regions - .iter() - .filter(|r| r.at < end && start < r.at.saturating_add(r.len)) - .map(|r| r.at.saturating_add(r.len)) - .max(); - match past { - None => return Ok((start, span)), - Some(next) => at = page_up(next), - } - } -} diff --git a/userland/capsule_linux/src/linux/call/mem/memory.rs b/userland/capsule_linux/src/linux/call/mem/memory.rs index fe1b4003b3..227da0d5c1 100644 --- a/userland/capsule_linux/src/linux/call/mem/memory.rs +++ b/userland/capsule_linux/src/linux/call/mem/memory.rs @@ -29,16 +29,18 @@ pub fn brk(guest: &mut Guest, want: u64) -> u64 { if want == 0 || want < BRK_BASE || want > BRK_LIMIT { return errno::ok(guest.brk); } - // Whole pages: the page the old break sits in is already held. + /* Whole pages: the page the old break sits in is already held. */ let (old, top) = (page_up(guest.brk), page_up(want)); if top > old { - // Linux refuses a break that would run into a mapping. + /* Linux refuses a break that would run into a mapping. */ if guest.overlaps(old, top - old) || guest.map(old, top - old, true, false) < 0 { return errno::ok(guest.brk); } } - // A lower break gives the pages above it back, so growing again reads - // zeroes, as on Linux. + /* + * A lower break gives the pages above it back, so growing again reads + * zeroes, as on Linux. + */ if top < old && guest.unmap(top, old - top) < 0 { return errno::ok(guest.brk); } @@ -48,7 +50,7 @@ pub fn brk(guest: &mut Guest, want: u64) -> u64 { /// The pages go back to the kernel and leave the guest's region list. pub fn munmap(guest: &mut Guest, addr: u64, len: u64) -> u64 { - // Linux takes an address on a page boundary, and rounds only the length. + /* Linux takes an address on a page boundary, and rounds only the length. */ if len == 0 || addr % PAGE != 0 { return errno::fail(errno::EINVAL); } diff --git a/userland/capsule_linux/src/linux/call/mem/mod.rs b/userland/capsule_linux/src/linux/call/mem/mod.rs index da5c5496ff..a8f37dd84e 100644 --- a/userland/capsule_linux/src/linux/call/mem/mod.rs +++ b/userland/capsule_linux/src/linux/call/mem/mod.rs @@ -23,13 +23,17 @@ mod map_anon; mod map_exec; mod map_file; mod map_fill; +mod map_free; +mod map_kind; mod map_place; mod map_req; mod memory; mod prot; mod prot_span; +mod prot_walk; mod remap; mod remap_move; +mod remap_one; mod resident; mod sync; diff --git a/userland/capsule_linux/src/linux/call/mem/prot.rs b/userland/capsule_linux/src/linux/call/mem/prot.rs index 4062b07461..ce3bebf235 100644 --- a/userland/capsule_linux/src/linux/call/mem/prot.rs +++ b/userland/capsule_linux/src/linux/call/mem/prot.rs @@ -19,8 +19,6 @@ use crate::linux::abi::errno; use crate::linux::guest::{span_within, Guest, PAGE, USER_MAX}; -use super::prot_span::protect_span; - pub const PROT_WRITE: u64 = 2; pub const PROT_EXEC: u64 = 4; /// PROT_READ, PROT_WRITE and PROT_EXEC together: any access at all. @@ -32,7 +30,7 @@ pub fn wx_refused(prot: u64) -> bool { } pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 { - // Linux takes an address on a page boundary, and rounds only the length. + /* Linux takes an address on a page boundary, and rounds only the length. */ if addr % PAGE != 0 { return errno::fail(errno::EINVAL); } @@ -58,33 +56,5 @@ pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 { if prot & PROT_EXEC != 0 && guest.span_unproven(start, span) { return errno::fail(errno::EPERM); } - let end = start + span; - let mut at = start; - while at < end { - let Some(r) = guest.regions.iter().find(|r| r.at <= at && at < r.at + r.len).copied() - else { - // Linux refuses a span with no mapping in it at all. - return errno::fail(errno::ENOMEM); - }; - let upto = end.min(r.at + r.len); - let piece = upto - at; - if !r.backed { - /* - * A PROT_NONE reservation has no pages for the kernel to - * reprotect. Asking for access commits it, which is how musl makes - * a thread stack: reserve with PROT_NONE, then mprotect the part - * it uses to read-write. PROT_NONE on it changes nothing. The - * commit maps the piece with `prot` and records it. - */ - if prot & PROT_ANY != 0 - && guest.commit(at, piece, prot & PROT_WRITE != 0, prot & PROT_EXEC != 0) < 0 - { - return errno::fail(errno::ENOMEM); - } - } else if protect_span(guest, at, piece, prot) < 0 { - return errno::fail(errno::EACCES); - } - at = upto; - } - errno::ok(0) + super::prot_walk::walk(guest, start, span, prot) } diff --git a/userland/capsule_linux/src/linux/call/mem/prot_walk.rs b/userland/capsule_linux/src/linux/call/mem/prot_walk.rs new file mode 100644 index 0000000000..ab11f99a58 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/prot_walk.rs @@ -0,0 +1,56 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Walking an mprotect span one mapping at a time. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::prot::{PROT_ANY, PROT_EXEC, PROT_WRITE}; +use super::prot_span::protect_span; + +/// Give `[start, start + span)` the protection `prot`, mapping by mapping. +pub(super) fn walk(guest: &mut Guest, start: u64, span: u64, prot: u64) -> u64 { + let end = start + span; + let mut at = start; + while at < end { + let Some(r) = guest.regions.iter().find(|r| r.at <= at && at < r.at + r.len).copied() + else { + /* Linux refuses a span with no mapping in it at all. */ + return errno::fail(errno::ENOMEM); + }; + let upto = end.min(r.at + r.len); + let piece = upto - at; + if !r.backed { + /* + * A PROT_NONE reservation has no pages for the kernel to + * reprotect. Asking for access commits it, which is how musl makes + * a thread stack: reserve with PROT_NONE, then mprotect the part + * it uses to read-write. PROT_NONE on it changes nothing. The + * commit maps the piece with `prot` and records it. + */ + if prot & PROT_ANY != 0 + && guest.commit(at, piece, prot & PROT_WRITE != 0, prot & PROT_EXEC != 0) < 0 + { + return errno::fail(errno::ENOMEM); + } + } else if protect_span(guest, at, piece, prot) < 0 { + return errno::fail(errno::EACCES); + } + at = upto; + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/mem/remap.rs b/userland/capsule_linux/src/linux/call/mem/remap.rs index a046279c90..85c41a2b0a 100644 --- a/userland/capsule_linux/src/linux/call/mem/remap.rs +++ b/userland/capsule_linux/src/linux/call/mem/remap.rs @@ -18,12 +18,12 @@ //! or move with MREMAP_MAYMOVE. glibc's realloc of a large block is this. use crate::linux::abi::errno; -use crate::linux::guest::{page_up, span_within, Guest, Region, PAGE, USER_MAX}; +use crate::linux::guest::{page_up, span_within, Guest, PAGE, USER_MAX}; const MAYMOVE: u64 = 1; pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u64) -> u64 { - // MREMAP_FIXED and DONTUNMAP choose the destination; neither is offered. + /* MREMAP_FIXED and DONTUNMAP choose the destination; neither is offered. */ if flags & !MAYMOVE != 0 || old % PAGE != 0 || old_len == 0 || new_len == 0 { return errno::fail(errno::EINVAL); } @@ -31,11 +31,11 @@ pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u6 let Some(r) = guest.regions.iter().find(|r| r.at <= old && old < r.at + r.len).copied() else { return errno::fail(errno::EFAULT); }; - // Linux moves one mapping at a time: the old span must lie inside one. - if !one_mapping(guest, old, old_len, &r) { + /* Linux moves one mapping at a time: the old span must lie inside one. */ + if !super::remap_one::one_mapping(guest, old, old_len, &r) { return errno::fail(errno::EFAULT); } - // Code was proved where it was mapped; a moved copy would not be. + /* Code was proved where it was mapped; a moved copy would not be. */ if r.exec { return errno::fail(errno::EPERM); } @@ -47,7 +47,7 @@ pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u6 } let tail = old + old_len; let grow = new_len - old_len; - // The grown part is the same mapping: its protection, its backing. + /* The grown part is the same mapping: its protection, its backing. */ if !guest.overlaps(tail, grow) && span_within(tail, grow, USER_MAX).is_some() && guest.map_like(tail, grow, &r) >= 0 @@ -59,22 +59,3 @@ pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u6 } super::remap_move::moved(guest, old, old_len, new_len, &r) } - -/// Every page of `[at, at + len)` is held with the same protection, backing -/// and provenance as `like`, which is what Linux keeps as one mapping. -fn one_mapping(guest: &Guest, at: u64, len: u64, like: &Region) -> bool { - let end = at + len; - let mut reach = at; - while reach < end { - let Some(r) = guest.regions.iter().find(|r| r.at <= reach && reach < r.at + r.len) else { - return false; - }; - let same = (r.write, r.exec, r.access, r.backed, r.unproven) - == (like.write, like.exec, like.access, like.backed, like.unproven); - if !same { - return false; - } - reach = r.at + r.len; - } - true -} diff --git a/userland/capsule_linux/src/linux/call/mem/remap_move.rs b/userland/capsule_linux/src/linux/call/mem/remap_move.rs index b1f7b7f6b4..2b10dad7aa 100644 --- a/userland/capsule_linux/src/linux/call/mem/remap_move.rs +++ b/userland/capsule_linux/src/linux/call/mem/remap_move.rs @@ -19,7 +19,7 @@ use crate::linux::abi::errno; use crate::linux::guest::{Guest, Region}; -use super::map_place::free_span; +use super::map_free::free_span; /// A fresh span at the mapping cursor held like the old one, with the old /// protection, backing and provenance; the old bytes copied in; the old span diff --git a/userland/capsule_linux/src/linux/call/mem/remap_one.rs b/userland/capsule_linux/src/linux/call/mem/remap_one.rs new file mode 100644 index 0000000000..40d0263151 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/remap_one.rs @@ -0,0 +1,38 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Whether an mremap span is one mapping, as Linux requires. + +use crate::linux::guest::{Guest, Region}; + +/// Every page of `[at, at + len)` is held with the same protection, backing +/// and provenance as `like`, which is what Linux keeps as one mapping. +pub(super) fn one_mapping(guest: &Guest, at: u64, len: u64, like: &Region) -> bool { + let end = at + len; + let mut reach = at; + while reach < end { + let Some(r) = guest.regions.iter().find(|r| r.at <= reach && reach < r.at + r.len) else { + return false; + }; + let same = (r.write, r.exec, r.access, r.backed, r.unproven) + == (like.write, like.exec, like.access, like.backed, like.unproven); + if !same { + return false; + } + reach = r.at + r.len; + } + true +} diff --git a/userland/capsule_linux/src/linux/call/mem/sync.rs b/userland/capsule_linux/src/linux/call/mem/sync.rs index 64b3834648..665ff0cfaa 100644 --- a/userland/capsule_linux/src/linux/call/mem/sync.rs +++ b/userland/capsule_linux/src/linux/call/mem/sync.rs @@ -36,7 +36,7 @@ pub fn msync(guest: &Guest, addr: u64, len: u64, flags: u64) -> u64 { let Some(end) = addr.checked_add(page_up(len)).filter(|_| len <= u64::MAX - PAGE) else { return errno::fail(errno::ENOMEM); }; - // Linux reports a span with a page nothing maps as ENOMEM. + /* Linux reports a span with a page nothing maps as ENOMEM. */ if end > addr && guest.mapped_from(addr) < end - addr { return errno::fail(errno::ENOMEM); } diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index 08e27b2b85..6b506d6eea 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -33,7 +33,7 @@ mod limits; mod limits_table; mod glibc; mod glibc_sched; -mod mem; +pub mod mem; mod pipe; mod pipe_dup; mod pipe_end; @@ -65,9 +65,7 @@ pub use life::{exit, exit_thread, set_tid_address}; pub use limits::{getrlimit, prlimit64}; pub use glibc::prctl; pub use glibc_sched::{clone3, getcpu, membarrier, sched_getaffinity}; -pub use mem::{ - brk, mincore, mlock, mlock2, mlockall, mmap, mprotect, mremap, msync, munmap, MapReq, -}; +pub use mem::{brk, mmap, mprotect, mremap, munmap, MapReq}; pub use pipe::pipe2; pub use pipe_dup::{dup, dup2}; pub use pipe_io::write as pipe_write; diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs index fa114ebf0e..c5d907932d 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs @@ -23,8 +23,10 @@ use nonos_libc::peer::{mk_peer_map, mk_peer_write}; pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool { let spans = guest.regions.clone(); for span in spans { - // An unbacked reservation has no frames to copy; the child holds the - // same reservation, and a touch there faults in the child as here. + /* + * An unbacked reservation has no frames to copy; the child holds the + * same reservation, and a touch there faults in the child as here. + */ if !span.backed { continue; } diff --git a/userland/capsule_linux/src/linux/guest/mem_like.rs b/userland/capsule_linux/src/linux/guest/mem_like.rs index 6c6ae90150..aa51cd4409 100644 --- a/userland/capsule_linux/src/linux/guest/mem_like.rs +++ b/userland/capsule_linux/src/linux/guest/mem_like.rs @@ -19,7 +19,7 @@ use super::handle::Guest; use super::mem::MAX_SPAN; -use super::mem_map::map_span; +use super::mem_span::map_span; use super::region::Region; impl Guest { diff --git a/userland/capsule_linux/src/linux/guest/mem_map.rs b/userland/capsule_linux/src/linux/guest/mem_map.rs index 6b1f484dda..9ae63e81f5 100644 --- a/userland/capsule_linux/src/linux/guest/mem_map.rs +++ b/userland/capsule_linux/src/linux/guest/mem_map.rs @@ -16,18 +16,17 @@ //! Backing a span of a guest with pages. -use nonos_libc::peer::mk_peer_map; - use super::handle::Guest; use super::layout::USER_MAX; -use super::mem::{span_within, MAX_SPAN}; +use super::mem::span_within; +use super::mem_span::map_span; use super::region::{peer_prot, Region}; use super::region_cut::cut; impl Guest { /// Pages covering `[addr, addr + len)`. pub fn map(&mut self, addr: u64, len: u64, write: bool, exec: bool) -> i64 { - // Bounded by the top of the guest's area, which is the stack. + /* Bounded by the top of the guest's area, which is the stack. */ let Some((start, span)) = span_within(addr, len, USER_MAX) else { return -1; }; @@ -73,37 +72,4 @@ impl Guest { }); 0 } - - /// Take `len` of address space at `addr` without backing it: a PROT_NONE - /// reservation. No page exists until a commit maps one; a touch before - /// that is a fault, as it is on Linux. - pub fn reserve(&mut self, addr: u64, len: u64) -> i64 { - let Some((start, span)) = span_within(addr, len, USER_MAX) else { - return -1; - }; - self.regions.push(Region { - at: start, - len: span, - write: false, - exec: false, - access: false, - unproven: false, - backed: false, - }); - 0 - } -} - -/// `MkPeerMap` over a span, a megabyte at a time. -pub(super) fn map_span(pid: u32, at: u64, len: u64, prot: u64) -> i64 { - let mut done = 0; - while done < len { - let take = (len - done).min(MAX_SPAN); - let rc = mk_peer_map(pid, at + done, take, prot); - if rc < 0 { - return rc; - } - done += take; - } - 0 } diff --git a/userland/capsule_linux/src/linux/guest/mem_reserve.rs b/userland/capsule_linux/src/linux/guest/mem_reserve.rs new file mode 100644 index 0000000000..bd533af92a --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/mem_reserve.rs @@ -0,0 +1,43 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Taking address space for a guest without backing it. + +use super::handle::Guest; +use super::layout::USER_MAX; +use super::mem::span_within; +use super::region::Region; + +impl Guest { + /// Take `len` of address space at `addr` without backing it: a PROT_NONE + /// reservation. No page exists until a commit maps one; a touch before + /// that is a fault, as it is on Linux. + pub fn reserve(&mut self, addr: u64, len: u64) -> i64 { + let Some((start, span)) = span_within(addr, len, USER_MAX) else { + return -1; + }; + self.regions.push(Region { + at: start, + len: span, + write: false, + exec: false, + access: false, + unproven: false, + backed: false, + }); + 0 + } +} diff --git a/userland/capsule_linux/src/linux/guest/mem_span.rs b/userland/capsule_linux/src/linux/guest/mem_span.rs new file mode 100644 index 0000000000..5208767926 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/mem_span.rs @@ -0,0 +1,35 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Backing a span of a guest, a megabyte at a time. + +use nonos_libc::peer::mk_peer_map; + +use super::mem::MAX_SPAN; + +/// `MkPeerMap` over a span, a megabyte at a time. +pub(super) fn map_span(pid: u32, at: u64, len: u64, prot: u64) -> i64 { + let mut done = 0; + while done < len { + let take = (len - done).min(MAX_SPAN); + let rc = mk_peer_map(pid, at + done, take, prot); + if rc < 0 { + return rc; + } + done += take; + } + 0 +} diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index 4994278f5d..b0e86ebb18 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -38,6 +38,8 @@ mod mem; mod mem_copy; mod mem_like; mod mem_map; +mod mem_reserve; +mod mem_span; mod mem_unmap; mod region; mod region_cut; diff --git a/userland/capsule_linux/src/linux/serve/table_mem.rs b/userland/capsule_linux/src/linux/serve/table_mem.rs index cded74bc72..71e3cf67c3 100644 --- a/userland/capsule_linux/src/linux/serve/table_mem.rs +++ b/userland/capsule_linux/src/linux/serve/table_mem.rs @@ -27,11 +27,11 @@ pub fn mem_ops(guest: &mut Guest, nr: u64, a: [u64; 6]) -> Option { nr::MUNMAP => call::munmap(guest, a[0], a[1]), nr::MPROTECT => call::mprotect(guest, a[0], a[1], a[2]), nr::MREMAP => call::mremap(guest, a[0], a[1], a[2], a[3]), - nr::MSYNC => call::msync(guest, a[0], a[1], a[2]), - nr::MINCORE => call::mincore(guest, a[0], a[1], a[2]), - nr::MLOCK | nr::MUNLOCK => call::mlock(guest, a[0], a[1]), - nr::MLOCK2 => call::mlock2(guest, a[0], a[1], a[2]), - nr::MLOCKALL => call::mlockall(a[0]), + nr::MSYNC => call::mem::msync(guest, a[0], a[1], a[2]), + nr::MINCORE => call::mem::mincore(guest, a[0], a[1], a[2]), + nr::MLOCK | nr::MUNLOCK => call::mem::mlock(guest, a[0], a[1]), + nr::MLOCK2 => call::mem::mlock2(guest, a[0], a[1], a[2]), + nr::MLOCKALL => call::mem::mlockall(a[0]), nr::MUNLOCKALL => errno::ok(0), // Advice, and this capsule takes none of it. nr::MADVISE => errno::ok(0), diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index 0f48028daf..f1966b6b72 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -129,13 +129,20 @@ $(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait)) # memcalls mmap placement, brk, alignment, mremap, and mlock, msync and # mincore, each against Linux's answer MEMPROOF_PARTS := guardpage protnone protfork touchfork memcalls -$(LINUX_GUESTS_C)/memproof: $(LINUX_GUESTS_DIR)/c/memproof.c \ - $(foreach p,$(MEMPROOF_PARTS),$(LINUX_GUESTS_DIR)/c/$(p).c) +# Files the proofs share, built as they are: no main of their own. +MEMPROOF_SHARED := memproof_run memcalls_map memcalls_remap memcalls_lock +MEMPROOF_SRCS := $(foreach p,memproof $(MEMPROOF_PARTS) $(MEMPROOF_SHARED),\ + $(LINUX_GUESTS_DIR)/c/$(p).c) $(LINUX_GUESTS_DIR)/c/memproof.h $(LINUX_GUESTS_DIR)/c/memcalls.h +$(LINUX_GUESTS_C)/memproof: $(MEMPROOF_SRCS) @mkdir -p $(@D)/memproof.o @for p in $(MEMPROOF_PARTS); do \ musl-gcc -O2 -c -Dmain=$${p}_main -o $(@D)/memproof.o/$$p.o $(LINUX_GUESTS_DIR)/c/$$p.c || exit 1; \ done - @musl-gcc -O2 -static -o $@ $< $(foreach p,$(MEMPROOF_PARTS),$(@D)/memproof.o/$(p).o) + @for p in $(MEMPROOF_SHARED); do \ + musl-gcc -O2 -c -o $(@D)/memproof.o/$$p.o $(LINUX_GUESTS_DIR)/c/$$p.c || exit 1; \ + done + @musl-gcc -O2 -static -o $@ $(LINUX_GUESTS_DIR)/c/memproof.c \ + $(foreach p,$(MEMPROOF_PARTS) $(MEMPROOF_SHARED),$(@D)/memproof.o/$(p).o) $(eval $(call LINUX_GUEST,memproof,4980,4981,$(LINUX_GUESTS_C)/memproof)) # The Linux-guest test store is about guests, not the desktop's media and demo diff --git a/userland/linux_guests/c/guardpage.c b/userland/linux_guests/c/guardpage.c index 716737e2f8..34e2ad1d81 100644 --- a/userland/linux_guests/c/guardpage.c +++ b/userland/linux_guests/c/guardpage.c @@ -1,10 +1,12 @@ -// A pthread recurses until it walks off the bottom of its stack. musl reserves -// each thread stack with PROT_NONE and opens all but the lowest part with -// mprotect, so the part it leaves closed is the guard. On Linux the first touch -// of the guard is SIGSEGV, which ends the whole process with status 139. The -// recursion stops by itself 64 KiB below the guard, so a guard that guards -// nothing prints the FAIL line with how far it ran; no PASS line exists, since -// the only correct outcome is that the process does not get to print one. +/* + * A pthread recurses until it walks off the bottom of its stack. musl reserves + * each thread stack with PROT_NONE and opens all but the lowest part with + * mprotect, so the part it leaves closed is the guard. On Linux the first touch + * of the guard is SIGSEGV, which ends the whole process with status 139. The + * recursion stops by itself 64 KiB below the guard, so a guard that guards + * nothing prints the FAIL line with how far it ran; no PASS line exists, since + * the only correct outcome is that the process does not get to print one. + */ #define _GNU_SOURCE #include #include @@ -12,16 +14,14 @@ #include #include +#include "memproof.h" + #define FRAME 1024 #define PAST (64 * 1024) static uintptr_t lo; static uintptr_t deepest; -static void say(const char *s) { - write(1, s, strlen(s)); -} - static unsigned dive(unsigned depth) { volatile char frame[FRAME]; uintptr_t here = (uintptr_t)frame; diff --git a/userland/linux_guests/c/memcalls.c b/userland/linux_guests/c/memcalls.c index d8fe1c6d73..437cd19dd8 100644 --- a/userland/linux_guests/c/memcalls.c +++ b/userland/linux_guests/c/memcalls.c @@ -1,43 +1,27 @@ -// The memory calls answered as Linux answers them: where mmap puts a hint, the -// break giving pages back, unaligned addresses refused, mremap keeping a -// mapping's protection, and mlock, msync and mincore with their errnos. Every -// part runs and prints one line, so one boot names every part that fails. -// Parts that must fault run in a forked child; the personality reports a -// signal death as exit status 128+signo, counted as the same SIGSEGV. -#define _GNU_SOURCE +/* + * The memory calls answered as Linux answers them: where mmap puts a hint, the + * break giving pages back, unaligned addresses refused, mremap keeping a + * mapping's protection, and mlock, msync and mincore with their errnos. Every + * part runs and prints one line, so one boot names every part that fails. + * Parts that must fault run in a forked child. + */ #include -#include -#include #include -#include #include -#include #include #include -#define PG 4096 -#define NOREPLACE 0x100000 +#include "memcalls.h" -static int failed, passed; -static volatile char *p; +volatile char *mc_p; -static void say(const char *s) { - write(1, s, strlen(s)); +void check(const char *name, int ok, long got) { + char detail[48]; + snprintf(detail, sizeof detail, "got %ld", got); + part_line("memcalls", name, ok, detail); } -static void check(const char *name, int ok, long got) { - char line[160]; - ok ? passed++ : failed++; - snprintf(line, sizeof line, "[C] memcalls %s: %s (got %ld)\n", name, ok ? "ok" : "FAIL", got); - say(line); -} - -// -errno of a call that returned -1, or its value. -static long rc(long v) { - return v == -1 ? -errno : v; -} - -static void faults(const char *name, void (*fn)(void)) { +void faults(const char *name, void (*fn)(void)) { pid_t c = fork(); if (c == 0) { fn(); @@ -45,170 +29,26 @@ static void faults(const char *name, void (*fn)(void)) { } int st = 0; waitpid(c, &st, 0); - int segv = (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || - (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); - check(name, segv, st); + check(name, segv(st), st); } -static void write_p(void) { - p[0] = 1; -} -static void read_p(void) { - (void)p[0]; +long rc(long v) { + return v == -1 ? -errno : v; } -static char *anon(long len, int prot) { +char *anon(long len, int prot) { return mmap(0, len, prot, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); } -static void placement(void) { - char *m = anon(PG, PROT_READ | PROT_WRITE); - m[0] = 0x11; - char *n = mmap(m, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); - check("hint on a mapping lands elsewhere, zeroed", n != m && n[0] == 0 && m[0] == 0x11, - (long)(n - m)); - void *q = mmap(m, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); - long e = q == MAP_FAILED ? -errno : 0; - check("MAP_FIXED_NOREPLACE on a mapping is EEXIST", e == -EEXIST && m[0] == 0x11, e); - // A mapping placed just above the last one mmap chose: the next mmap that - // leaves the choice to the system must not land on it. On a system that - // already holds that address the probe cannot be placed, and says so. - char *a = anon(PG, PROT_READ | PROT_WRITE); - char *f = mmap(a + PG, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); - if (f == MAP_FAILED) { - check("next mmap skips a mapping placed above the last (probe address taken)", - errno == EEXIST, -errno); - return; - } - f[0] = 0x44; - char *b = anon(PG, PROT_READ | PROT_WRITE); - check("next mmap skips a mapping placed above the last", b != f && b[0] == 0 && f[0] == 0x44, - (long)(b - f)); -} - -static void breaks(void) { - long cur = syscall(SYS_brk, 0); - long up = syscall(SYS_brk, cur + 2 * PG); - ((volatile char *)cur)[PG] = 0x22; - syscall(SYS_brk, cur); - syscall(SYS_brk, cur + 2 * PG); - char b = ((volatile char *)cur)[PG]; - check("brk down and up again reads zero", up == cur + 2 * PG && b == 0, b); - syscall(SYS_brk, cur); - long top = (cur + PG - 1) & ~(long)(PG - 1); - void *in = mmap((void *)(top + PG), PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); - long got = syscall(SYS_brk, top + 4 * PG); - check("brk into a mapping is refused", in != MAP_FAILED && got == cur, got - cur); - munmap(in, PG); -} - -static void alignment(void) { - char *m = anon(2 * PG, PROT_READ | PROT_WRITE); - check("munmap unaligned is EINVAL", rc(munmap(m + 1, PG)) == -EINVAL, rc(munmap(m + 1, PG))); - // musl's mprotect rounds the address down itself; the kernel's does not. - long e = rc(syscall(SYS_mprotect, m + 1, PG, PROT_READ)); - check("mprotect unaligned is EINVAL", e == -EINVAL, e); - void *q = mmap(m + 1, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); - e = q == MAP_FAILED ? -errno : 0; - check("MAP_FIXED unaligned is EINVAL", e == -EINVAL, e); - // musl's mmap refuses an unaligned offset itself; the kernel's must too. - e = rc(syscall(SYS_mmap, 0, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS, -1, 1)); - check("mmap offset unaligned is EINVAL", e == -EINVAL, e); -} - -static void remaps(void) { - char *m = anon(PG, PROT_READ | PROT_WRITE); - m[0] = 0x33; - mprotect(m, PG, PROT_READ); - char *g = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE); - check("mremap of a read-only page keeps the byte", g != MAP_FAILED && g[0] == 0x33, g[0]); - p = g + PG; - faults("mremap grown part of a read-only page faults on write", write_p); - char *r = anon(2 * PG, PROT_NONE); - anon(PG, PROT_NONE); - char *q = mremap(r, 2 * PG, 4 * PG, MREMAP_MAYMOVE); - check("mremap of a reservation succeeds", q != MAP_FAILED, (long)(q == MAP_FAILED)); - p = q + 3 * PG; - faults("mremap grown reservation still faults on read", read_p); - char *two = anon(2 * PG, PROT_READ | PROT_WRITE); - mprotect(two + PG, PG, PROT_READ); - void *x = mremap(two, 2 * PG, 3 * PG, MREMAP_MAYMOVE); - long e = x == MAP_FAILED ? -errno : 0; - check("mremap across two mappings is EFAULT", e == -EFAULT, e); -} - -// A file mapped without exec was never proved; where mprotect refuses to make -// it executable, it must refuse the copy mremap moved too. Host Linux allows -// both, and the part checks only that the two answers agree. -static void provenance(void) { - // This program's own file: /bin/memproof in the store, itself on a host. - int fd = open("/bin/memproof", O_RDONLY); - if (fd < 0) { - fd = open("/proc/self/exe", O_RDONLY); - } - char *m = mmap(0, PG, PROT_READ, MAP_PRIVATE, fd, 0); - if (m == MAP_FAILED) { - check("file mapping for the provenance part", 0, -errno); - return; - } - long first = rc(mprotect(m, PG, PROT_READ | PROT_EXEC)); - mprotect(m, PG, PROT_READ); - mmap(m + PG, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); - char *n = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE); - long moved = n == MAP_FAILED ? -1000 : rc(mprotect(n, PG, PROT_READ | PROT_EXEC)); - check("moved unproven file bytes refused exec as before the move", n != m && moved == first, - moved); - close(fd); -} - -static void locks(void) { - char *m = anon(3 * PG, PROT_READ | PROT_WRITE); - check("mlock of a mapping", rc(mlock(m, 3 * PG)) == 0, rc(mlock(m, 3 * PG))); - check("munlock of a mapping", rc(munlock(m, 3 * PG)) == 0, rc(munlock(m, 3 * PG))); - munmap(m + PG, PG); - check("mlock over a hole is ENOMEM", rc(mlock(m, 3 * PG)) == -ENOMEM, rc(mlock(m, 3 * PG))); - check("mlock2 with an unknown flag is EINVAL", rc(syscall(SYS_mlock2, m, PG, 2)) == -EINVAL, - rc(syscall(SYS_mlock2, m, PG, 2))); - check("mlock2 MLOCK_ONFAULT", rc(syscall(SYS_mlock2, m, PG, 1)) == 0, - rc(syscall(SYS_mlock2, m, PG, 1))); - check("mlockall(0) is EINVAL", rc(mlockall(0)) == -EINVAL, rc(mlockall(0))); - check("mlockall(MCL_ONFAULT) alone is EINVAL", rc(mlockall(MCL_ONFAULT)) == -EINVAL, - rc(mlockall(MCL_ONFAULT))); - check("mlockall(MCL_CURRENT)", rc(mlockall(MCL_CURRENT)) == 0, rc(mlockall(MCL_CURRENT))); - check("munlockall", rc(munlockall()) == 0, rc(munlockall())); -} - -static void syncs(void) { - char *m = anon(3 * PG, PROT_READ | PROT_WRITE); - check("msync MS_SYNC", rc(msync(m, 3 * PG, MS_SYNC)) == 0, rc(msync(m, 3 * PG, MS_SYNC))); - check("msync unaligned is EINVAL", rc(msync(m + 1, PG, MS_SYNC)) == -EINVAL, - rc(msync(m + 1, PG, MS_SYNC))); - check("msync MS_ASYNC|MS_SYNC is EINVAL", rc(msync(m, PG, MS_ASYNC | MS_SYNC)) == -EINVAL, - rc(msync(m, PG, MS_ASYNC | MS_SYNC))); - check("msync unknown flag is EINVAL", rc(msync(m, PG, 8)) == -EINVAL, rc(msync(m, PG, 8))); - munmap(m + PG, PG); - check("msync over a hole is ENOMEM", rc(msync(m, 3 * PG, MS_SYNC)) == -ENOMEM, - rc(msync(m, 3 * PG, MS_SYNC))); +void mc_write(void) { + mc_p[0] = 1; } -static void cores(void) { - char *r = anon(4 * PG, PROT_NONE); - mprotect(r + PG, PG, PROT_READ | PROT_WRITE); - r[PG] = 1; - unsigned char v[4] = { 9, 9, 9, 9 }; - long got = rc(mincore(r, 4 * PG, v)); - check("mincore of a reservation with one page opened is 0,1,0,0", - got == 0 && v[0] == 0 && v[1] == 1 && v[2] == 0 && v[3] == 0, - got ? got : v[0] | v[1] << 8 | v[2] << 16 | (long)v[3] << 24); - check("mincore unaligned is EINVAL", rc(mincore(r + 1, PG, v)) == -EINVAL, - rc(mincore(r + 1, PG, v))); - munmap(r + 2 * PG, PG); - check("mincore over a hole is ENOMEM", rc(mincore(r, 4 * PG, v)) == -ENOMEM, - rc(mincore(r, 4 * PG, v))); +void mc_read(void) { + (void)mc_p[0]; } int main(void) { - char line[160]; placement(); breaks(); alignment(); @@ -217,8 +57,5 @@ int main(void) { locks(); syncs(); cores(); - snprintf(line, sizeof line, "[C] memcalls %s: %d parts ok, %d failed\n", - failed ? "FAIL" : "PASS", passed, failed); - say(line); - return failed ? 1 : 0; + return finish("memcalls"); } diff --git a/userland/linux_guests/c/memcalls.h b/userland/linux_guests/c/memcalls.h new file mode 100644 index 0000000000..4839239261 --- /dev/null +++ b/userland/linux_guests/c/memcalls.h @@ -0,0 +1,30 @@ +/* What the memcalls files share. */ +#ifndef MEMCALLS_H +#define MEMCALLS_H + +#include "memproof.h" + +#define NOREPLACE 0x100000 + +/* The page a part run in a child touches. */ +extern volatile char *mc_p; + +void check(const char *name, int ok, long got); +/* A part run in a forked child that must die of SIGSEGV. */ +void faults(const char *name, void (*fn)(void)); +/* -errno of a call that returned -1, or its value. */ +long rc(long v); +char *anon(long len, int prot); +void mc_write(void); +void mc_read(void); + +void placement(void); +void breaks(void); +void alignment(void); +void remaps(void); +void provenance(void); +void locks(void); +void syncs(void); +void cores(void); + +#endif diff --git a/userland/linux_guests/c/memcalls_lock.c b/userland/linux_guests/c/memcalls_lock.c new file mode 100644 index 0000000000..b8b74ce02a --- /dev/null +++ b/userland/linux_guests/c/memcalls_lock.c @@ -0,0 +1,55 @@ +/* memcalls: mlock, msync and mincore with Linux's errnos. */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "memcalls.h" + +void locks(void) { + char *m = anon(3 * PG, PROT_READ | PROT_WRITE); + check("mlock of a mapping", rc(mlock(m, 3 * PG)) == 0, rc(mlock(m, 3 * PG))); + check("munlock of a mapping", rc(munlock(m, 3 * PG)) == 0, rc(munlock(m, 3 * PG))); + munmap(m + PG, PG); + check("mlock over a hole is ENOMEM", rc(mlock(m, 3 * PG)) == -ENOMEM, rc(mlock(m, 3 * PG))); + check("mlock2 with an unknown flag is EINVAL", rc(syscall(SYS_mlock2, m, PG, 2)) == -EINVAL, + rc(syscall(SYS_mlock2, m, PG, 2))); + check("mlock2 MLOCK_ONFAULT", rc(syscall(SYS_mlock2, m, PG, 1)) == 0, + rc(syscall(SYS_mlock2, m, PG, 1))); + check("mlockall(0) is EINVAL", rc(mlockall(0)) == -EINVAL, rc(mlockall(0))); + check("mlockall(MCL_ONFAULT) alone is EINVAL", rc(mlockall(MCL_ONFAULT)) == -EINVAL, + rc(mlockall(MCL_ONFAULT))); + check("mlockall(MCL_CURRENT)", rc(mlockall(MCL_CURRENT)) == 0, rc(mlockall(MCL_CURRENT))); + check("munlockall", rc(munlockall()) == 0, rc(munlockall())); +} + +void syncs(void) { + char *m = anon(3 * PG, PROT_READ | PROT_WRITE); + check("msync MS_SYNC", rc(msync(m, 3 * PG, MS_SYNC)) == 0, rc(msync(m, 3 * PG, MS_SYNC))); + check("msync unaligned is EINVAL", rc(msync(m + 1, PG, MS_SYNC)) == -EINVAL, + rc(msync(m + 1, PG, MS_SYNC))); + check("msync MS_ASYNC|MS_SYNC is EINVAL", rc(msync(m, PG, MS_ASYNC | MS_SYNC)) == -EINVAL, + rc(msync(m, PG, MS_ASYNC | MS_SYNC))); + check("msync unknown flag is EINVAL", rc(msync(m, PG, 8)) == -EINVAL, rc(msync(m, PG, 8))); + munmap(m + PG, PG); + check("msync over a hole is ENOMEM", rc(msync(m, 3 * PG, MS_SYNC)) == -ENOMEM, + rc(msync(m, 3 * PG, MS_SYNC))); +} + +void cores(void) { + char *r = anon(4 * PG, PROT_NONE); + mprotect(r + PG, PG, PROT_READ | PROT_WRITE); + r[PG] = 1; + unsigned char v[4] = { 9, 9, 9, 9 }; + long got = rc(mincore(r, 4 * PG, v)); + check("mincore of a reservation with one page opened is 0,1,0,0", + got == 0 && v[0] == 0 && v[1] == 1 && v[2] == 0 && v[3] == 0, + got ? got : v[0] | v[1] << 8 | v[2] << 16 | (long)v[3] << 24); + check("mincore unaligned is EINVAL", rc(mincore(r + 1, PG, v)) == -EINVAL, + rc(mincore(r + 1, PG, v))); + munmap(r + 2 * PG, PG); + check("mincore over a hole is ENOMEM", rc(mincore(r, 4 * PG, v)) == -ENOMEM, + rc(mincore(r, 4 * PG, v))); +} diff --git a/userland/linux_guests/c/memcalls_map.c b/userland/linux_guests/c/memcalls_map.c new file mode 100644 index 0000000000..2adb3be11f --- /dev/null +++ b/userland/linux_guests/c/memcalls_map.c @@ -0,0 +1,66 @@ +/* memcalls: where mmap puts a mapping, the break, and unaligned addresses. */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "memcalls.h" + +void placement(void) { + char *m = anon(PG, PROT_READ | PROT_WRITE); + m[0] = 0x11; + char *n = mmap(m, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + check("hint on a mapping lands elsewhere, zeroed", n != m && n[0] == 0 && m[0] == 0x11, + (long)(n - m)); + void *q = mmap(m, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + long e = q == MAP_FAILED ? -errno : 0; + check("MAP_FIXED_NOREPLACE on a mapping is EEXIST", e == -EEXIST && m[0] == 0x11, e); + /* + * A mapping placed just above the last one mmap chose: the next mmap that + * leaves the choice to the system must not land on it. On a system that + * already holds that address the probe cannot be placed, and says so. + */ + char *a = anon(PG, PROT_READ | PROT_WRITE); + char *f = mmap(a + PG, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + if (f == MAP_FAILED) { + check("next mmap skips a mapping placed above the last (probe address taken)", + errno == EEXIST, -errno); + return; + } + f[0] = 0x44; + char *b = anon(PG, PROT_READ | PROT_WRITE); + check("next mmap skips a mapping placed above the last", b != f && b[0] == 0 && f[0] == 0x44, + (long)(b - f)); +} + +void breaks(void) { + long cur = syscall(SYS_brk, 0); + long up = syscall(SYS_brk, cur + 2 * PG); + ((volatile char *)cur)[PG] = 0x22; + syscall(SYS_brk, cur); + syscall(SYS_brk, cur + 2 * PG); + char b = ((volatile char *)cur)[PG]; + check("brk down and up again reads zero", up == cur + 2 * PG && b == 0, b); + syscall(SYS_brk, cur); + long top = (cur + PG - 1) & ~(long)(PG - 1); + void *in = mmap((void *)(top + PG), PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + long got = syscall(SYS_brk, top + 4 * PG); + check("brk into a mapping is refused", in != MAP_FAILED && got == cur, got - cur); + munmap(in, PG); +} + +void alignment(void) { + char *m = anon(2 * PG, PROT_READ | PROT_WRITE); + check("munmap unaligned is EINVAL", rc(munmap(m + 1, PG)) == -EINVAL, rc(munmap(m + 1, PG))); + /* musl's mprotect rounds the address down itself; the kernel's does not. */ + long e = rc(syscall(SYS_mprotect, m + 1, PG, PROT_READ)); + check("mprotect unaligned is EINVAL", e == -EINVAL, e); + void *q = mmap(m + 1, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); + e = q == MAP_FAILED ? -errno : 0; + check("MAP_FIXED unaligned is EINVAL", e == -EINVAL, e); + /* musl's mmap refuses an unaligned offset itself; the kernel's must too. */ + e = rc(syscall(SYS_mmap, 0, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS, -1, 1)); + check("mmap offset unaligned is EINVAL", e == -EINVAL, e); +} diff --git a/userland/linux_guests/c/memcalls_remap.c b/userland/linux_guests/c/memcalls_remap.c new file mode 100644 index 0000000000..add9b6d0e4 --- /dev/null +++ b/userland/linux_guests/c/memcalls_remap.c @@ -0,0 +1,56 @@ +/* memcalls: mremap keeping a mapping's protection, backing and provenance. */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "memcalls.h" + +void remaps(void) { + char *m = anon(PG, PROT_READ | PROT_WRITE); + m[0] = 0x33; + mprotect(m, PG, PROT_READ); + char *g = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE); + check("mremap of a read-only page keeps the byte", g != MAP_FAILED && g[0] == 0x33, g[0]); + mc_p = g + PG; + faults("mremap grown part of a read-only page faults on write", mc_write); + char *r = anon(2 * PG, PROT_NONE); + anon(PG, PROT_NONE); + char *q = mremap(r, 2 * PG, 4 * PG, MREMAP_MAYMOVE); + check("mremap of a reservation succeeds", q != MAP_FAILED, (long)(q == MAP_FAILED)); + mc_p = q + 3 * PG; + faults("mremap grown reservation still faults on read", mc_read); + char *two = anon(2 * PG, PROT_READ | PROT_WRITE); + mprotect(two + PG, PG, PROT_READ); + void *x = mremap(two, 2 * PG, 3 * PG, MREMAP_MAYMOVE); + long e = x == MAP_FAILED ? -errno : 0; + check("mremap across two mappings is EFAULT", e == -EFAULT, e); +} + +/* + * A file mapped without exec was never proved; where mprotect refuses to make + * it executable, it must refuse the copy mremap moved too. Host Linux allows + * both, and the part checks only that the two answers agree. + */ +void provenance(void) { + /* This program's own file: /bin/memproof in the store, itself on a host. */ + int fd = open("/bin/memproof", O_RDONLY); + if (fd < 0) { + fd = open("/proc/self/exe", O_RDONLY); + } + char *m = mmap(0, PG, PROT_READ, MAP_PRIVATE, fd, 0); + if (m == MAP_FAILED) { + check("file mapping for the provenance part", 0, -errno); + return; + } + long first = rc(mprotect(m, PG, PROT_READ | PROT_EXEC)); + mprotect(m, PG, PROT_READ); + mmap(m + PG, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + char *n = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE); + long moved = n == MAP_FAILED ? -1000 : rc(mprotect(n, PG, PROT_READ | PROT_EXEC)); + check("moved unproven file bytes refused exec as before the move", n != m && moved == first, + moved); + close(fd); +} diff --git a/userland/linux_guests/c/memproof.c b/userland/linux_guests/c/memproof.c index fc486b76a0..bb094f7315 100644 --- a/userland/linux_guests/c/memproof.c +++ b/userland/linux_guests/c/memproof.c @@ -1,8 +1,10 @@ -// The memory proofs as one program, so the test store carries one binary and -// one set of proofs for all of them instead of one per proof: the store has a -// fixed load budget and each proof set is most of a guest's size there. The -// first argument names the proof; each is its own file, built with its main -// renamed, and runs exactly as it would as a program of its own. +/* + * The memory proofs as one program, so the test store carries one binary and + * one set of proofs for all of them instead of one per proof: the store has a + * fixed load budget and each proof set is most of a guest's size there. The + * first argument names the proof; each is its own file, built with its main + * renamed, and runs exactly as it would as a program of its own. + */ #include #include diff --git a/userland/linux_guests/c/memproof.h b/userland/linux_guests/c/memproof.h new file mode 100644 index 0000000000..69dd071520 --- /dev/null +++ b/userland/linux_guests/c/memproof.h @@ -0,0 +1,27 @@ +/* + * What every memory proof shares: printing a line, running a part in a forked + * child, reading how the child ended, and counting parts for the last line. + */ +#ifndef MEMPROOF_H +#define MEMPROOF_H + +#define PG 4096 + +void say(const char *s); + +/* + * A SIGSEGV death. The personality reports a signal death as exit status + * 128+signo, which counts as the same SIGSEGV; the raw status is printed. + */ +int segv(int st); + +/* Count one part and print its line: "[C] : ok|FAIL ()". */ +void part_line(const char *proof, const char *name, int ok, const char *detail); + +/* Run `fn` in a forked child that must, or must not, die of SIGSEGV. */ +void part(const char *proof, const char *name, void (*fn)(void), int must_fault); + +/* Print the PASS or FAIL line and give the exit code. */ +int finish(const char *proof); + +#endif diff --git a/userland/linux_guests/c/memproof_run.c b/userland/linux_guests/c/memproof_run.c new file mode 100644 index 0000000000..88f3b59939 --- /dev/null +++ b/userland/linux_guests/c/memproof_run.c @@ -0,0 +1,49 @@ +/* The parts of memproof.h every proof shares. */ +#include +#include +#include +#include +#include + +#include "memproof.h" + +static int passed, failed; + +void say(const char *s) { + write(1, s, strlen(s)); +} + +int segv(int st) { + return (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || + (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); +} + +void part_line(const char *proof, const char *name, int ok, const char *detail) { + char line[200]; + ok ? passed++ : failed++; + snprintf(line, sizeof line, "[C] %s %s: %s (%s)\n", proof, name, ok ? "ok" : "FAIL", detail); + say(line); +} + +void part(const char *proof, const char *name, void (*fn)(void), int must_fault) { + char detail[64]; + pid_t c = fork(); + if (c == 0) { + fn(); + _exit(0); + } + int st = 0; + waitpid(c, &st, 0); + int clean = WIFEXITED(st) && WEXITSTATUS(st) == 0; + snprintf(detail, sizeof detail, "%s, status 0x%x", must_fault ? "must fault" : "must not fault", + st); + part_line(proof, name, must_fault ? segv(st) : clean, detail); +} + +int finish(const char *proof) { + char line[160]; + snprintf(line, sizeof line, "[C] %s %s: %d parts ok, %d failed\n", proof, + failed ? "FAIL" : "PASS", passed, failed); + say(line); + return failed ? 1 : 0; +} diff --git a/userland/linux_guests/c/protfork.c b/userland/linux_guests/c/protfork.c index 8b0463caa9..634db301e5 100644 --- a/userland/linux_guests/c/protfork.c +++ b/userland/linux_guests/c/protfork.c @@ -1,43 +1,16 @@ -// A fork gives the child the parent's mappings with the protection they have -// now, not the one they were made with. Each part changes a protection with -// mprotect, forks, and the child tries one access; the parent reads how the -// child ended. The personality reports a signal death as exit status -// 128+signo, which is counted as the same SIGSEGV; the raw status is printed. -#include -#include -#include +/* + * A fork gives the child the parent's mappings with the protection they have + * now, not the one they were made with. Each part changes a protection with + * mprotect, forks, and the child tries one access; the parent reads how the + * child ended. + */ #include -#include #include -#define PG 4096 +#include "memproof.h" -static int failed, passed; static volatile char *p; -static void say(const char *s) { - write(1, s, strlen(s)); -} - -static void part(const char *name, void (*fn)(void), int must_fault) { - char line[160]; - pid_t c = fork(); - if (c == 0) { - fn(); - _exit(0); - } - int st = 0; - waitpid(c, &st, 0); - int segv = (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || - (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); - int clean = WIFEXITED(st) && WEXITSTATUS(st) == 0; - int ok = must_fault ? segv : clean; - ok ? passed++ : failed++; - snprintf(line, sizeof line, "[C] protfork %s: %s (%s, status 0x%x)\n", name, - ok ? "ok" : "FAIL", must_fault ? "must fault" : "must not fault", st); - say(line); -} - static void write_it(void) { p[0] = 1; if (p[0] != 1) { @@ -60,32 +33,28 @@ static volatile char *fresh(int pages) { } int main(void) { - char line[160]; + const char *me = "protfork"; p = fresh(1); mprotect((void *)p, PG, PROT_READ); - part("RW to R, child writes", write_it, 1); - part("RW to R, child reads the byte", read_it, 0); + part(me, "RW to R, child writes", write_it, 1); + part(me, "RW to R, child reads the byte", read_it, 0); p = fresh(1); mprotect((void *)p, PG, PROT_NONE); - part("RW to NONE, child reads", read_it, 1); + part(me, "RW to NONE, child reads", read_it, 1); p = fresh(1); mprotect((void *)p, PG, PROT_READ); mprotect((void *)p, PG, PROT_READ | PROT_WRITE); - part("RW to R to RW, child writes", write_it, 0); + part(me, "RW to R to RW, child writes", write_it, 0); volatile char *m = fresh(3); mprotect((void *)(m + PG), PG, PROT_READ); p = m; - part("middle page R, child writes the first", write_it, 0); + part(me, "middle page R, child writes the first", write_it, 0); p = m + PG; - part("middle page R, child writes the middle", write_it, 1); + part(me, "middle page R, child writes the middle", write_it, 1); p = m + 2 * PG; - part("middle page R, child writes the last", write_it, 0); - - snprintf(line, sizeof line, "[C] protfork %s: %d parts ok, %d failed\n", - failed ? "FAIL" : "PASS", passed, failed); - say(line); - return failed ? 1 : 0; + part(me, "middle page R, child writes the last", write_it, 0); + return finish(me); } diff --git a/userland/linux_guests/c/protnone.c b/userland/linux_guests/c/protnone.c index dac6a37fc4..dfc9f52754 100644 --- a/userland/linux_guests/c/protnone.c +++ b/userland/linux_guests/c/protnone.c @@ -1,44 +1,16 @@ -// PROT_NONE means no access. Each part runs in a forked child and the parent -// reads how the child ended: a part that must fault passes only when the child -// dies of SIGSEGV, a part that must not fault passes only when it exits 0. -// Every part runs, so one boot names every part that fails. The personality -// reports a signal death as exit status 128+signo, which is counted as the -// same SIGSEGV; the raw status is printed either way. -#include -#include -#include +/* + * PROT_NONE means no access. Each part runs in a forked child and the parent + * reads how the child ended: a part that must fault passes only when the child + * dies of SIGSEGV, a part that must not fault passes only when it exits 0. + * Every part runs, so one boot names every part that fails. + */ #include -#include #include -#define PG 4096 +#include "memproof.h" -static int failed, passed; static volatile char *p; -static void say(const char *s) { - write(1, s, strlen(s)); -} - -static void part(const char *name, void (*fn)(void), int must_fault) { - char line[160]; - pid_t c = fork(); - if (c == 0) { - fn(); - _exit(0); - } - int st = 0; - waitpid(c, &st, 0); - int segv = (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || - (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); - int clean = WIFEXITED(st) && WEXITSTATUS(st) == 0; - int ok = must_fault ? segv : clean; - ok ? passed++ : failed++; - snprintf(line, sizeof line, "[C] protnone %s: %s (%s, status 0x%x)\n", name, - ok ? "ok" : "FAIL", must_fault ? "must fault" : "must not fault", st); - say(line); -} - static void read_it(void) { if (p[0] != 0x5a) { _exit(2); @@ -52,28 +24,24 @@ static void read_below(void) { } int main(void) { - char line[160]; + const char *me = "protnone"; p = mmap(0, PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); - part("read of an mmap PROT_NONE page", read_it, 1); - part("write to an mmap PROT_NONE page", write_it, 1); + part(me, "read of an mmap PROT_NONE page", read_it, 1); + part(me, "write to an mmap PROT_NONE page", write_it, 1); p = mmap(0, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); p[0] = 0x5a; mprotect((void *)p, PG, PROT_NONE); - part("read after mprotect RW to PROT_NONE", read_it, 1); + part(me, "read after mprotect RW to PROT_NONE", read_it, 1); mprotect((void *)p, PG, PROT_READ); - part("read after PROT_NONE back to R keeps the byte", read_it, 0); - part("write to a PROT_READ page", write_it, 1); + part(me, "read after PROT_NONE back to R keeps the byte", read_it, 0); + part(me, "write to a PROT_READ page", write_it, 1); char *r = mmap(0, 3 * PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); mprotect(r + PG, PG, PROT_READ | PROT_WRITE); p = (volatile char *)(r + PG); p[0] = 0x5a; - part("read of the opened page of a reservation", read_it, 0); - part("read of the closed page below it", read_below, 1); - - snprintf(line, sizeof line, "[C] protnone %s: %d parts ok, %d failed\n", - failed ? "FAIL" : "PASS", passed, failed); - say(line); - return failed ? 1 : 0; + part(me, "read of the opened page of a reservation", read_it, 0); + part(me, "read of the closed page below it", read_below, 1); + return finish(me); } diff --git a/userland/linux_guests/c/touchfork.c b/userland/linux_guests/c/touchfork.c index 133e2b4e42..447204f89a 100644 --- a/userland/linux_guests/c/touchfork.c +++ b/userland/linux_guests/c/touchfork.c @@ -1,44 +1,19 @@ -// Bytes a guest wrote into a reservation survive a fork. A reservation is a -// PROT_NONE mapping; a program opens part of it with mprotect, writes, and may -// close it again. Linux keeps the bytes through all of that and gives the child -// a copy. Touching a page never opened is SIGSEGV, in the child as anywhere. -// MAP_FIXED over a mapping replaces it: the new pages read zero, as Linux says. -#include -#include -#include +/* + * Bytes a guest wrote into a reservation survive a fork. A reservation is a + * PROT_NONE mapping; a program opens part of it with mprotect, writes, and may + * close it again. Linux keeps the bytes through all of that and gives the child + * a copy. Touching a page never opened is SIGSEGV, in the child as anywhere. + * MAP_FIXED over a mapping replaces it: the new pages read zero, as Linux says. + */ #include -#include #include -#define PG 4096 +#include "memproof.h" + #define PAGES 16 -static int failed, passed; static volatile char *r; -static void say(const char *s) { - write(1, s, strlen(s)); -} - -static void part(const char *name, void (*fn)(void), int must_fault) { - char line[160]; - pid_t c = fork(); - if (c == 0) { - fn(); - _exit(0); - } - int st = 0; - waitpid(c, &st, 0); - int segv = (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || - (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); - int clean = WIFEXITED(st) && WEXITSTATUS(st) == 0; - int ok = must_fault ? segv : clean; - ok ? passed++ : failed++; - snprintf(line, sizeof line, "[C] touchfork %s: %s (%s, status 0x%x)\n", name, - ok ? "ok" : "FAIL", must_fault ? "must fault" : "must not fault", st); - say(line); -} - static void check_bytes(void) { for (int i = 4; i < 8; i++) { if (r[i * PG] != (char)(0x40 + i) || r[i * PG + PG - 1] != (char)(0x50 + i)) { @@ -60,26 +35,23 @@ static void check_zero(void) { } int main(void) { - char line[160]; - r = mmap(0, PAGES * PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + const char *me = "touchfork"; + int anon = MAP_PRIVATE | MAP_ANONYMOUS; + r = mmap(0, PAGES * PG, PROT_NONE, anon, -1, 0); mprotect((void *)(r + 4 * PG), 4 * PG, PROT_READ | PROT_WRITE); for (int i = 4; i < 8; i++) { r[i * PG] = (char)(0x40 + i); r[i * PG + PG - 1] = (char)(0x50 + i); } - part("opened part of a reservation, child reads the bytes", check_bytes, 0); + part(me, "opened part of a reservation, child reads the bytes", check_bytes, 0); mprotect((void *)(r + 4 * PG), 4 * PG, PROT_NONE); - part("closed again, child opens it and reads the bytes", open_then_check, 0); - part("child touches a page never opened", touch_unopened, 1); + part(me, "closed again, child opens it and reads the bytes", open_then_check, 0); + part(me, "child touches a page never opened", touch_unopened, 1); - r = mmap(0, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + r = mmap(0, PG, PROT_READ | PROT_WRITE, anon, -1, 0); r[0] = 0x77; - mmap((void *)r, PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); + mmap((void *)r, PG, PROT_NONE, anon | MAP_FIXED, -1, 0); mprotect((void *)r, PG, PROT_READ); - part("MAP_FIXED PROT_NONE over written page, child reads zero", check_zero, 0); - - snprintf(line, sizeof line, "[C] touchfork %s: %d parts ok, %d failed\n", - failed ? "FAIL" : "PASS", passed, failed); - say(line); - return failed ? 1 : 0; + part(me, "MAP_FIXED PROT_NONE over written page, child reads zero", check_zero, 0); + return finish(me); } From a1112700a3524785156d73fe7774f5a0b1d670fd Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 09:55:33 +0000 Subject: [PATCH 12/13] linux: fork copies a large span in pieces, with its own protection Fork mapped each span into the child with one MkPeerMap over the whole span. The kernel refuses a peer call longer than 1 MiB, so forking a guest that held a larger span, as a Go heap is, failed. Another branch fixes that by mapping in pieces, but maps each piece with write and exec only, so a span the parent closed with PROT_NONE comes back open in the child. Fork now maps and copies each span a megabyte at a time, and gives every piece the protection the span has now, PROT_NONE included. The proof helper also counts a fork that fails as a failed part; it used to wait on nothing and report a pass. protfork gains two parts on a 2 MiB mapping: read-write, with the child reading the last page, and closed with PROT_NONE, where the child's read must fault. --- .../src/linux/call/spawn/fork_copy.rs | 21 ++++++++++++------- userland/linux_guests/c/memproof_run.c | 4 ++++ userland/linux_guests/c/protfork.c | 8 +++++++ 3 files changed, 26 insertions(+), 7 deletions(-) diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs index c5d907932d..e7b7b3a879 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs @@ -16,7 +16,7 @@ //! Copying a parent's spans into the child it just made. -use crate::linux::guest::Guest; +use crate::linux::guest::{Guest, MAX_SPAN}; use nonos_libc::peer::{mk_peer_map, mk_peer_write}; /// Every span, mapped into the child and then filled from the parent. @@ -31,15 +31,22 @@ pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool { continue; } /* - * The protection the span has now, PROT_NONE included: the kernel + * The kernel maps and copies at most MAX_SPAN in one peer call, so a + * larger span, which a Go heap is, crosses in pieces. Each piece gets + * the protection the span has now, PROT_NONE included: the kernel * copies into a page whatever its protection, so the bytes still go * in, and the child can do no more with them than the parent can. */ - if mk_peer_map(child, span.at, span.len, span.peer_prot()) < 0 { - return false; - } - if !copy_one(guest, child, span.at, span.len) { - return false; + let mut done = 0; + while done < span.len { + let take = (span.len - done).min(MAX_SPAN); + if mk_peer_map(child, span.at + done, take, span.peer_prot()) < 0 { + return false; + } + if !copy_one(guest, child, span.at + done, take) { + return false; + } + done += take; } } true diff --git a/userland/linux_guests/c/memproof_run.c b/userland/linux_guests/c/memproof_run.c index 88f3b59939..4d6a1100f9 100644 --- a/userland/linux_guests/c/memproof_run.c +++ b/userland/linux_guests/c/memproof_run.c @@ -28,6 +28,10 @@ void part_line(const char *proof, const char *name, int ok, const char *detail) void part(const char *proof, const char *name, void (*fn)(void), int must_fault) { char detail[64]; pid_t c = fork(); + if (c < 0) { + part_line(proof, name, 0, "fork failed"); + return; + } if (c == 0) { fn(); _exit(0); diff --git a/userland/linux_guests/c/protfork.c b/userland/linux_guests/c/protfork.c index 634db301e5..5f40603222 100644 --- a/userland/linux_guests/c/protfork.c +++ b/userland/linux_guests/c/protfork.c @@ -56,5 +56,13 @@ int main(void) { part(me, "middle page R, child writes the middle", write_it, 1); p = m + 2 * PG; part(me, "middle page R, child writes the last", write_it, 0); + /* Larger than the kernel's 1 MiB per peer call, so fork copies it in pieces. */ + long big = 2 * 1024 * 1024; + p = mmap(0, big, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + p += big - PG; + p[0] = 0x5a; + part(me, "2 MiB RW, child reads the last page", read_it, 0); + mprotect((void *)(p - big + PG), big, PROT_NONE); + part(me, "2 MiB RW to NONE, child reads the last page", read_it, 1); return finish(me); } From 9ac41dbda0e03d193f155165bc471930c78eb429 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Tue, 29 Sep 2026 11:59:38 +0000 Subject: [PATCH 13/13] linux: an attack guest that tries to leave its address space A guest that turns the isolation proofs around: instead of showing a mapping behaves, it attacks its own confinement through the syscalls it has. It maps at the kernel half, wraps a span, asks for a write-and- execute page, adds execute to a file it never proved, calls a number the personality does not serve, and steps into an unmapped hole. Each part passes when the machine refuses. It is a memproof part, so it shares the one guest binary and its store trailer. On NONOS all seven are refused. On native Linux two are allowed, a write-and-execute mapping and adding execute to an unproven file, which are the two the personality enforces and stock Linux does not. --- userland/linux_guests/Guests.mk | 6 ++-- userland/linux_guests/c/escape.c | 31 +++++++++++++++++ userland/linux_guests/c/escape.h | 19 ++++++++++ userland/linux_guests/c/escape_exec.c | 50 +++++++++++++++++++++++++++ userland/linux_guests/c/escape_map.c | 40 +++++++++++++++++++++ userland/linux_guests/c/memproof.c | 4 ++- 6 files changed, 146 insertions(+), 4 deletions(-) create mode 100644 userland/linux_guests/c/escape.c create mode 100644 userland/linux_guests/c/escape.h create mode 100644 userland/linux_guests/c/escape_exec.c create mode 100644 userland/linux_guests/c/escape_map.c diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index f1966b6b72..3848f6f20e 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -128,11 +128,11 @@ $(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait)) # over a written page replaces it with zeroes # memcalls mmap placement, brk, alignment, mremap, and mlock, msync and # mincore, each against Linux's answer -MEMPROOF_PARTS := guardpage protnone protfork touchfork memcalls +MEMPROOF_PARTS := guardpage protnone protfork touchfork memcalls escape # Files the proofs share, built as they are: no main of their own. -MEMPROOF_SHARED := memproof_run memcalls_map memcalls_remap memcalls_lock +MEMPROOF_SHARED := memproof_run memcalls_map memcalls_remap memcalls_lock escape_map escape_exec MEMPROOF_SRCS := $(foreach p,memproof $(MEMPROOF_PARTS) $(MEMPROOF_SHARED),\ - $(LINUX_GUESTS_DIR)/c/$(p).c) $(LINUX_GUESTS_DIR)/c/memproof.h $(LINUX_GUESTS_DIR)/c/memcalls.h + $(LINUX_GUESTS_DIR)/c/$(p).c) $(LINUX_GUESTS_DIR)/c/memproof.h $(LINUX_GUESTS_DIR)/c/memcalls.h $(LINUX_GUESTS_DIR)/c/escape.h $(LINUX_GUESTS_C)/memproof: $(MEMPROOF_SRCS) @mkdir -p $(@D)/memproof.o @for p in $(MEMPROOF_PARTS); do \ diff --git a/userland/linux_guests/c/escape.c b/userland/linux_guests/c/escape.c new file mode 100644 index 0000000000..4253fb8e02 --- /dev/null +++ b/userland/linux_guests/c/escape.c @@ -0,0 +1,31 @@ +/* + * A Linux guest trying to break out of its own address space through the + * syscall surface it is given: reach the kernel half, wrap a span, map + * write-and-execute, add execute to a file it never proved, call a number + * that is not served, and step past the end of what it holds. Each part + * passes when the machine refuses; one boot names any that got through. + */ +#include +#include + +#include "escape.h" + +void held(const char *name, int ok, long got) { + char detail[48]; + snprintf(detail, sizeof detail, "got %ld", got); + part_line("escape", name, ok, detail); +} + +long erc(long v) { + return v == -1 ? -errno : v; +} + +int main(void) { + reach_kernel(); + wrap_span(); + wx_map(); + exec_escalate(); + forged_call(); + past_end(); + return finish("escape"); +} diff --git a/userland/linux_guests/c/escape.h b/userland/linux_guests/c/escape.h new file mode 100644 index 0000000000..355e7253d9 --- /dev/null +++ b/userland/linux_guests/c/escape.h @@ -0,0 +1,19 @@ +/* What the escape attack files share. Each part passes when NONOS refuses. */ +#ifndef ESCAPE_H +#define ESCAPE_H + +#include "memproof.h" + +/* Record an attack: ok means the machine held (refused it). */ +void held(const char *name, int ok, long got); +/* -errno of a call (escape) that returned -1, or its value. */ +long erc(long v); + +void reach_kernel(void); +void wrap_span(void); +void wx_map(void); +void exec_escalate(void); +void forged_call(void); +void past_end(void); + +#endif diff --git a/userland/linux_guests/c/escape_exec.c b/userland/linux_guests/c/escape_exec.c new file mode 100644 index 0000000000..5fa27175a6 --- /dev/null +++ b/userland/linux_guests/c/escape_exec.c @@ -0,0 +1,50 @@ +/* Attacks on what a guest may run and call: adding execute to a file it never + * proved, a syscall number that is not served, and a step past a mapping. */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "escape.h" + +static volatile char *edge; + +void exec_escalate(void) { + /* Map this program read-only, then try to make it executable. It was + * mapped without execute, so it was never proved; NONOS refuses. */ + int fd = open("/bin/memproof", O_RDONLY); + if (fd < 0) { + fd = open("/proc/self/exe", O_RDONLY); + } + if (fd < 0) { + held("exec added to an unproven file mapping is refused", 0, -errno); + return; + } + void *p = mmap(0, PG, PROT_READ, MAP_PRIVATE, fd, 0); + close(fd); + if (p == MAP_FAILED) { + held("exec added to an unproven file mapping is refused", 0, -errno); + return; + } + long r = erc(mprotect(p, PG, PROT_READ | PROT_EXEC)); + held("exec added to an unproven file mapping is refused", r == -EPERM, r); +} + +void forged_call(void) { + /* A syscall number the personality does not serve. */ + long r = erc(syscall(0x462)); + held("an unserved syscall number answers ENOSYS", r == -ENOSYS, r); +} + +static void step_past(void) { + edge[PG] = 1; +} + +void past_end(void) { + /* Two pages, second given back, so the page written is a certain hole. */ + edge = mmap(0, 2 * PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + munmap((void *)(edge + PG), PG); + part("escape", "a write into an unmapped hole faults", step_past, 1); +} diff --git a/userland/linux_guests/c/escape_map.c b/userland/linux_guests/c/escape_map.c new file mode 100644 index 0000000000..0cecc8ea3d --- /dev/null +++ b/userland/linux_guests/c/escape_map.c @@ -0,0 +1,40 @@ +/* Attacks on where and how a guest may map: the kernel half, a wrapping + * span, and a page that is both writable and executable. */ +#include +#include +#include + +#include "escape.h" + +/* The first address of the kernel half, which no guest mapping may reach. */ +#define KERNEL_HALF 0x0000800000000000UL + +static int fixed_refused(uintptr_t at) { + void *p = mmap((void *)at, PG, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); + if (p == MAP_FAILED) { + return 1; + } + /* It answered with an address; it must at least not be the one asked. */ + munmap(p, PG); + return (uintptr_t)p != at; +} + +void reach_kernel(void) { + held("MAP_FIXED into the kernel half is refused", fixed_refused(KERNEL_HALF), 0); + held("MAP_FIXED one page below the kernel half is refused", fixed_refused(KERNEL_HALF - PG), 0); +} + +void wrap_span(void) { + /* A length that wraps past the top of the address space. */ + void *p = mmap(0, (size_t)-PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + long got = p == MAP_FAILED ? -errno : 0; + held("a mapping whose length wraps is refused", p == MAP_FAILED, got); +} + +void wx_map(void) { + /* Write and execute at once: NONOS refuses it, Linux allows it. */ + void *p = mmap(0, PG, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + long got = p == MAP_FAILED ? -errno : 0; + held("a write-and-execute mapping is refused", p == MAP_FAILED, got); +} diff --git a/userland/linux_guests/c/memproof.c b/userland/linux_guests/c/memproof.c index bb094f7315..2f154eccff 100644 --- a/userland/linux_guests/c/memproof.c +++ b/userland/linux_guests/c/memproof.c @@ -13,6 +13,7 @@ int protnone_main(void); int protfork_main(void); int touchfork_main(void); int memcalls_main(void); +int escape_main(void); static const struct { const char *name; @@ -23,6 +24,7 @@ static const struct { { "protfork", protfork_main }, { "touchfork", touchfork_main }, { "memcalls", memcalls_main }, + { "escape", escape_main }, }; int main(int argc, char **argv) { @@ -31,7 +33,7 @@ int main(int argc, char **argv) { return proofs[i].run(); } } - fputs("[C] memproof FAIL: name a proof: guardpage protnone protfork touchfork memcalls\n", + fputs("[C] memproof FAIL: name a proof: guardpage protnone protfork touchfork memcalls escape\n", stdout); return 2; }