diff --git a/src/memory/paging/manager/faults/demand.rs b/src/memory/paging/manager/faults/demand.rs index 8dc741b0f..3a6e64f77 100644 --- a/src/memory/paging/manager/faults/demand.rs +++ b/src/memory/paging/manager/faults/demand.rs @@ -29,27 +29,16 @@ impl PagingManager { virtual_addr: VirtAddr, stats: &PagingStatistics, ) -> PagingResult<()> { - // Only user-space addresses may be demand-backed. A not-present fault - // in the kernel half is never a legitimate lazy mapping; backing it - // silently would hand a capsule kernel-range memory. Surface it as an - // unhandled fault so the fault path kills the offender (user) or traps - // the real kernel bug, instead of papering over it. - if !layout::in_user_space(virtual_addr.as_u64()) { - return Err(PagingError::UnhandledPageFault); - } - - // Never demand-back the null page. A fault in the lowest page is a null - // or near-null dereference; backing it would silently satisfy the bug - // instead of trapping it. Leave the page unmapped as a guard so the - // fault path kills the offending capsule. - if virtual_addr.as_u64() < PAGE_SIZE_4K as u64 { + let pid = crate::process::current_pid().unwrap_or(0); + if super::demand_refuse::refused(virtual_addr.as_u64(), pid) { return Err(PagingError::UnhandledPageFault); } - // Charge the page against the faulting process's demand budget. A - // runaway capsule is refused here and killed by the fault path instead - // of exhausting physical memory. - let pid = crate::process::current_pid().unwrap_or(0); + /* + * Charge the page against the faulting process's demand budget. A + * runaway capsule is refused here and killed by the fault path instead + * of exhausting physical memory. + */ if !super::demand_cap::charge(pid) { return Err(PagingError::UnhandledPageFault); } diff --git a/src/memory/paging/manager/faults/demand_refuse.rs b/src/memory/paging/manager/faults/demand_refuse.rs new file mode 100644 index 000000000..0c92672cc --- /dev/null +++ b/src/memory/paging/manager/faults/demand_refuse.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The pages the kernel never fills on a fault. + +use crate::memory::layout; +use crate::memory::paging::constants::PAGE_SIZE_4K; + +/// True when a not-present fault at `addr` in `pid` must not be filled. +pub(super) fn refused(addr: u64, pid: u32) -> bool { + /* + * Only user-space addresses may be demand-backed. A not-present fault + * in the kernel half is never a legitimate lazy mapping; backing it + * silently would hand a capsule kernel-range memory. Surface it as an + * unhandled fault so the fault path kills the offender (user) or traps + * the real kernel bug, instead of papering over it. + */ + if !layout::in_user_space(addr) { + return true; + } + /* + * Never demand-back the null page. A fault in the lowest page is a null + * or near-null dereference; backing it would silently satisfy the bug + * instead of trapping it. Leave the page unmapped as a guard so the + * fault path kills the offending capsule. + */ + if addr < PAGE_SIZE_4K as u64 { + return true; + } + /* + * A foreign guest's pages are exactly the ones its supervisor mapped for + * it. Filling any other page would hand the guest memory nobody gave it: + * a PROT_NONE reservation, a guard page, a hole. So the fault is refused, + * the fault path ends the thread, and its supervisor is told and decides + * what that means for the guest. + */ + crate::process::foreign::is_foreign(pid) +} diff --git a/src/memory/paging/manager/faults/mod.rs b/src/memory/paging/manager/faults/mod.rs index de2ee927f..87fa92be2 100644 --- a/src/memory/paging/manager/faults/mod.rs +++ b/src/memory/paging/manager/faults/mod.rs @@ -17,4 +17,5 @@ mod cow; mod demand; mod demand_cap; +mod demand_refuse; mod handler; diff --git a/src/process/foreign/peer_guard.rs b/src/process/foreign/peer_guard.rs index 4fc9aa275..596dd00c1 100644 --- a/src/process/foreign/peer_guard.rs +++ b/src/process/foreign/peer_guard.rs @@ -21,11 +21,13 @@ use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_PERM}; pub(super) const PAGE: u64 = 4096; -// One call maps or copies at most this much, so a guest image crosses in -// bounded pieces and no single call holds the processor. +/* + * One call maps or copies at most this much, so a guest image crosses in + * bounded pieces and no single call holds the processor. + */ pub(super) const MAX_SPAN: u64 = 1 << 20; -// The first address of the kernel half. +/* The first address of the kernel half. */ pub(super) const USER_VA_END: u64 = 0x0000_8000_0000_0000; /// True when `[addr, addr + len)` lies wholly in the guest's own half. @@ -38,6 +40,12 @@ pub(super) fn in_user_half(addr: u64, len: u64) -> bool { pub const PROT_WRITE: u64 = 1 << 0; pub const PROT_EXEC: u64 = 1 << 1; +/* + * No access from the guest at all. The page stays present with the user bit + * clear, so every guest access faults and the frame keeps its bytes for a + * later protection that allows access, as Linux keeps them. + */ +pub(super) const PROT_NONE: u64 = 1 << 2; /// The pid a syscall argument names. Refused rather than truncated: `as u32` diff --git a/src/process/foreign/peer_map.rs b/src/process/foreign/peer_map.rs index 3740c267a..2a309a4a5 100644 --- a/src/process/foreign/peer_map.rs +++ b/src/process/foreign/peer_map.rs @@ -18,26 +18,15 @@ use crate::memory::addr::VirtAddr; use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid}; -use crate::memory::paging::types::PagePermissions; use crate::syscall::microkernel::errnos::{ERRNO_INVAL, ERRNO_NOMEM}; -use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_WRITE}; +use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE}; +use super::peer_protect::perms_of; fn span_ok(addr: u64, len: u64) -> bool { len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len) } -pub(super) fn perms_of(prot: u64) -> PagePermissions { - let mut perms = PagePermissions::READ | PagePermissions::USER; - if prot & PROT_WRITE != 0 { - perms = perms | PagePermissions::WRITE; - } - if prot & PROT_EXEC != 0 { - perms = perms | PagePermissions::EXECUTE; - } - perms -} - /// `MkPeerMap`: map `[addr, addr + len)` in a guest the caller supervises. pub fn sys_peer_map(pid: u64, addr: u64, len: u64, prot: u64) -> i64 { let Some(caller) = crate::process::current_pid() else { diff --git a/src/process/foreign/peer_protect.rs b/src/process/foreign/peer_protect.rs index 79d74e1ce..cfde6b0b7 100644 --- a/src/process/foreign/peer_protect.rs +++ b/src/process/foreign/peer_protect.rs @@ -19,10 +19,12 @@ use crate::memory::addr::VirtAddr; use crate::memory::paging::manager::{map_page_in_asid, translate_in_asid}; +use crate::memory::paging::types::PagePermissions; use crate::syscall::microkernel::errnos::{ERRNO_FAULT, ERRNO_INVAL}; -use super::peer_guard::{in_user_half, supervised_asid, MAX_SPAN, PAGE}; -use super::peer_map::perms_of; +use super::peer_guard::{ + in_user_half, supervised_asid, MAX_SPAN, PAGE, PROT_EXEC, PROT_NONE, PROT_WRITE, +}; fn span_ok(addr: u64, len: u64) -> bool { len != 0 && len <= MAX_SPAN && addr % PAGE == 0 && in_user_half(addr, len) @@ -53,3 +55,21 @@ pub fn sys_peer_protect(pid: u64, addr: u64, len: u64, prot: u64) -> i64 { } 0 } + +pub(super) fn perms_of(prot: u64) -> PagePermissions { + /* + * Not USER: present for the kernel, which copies it at fork and frees it + * at teardown, and absent for every access the guest makes. + */ + if prot & PROT_NONE != 0 { + return PagePermissions::READ; + } + let mut perms = PagePermissions::READ | PagePermissions::USER; + if prot & PROT_WRITE != 0 { + perms = perms | PagePermissions::WRITE; + } + if prot & PROT_EXEC != 0 { + perms = perms | PagePermissions::EXECUTE; + } + perms +} diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs index 68e04090a..667abbfa1 100644 --- a/userland/capsule_linux/src/linux/abi/mod.rs +++ b/userland/capsule_linux/src/linux/abi/mod.rs @@ -23,4 +23,5 @@ pub mod name; pub mod nr; pub mod nr_path; pub mod nr_high; +pub mod nr_mem; pub mod nr_sched; diff --git a/userland/capsule_linux/src/linux/abi/nr.rs b/userland/capsule_linux/src/linux/abi/nr.rs index 1e800243b..03bd5c153 100644 --- a/userland/capsule_linux/src/linux/abi/nr.rs +++ b/userland/capsule_linux/src/linux/abi/nr.rs @@ -18,6 +18,7 @@ //! Linux x86_64 syscall numbers, by family. pub use super::nr_high::*; +pub use super::nr_mem::*; pub use super::nr_sched::*; pub const READ: u64 = 0; diff --git a/userland/capsule_linux/src/linux/abi/nr_mem.rs b/userland/capsule_linux/src/linux/abi/nr_mem.rs new file mode 100644 index 000000000..a6d3d83d7 --- /dev/null +++ b/userland/capsule_linux/src/linux/abi/nr_mem.rs @@ -0,0 +1,24 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . +//! Memory calls the Linux x86_64 table has that the first families lacked. + +pub const MSYNC: u64 = 26; +pub const MINCORE: u64 = 27; +pub const MLOCK: u64 = 149; +pub const MUNLOCK: u64 = 150; +pub const MLOCKALL: u64 = 151; +pub const MUNLOCKALL: u64 = 152; +pub const MLOCK2: u64 = 325; diff --git a/userland/capsule_linux/src/linux/call/mem/lock.rs b/userland/capsule_linux/src/linux/call/mem/lock.rs new file mode 100644 index 000000000..dfbaa21d7 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/lock.rs @@ -0,0 +1,64 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . +//! `mlock`, `mlock2`, `munlock`, `mlockall` and `munlockall`. +//! +//! Locking keeps pages resident. Every page a guest holds here is resident +//! from the moment it is mapped until it is unmapped, and none is ever paged +//! out, so every page is already as locked as Linux can make it. What is left +//! of each call is Linux's checking of its arguments, answered the same way: +//! the span must be held, and the flags known. RLIMIT_MEMLOCK is reported +//! unlimited, so no lock is refused for its size. + +use crate::linux::abi::errno; +use crate::linux::guest::{page_down, page_up, Guest, PAGE}; + +const MLOCK_ONFAULT: u64 = 1; +const MCL_CURRENT: u64 = 1; +const MCL_FUTURE: u64 = 2; +const MCL_ONFAULT: u64 = 4; + +/// `mlock` and `munlock` alike: the address is rounded down, the length up, +/// and a span with a page the guest does not hold is ENOMEM. +pub fn mlock(guest: &Guest, addr: u64, len: u64) -> u64 { + let start = page_down(addr); + let Some(end) = addr.checked_add(len).filter(|e| *e <= u64::MAX - PAGE) else { + return errno::fail(errno::EINVAL); + }; + let end = page_up(end); + if end == start { + return errno::ok(0); + } + if guest.mapped_from(start) < end - start { + return errno::fail(errno::ENOMEM); + } + errno::ok(0) +} + +pub fn mlock2(guest: &Guest, addr: u64, len: u64, flags: u64) -> u64 { + if flags & !MLOCK_ONFAULT != 0 { + return errno::fail(errno::EINVAL); + } + mlock(guest, addr, len) +} + +/// Linux refuses no flags, unknown flags, and MCL_ONFAULT on its own. +pub fn mlockall(flags: u64) -> u64 { + let known = MCL_CURRENT | MCL_FUTURE | MCL_ONFAULT; + if flags == 0 || flags & !known != 0 || flags == MCL_ONFAULT { + return errno::fail(errno::EINVAL); + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/mem/map.rs b/userland/capsule_linux/src/linux/call/mem/map.rs index 904072cdb..1baa5ce4b 100644 --- a/userland/capsule_linux/src/linux/call/mem/map.rs +++ b/userland/capsule_linux/src/linux/call/mem/map.rs @@ -17,50 +17,43 @@ //! `mmap`: anonymous pages, or a private mapping of a file. use crate::linux::abi::errno; -use crate::linux::guest::{span_within, Guest, MMAP_LIMIT, USER_MAX}; +use crate::linux::guest::{Guest, PAGE}; -use super::map_anon::{anonymous, memfd}; -use super::map_file::file; +use super::map_place::place; use super::map_req::MapReq; use super::prot::wx_refused; -const MAP_SHARED: u64 = 0x01; -const MAP_ANONYMOUS: u64 = 0x20; const MAP_FIXED: u64 = 0x10; +const MAP_FIXED_NOREPLACE: u64 = 0x10_0000; pub fn mmap(guest: &mut Guest, req: MapReq) -> u64 { - if req.len == 0 { + /* + * Linux takes a file offset on a page boundary, and an exact address too; + * only a hint is rounded. + */ + let exact = req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0; + if req.len == 0 || req.off % PAGE != 0 || (exact && req.addr % PAGE != 0) { return errno::fail(errno::EINVAL); } if wx_refused(req.prot) { return errno::fail(errno::EPERM); } - // MAP_FIXED is the exact address or failure. Page zero is never in the - // plan, and landing elsewhere would hand back memory the guest did not - // ask for, so it is refused, as Linux refuses it below mmap_min_addr. - if req.flags & MAP_FIXED != 0 && req.addr == 0 { + /* + * MAP_FIXED and MAP_FIXED_NOREPLACE are the exact address or failure. + * Page zero is never in the plan, and landing elsewhere would hand back + * memory the guest did not ask for, so it is refused, as Linux refuses it + * below mmap_min_addr. + */ + if exact && req.addr == 0 { return errno::fail(errno::EPERM); } - // The ceiling differs by who chose the address. - let (at, limit) = match req.fixed() { - Some(addr) => (addr, USER_MAX), - None => (guest.mmap_next, MMAP_LIMIT), + let spot = match place(guest, &req) { + Ok(spot) => spot, + Err(e) => return errno::fail(e), }; - let Some((at, span)) = span_within(at, req.len, limit) else { - return errno::fail(errno::ENOMEM); - }; - if req.flags & MAP_ANONYMOUS != 0 { - return anonymous(guest, &req, at, span); - } - if crate::linux::file::is_memfd(guest, req.fd) { - return memfd(guest, &req, at, span); - } - if req.flags & MAP_SHARED != 0 { - /* - * Sharing a file between processes needs frames that two address - * spaces both point at, which no peer call offers. - */ - return errno::fail(errno::ENOSYS); + let out = super::map_kind::map_at(guest, &req, spot.at, spot.span); + if spot.from_cursor && (out as i64) >= 0 { + guest.mmap_next = spot.at + spot.span; } - file(guest, &req, at, span) + out } diff --git a/userland/capsule_linux/src/linux/call/mem/map_anon.rs b/userland/capsule_linux/src/linux/call/mem/map_anon.rs index 0158b7f26..d3c8ef994 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_anon.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_anon.rs @@ -44,10 +44,15 @@ pub fn memfd(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { } pub fn anonymous(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { - // A PROT_NONE anonymous mapping is a reservation: the runtime that makes it - // (Go's, for one) commits a fraction of it later with a fixed RW mapping. - // Backing the whole span here would spend real frames on address space no - // one has touched, so reserve it and let the first access fault a page in. + if !req.make_room(guest, at, span) { + return errno::fail(errno::ENOMEM); + } + /* + * A PROT_NONE anonymous mapping is a reservation: the runtime that makes it + * (Go's, for one) commits a fraction of it later with a fixed RW mapping. + * Backing the whole span here would spend real frames on address space no + * one may touch, so reserve it; a commit maps the part that is opened. + */ let backed = if req.prot == 0 { guest.reserve(at, span) } else { @@ -56,8 +61,5 @@ pub fn anonymous(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if backed < 0 { return errno::fail(errno::ENOMEM); } - if req.fixed().is_none() { - guest.mmap_next += span; - } errno::ok(at) } diff --git a/userland/capsule_linux/src/linux/call/mem/map_file.rs b/userland/capsule_linux/src/linux/call/mem/map_file.rs index 2961e9123..894d39531 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_file.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_file.rs @@ -36,7 +36,7 @@ pub fn file(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if let Some(None) = proved { return errno::fail(errno::EPERM); } - if guest.map(at, span, true, false) < 0 { + if !req.make_room(guest, at, span) || guest.map(at, span, true, false) < 0 { return errno::fail(errno::ENOMEM); } if let Some(Some(bytes)) = proved { @@ -48,7 +48,7 @@ pub fn file(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if fill_read(guest, req, at) < 0 { return errno::fail(errno::EACCES); } - // Not proved, since nothing asked to run it: it stays that way. + /* Not proved, since nothing asked to run it: it stays that way. */ guest.mark_unproven(at, span); finish(guest, req, at, span) } @@ -57,8 +57,5 @@ fn finish(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { if protect_span(guest, at, span, req.prot) < 0 { return errno::fail(errno::EACCES); } - if req.fixed().is_none() { - guest.mmap_next += span; - } errno::ok(at) } diff --git a/userland/capsule_linux/src/linux/call/mem/map_free.rs b/userland/capsule_linux/src/linux/call/mem/map_free.rs new file mode 100644 index 000000000..ccb509375 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/map_free.rs @@ -0,0 +1,40 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where the mapping cursor finds room. + +use crate::linux::abi::errno; +use crate::linux::guest::{page_up, span_within, Guest, MMAP_LIMIT}; + +/// The first span of `len` at or above the mapping cursor that meets +/// nothing the guest holds. +pub fn free_span(guest: &Guest, len: u64) -> Result<(u64, u64), i64> { + let mut at = guest.mmap_next; + loop { + let (start, span) = span_within(at, len, MMAP_LIMIT).ok_or(errno::ENOMEM)?; + let end = start + span; + let past = guest + .regions + .iter() + .filter(|r| r.at < end && start < r.at.saturating_add(r.len)) + .map(|r| r.at.saturating_add(r.len)) + .max(); + match past { + None => return Ok((start, span)), + Some(next) => at = page_up(next), + } + } +} diff --git a/userland/capsule_linux/src/linux/call/mem/map_kind.rs b/userland/capsule_linux/src/linux/call/mem/map_kind.rs new file mode 100644 index 000000000..b0b45e11c --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/map_kind.rs @@ -0,0 +1,44 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which kind of mapping an mmap makes once it has a place. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::map_anon::{anonymous, memfd}; +use super::map_file::file; +use super::map_req::MapReq; + +const MAP_SHARED: u64 = 0x01; +const MAP_ANONYMOUS: u64 = 0x20; + +pub(super) fn map_at(guest: &mut Guest, req: &MapReq, at: u64, span: u64) -> u64 { + if req.flags & MAP_ANONYMOUS != 0 { + return anonymous(guest, req, at, span); + } + if crate::linux::file::is_memfd(guest, req.fd) { + return memfd(guest, req, at, span); + } + if req.flags & MAP_SHARED != 0 { + /* + * Sharing a file between processes needs frames that two address + * spaces both point at, which no peer call offers. + */ + return errno::fail(errno::ENOSYS); + } + file(guest, req, at, span) +} diff --git a/userland/capsule_linux/src/linux/call/mem/map_place.rs b/userland/capsule_linux/src/linux/call/mem/map_place.rs new file mode 100644 index 000000000..830447d2b --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/map_place.rs @@ -0,0 +1,58 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where an mmap goes. MAP_FIXED is the exact address; MAP_FIXED_NOREPLACE +//! is the exact address or EEXIST when something is there; any other address +//! is a hint taken only when nothing is there. Otherwise the mapping cursor +//! chooses, skipping every span the guest already holds, since a new mapping +//! laid over an old one would hand back the old pages. + +use crate::linux::abi::errno; +use crate::linux::guest::{page_up, span_within, Guest, USER_MAX}; + +use super::map_req::MapReq; + +const MAP_FIXED: u64 = 0x10; +const MAP_FIXED_NOREPLACE: u64 = 0x10_0000; + +pub struct Place { + pub at: u64, + pub span: u64, + /// Chosen by the cursor, which then moves past it. + pub from_cursor: bool, +} + +/// The span the mapping takes, or the errno refusing it. +pub fn place(guest: &Guest, req: &MapReq) -> Result { + let exact = |(at, span)| Place { at, span, from_cursor: false }; + if req.flags & (MAP_FIXED | MAP_FIXED_NOREPLACE) != 0 { + let (at, span) = span_within(req.addr, req.len, USER_MAX).ok_or(errno::ENOMEM)?; + if req.flags & MAP_FIXED == 0 && guest.overlaps(at, span) { + return Err(errno::EEXIST); + } + return Ok(exact((at, span))); + } + /* Linux rounds a hint up to a page. */ + if req.addr != 0 { + if let Some(got) = span_within(page_up(req.addr), req.len, USER_MAX) { + if !guest.overlaps(got.0, got.1) { + return Ok(exact(got)); + } + } + } + let (at, span) = super::map_free::free_span(guest, req.len)?; + Ok(Place { at, span, from_cursor: true }) +} diff --git a/userland/capsule_linux/src/linux/call/mem/map_req.rs b/userland/capsule_linux/src/linux/call/mem/map_req.rs index c426dcb11..7e278ccd2 100644 --- a/userland/capsule_linux/src/linux/call/mem/map_req.rs +++ b/userland/capsule_linux/src/linux/call/mem/map_req.rs @@ -17,6 +17,10 @@ //! What a guest asked `mmap` for, in one value. +use crate::linux::guest::Guest; + +const MAP_FIXED: u64 = 0x10; + pub struct MapReq { pub addr: u64, pub len: u64, @@ -31,12 +35,11 @@ impl MapReq { MapReq { addr: a[0], len: a[1], prot: a[2], flags: a[3], fd: a[4], off: a[5] } } - /// The address the guest named, or nothing when it left the choice - /// to this capsule, which is the case the mapping cursor advances on. - pub fn fixed(&self) -> Option { - match self.addr { - 0 => None, - addr => Some(crate::linux::guest::page_down(addr)), - } + /// Make room for a MAP_FIXED mapping at `[at, at + span)`. Linux replaces + /// whatever was there: the old pages go and the new mapping starts from + /// zeroes with its own protection. Called just before the new pages go in, + /// so a mapping refused earlier leaves the old one where it was. + pub fn make_room(&self, guest: &mut Guest, at: u64, span: u64) -> bool { + self.flags & MAP_FIXED == 0 || guest.unmap(at, span) >= 0 } } diff --git a/userland/capsule_linux/src/linux/call/mem/memory.rs b/userland/capsule_linux/src/linux/call/mem/memory.rs index 28a2cfb6a..227da0d5c 100644 --- a/userland/capsule_linux/src/linux/call/mem/memory.rs +++ b/userland/capsule_linux/src/linux/call/mem/memory.rs @@ -17,7 +17,7 @@ //! `brk` and `munmap`. use crate::linux::abi::errno; -use crate::linux::guest::{page_up, Guest, BRK_BASE, BRK_LIMIT}; +use crate::linux::guest::{page_up, Guest, BRK_BASE, BRK_LIMIT, PAGE}; /// `brk(0)` reports the break; any other value moves it and reports where /// it landed, which is Linux's contract and not an error channel. @@ -29,20 +29,29 @@ pub fn brk(guest: &mut Guest, want: u64) -> u64 { if want == 0 || want < BRK_BASE || want > BRK_LIMIT { return errno::ok(guest.brk); } - let top = page_up(want); - if top > guest.brk { - let len = top - guest.brk; - if guest.map(guest.brk, len, true, false) < 0 { + /* Whole pages: the page the old break sits in is already held. */ + let (old, top) = (page_up(guest.brk), page_up(want)); + if top > old { + /* Linux refuses a break that would run into a mapping. */ + if guest.overlaps(old, top - old) || guest.map(old, top - old, true, false) < 0 { return errno::ok(guest.brk); } } + /* + * A lower break gives the pages above it back, so growing again reads + * zeroes, as on Linux. + */ + if top < old && guest.unmap(top, old - top) < 0 { + return errno::ok(guest.brk); + } guest.brk = want; errno::ok(guest.brk) } /// The pages go back to the kernel and leave the guest's region list. pub fn munmap(guest: &mut Guest, addr: u64, len: u64) -> u64 { - if len == 0 { + /* Linux takes an address on a page boundary, and rounds only the length. */ + if len == 0 || addr % PAGE != 0 { return errno::fail(errno::EINVAL); } match guest.unmap(addr, len) { diff --git a/userland/capsule_linux/src/linux/call/mem/mod.rs b/userland/capsule_linux/src/linux/call/mem/mod.rs index 423f1a164..a8f37dd84 100644 --- a/userland/capsule_linux/src/linux/call/mem/mod.rs +++ b/userland/capsule_linux/src/linux/call/mem/mod.rs @@ -14,23 +14,34 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! The calls that shape a guest's address space: `mmap`, `munmap`, `brk` and -//! `mprotect`. +//! The calls that shape a guest's address space: `mmap`, `munmap`, `brk`, +//! `mprotect` and `mremap`, and the ones that ask about it or lock it. +mod lock; mod map; mod map_anon; mod map_exec; mod map_file; mod map_fill; +mod map_free; +mod map_kind; +mod map_place; mod map_req; mod memory; mod prot; mod prot_span; +mod prot_walk; mod remap; mod remap_move; +mod remap_one; +mod resident; +mod sync; +pub use lock::{mlock, mlock2, mlockall}; pub use map::mmap; pub use map_req::MapReq; pub use memory::{brk, munmap}; pub use prot::mprotect; pub use remap::mremap; +pub use resident::mincore; +pub use sync::msync; diff --git a/userland/capsule_linux/src/linux/call/mem/prot.rs b/userland/capsule_linux/src/linux/call/mem/prot.rs index 9d060a23e..ce3bebf23 100644 --- a/userland/capsule_linux/src/linux/call/mem/prot.rs +++ b/userland/capsule_linux/src/linux/call/mem/prot.rs @@ -17,14 +17,12 @@ //! `mprotect`, and the rule that makes it necessary. use crate::linux::abi::errno; -use crate::linux::guest::{span_within, Guest, USER_MAX}; - -use super::prot_span::protect_span; +use crate::linux::guest::{span_within, Guest, PAGE, USER_MAX}; pub const PROT_WRITE: u64 = 2; pub const PROT_EXEC: u64 = 4; /// PROT_READ, PROT_WRITE and PROT_EXEC together: any access at all. -const PROT_ANY: u64 = 7; +pub const PROT_ANY: u64 = 7; /// A request for both at once. pub fn wx_refused(prot: u64) -> bool { @@ -32,6 +30,10 @@ pub fn wx_refused(prot: u64) -> bool { } pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 { + /* Linux takes an address on a page boundary, and rounds only the length. */ + if addr % PAGE != 0 { + return errno::fail(errno::EINVAL); + } if len == 0 { return errno::ok(0); } @@ -54,37 +56,5 @@ pub fn mprotect(guest: &mut Guest, addr: u64, len: u64, prot: u64) -> u64 { if prot & PROT_EXEC != 0 && guest.span_unproven(start, span) { return errno::fail(errno::EPERM); } - let end = start + span; - let mut at = start; - while at < end { - let Some(r) = guest.regions.iter().find(|r| r.at <= at && at < r.at + r.len).copied() - else { - // Linux refuses a span with no mapping in it at all. - return errno::fail(errno::ENOMEM); - }; - let upto = end.min(r.at + r.len); - let piece = upto - at; - if !r.backed { - /* - * A PROT_NONE reservation has no pages for the kernel to - * reprotect. Asking for access commits it, which is how musl makes - * a thread stack: reserve with PROT_NONE, then mprotect the part - * it uses to read-write. PROT_NONE on it changes nothing. - */ - if prot & PROT_ANY == 0 { - at = upto; - continue; - } - if guest.commit(at, piece, prot & PROT_WRITE != 0, prot & PROT_EXEC != 0) < 0 { - return errno::fail(errno::ENOMEM); - } - } - // Every page is present now; this sets `prot` on all of them, - // including any the guest touched while the span was reserved. - if protect_span(guest, at, piece, prot) < 0 { - return errno::fail(errno::EACCES); - } - at = upto; - } - errno::ok(0) + super::prot_walk::walk(guest, start, span, prot) } diff --git a/userland/capsule_linux/src/linux/call/mem/prot_span.rs b/userland/capsule_linux/src/linux/call/mem/prot_span.rs index aac1b7043..bbe49caf6 100644 --- a/userland/capsule_linux/src/linux/call/mem/prot_span.rs +++ b/userland/capsule_linux/src/linux/call/mem/prot_span.rs @@ -16,21 +16,18 @@ //! Reprotecting a span, a peer call at a time. -use nonos_libc::peer::{mk_peer_protect, PEER_PROT_EXEC, PEER_PROT_WRITE}; +use nonos_libc::peer::mk_peer_protect; -use crate::linux::guest::{Guest, MAX_SPAN}; +use crate::linux::guest::{peer_prot, Guest, MAX_SPAN}; -use super::prot::{PROT_EXEC, PROT_WRITE}; +use super::prot::{PROT_ANY, PROT_EXEC, PROT_WRITE}; -/// Set the protection of a span already mapped in the guest. -pub fn protect_span(guest: &Guest, addr: u64, span: u64, prot: u64) -> i64 { - let mut bits = 0; - if prot & PROT_WRITE != 0 { - bits |= PEER_PROT_WRITE; - } - if prot & PROT_EXEC != 0 { - bits |= PEER_PROT_EXEC; - } +/// Set the protection of a span already mapped in the guest, and record it +/// on the spans it covers. +pub fn protect_span(guest: &mut Guest, addr: u64, span: u64, prot: u64) -> i64 { + let (write, exec, access) = + (prot & PROT_WRITE != 0, prot & PROT_EXEC != 0, prot & PROT_ANY != 0); + let bits = peer_prot(write, exec, access); let mut done = 0; while done < span { let take = (span - done).min(MAX_SPAN); @@ -40,5 +37,6 @@ pub fn protect_span(guest: &Guest, addr: u64, span: u64, prot: u64) -> i64 { } done += take; } + guest.set_prot(addr, span, write, exec, access); 0 } diff --git a/userland/capsule_linux/src/linux/call/mem/prot_walk.rs b/userland/capsule_linux/src/linux/call/mem/prot_walk.rs new file mode 100644 index 000000000..ab11f99a5 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/prot_walk.rs @@ -0,0 +1,56 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Walking an mprotect span one mapping at a time. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::prot::{PROT_ANY, PROT_EXEC, PROT_WRITE}; +use super::prot_span::protect_span; + +/// Give `[start, start + span)` the protection `prot`, mapping by mapping. +pub(super) fn walk(guest: &mut Guest, start: u64, span: u64, prot: u64) -> u64 { + let end = start + span; + let mut at = start; + while at < end { + let Some(r) = guest.regions.iter().find(|r| r.at <= at && at < r.at + r.len).copied() + else { + /* Linux refuses a span with no mapping in it at all. */ + return errno::fail(errno::ENOMEM); + }; + let upto = end.min(r.at + r.len); + let piece = upto - at; + if !r.backed { + /* + * A PROT_NONE reservation has no pages for the kernel to + * reprotect. Asking for access commits it, which is how musl makes + * a thread stack: reserve with PROT_NONE, then mprotect the part + * it uses to read-write. PROT_NONE on it changes nothing. The + * commit maps the piece with `prot` and records it. + */ + if prot & PROT_ANY != 0 + && guest.commit(at, piece, prot & PROT_WRITE != 0, prot & PROT_EXEC != 0) < 0 + { + return errno::fail(errno::ENOMEM); + } + } else if protect_span(guest, at, piece, prot) < 0 { + return errno::fail(errno::EACCES); + } + at = upto; + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/mem/remap.rs b/userland/capsule_linux/src/linux/call/mem/remap.rs index add8ae3a4..85c41a2b0 100644 --- a/userland/capsule_linux/src/linux/call/mem/remap.rs +++ b/userland/capsule_linux/src/linux/call/mem/remap.rs @@ -18,23 +18,24 @@ //! or move with MREMAP_MAYMOVE. glibc's realloc of a large block is this. use crate::linux::abi::errno; -use crate::linux::guest::{page_up, span_within, Guest, MMAP_LIMIT, PAGE}; +use crate::linux::guest::{page_up, span_within, Guest, PAGE, USER_MAX}; const MAYMOVE: u64 = 1; pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u64) -> u64 { - // MREMAP_FIXED and DONTUNMAP choose the destination; neither is offered. + /* MREMAP_FIXED and DONTUNMAP choose the destination; neither is offered. */ if flags & !MAYMOVE != 0 || old % PAGE != 0 || old_len == 0 || new_len == 0 { return errno::fail(errno::EINVAL); } let (old_len, new_len) = (page_up(old_len), page_up(new_len)); - if guest.mapped_from(old) < old_len { - return errno::fail(errno::EFAULT); - } let Some(r) = guest.regions.iter().find(|r| r.at <= old && old < r.at + r.len).copied() else { return errno::fail(errno::EFAULT); }; - // Code was proved where it was mapped; a moved copy would not be. + /* Linux moves one mapping at a time: the old span must lie inside one. */ + if !super::remap_one::one_mapping(guest, old, old_len, &r) { + return errno::fail(errno::EFAULT); + } + /* Code was proved where it was mapped; a moved copy would not be. */ if r.exec { return errno::fail(errno::EPERM); } @@ -46,16 +47,15 @@ pub fn mremap(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, flags: u6 } let tail = old + old_len; let grow = new_len - old_len; - let free = !guest.regions.iter().any(|g| g.at < tail + grow && tail < g.at + g.len); - if free - && span_within(tail, grow, MMAP_LIMIT).is_some() - && guest.map(tail, grow, r.write, false) >= 0 + /* The grown part is the same mapping: its protection, its backing. */ + if !guest.overlaps(tail, grow) + && span_within(tail, grow, USER_MAX).is_some() + && guest.map_like(tail, grow, &r) >= 0 { - guest.mmap_next = guest.mmap_next.max(tail + grow); return errno::ok(old); } if flags & MAYMOVE == 0 { return errno::fail(errno::ENOMEM); } - super::remap_move::moved(guest, old, old_len, new_len, r.write) + super::remap_move::moved(guest, old, old_len, new_len, &r) } diff --git a/userland/capsule_linux/src/linux/call/mem/remap_move.rs b/userland/capsule_linux/src/linux/call/mem/remap_move.rs index b8b3930d7..2b10dad7a 100644 --- a/userland/capsule_linux/src/linux/call/mem/remap_move.rs +++ b/userland/capsule_linux/src/linux/call/mem/remap_move.rs @@ -17,29 +17,26 @@ //! `mremap` when the block cannot grow where it is: moved to fresh pages. use crate::linux::abi::errno; -use crate::linux::guest::{span_within, Guest, MMAP_LIMIT}; +use crate::linux::guest::{Guest, Region}; -const PROT_READ: u64 = 1; +use super::map_free::free_span; -// A fresh span at the mapping cursor, the old bytes copied in, the old span gone. -pub(super) fn moved(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, write: bool) -> u64 { - let Some((at, span)) = span_within(guest.mmap_next, new_len, MMAP_LIMIT) else { - return errno::fail(errno::ENOMEM); - }; - let Some(bytes) = guest.read(old, old_len as usize) else { - return errno::fail(errno::EFAULT); +/// A fresh span at the mapping cursor held like the old one, with the old +/// protection, backing and provenance; the old bytes copied in; the old span +/// gone. A move that fails part way leaves nothing new behind. +pub(super) fn moved(guest: &mut Guest, old: u64, old_len: u64, new_len: u64, like: &Region) -> u64 { + let (at, span) = match free_span(guest, new_len) { + Ok(got) => got, + Err(e) => return errno::fail(e), }; - // Writable while the bytes go in; the old protection after. - if guest.map(at, span, true, false) < 0 { + if guest.map_like(at, span, like) < 0 { return errno::fail(errno::ENOMEM); } - guest.mmap_next += span; - if guest.write(at, &bytes) < bytes.len() as i64 { + if like.backed && !guest.copy_within(old, at, old_len) { + let _ = guest.unmap(at, span); return errno::fail(errno::EFAULT); } - if !write { - let _ = super::prot::mprotect(guest, at, span, PROT_READ); - } let _ = guest.unmap(old, old_len); + guest.mmap_next = at + span; errno::ok(at) } diff --git a/userland/capsule_linux/src/linux/call/mem/remap_one.rs b/userland/capsule_linux/src/linux/call/mem/remap_one.rs new file mode 100644 index 000000000..40d026315 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/remap_one.rs @@ -0,0 +1,38 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Whether an mremap span is one mapping, as Linux requires. + +use crate::linux::guest::{Guest, Region}; + +/// Every page of `[at, at + len)` is held with the same protection, backing +/// and provenance as `like`, which is what Linux keeps as one mapping. +pub(super) fn one_mapping(guest: &Guest, at: u64, len: u64, like: &Region) -> bool { + let end = at + len; + let mut reach = at; + while reach < end { + let Some(r) = guest.regions.iter().find(|r| r.at <= reach && reach < r.at + r.len) else { + return false; + }; + let same = (r.write, r.exec, r.access, r.backed, r.unproven) + == (like.write, like.exec, like.access, like.backed, like.unproven); + if !same { + return false; + } + reach = r.at + r.len; + } + true +} diff --git a/userland/capsule_linux/src/linux/call/mem/resident.rs b/userland/capsule_linux/src/linux/call/mem/resident.rs new file mode 100644 index 000000000..dba74b088 --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/resident.rs @@ -0,0 +1,54 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . +//! `mincore`: which pages of a span are resident. +//! +//! A backed span is resident page for page, since the personality maps every +//! page of it when it is committed and the kernel pages nothing out, whatever +//! its protection now. A reservation has no page at all. So the region list +//! answers exactly, one byte per page, 1 for resident. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::{page_up, Guest, MAX_SPAN, PAGE, USER_MAX}; + +pub fn mincore(guest: &Guest, addr: u64, len: u64, vec: u64) -> u64 { + if addr % PAGE != 0 { + return errno::fail(errno::EINVAL); + } + let Some(end) = addr.checked_add(len).filter(|e| *e <= USER_MAX) else { + return errno::fail(errno::ENOMEM); + }; + let end = page_up(end); + let mut at = addr; + let mut out: Vec = Vec::new(); + let mut written = 0u64; + while at < end { + let Some(r) = guest.regions.iter().find(|r| r.at <= at && at < r.at + r.len) else { + return errno::fail(errno::ENOMEM); + }; + out.push(r.backed as u8); + at += PAGE; + if out.len() as u64 == MAX_SPAN || at >= end { + if guest.write(vec + written, &out) < out.len() as i64 { + return errno::fail(errno::EFAULT); + } + written += out.len() as u64; + out.clear(); + } + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/mem/sync.rs b/userland/capsule_linux/src/linux/call/mem/sync.rs new file mode 100644 index 000000000..665ff0cfa --- /dev/null +++ b/userland/capsule_linux/src/linux/call/mem/sync.rs @@ -0,0 +1,44 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . +//! `msync`: write a shared file mapping back to its file. +//! +//! Every file mapping here is private, since MAP_SHARED of a file is refused, +//! and a private mapping has nothing to write back: Linux answers msync on +//! one with its argument checks alone, and so does this. + +use crate::linux::abi::errno; +use crate::linux::guest::{page_up, Guest, PAGE}; + +const MS_ASYNC: u64 = 1; +const MS_INVALIDATE: u64 = 2; +const MS_SYNC: u64 = 4; + +pub fn msync(guest: &Guest, addr: u64, len: u64, flags: u64) -> u64 { + if flags & !(MS_ASYNC | MS_INVALIDATE | MS_SYNC) != 0 || addr % PAGE != 0 { + return errno::fail(errno::EINVAL); + } + if flags & MS_ASYNC != 0 && flags & MS_SYNC != 0 { + return errno::fail(errno::EINVAL); + } + let Some(end) = addr.checked_add(page_up(len)).filter(|_| len <= u64::MAX - PAGE) else { + return errno::fail(errno::ENOMEM); + }; + /* Linux reports a span with a page nothing maps as ENOMEM. */ + if end > addr && guest.mapped_from(addr) < end - addr { + return errno::fail(errno::ENOMEM); + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/call/mod.rs b/userland/capsule_linux/src/linux/call/mod.rs index ab445d17c..6b506d6ee 100644 --- a/userland/capsule_linux/src/linux/call/mod.rs +++ b/userland/capsule_linux/src/linux/call/mod.rs @@ -33,7 +33,7 @@ mod limits; mod limits_table; mod glibc; mod glibc_sched; -mod mem; +pub mod mem; mod pipe; mod pipe_dup; mod pipe_end; diff --git a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs index 3cf86191a..e7b7b3a87 100644 --- a/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs +++ b/userland/capsule_linux/src/linux/call/spawn/fork_copy.rs @@ -16,39 +16,42 @@ //! Copying a parent's spans into the child it just made. -use crate::linux::guest::{Guest, Region}; -use nonos_libc::peer::{mk_peer_map, mk_peer_write, PEER_PROT_EXEC, PEER_PROT_WRITE}; +use crate::linux::guest::{Guest, MAX_SPAN}; +use nonos_libc::peer::{mk_peer_map, mk_peer_write}; /// Every span, mapped into the child and then filled from the parent. pub(super) fn copy_spans(guest: &mut Guest, child: u32) -> bool { let spans = guest.regions.clone(); for span in spans { - // An unbacked reservation has no frames to copy; the child reserves it - // the same way, and its own first access faults a page in. + /* + * An unbacked reservation has no frames to copy; the child holds the + * same reservation, and a touch there faults in the child as here. + */ if !span.backed { continue; } - if mk_peer_map(child, span.at, span.len, prot_of(&span)) < 0 { - return false; - } - if !copy_one(guest, child, span.at, span.len) { - return false; + /* + * The kernel maps and copies at most MAX_SPAN in one peer call, so a + * larger span, which a Go heap is, crosses in pieces. Each piece gets + * the protection the span has now, PROT_NONE included: the kernel + * copies into a page whatever its protection, so the bytes still go + * in, and the child can do no more with them than the parent can. + */ + let mut done = 0; + while done < span.len { + let take = (span.len - done).min(MAX_SPAN); + if mk_peer_map(child, span.at + done, take, span.peer_prot()) < 0 { + return false; + } + if !copy_one(guest, child, span.at + done, take) { + return false; + } + done += take; } } true } -fn prot_of(span: &Region) -> u64 { - let mut prot = 0; - if span.write { - prot |= PEER_PROT_WRITE; - } - if span.exec { - prot |= PEER_PROT_EXEC; - } - prot -} - fn copy_one(guest: &Guest, child: u32, at: u64, len: u64) -> bool { let Some(bytes) = guest.read(at, len as usize) else { return false; diff --git a/userland/capsule_linux/src/linux/guest/mem_like.rs b/userland/capsule_linux/src/linux/guest/mem_like.rs new file mode 100644 index 000000000..aa51cd440 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/mem_like.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A span that takes after another: what mremap makes when a mapping grows +//! or moves, since Linux gives the new part the mapping's own protection. + +use super::handle::Guest; +use super::mem::MAX_SPAN; +use super::mem_span::map_span; +use super::region::Region; + +impl Guest { + /// `[at, at + len)` held like `like`: the same protection, provenance and + /// backing. A reservation stays one, with no pages. + pub fn map_like(&mut self, at: u64, len: u64, like: &Region) -> i64 { + if like.backed { + let rc = map_span(self.pid, at, len, like.peer_prot()); + if rc < 0 { + return rc; + } + } + self.regions.push(Region { at, len, ..*like }); + 0 + } + + /// Copy `len` bytes from `from` to `to` inside the guest, a megabyte at a + /// time. The kernel copies whatever the pages' protection is. + pub fn copy_within(&self, from: u64, to: u64, len: u64) -> bool { + let mut done = 0; + while done < len { + let take = (len - done).min(MAX_SPAN); + let Some(bytes) = self.read(from + done, take as usize) else { + return false; + }; + if self.write(to + done, &bytes) < bytes.len() as i64 { + return false; + } + done += take; + } + true + } +} diff --git a/userland/capsule_linux/src/linux/guest/mem_map.rs b/userland/capsule_linux/src/linux/guest/mem_map.rs index 83404d6f4..9ae63e81f 100644 --- a/userland/capsule_linux/src/linux/guest/mem_map.rs +++ b/userland/capsule_linux/src/linux/guest/mem_map.rs @@ -16,36 +16,23 @@ //! Backing a span of a guest with pages. -use nonos_libc::peer::{mk_peer_map, PEER_PROT_EXEC, PEER_PROT_WRITE}; - use super::handle::Guest; use super::layout::USER_MAX; -use super::mem::{span_within, MAX_SPAN}; -use super::region::Region; +use super::mem::span_within; +use super::mem_span::map_span; +use super::region::{peer_prot, Region}; use super::region_cut::cut; impl Guest { /// Pages covering `[addr, addr + len)`. pub fn map(&mut self, addr: u64, len: u64, write: bool, exec: bool) -> i64 { - // Bounded by the top of the guest's area, which is the stack. + /* Bounded by the top of the guest's area, which is the stack. */ let Some((start, span)) = span_within(addr, len, USER_MAX) else { return -1; }; - let mut prot = 0; - if write { - prot |= PEER_PROT_WRITE; - } - if exec { - prot |= PEER_PROT_EXEC; - } - let mut done = 0; - while done < span { - let take = (span - done).min(MAX_SPAN); - let rc = mk_peer_map(self.pid, start + done, take, prot); - if rc < 0 { - return rc; - } - done += take; + let rc = map_span(self.pid, start, span, peer_prot(write, exec, true)); + if rc < 0 { + return rc; } /* * Remembered because fork copies a guest by walking what its @@ -56,6 +43,7 @@ impl Guest { len: span, write, exec, + access: true, unproven: false, backed: true, }); @@ -63,45 +51,24 @@ impl Guest { } /// Back `[at, at + len)` of a reservation with the given protection, the - /// commit a fixed mmap makes. Pages the guest has not touched get zeroed - /// frames; pages it has touched keep their contents, since peer_map skips - /// a page that is already there. The span is then recorded as backed, in - /// place of the reservation it came from, so fork copies it. + /// commit an mprotect that asks for access makes. The kernel fills no page + /// a guest touches on its own, so every page here is new and zeroed. The + /// span is then recorded as backed, in place of the reservation it came + /// from, so fork copies it. pub fn commit(&mut self, at: u64, len: u64, write: bool, exec: bool) -> i64 { - let mut prot = 0; - if write { - prot |= PEER_PROT_WRITE; - } - if exec { - prot |= PEER_PROT_EXEC; - } - let mut done = 0; - while done < len { - let take = (len - done).min(MAX_SPAN); - let rc = mk_peer_map(self.pid, at + done, take, prot); - if rc < 0 { - return rc; - } - done += take; + let rc = map_span(self.pid, at, len, peer_prot(write, exec, true)); + if rc < 0 { + return rc; } self.regions = cut(&self.regions, at, len); - self.regions.push(Region { at, len, write, exec, unproven: false, backed: true }); - 0 - } - - /// Take `len` of address space at `addr` without backing it: a PROT_NONE - /// reservation. Bytes appear, zeroed, when the guest first touches them. - pub fn reserve(&mut self, addr: u64, len: u64) -> i64 { - let Some((start, span)) = span_within(addr, len, USER_MAX) else { - return -1; - }; self.regions.push(Region { - at: start, - len: span, - write: true, - exec: false, + at, + len, + write, + exec, + access: true, unproven: false, - backed: false, + backed: true, }); 0 } diff --git a/userland/capsule_linux/src/linux/guest/mem_reserve.rs b/userland/capsule_linux/src/linux/guest/mem_reserve.rs new file mode 100644 index 000000000..bd533af92 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/mem_reserve.rs @@ -0,0 +1,43 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Taking address space for a guest without backing it. + +use super::handle::Guest; +use super::layout::USER_MAX; +use super::mem::span_within; +use super::region::Region; + +impl Guest { + /// Take `len` of address space at `addr` without backing it: a PROT_NONE + /// reservation. No page exists until a commit maps one; a touch before + /// that is a fault, as it is on Linux. + pub fn reserve(&mut self, addr: u64, len: u64) -> i64 { + let Some((start, span)) = span_within(addr, len, USER_MAX) else { + return -1; + }; + self.regions.push(Region { + at: start, + len: span, + write: false, + exec: false, + access: false, + unproven: false, + backed: false, + }); + 0 + } +} diff --git a/userland/capsule_linux/src/linux/guest/mem_span.rs b/userland/capsule_linux/src/linux/guest/mem_span.rs new file mode 100644 index 000000000..520876792 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/mem_span.rs @@ -0,0 +1,35 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Backing a span of a guest, a megabyte at a time. + +use nonos_libc::peer::mk_peer_map; + +use super::mem::MAX_SPAN; + +/// `MkPeerMap` over a span, a megabyte at a time. +pub(super) fn map_span(pid: u32, at: u64, len: u64, prot: u64) -> i64 { + let mut done = 0; + while done < len { + let take = (len - done).min(MAX_SPAN); + let rc = mk_peer_map(pid, at + done, take, prot); + if rc < 0 { + return rc; + } + done += take; + } + 0 +} diff --git a/userland/capsule_linux/src/linux/guest/mod.rs b/userland/capsule_linux/src/linux/guest/mod.rs index 514a26bdd..b0e86ebb1 100644 --- a/userland/capsule_linux/src/linux/guest/mod.rs +++ b/userland/capsule_linux/src/linux/guest/mod.rs @@ -36,12 +36,16 @@ mod links_list; mod links_load; mod mem; mod mem_copy; +mod mem_like; mod mem_map; +mod mem_reserve; +mod mem_span; mod mem_unmap; mod region; mod region_cut; mod region_find; mod region_mark; +mod region_prot; mod threads; mod timer; mod watch; @@ -57,6 +61,6 @@ pub use layout::{ STACK_TOP, USER_MAX, }; pub use mem::{page_down, page_up, span_within, MAX_SPAN, PAGE}; -pub use region::Region; +pub use region::{peer_prot, Region}; pub use timer::Timer; pub use watch::{Watch, EPOLLET, EPOLLONESHOT}; diff --git a/userland/capsule_linux/src/linux/guest/region.rs b/userland/capsule_linux/src/linux/guest/region.rs index 8caf7b1a1..9ed68d6c9 100644 --- a/userland/capsule_linux/src/linux/guest/region.rs +++ b/userland/capsule_linux/src/linux/guest/region.rs @@ -16,17 +16,44 @@ //! One span of a guest's address space, as this capsule laid it down. +use nonos_libc::peer::{PEER_PROT_EXEC, PEER_PROT_NONE, PEER_PROT_WRITE}; + #[derive(Clone, Copy)] pub struct Region { pub at: u64, pub len: u64, pub write: bool, pub exec: bool, + /// False for PROT_NONE: the guest may not touch the span at all. A backed + /// span keeps its pages and their bytes, present to the kernel only. + pub access: bool, /// File bytes mapped without exec, so never proved: mprotect may not /// make them executable later. pub unproven: bool, - /// False for a PROT_NONE reservation: address space taken, no frames yet. - /// The kernel demand-fills a page on first access, so reserving a large - /// span and committing a little costs only what is touched; fork skips it. + /// False for a PROT_NONE reservation: address space taken, no frames. + /// The kernel fills no page for a guest on its own, so a touch of one is a + /// fault; a commit maps the part asked for and records it backed. pub backed: bool, } + +impl Region { + /// The protection the kernel is asked to give this span's pages. + pub fn peer_prot(&self) -> u64 { + peer_prot(self.write, self.exec, self.access) + } +} + +/// Peer protection bits for an access, a write and an exec permission. +pub fn peer_prot(write: bool, exec: bool, access: bool) -> u64 { + if !access { + return PEER_PROT_NONE; + } + let mut prot = 0; + if write { + prot |= PEER_PROT_WRITE; + } + if exec { + prot |= PEER_PROT_EXEC; + } + prot +} diff --git a/userland/capsule_linux/src/linux/guest/region_find.rs b/userland/capsule_linux/src/linux/guest/region_find.rs index aabfe4cd0..e557af1a7 100644 --- a/userland/capsule_linux/src/linux/guest/region_find.rs +++ b/userland/capsule_linux/src/linux/guest/region_find.rs @@ -36,4 +36,10 @@ impl Guest { } reach - addr } + + /// Whether any span the guest holds meets `[at, at + len)`. + pub fn overlaps(&self, at: u64, len: u64) -> bool { + let end = at.saturating_add(len); + self.regions.iter().any(|r| r.at < end && at < r.at.saturating_add(r.len)) + } } diff --git a/userland/capsule_linux/src/linux/guest/region_prot.rs b/userland/capsule_linux/src/linux/guest/region_prot.rs new file mode 100644 index 000000000..705ddf967 --- /dev/null +++ b/userland/capsule_linux/src/linux/guest/region_prot.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Recording a protection change on the spans a guest holds. + +use alloc::vec::Vec; + +use super::handle::Guest; +use super::region::Region; +use super::region_cut::cut; + +impl Guest { + /// Every backed part of `[at, at + len)` now has this protection. The + /// list is what fork maps the child from, so a list that kept the old + /// protection would give the child access the parent gave up. + pub fn set_prot(&mut self, at: u64, len: u64, write: bool, exec: bool, access: bool) { + let end = at.saturating_add(len); + let changed: Vec = self + .regions + .iter() + .filter(|r| r.backed && r.at < end && at < r.at.saturating_add(r.len)) + .map(|r| { + let from = r.at.max(at); + let to = r.at.saturating_add(r.len).min(end); + Region { at: from, len: to - from, write, exec, access, ..*r } + }) + .collect(); + for piece in &changed { + self.regions = cut(&self.regions, piece.at, piece.len); + } + self.regions.extend(changed); + } +} diff --git a/userland/capsule_linux/src/linux/serve/table_mem.rs b/userland/capsule_linux/src/linux/serve/table_mem.rs index 4ca19272b..71e3cf67c 100644 --- a/userland/capsule_linux/src/linux/serve/table_mem.rs +++ b/userland/capsule_linux/src/linux/serve/table_mem.rs @@ -27,6 +27,12 @@ pub fn mem_ops(guest: &mut Guest, nr: u64, a: [u64; 6]) -> Option { nr::MUNMAP => call::munmap(guest, a[0], a[1]), nr::MPROTECT => call::mprotect(guest, a[0], a[1], a[2]), nr::MREMAP => call::mremap(guest, a[0], a[1], a[2], a[3]), + nr::MSYNC => call::mem::msync(guest, a[0], a[1], a[2]), + nr::MINCORE => call::mem::mincore(guest, a[0], a[1], a[2]), + nr::MLOCK | nr::MUNLOCK => call::mem::mlock(guest, a[0], a[1]), + nr::MLOCK2 => call::mem::mlock2(guest, a[0], a[1], a[2]), + nr::MLOCKALL => call::mem::mlockall(a[0]), + nr::MUNLOCKALL => errno::ok(0), // Advice, and this capsule takes none of it. nr::MADVISE => errno::ok(0), _ => return None, diff --git a/userland/libc/src/peer.rs b/userland/libc/src/peer.rs index 988f4cff9..29bb9ed16 100644 --- a/userland/libc/src/peer.rs +++ b/userland/libc/src/peer.rs @@ -25,6 +25,9 @@ use crate::syscall::{ /// Pages of a guest may be written, and may be executed. pub const PEER_PROT_WRITE: u64 = 1 << 0; pub const PEER_PROT_EXEC: u64 = 1 << 1; +/// Pages the guest may not touch at all, their bytes kept for a later +/// protection that opens them: PROT_NONE. +pub const PEER_PROT_NONE: u64 = 1 << 2; /// Back a span of a guest's address space with fresh zeroed frames. pub fn mk_peer_map(pid: u32, addr: u64, len: u64, prot: u64) -> i64 { diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index c42c8e873..3848f6f20 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -119,6 +119,32 @@ $(LINUX_GUESTS_C)/cwait: $(LINUX_GUESTS_DIR)/c/cwait.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait)) +# The memory proofs, one program whose first argument names the proof, so the +# store carries one binary and one set of proofs for all five: +# guardpage a pthread recursing into its guard page ends on SIGSEGV, 139 +# protnone PROT_NONE means no access, and bytes survive a close and reopen +# protfork a fork after mprotect gives the child the protection set now +# touchfork bytes written into a reservation survive a fork; MAP_FIXED +# over a written page replaces it with zeroes +# memcalls mmap placement, brk, alignment, mremap, and mlock, msync and +# mincore, each against Linux's answer +MEMPROOF_PARTS := guardpage protnone protfork touchfork memcalls escape +# Files the proofs share, built as they are: no main of their own. +MEMPROOF_SHARED := memproof_run memcalls_map memcalls_remap memcalls_lock escape_map escape_exec +MEMPROOF_SRCS := $(foreach p,memproof $(MEMPROOF_PARTS) $(MEMPROOF_SHARED),\ + $(LINUX_GUESTS_DIR)/c/$(p).c) $(LINUX_GUESTS_DIR)/c/memproof.h $(LINUX_GUESTS_DIR)/c/memcalls.h $(LINUX_GUESTS_DIR)/c/escape.h +$(LINUX_GUESTS_C)/memproof: $(MEMPROOF_SRCS) + @mkdir -p $(@D)/memproof.o + @for p in $(MEMPROOF_PARTS); do \ + musl-gcc -O2 -c -Dmain=$${p}_main -o $(@D)/memproof.o/$$p.o $(LINUX_GUESTS_DIR)/c/$$p.c || exit 1; \ + done + @for p in $(MEMPROOF_SHARED); do \ + musl-gcc -O2 -c -o $(@D)/memproof.o/$$p.o $(LINUX_GUESTS_DIR)/c/$$p.c || exit 1; \ + done + @musl-gcc -O2 -static -o $@ $(LINUX_GUESTS_DIR)/c/memproof.c \ + $(foreach p,$(MEMPROOF_PARTS) $(MEMPROOF_SHARED),$(@D)/memproof.o/$(p).o) +$(eval $(call LINUX_GUEST,memproof,4980,4981,$(LINUX_GUESTS_C)/memproof)) + # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the # normal image, which does not set NONOS_LINUX_GUESTS, still ships them. diff --git a/userland/linux_guests/c/escape.c b/userland/linux_guests/c/escape.c new file mode 100644 index 000000000..4253fb8e0 --- /dev/null +++ b/userland/linux_guests/c/escape.c @@ -0,0 +1,31 @@ +/* + * A Linux guest trying to break out of its own address space through the + * syscall surface it is given: reach the kernel half, wrap a span, map + * write-and-execute, add execute to a file it never proved, call a number + * that is not served, and step past the end of what it holds. Each part + * passes when the machine refuses; one boot names any that got through. + */ +#include +#include + +#include "escape.h" + +void held(const char *name, int ok, long got) { + char detail[48]; + snprintf(detail, sizeof detail, "got %ld", got); + part_line("escape", name, ok, detail); +} + +long erc(long v) { + return v == -1 ? -errno : v; +} + +int main(void) { + reach_kernel(); + wrap_span(); + wx_map(); + exec_escalate(); + forged_call(); + past_end(); + return finish("escape"); +} diff --git a/userland/linux_guests/c/escape.h b/userland/linux_guests/c/escape.h new file mode 100644 index 000000000..355e7253d --- /dev/null +++ b/userland/linux_guests/c/escape.h @@ -0,0 +1,19 @@ +/* What the escape attack files share. Each part passes when NONOS refuses. */ +#ifndef ESCAPE_H +#define ESCAPE_H + +#include "memproof.h" + +/* Record an attack: ok means the machine held (refused it). */ +void held(const char *name, int ok, long got); +/* -errno of a call (escape) that returned -1, or its value. */ +long erc(long v); + +void reach_kernel(void); +void wrap_span(void); +void wx_map(void); +void exec_escalate(void); +void forged_call(void); +void past_end(void); + +#endif diff --git a/userland/linux_guests/c/escape_exec.c b/userland/linux_guests/c/escape_exec.c new file mode 100644 index 000000000..5fa27175a --- /dev/null +++ b/userland/linux_guests/c/escape_exec.c @@ -0,0 +1,50 @@ +/* Attacks on what a guest may run and call: adding execute to a file it never + * proved, a syscall number that is not served, and a step past a mapping. */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "escape.h" + +static volatile char *edge; + +void exec_escalate(void) { + /* Map this program read-only, then try to make it executable. It was + * mapped without execute, so it was never proved; NONOS refuses. */ + int fd = open("/bin/memproof", O_RDONLY); + if (fd < 0) { + fd = open("/proc/self/exe", O_RDONLY); + } + if (fd < 0) { + held("exec added to an unproven file mapping is refused", 0, -errno); + return; + } + void *p = mmap(0, PG, PROT_READ, MAP_PRIVATE, fd, 0); + close(fd); + if (p == MAP_FAILED) { + held("exec added to an unproven file mapping is refused", 0, -errno); + return; + } + long r = erc(mprotect(p, PG, PROT_READ | PROT_EXEC)); + held("exec added to an unproven file mapping is refused", r == -EPERM, r); +} + +void forged_call(void) { + /* A syscall number the personality does not serve. */ + long r = erc(syscall(0x462)); + held("an unserved syscall number answers ENOSYS", r == -ENOSYS, r); +} + +static void step_past(void) { + edge[PG] = 1; +} + +void past_end(void) { + /* Two pages, second given back, so the page written is a certain hole. */ + edge = mmap(0, 2 * PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + munmap((void *)(edge + PG), PG); + part("escape", "a write into an unmapped hole faults", step_past, 1); +} diff --git a/userland/linux_guests/c/escape_map.c b/userland/linux_guests/c/escape_map.c new file mode 100644 index 000000000..0cecc8ea3 --- /dev/null +++ b/userland/linux_guests/c/escape_map.c @@ -0,0 +1,40 @@ +/* Attacks on where and how a guest may map: the kernel half, a wrapping + * span, and a page that is both writable and executable. */ +#include +#include +#include + +#include "escape.h" + +/* The first address of the kernel half, which no guest mapping may reach. */ +#define KERNEL_HALF 0x0000800000000000UL + +static int fixed_refused(uintptr_t at) { + void *p = mmap((void *)at, PG, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); + if (p == MAP_FAILED) { + return 1; + } + /* It answered with an address; it must at least not be the one asked. */ + munmap(p, PG); + return (uintptr_t)p != at; +} + +void reach_kernel(void) { + held("MAP_FIXED into the kernel half is refused", fixed_refused(KERNEL_HALF), 0); + held("MAP_FIXED one page below the kernel half is refused", fixed_refused(KERNEL_HALF - PG), 0); +} + +void wrap_span(void) { + /* A length that wraps past the top of the address space. */ + void *p = mmap(0, (size_t)-PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + long got = p == MAP_FAILED ? -errno : 0; + held("a mapping whose length wraps is refused", p == MAP_FAILED, got); +} + +void wx_map(void) { + /* Write and execute at once: NONOS refuses it, Linux allows it. */ + void *p = mmap(0, PG, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + long got = p == MAP_FAILED ? -errno : 0; + held("a write-and-execute mapping is refused", p == MAP_FAILED, got); +} diff --git a/userland/linux_guests/c/guardpage.c b/userland/linux_guests/c/guardpage.c new file mode 100644 index 000000000..34e2ad1d8 --- /dev/null +++ b/userland/linux_guests/c/guardpage.c @@ -0,0 +1,66 @@ +/* + * A pthread recurses until it walks off the bottom of its stack. musl reserves + * each thread stack with PROT_NONE and opens all but the lowest part with + * mprotect, so the part it leaves closed is the guard. On Linux the first touch + * of the guard is SIGSEGV, which ends the whole process with status 139. The + * recursion stops by itself 64 KiB below the guard, so a guard that guards + * nothing prints the FAIL line with how far it ran; no PASS line exists, since + * the only correct outcome is that the process does not get to print one. + */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "memproof.h" + +#define FRAME 1024 +#define PAST (64 * 1024) + +static uintptr_t lo; +static uintptr_t deepest; + +static unsigned dive(unsigned depth) { + volatile char frame[FRAME]; + uintptr_t here = (uintptr_t)frame; + frame[0] = (char)depth; + frame[FRAME - 1] = (char)depth; + deepest = here; + if (here < lo - PAST) { + return depth; + } + return dive(depth + 1) + frame[0] - frame[FRAME - 1]; +} + +static void *worker(void *arg) { + (void)arg; + pthread_attr_t a; + void *base; + size_t size, guard; + char line[160]; + pthread_getattr_np(pthread_self(), &a); + pthread_attr_getstack(&a, &base, &size); + pthread_attr_getguardsize(&a, &guard); + lo = (uintptr_t)base; + snprintf(line, sizeof line, "[C] guardpage: stack %zu KiB, guard %zu KiB below 0x%lx; recursing\n", + size / 1024, guard / 1024, (unsigned long)lo); + say(line); + unsigned depth = dive(0); + snprintf(line, sizeof line, + "[C] guardpage FAIL: %u frames, ran %lu KiB below the stack with no fault\n", depth, + (unsigned long)((lo - deepest) / 1024)); + say(line); + return 0; +} + +int main(void) { + pthread_t t; + if (pthread_create(&t, 0, worker, 0) != 0) { + say("[C] guardpage FAIL: no thread\n"); + return 1; + } + pthread_join(t, 0); + return 1; +} diff --git a/userland/linux_guests/c/memcalls.c b/userland/linux_guests/c/memcalls.c new file mode 100644 index 000000000..437cd19dd --- /dev/null +++ b/userland/linux_guests/c/memcalls.c @@ -0,0 +1,61 @@ +/* + * The memory calls answered as Linux answers them: where mmap puts a hint, the + * break giving pages back, unaligned addresses refused, mremap keeping a + * mapping's protection, and mlock, msync and mincore with their errnos. Every + * part runs and prints one line, so one boot names every part that fails. + * Parts that must fault run in a forked child. + */ +#include +#include +#include +#include +#include + +#include "memcalls.h" + +volatile char *mc_p; + +void check(const char *name, int ok, long got) { + char detail[48]; + snprintf(detail, sizeof detail, "got %ld", got); + part_line("memcalls", name, ok, detail); +} + +void faults(const char *name, void (*fn)(void)) { + pid_t c = fork(); + if (c == 0) { + fn(); + _exit(0); + } + int st = 0; + waitpid(c, &st, 0); + check(name, segv(st), st); +} + +long rc(long v) { + return v == -1 ? -errno : v; +} + +char *anon(long len, int prot) { + return mmap(0, len, prot, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); +} + +void mc_write(void) { + mc_p[0] = 1; +} + +void mc_read(void) { + (void)mc_p[0]; +} + +int main(void) { + placement(); + breaks(); + alignment(); + remaps(); + provenance(); + locks(); + syncs(); + cores(); + return finish("memcalls"); +} diff --git a/userland/linux_guests/c/memcalls.h b/userland/linux_guests/c/memcalls.h new file mode 100644 index 000000000..483923926 --- /dev/null +++ b/userland/linux_guests/c/memcalls.h @@ -0,0 +1,30 @@ +/* What the memcalls files share. */ +#ifndef MEMCALLS_H +#define MEMCALLS_H + +#include "memproof.h" + +#define NOREPLACE 0x100000 + +/* The page a part run in a child touches. */ +extern volatile char *mc_p; + +void check(const char *name, int ok, long got); +/* A part run in a forked child that must die of SIGSEGV. */ +void faults(const char *name, void (*fn)(void)); +/* -errno of a call that returned -1, or its value. */ +long rc(long v); +char *anon(long len, int prot); +void mc_write(void); +void mc_read(void); + +void placement(void); +void breaks(void); +void alignment(void); +void remaps(void); +void provenance(void); +void locks(void); +void syncs(void); +void cores(void); + +#endif diff --git a/userland/linux_guests/c/memcalls_lock.c b/userland/linux_guests/c/memcalls_lock.c new file mode 100644 index 000000000..b8b74ce02 --- /dev/null +++ b/userland/linux_guests/c/memcalls_lock.c @@ -0,0 +1,55 @@ +/* memcalls: mlock, msync and mincore with Linux's errnos. */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "memcalls.h" + +void locks(void) { + char *m = anon(3 * PG, PROT_READ | PROT_WRITE); + check("mlock of a mapping", rc(mlock(m, 3 * PG)) == 0, rc(mlock(m, 3 * PG))); + check("munlock of a mapping", rc(munlock(m, 3 * PG)) == 0, rc(munlock(m, 3 * PG))); + munmap(m + PG, PG); + check("mlock over a hole is ENOMEM", rc(mlock(m, 3 * PG)) == -ENOMEM, rc(mlock(m, 3 * PG))); + check("mlock2 with an unknown flag is EINVAL", rc(syscall(SYS_mlock2, m, PG, 2)) == -EINVAL, + rc(syscall(SYS_mlock2, m, PG, 2))); + check("mlock2 MLOCK_ONFAULT", rc(syscall(SYS_mlock2, m, PG, 1)) == 0, + rc(syscall(SYS_mlock2, m, PG, 1))); + check("mlockall(0) is EINVAL", rc(mlockall(0)) == -EINVAL, rc(mlockall(0))); + check("mlockall(MCL_ONFAULT) alone is EINVAL", rc(mlockall(MCL_ONFAULT)) == -EINVAL, + rc(mlockall(MCL_ONFAULT))); + check("mlockall(MCL_CURRENT)", rc(mlockall(MCL_CURRENT)) == 0, rc(mlockall(MCL_CURRENT))); + check("munlockall", rc(munlockall()) == 0, rc(munlockall())); +} + +void syncs(void) { + char *m = anon(3 * PG, PROT_READ | PROT_WRITE); + check("msync MS_SYNC", rc(msync(m, 3 * PG, MS_SYNC)) == 0, rc(msync(m, 3 * PG, MS_SYNC))); + check("msync unaligned is EINVAL", rc(msync(m + 1, PG, MS_SYNC)) == -EINVAL, + rc(msync(m + 1, PG, MS_SYNC))); + check("msync MS_ASYNC|MS_SYNC is EINVAL", rc(msync(m, PG, MS_ASYNC | MS_SYNC)) == -EINVAL, + rc(msync(m, PG, MS_ASYNC | MS_SYNC))); + check("msync unknown flag is EINVAL", rc(msync(m, PG, 8)) == -EINVAL, rc(msync(m, PG, 8))); + munmap(m + PG, PG); + check("msync over a hole is ENOMEM", rc(msync(m, 3 * PG, MS_SYNC)) == -ENOMEM, + rc(msync(m, 3 * PG, MS_SYNC))); +} + +void cores(void) { + char *r = anon(4 * PG, PROT_NONE); + mprotect(r + PG, PG, PROT_READ | PROT_WRITE); + r[PG] = 1; + unsigned char v[4] = { 9, 9, 9, 9 }; + long got = rc(mincore(r, 4 * PG, v)); + check("mincore of a reservation with one page opened is 0,1,0,0", + got == 0 && v[0] == 0 && v[1] == 1 && v[2] == 0 && v[3] == 0, + got ? got : v[0] | v[1] << 8 | v[2] << 16 | (long)v[3] << 24); + check("mincore unaligned is EINVAL", rc(mincore(r + 1, PG, v)) == -EINVAL, + rc(mincore(r + 1, PG, v))); + munmap(r + 2 * PG, PG); + check("mincore over a hole is ENOMEM", rc(mincore(r, 4 * PG, v)) == -ENOMEM, + rc(mincore(r, 4 * PG, v))); +} diff --git a/userland/linux_guests/c/memcalls_map.c b/userland/linux_guests/c/memcalls_map.c new file mode 100644 index 000000000..2adb3be11 --- /dev/null +++ b/userland/linux_guests/c/memcalls_map.c @@ -0,0 +1,66 @@ +/* memcalls: where mmap puts a mapping, the break, and unaligned addresses. */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "memcalls.h" + +void placement(void) { + char *m = anon(PG, PROT_READ | PROT_WRITE); + m[0] = 0x11; + char *n = mmap(m, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + check("hint on a mapping lands elsewhere, zeroed", n != m && n[0] == 0 && m[0] == 0x11, + (long)(n - m)); + void *q = mmap(m, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + long e = q == MAP_FAILED ? -errno : 0; + check("MAP_FIXED_NOREPLACE on a mapping is EEXIST", e == -EEXIST && m[0] == 0x11, e); + /* + * A mapping placed just above the last one mmap chose: the next mmap that + * leaves the choice to the system must not land on it. On a system that + * already holds that address the probe cannot be placed, and says so. + */ + char *a = anon(PG, PROT_READ | PROT_WRITE); + char *f = mmap(a + PG, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + if (f == MAP_FAILED) { + check("next mmap skips a mapping placed above the last (probe address taken)", + errno == EEXIST, -errno); + return; + } + f[0] = 0x44; + char *b = anon(PG, PROT_READ | PROT_WRITE); + check("next mmap skips a mapping placed above the last", b != f && b[0] == 0 && f[0] == 0x44, + (long)(b - f)); +} + +void breaks(void) { + long cur = syscall(SYS_brk, 0); + long up = syscall(SYS_brk, cur + 2 * PG); + ((volatile char *)cur)[PG] = 0x22; + syscall(SYS_brk, cur); + syscall(SYS_brk, cur + 2 * PG); + char b = ((volatile char *)cur)[PG]; + check("brk down and up again reads zero", up == cur + 2 * PG && b == 0, b); + syscall(SYS_brk, cur); + long top = (cur + PG - 1) & ~(long)(PG - 1); + void *in = mmap((void *)(top + PG), PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + long got = syscall(SYS_brk, top + 4 * PG); + check("brk into a mapping is refused", in != MAP_FAILED && got == cur, got - cur); + munmap(in, PG); +} + +void alignment(void) { + char *m = anon(2 * PG, PROT_READ | PROT_WRITE); + check("munmap unaligned is EINVAL", rc(munmap(m + 1, PG)) == -EINVAL, rc(munmap(m + 1, PG))); + /* musl's mprotect rounds the address down itself; the kernel's does not. */ + long e = rc(syscall(SYS_mprotect, m + 1, PG, PROT_READ)); + check("mprotect unaligned is EINVAL", e == -EINVAL, e); + void *q = mmap(m + 1, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); + e = q == MAP_FAILED ? -errno : 0; + check("MAP_FIXED unaligned is EINVAL", e == -EINVAL, e); + /* musl's mmap refuses an unaligned offset itself; the kernel's must too. */ + e = rc(syscall(SYS_mmap, 0, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS, -1, 1)); + check("mmap offset unaligned is EINVAL", e == -EINVAL, e); +} diff --git a/userland/linux_guests/c/memcalls_remap.c b/userland/linux_guests/c/memcalls_remap.c new file mode 100644 index 000000000..add9b6d0e --- /dev/null +++ b/userland/linux_guests/c/memcalls_remap.c @@ -0,0 +1,56 @@ +/* memcalls: mremap keeping a mapping's protection, backing and provenance. */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "memcalls.h" + +void remaps(void) { + char *m = anon(PG, PROT_READ | PROT_WRITE); + m[0] = 0x33; + mprotect(m, PG, PROT_READ); + char *g = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE); + check("mremap of a read-only page keeps the byte", g != MAP_FAILED && g[0] == 0x33, g[0]); + mc_p = g + PG; + faults("mremap grown part of a read-only page faults on write", mc_write); + char *r = anon(2 * PG, PROT_NONE); + anon(PG, PROT_NONE); + char *q = mremap(r, 2 * PG, 4 * PG, MREMAP_MAYMOVE); + check("mremap of a reservation succeeds", q != MAP_FAILED, (long)(q == MAP_FAILED)); + mc_p = q + 3 * PG; + faults("mremap grown reservation still faults on read", mc_read); + char *two = anon(2 * PG, PROT_READ | PROT_WRITE); + mprotect(two + PG, PG, PROT_READ); + void *x = mremap(two, 2 * PG, 3 * PG, MREMAP_MAYMOVE); + long e = x == MAP_FAILED ? -errno : 0; + check("mremap across two mappings is EFAULT", e == -EFAULT, e); +} + +/* + * A file mapped without exec was never proved; where mprotect refuses to make + * it executable, it must refuse the copy mremap moved too. Host Linux allows + * both, and the part checks only that the two answers agree. + */ +void provenance(void) { + /* This program's own file: /bin/memproof in the store, itself on a host. */ + int fd = open("/bin/memproof", O_RDONLY); + if (fd < 0) { + fd = open("/proc/self/exe", O_RDONLY); + } + char *m = mmap(0, PG, PROT_READ, MAP_PRIVATE, fd, 0); + if (m == MAP_FAILED) { + check("file mapping for the provenance part", 0, -errno); + return; + } + long first = rc(mprotect(m, PG, PROT_READ | PROT_EXEC)); + mprotect(m, PG, PROT_READ); + mmap(m + PG, PG, PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS | NOREPLACE, -1, 0); + char *n = mremap(m, PG, 2 * PG, MREMAP_MAYMOVE); + long moved = n == MAP_FAILED ? -1000 : rc(mprotect(n, PG, PROT_READ | PROT_EXEC)); + check("moved unproven file bytes refused exec as before the move", n != m && moved == first, + moved); + close(fd); +} diff --git a/userland/linux_guests/c/memproof.c b/userland/linux_guests/c/memproof.c new file mode 100644 index 000000000..2f154eccf --- /dev/null +++ b/userland/linux_guests/c/memproof.c @@ -0,0 +1,39 @@ +/* + * The memory proofs as one program, so the test store carries one binary and + * one set of proofs for all of them instead of one per proof: the store has a + * fixed load budget and each proof set is most of a guest's size there. The + * first argument names the proof; each is its own file, built with its main + * renamed, and runs exactly as it would as a program of its own. + */ +#include +#include + +int guardpage_main(void); +int protnone_main(void); +int protfork_main(void); +int touchfork_main(void); +int memcalls_main(void); +int escape_main(void); + +static const struct { + const char *name; + int (*run)(void); +} proofs[] = { + { "guardpage", guardpage_main }, + { "protnone", protnone_main }, + { "protfork", protfork_main }, + { "touchfork", touchfork_main }, + { "memcalls", memcalls_main }, + { "escape", escape_main }, +}; + +int main(int argc, char **argv) { + for (unsigned i = 0; argc > 1 && i < sizeof proofs / sizeof proofs[0]; i++) { + if (strcmp(argv[1], proofs[i].name) == 0) { + return proofs[i].run(); + } + } + fputs("[C] memproof FAIL: name a proof: guardpage protnone protfork touchfork memcalls escape\n", + stdout); + return 2; +} diff --git a/userland/linux_guests/c/memproof.h b/userland/linux_guests/c/memproof.h new file mode 100644 index 000000000..69dd07152 --- /dev/null +++ b/userland/linux_guests/c/memproof.h @@ -0,0 +1,27 @@ +/* + * What every memory proof shares: printing a line, running a part in a forked + * child, reading how the child ended, and counting parts for the last line. + */ +#ifndef MEMPROOF_H +#define MEMPROOF_H + +#define PG 4096 + +void say(const char *s); + +/* + * A SIGSEGV death. The personality reports a signal death as exit status + * 128+signo, which counts as the same SIGSEGV; the raw status is printed. + */ +int segv(int st); + +/* Count one part and print its line: "[C] : ok|FAIL ()". */ +void part_line(const char *proof, const char *name, int ok, const char *detail); + +/* Run `fn` in a forked child that must, or must not, die of SIGSEGV. */ +void part(const char *proof, const char *name, void (*fn)(void), int must_fault); + +/* Print the PASS or FAIL line and give the exit code. */ +int finish(const char *proof); + +#endif diff --git a/userland/linux_guests/c/memproof_run.c b/userland/linux_guests/c/memproof_run.c new file mode 100644 index 000000000..4d6a1100f --- /dev/null +++ b/userland/linux_guests/c/memproof_run.c @@ -0,0 +1,53 @@ +/* The parts of memproof.h every proof shares. */ +#include +#include +#include +#include +#include + +#include "memproof.h" + +static int passed, failed; + +void say(const char *s) { + write(1, s, strlen(s)); +} + +int segv(int st) { + return (WIFSIGNALED(st) && WTERMSIG(st) == SIGSEGV) || + (WIFEXITED(st) && WEXITSTATUS(st) == 128 + SIGSEGV); +} + +void part_line(const char *proof, const char *name, int ok, const char *detail) { + char line[200]; + ok ? passed++ : failed++; + snprintf(line, sizeof line, "[C] %s %s: %s (%s)\n", proof, name, ok ? "ok" : "FAIL", detail); + say(line); +} + +void part(const char *proof, const char *name, void (*fn)(void), int must_fault) { + char detail[64]; + pid_t c = fork(); + if (c < 0) { + part_line(proof, name, 0, "fork failed"); + return; + } + if (c == 0) { + fn(); + _exit(0); + } + int st = 0; + waitpid(c, &st, 0); + int clean = WIFEXITED(st) && WEXITSTATUS(st) == 0; + snprintf(detail, sizeof detail, "%s, status 0x%x", must_fault ? "must fault" : "must not fault", + st); + part_line(proof, name, must_fault ? segv(st) : clean, detail); +} + +int finish(const char *proof) { + char line[160]; + snprintf(line, sizeof line, "[C] %s %s: %d parts ok, %d failed\n", proof, + failed ? "FAIL" : "PASS", passed, failed); + say(line); + return failed ? 1 : 0; +} diff --git a/userland/linux_guests/c/protfork.c b/userland/linux_guests/c/protfork.c new file mode 100644 index 000000000..5f4060322 --- /dev/null +++ b/userland/linux_guests/c/protfork.c @@ -0,0 +1,68 @@ +/* + * A fork gives the child the parent's mappings with the protection they have + * now, not the one they were made with. Each part changes a protection with + * mprotect, forks, and the child tries one access; the parent reads how the + * child ended. + */ +#include +#include + +#include "memproof.h" + +static volatile char *p; + +static void write_it(void) { + p[0] = 1; + if (p[0] != 1) { + _exit(2); + } +} +static void read_it(void) { + if (p[0] != 0x5a) { + _exit(2); + } +} + +static volatile char *fresh(int pages) { + volatile char *m = + mmap(0, pages * PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + for (int i = 0; i < pages; i++) { + m[i * PG] = 0x5a; + } + return m; +} + +int main(void) { + const char *me = "protfork"; + p = fresh(1); + mprotect((void *)p, PG, PROT_READ); + part(me, "RW to R, child writes", write_it, 1); + part(me, "RW to R, child reads the byte", read_it, 0); + + p = fresh(1); + mprotect((void *)p, PG, PROT_NONE); + part(me, "RW to NONE, child reads", read_it, 1); + + p = fresh(1); + mprotect((void *)p, PG, PROT_READ); + mprotect((void *)p, PG, PROT_READ | PROT_WRITE); + part(me, "RW to R to RW, child writes", write_it, 0); + + volatile char *m = fresh(3); + mprotect((void *)(m + PG), PG, PROT_READ); + p = m; + part(me, "middle page R, child writes the first", write_it, 0); + p = m + PG; + part(me, "middle page R, child writes the middle", write_it, 1); + p = m + 2 * PG; + part(me, "middle page R, child writes the last", write_it, 0); + /* Larger than the kernel's 1 MiB per peer call, so fork copies it in pieces. */ + long big = 2 * 1024 * 1024; + p = mmap(0, big, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + p += big - PG; + p[0] = 0x5a; + part(me, "2 MiB RW, child reads the last page", read_it, 0); + mprotect((void *)(p - big + PG), big, PROT_NONE); + part(me, "2 MiB RW to NONE, child reads the last page", read_it, 1); + return finish(me); +} diff --git a/userland/linux_guests/c/protnone.c b/userland/linux_guests/c/protnone.c new file mode 100644 index 000000000..dfc9f5275 --- /dev/null +++ b/userland/linux_guests/c/protnone.c @@ -0,0 +1,47 @@ +/* + * PROT_NONE means no access. Each part runs in a forked child and the parent + * reads how the child ended: a part that must fault passes only when the child + * dies of SIGSEGV, a part that must not fault passes only when it exits 0. + * Every part runs, so one boot names every part that fails. + */ +#include +#include + +#include "memproof.h" + +static volatile char *p; + +static void read_it(void) { + if (p[0] != 0x5a) { + _exit(2); + } +} +static void write_it(void) { + p[0] = 1; +} +static void read_below(void) { + (void)p[-1]; +} + +int main(void) { + const char *me = "protnone"; + p = mmap(0, PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + part(me, "read of an mmap PROT_NONE page", read_it, 1); + part(me, "write to an mmap PROT_NONE page", write_it, 1); + + p = mmap(0, PG, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + p[0] = 0x5a; + mprotect((void *)p, PG, PROT_NONE); + part(me, "read after mprotect RW to PROT_NONE", read_it, 1); + mprotect((void *)p, PG, PROT_READ); + part(me, "read after PROT_NONE back to R keeps the byte", read_it, 0); + part(me, "write to a PROT_READ page", write_it, 1); + + char *r = mmap(0, 3 * PG, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + mprotect(r + PG, PG, PROT_READ | PROT_WRITE); + p = (volatile char *)(r + PG); + p[0] = 0x5a; + part(me, "read of the opened page of a reservation", read_it, 0); + part(me, "read of the closed page below it", read_below, 1); + return finish(me); +} diff --git a/userland/linux_guests/c/touchfork.c b/userland/linux_guests/c/touchfork.c new file mode 100644 index 000000000..447204f89 --- /dev/null +++ b/userland/linux_guests/c/touchfork.c @@ -0,0 +1,57 @@ +/* + * Bytes a guest wrote into a reservation survive a fork. A reservation is a + * PROT_NONE mapping; a program opens part of it with mprotect, writes, and may + * close it again. Linux keeps the bytes through all of that and gives the child + * a copy. Touching a page never opened is SIGSEGV, in the child as anywhere. + * MAP_FIXED over a mapping replaces it: the new pages read zero, as Linux says. + */ +#include +#include + +#include "memproof.h" + +#define PAGES 16 + +static volatile char *r; + +static void check_bytes(void) { + for (int i = 4; i < 8; i++) { + if (r[i * PG] != (char)(0x40 + i) || r[i * PG + PG - 1] != (char)(0x50 + i)) { + _exit(2); + } + } +} +static void open_then_check(void) { + mprotect((void *)(r + 4 * PG), 4 * PG, PROT_READ); + check_bytes(); +} +static void touch_unopened(void) { + (void)r[0]; +} +static void check_zero(void) { + if (r[0] != 0) { + _exit(2); + } +} + +int main(void) { + const char *me = "touchfork"; + int anon = MAP_PRIVATE | MAP_ANONYMOUS; + r = mmap(0, PAGES * PG, PROT_NONE, anon, -1, 0); + mprotect((void *)(r + 4 * PG), 4 * PG, PROT_READ | PROT_WRITE); + for (int i = 4; i < 8; i++) { + r[i * PG] = (char)(0x40 + i); + r[i * PG + PG - 1] = (char)(0x50 + i); + } + part(me, "opened part of a reservation, child reads the bytes", check_bytes, 0); + mprotect((void *)(r + 4 * PG), 4 * PG, PROT_NONE); + part(me, "closed again, child opens it and reads the bytes", open_then_check, 0); + part(me, "child touches a page never opened", touch_unopened, 1); + + r = mmap(0, PG, PROT_READ | PROT_WRITE, anon, -1, 0); + r[0] = 0x77; + mmap((void *)r, PG, PROT_NONE, anon | MAP_FIXED, -1, 0); + mprotect((void *)r, PG, PROT_READ); + part(me, "MAP_FIXED PROT_NONE over written page, child reads zero", check_zero, 0); + return finish(me); +}