diff --git a/userland/capsule_linux/src/linux/abi/errno.rs b/userland/capsule_linux/src/linux/abi/errno.rs index 448c49fa7..188d8ca83 100644 --- a/userland/capsule_linux/src/linux/abi/errno.rs +++ b/userland/capsule_linux/src/linux/abi/errno.rs @@ -16,6 +16,8 @@ //! Linux errno values, and the convention for returning them. +pub use super::errno_sock::*; + pub const EPERM: i64 = 1; pub const ENOENT: i64 = 2; pub const EINTR: i64 = 4; diff --git a/userland/capsule_linux/src/linux/abi/errno_sock.rs b/userland/capsule_linux/src/linux/abi/errno_sock.rs new file mode 100644 index 000000000..d06e632ef --- /dev/null +++ b/userland/capsule_linux/src/linux/abi/errno_sock.rs @@ -0,0 +1,33 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Linux errno values the socket calls answer with, from +//! include/uapi/asm-generic/errno-base.h and errno.h. + +pub const EDOM: i64 = 33; +pub const EDESTADDRREQ: i64 = 89; +pub const EMSGSIZE: i64 = 90; +pub const EPROTOTYPE: i64 = 91; +pub const ENOPROTOOPT: i64 = 92; +pub const EPROTONOSUPPORT: i64 = 93; +pub const ESOCKTNOSUPPORT: i64 = 94; +pub const EOPNOTSUPP: i64 = 95; +pub const EADDRINUSE: i64 = 98; +pub const EADDRNOTAVAIL: i64 = 99; +pub const ENETUNREACH: i64 = 101; +pub const EISCONN: i64 = 106; +pub const ECONNABORTED: i64 = 103; +pub const EALREADY: i64 = 114; diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs index 68e04090a..319e970d7 100644 --- a/userland/capsule_linux/src/linux/abi/mod.rs +++ b/userland/capsule_linux/src/linux/abi/mod.rs @@ -19,8 +19,10 @@ #![allow(dead_code)] pub mod errno; +pub mod errno_sock; pub mod name; pub mod nr; -pub mod nr_path; pub mod nr_high; +pub mod nr_path; pub mod nr_sched; +pub mod nr_sock; diff --git a/userland/capsule_linux/src/linux/abi/nr.rs b/userland/capsule_linux/src/linux/abi/nr.rs index 1e800243b..b489413e7 100644 --- a/userland/capsule_linux/src/linux/abi/nr.rs +++ b/userland/capsule_linux/src/linux/abi/nr.rs @@ -14,11 +14,11 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - //! Linux x86_64 syscall numbers, by family. pub use super::nr_high::*; pub use super::nr_sched::*; +pub use super::nr_sock::*; pub const READ: u64 = 0; pub const WRITE: u64 = 1; diff --git a/userland/capsule_linux/src/linux/abi/nr_sock.rs b/userland/capsule_linux/src/linux/abi/nr_sock.rs new file mode 100644 index 000000000..3a8a13fba --- /dev/null +++ b/userland/capsule_linux/src/linux/abi/nr_sock.rs @@ -0,0 +1,28 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Linux x86_64 syscall numbers for sockets, from +//! arch/x86/entry/syscalls/syscall_64.tbl. Same contract as `nr`. + +pub const BIND: u64 = 49; +pub const LISTEN: u64 = 50; +pub const GETSOCKNAME: u64 = 51; +pub const GETPEERNAME: u64 = 52; +pub const SOCKETPAIR: u64 = 53; +pub const SETSOCKOPT: u64 = 54; +pub const GETSOCKOPT: u64 = 55; +pub const RECVMMSG: u64 = 299; +pub const SENDMMSG: u64 = 307; diff --git a/userland/capsule_linux/src/linux/call/io.rs b/userland/capsule_linux/src/linux/call/io.rs index f5145519b..07d8a2725 100644 --- a/userland/capsule_linux/src/linux/call/io.rs +++ b/userland/capsule_linux/src/linux/call/io.rs @@ -59,8 +59,6 @@ pub fn read(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { } pub fn close(guest: &mut Guest, fd: u64) -> u64 { - if let Some(h) = guest.socket_handle(fd) { - net::close(h); - } + net::close(guest, fd); file::close(guest, fd) } diff --git a/userland/capsule_linux/src/linux/guest/fork_state.rs b/userland/capsule_linux/src/linux/guest/fork_state.rs index 54b26502a..39074ba25 100644 --- a/userland/capsule_linux/src/linux/guest/fork_state.rs +++ b/userland/capsule_linux/src/linux/guest/fork_state.rs @@ -46,6 +46,7 @@ impl Guest { g.sid = self.sid; g.umask = self.umask; g.links = self.links.clone(); + self.sockets.fork(child, &self.fds); g } } diff --git a/userland/capsule_linux/src/linux/guest/handle.rs b/userland/capsule_linux/src/linux/guest/handle.rs index 1f12ca784..8aaedd7c8 100644 --- a/userland/capsule_linux/src/linux/guest/handle.rs +++ b/userland/capsule_linux/src/linux/guest/handle.rs @@ -86,4 +86,6 @@ pub struct Guest { pub blocked: Vec, /// The image's symbolic links, read once and shared by the family. pub links: alloc::rc::Rc, + /// Lets go of this process's family sockets when it is dropped (net::sock). + pub sockets: crate::linux::net::sock::Holder, } diff --git a/userland/capsule_linux/src/linux/guest/handle_new.rs b/userland/capsule_linux/src/linux/guest/handle_new.rs index 0d8b798db..103c11717 100644 --- a/userland/capsule_linux/src/linux/guest/handle_new.rs +++ b/userland/capsule_linux/src/linux/guest/handle_new.rs @@ -61,6 +61,7 @@ impl Guest { sleepers: Vec::new(), blocked: Vec::new(), links: Default::default(), + sockets: crate::linux::net::sock::Holder::new(pid), } } } diff --git a/userland/capsule_linux/src/linux/heap.rs b/userland/capsule_linux/src/linux/heap.rs index b669aa944..e61776739 100644 --- a/userland/capsule_linux/src/linux/heap.rs +++ b/userland/capsule_linux/src/linux/heap.rs @@ -21,9 +21,14 @@ use nonos_libc::{heap_init, heap_init_sized, mk_args}; /// An install holds a distribution's index while it resolves a closure. /// Kali's main is 21 MB fetched and 85 MB inflated, parsed into records -/// beside it; Alpine's is a few. A run takes the default. +/// beside it; Alpine's is a few. const INSTALL_HEAP: usize = 320 << 20; +/// A run holds the program it loads, read whole from the store, beside the +/// family's own state. A 6 MB Go program outgrew the 16 MiB default while it +/// was read; this is the most a program may be (`source::MAX_IMAGE`). +const RUN_HEAP: usize = 64 << 20; + pub fn init() { let mut buf = [0u8; 256]; let n = mk_args(buf.as_mut_ptr(), buf.len()); @@ -35,6 +40,8 @@ pub fn init() { // it is read, with that reason, instead of here without one. let line = b"[LINUX] no room for a large index, installing in the default heap\n"; let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + } else if heap_init_sized(RUN_HEAP).is_ok() { + return; } let _ = heap_init(); } diff --git a/userland/capsule_linux/src/linux/net/accept.rs b/userland/capsule_linux/src/linux/net/accept.rs new file mode 100644 index 000000000..bbf6f926c --- /dev/null +++ b/userland/capsule_linux/src/linux/net/accept.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `accept` and `accept4`: the oldest connection a listener has queued, as +//! a new descriptor held by the caller alone. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::close::discard; +use super::fd::{install, sock_of, SOCK_CLOEXEC, SOCK_NONBLOCK}; +use super::sock::{self, Domain, Peer, Proto}; + +pub fn accept4(guest: &mut Guest, fd: u64, at: u64, lenp: u64, flags: u64) -> u64 { + if flags & !(SOCK_NONBLOCK | SOCK_CLOEXEC) != 0 { + return errno::fail(errno::EINVAL); + } + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let pid = guest.pid; + let taken = sock::with(|t| { + let s = t.get_mut(id).ok_or(errno::EBADF)?; + if s.proto == Proto::Dgram { + return Err(errno::EOPNOTSUPP); + } + if !s.listening { + return Err(errno::EINVAL); + } + let child = s.pending.pop_front().ok_or(errno::EAGAIN)?; + t.make_room(id); + let c = t.get_mut(child).ok_or(errno::ECONNABORTED)?; + c.holders.push(pid); + let from = match c.domain { + Domain::Inet => Peer::Inet(c.remote.unwrap_or_default()), + Domain::Unix => Peer::Unix(c.upeer.clone()), + }; + Ok((child, from)) + }); + let (child, from) = match taken { + Ok(v) => v, + Err(e) => return errno::fail(e), + }; + let n = install(guest, child, flags); + let Some(slot) = errno::slot(n) else { + return n; + }; + let wrote = super::sockaddr_out::write(guest, at, lenp, &from); + if errno::slot(wrote).is_none() { + discard(guest, slot as u64); + return wrote; + } + n +} diff --git a/userland/capsule_linux/src/linux/net/api.rs b/userland/capsule_linux/src/linux/net/api.rs new file mode 100644 index 000000000..455637135 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/api.rs @@ -0,0 +1,42 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What the rest of the personality calls on sockets. + +pub use super::accept::accept4; +pub use super::bind::bind; +pub use super::call_kind::{flags as call_flags, wants_all}; +pub use super::close::close; +pub use super::connect::connect; +pub use super::dgram::sendto; +pub use super::fd::{is_stream, sock_id}; +pub use super::listen::listen; +pub use super::mmsg::{recvmmsg, sendmmsg}; +pub use super::msg::sendmsg; +pub use super::msg_recv::recvmsg; +pub use super::name::{getpeername, getsockname}; +pub use super::opt::{getsockopt, limit_ms, setsockopt}; +pub use super::pair::socketpair; +pub use super::poll::{ready, POLLERR, POLLHUP}; +pub use super::poll_set::poll; +pub use super::poll_socket::outside; +pub use super::recvfrom::recvfrom; +pub use super::select::{clear as select_clear, select}; +pub use super::shutdown::shutdown; +pub use super::socket::socket; +pub use super::try_call::try_call; +pub use super::xfer_in::read as recv; +pub use super::xfer_out::write as send; diff --git a/userland/capsule_linux/src/linux/net/bind.rs b/userland/capsule_linux/src/linux/net/bind.rs new file mode 100644 index 000000000..b8e002b26 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/bind.rs @@ -0,0 +1,65 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `bind` and `listen`, on 127.0.0.0/8 only (`policy`). + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::policy::not_loopback; +use super::sock::{self, Domain}; +use super::sockaddr::{self, is_loopback, AF_INET}; + +pub fn bind(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + if sock::with(|t| t.get(id).is_some_and(|s| s.domain == Domain::Unix)) { + return super::named::bind(guest, id, at, len); + } + let (family, mut want) = match sockaddr::read(guest, at, len) { + Ok(v) => v, + Err(e) => return e, + }; + if family != AF_INET { + return errno::fail(errno::EAFNOSUPPORT); + } + if !is_loopback(want.ip) { + return not_loopback("bind", want); + } + sock::with(|t| { + let Some(s) = t.get(id) else { + return errno::fail(errno::EBADF); + }; + if s.domain != Domain::Inet || s.local.is_some() || s.svc.is_some() { + return errno::fail(errno::EINVAL); + } + if want.port == 0 { + match t.ephemeral(s.proto, want.ip) { + Some(port) => want.port = port, + None => return errno::fail(errno::EADDRINUSE), + } + } else if t.in_use(id, want) { + return errno::fail(errno::EADDRINUSE); + } + if let Some(s) = t.get_mut(id) { + s.local = Some(want); + } + errno::ok(0) + }) +} diff --git a/userland/capsule_linux/src/linux/net/call_kind.rs b/userland/capsule_linux/src/linux/net/call_kind.rs new file mode 100644 index 000000000..6a28e0a56 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/call_kind.rs @@ -0,0 +1,61 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What kind of wait a socket call makes: its flags, and whether a +//! blocking one waits until it has moved everything. + +use crate::linux::abi::{errno, nr}; +use crate::linux::guest::Guest; + +use super::flags::{MSG_DONTWAIT, MSG_WAITALL}; +use super::{iov, mmsg}; + +/// The call's flags, where it has them. +pub fn flags(n: u64, a: [u64; 6]) -> u64 { + match n { + nr::RECVFROM | nr::SENDTO | nr::RECVMMSG | nr::SENDMMSG | nr::ACCEPT4 => a[3], + nr::RECVMSG | nr::SENDMSG => a[2], + _ => 0, + } +} + +/// True when a blocking call waits until it has moved everything it asked +/// for: a send on a stream, a receive with MSG_WAITALL, and recvmmsg +/// without MSG_WAITFORONE. +pub fn wants_all(stream: bool, n: u64, flags: u64) -> bool { + match n { + nr::WRITE | nr::WRITEV | nr::SENDTO | nr::SENDMSG => stream, + nr::RECVFROM | nr::RECVMSG => stream && flags & MSG_WAITALL != 0, + nr::RECVMMSG => flags & (mmsg::MSG_WAITFORONE | MSG_DONTWAIT) == 0, + _ => false, + } +} + +/// A receive's answer: the count, or the errno. +pub(super) fn bytes_in(guest: &Guest, id: u32, v: &iov::Iov, done: usize, flags: u64) -> u64 { + match super::xfer_in::recv(guest, id, v, done, flags) { + Ok(got) => errno::ok(got.n as u64), + Err(e) => e, + } +} + +/// The bytes a msghdr's iovecs ask for. +pub(super) fn msg_len(guest: &Guest, msg: u64) -> usize { + let word = |at: u64| { + guest.read(at, 8).map_or(0, |b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8]))) + }; + iov::read(guest, word(msg + 16), word(msg + 24)).map_or(0, |v| iov::total(&v)) +} diff --git a/userland/capsule_linux/src/linux/net/cap.rs b/userland/capsule_linux/src/linux/net/cap.rs new file mode 100644 index 000000000..e51aa5d03 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/cap.rs @@ -0,0 +1,33 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! How many bytes one send takes from the guest. + +/// What one send takes from the guest at most: the default receive buffer +/// of a stream's peer, so a single call can fill it. +const STREAM_CAP: usize = 128 << 10; +/// One byte past the largest datagram, so a larger one is seen and refused. +const GRAM_CAP: usize = 65508; + +/// The most one send gathers: what net.sockets carries in a call, a +/// stream peer's default queue, or one byte past the largest datagram. +pub fn cap(outside: bool, stream: bool) -> usize { + match (outside, stream) { + (true, _) => super::stream::MAX_IO, + (false, true) => STREAM_CAP, + (false, false) => GRAM_CAP, + } +} diff --git a/userland/capsule_linux/src/linux/net/close.rs b/userland/capsule_linux/src/linux/net/close.rs new file mode 100644 index 000000000..0dbf62ee9 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/close.rs @@ -0,0 +1,45 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Closing a socket descriptor. The socket goes with the last descriptor +//! naming it in the last process holding it. + +use crate::linux::guest::{Guest, Kind}; + +use super::fd::sock_of; +use super::sock; + +/// Called by close before it clears `fd`: this process lets go of the +/// socket unless another of its descriptors still names it. +pub fn close(guest: &mut Guest, fd: u64) { + let Ok(id) = sock_of(guest, fd) else { + return; + }; + let named_again = guest + .fds + .iter() + .enumerate() + .any(|(i, f)| i as u64 != fd && f.kind == Kind::Socket && f.handle == id); + if !named_again { + sock::with(|t| t.release(id, guest.pid)); + } +} + +/// Close a descriptor this module opened and cannot hand out after all. +pub(super) fn discard(guest: &mut Guest, fd: u64) { + close(guest, fd); + let _ = crate::linux::file::close(guest, fd); +} diff --git a/userland/capsule_linux/src/linux/net/connect.rs b/userland/capsule_linux/src/linux/net/connect.rs index 51f727956..8d630bc56 100644 --- a/userland/capsule_linux/src/linux/net/connect.rs +++ b/userland/capsule_linux/src/linux/net/connect.rs @@ -14,54 +14,38 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `connect`, for a socket the guest opened here. - -use alloc::vec::Vec; +//! `connect`. On 127.0.0.0/8 the family links the two ends in the caller's +//! own call, as Linux's loopback does, and a non-blocking socket answers +//! EINPROGRESS all the same; anywhere else a stream goes over the mixnet. use crate::linux::abi::errno; use crate::linux::guest::{Guest, Kind}; -use super::addr::inet; -use super::call::call; -use super::dns::host_for; -use super::ops::{OP_CONNECT, OP_CONNECT_HOST}; +use super::fd::{nonblock, sock_of}; +use super::sock::{self, Domain, Proto}; +use super::sockaddr::{self, is_loopback, AF_INET}; pub fn connect(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { - /* - * A program may connect its nameserver socket before writing to - * it. There is nothing to reach: this capsule is the nameserver. - */ + /* This capsule is the nameserver, so its socket has nothing to reach. */ if guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Resolver) { return errno::ok(0); } - let Some(handle) = guest.socket_handle(fd) else { - return errno::fail(errno::ENOTSOCK); + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, }; - let Some((port, ip)) = inet(guest, at, len) else { - return errno::fail(errno::EAFNOSUPPORT); + let (family, to) = match sockaddr::read(guest, at, len) { + Ok(v) => v, + Err(e) => return e, }; - // An address this capsule invented for a name goes back to being the name. - if let Some(host) = host_for(guest, ip) { - return by_host(handle, &host, port); - } - let mut body = Vec::with_capacity(10); - body.extend_from_slice(&handle.to_le_bytes()); - body.extend_from_slice(&ip); - body.extend_from_slice(&port.to_le_bytes()); - match call(OP_CONNECT, &body, 0) { - Some((0, _)) => errno::ok(0), - Some(_) => errno::fail(errno::ECONNREFUSED), - None => errno::fail(errno::EIO), - } -} - -fn by_host(handle: u32, host: &[u8], port: u16) -> u64 { - let Some(body) = super::host_body::host_body(handle, port, host) else { - return errno::fail(errno::EINVAL); + let Some((proto, domain)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.domain))) else { + return errno::fail(errno::EBADF); }; - match call(OP_CONNECT_HOST, &body, 0) { - Some((0, _)) => errno::ok(0), - Some(_) => errno::fail(errno::ECONNREFUSED), - None => errno::fail(errno::EIO), + match (proto, domain) { + (_, Domain::Unix) => super::named::connect(guest, fd, id, proto, at, len), + (Proto::Dgram, _) => super::connect_dgram::connect(guest, fd, id, family, to), + _ if family != AF_INET => errno::fail(errno::EAFNOSUPPORT), + _ if is_loopback(to.ip) => super::connect_lo::loopback(id, to, nonblock(guest, fd)), + _ => super::connect_out::connect(guest, id, to), } } diff --git a/userland/capsule_linux/src/linux/net/connect_dgram.rs b/userland/capsule_linux/src/linux/net/connect_dgram.rs new file mode 100644 index 000000000..ee74eb7f9 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/connect_dgram.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `connect` on a datagram socket: it only names where sends go and whose +//! datagrams are kept. AF_UNSPEC forgets it again. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::policy::refuse_out; +use super::sock::{self, Addr}; +use super::sockaddr::{is_loopback, AF_INET, AF_UNSPEC}; + +pub fn connect(guest: &mut Guest, fd: u64, id: u32, family: u16, to: Addr) -> u64 { + if family == AF_UNSPEC { + sock::with(|t| t.get_mut(id).map(|s| s.remote = None)); + return errno::ok(0); + } + if family != AF_INET { + return errno::fail(errno::EAFNOSUPPORT); + } + if super::resolver::is_nameserver(to) { + super::resolver::become_resolver(guest, fd); + return errno::ok(0); + } + if !is_loopback(to.ip) { + return refuse_out("connect", to); + } + sock::with(|t| { + t.autobind(id)?; + if let Some(s) = t.get_mut(id) { + s.remote = Some(to); + s.error = 0; + } + Ok(()) + }) + .map_or_else(errno::fail, |()| errno::ok(0)) +} diff --git a/userland/capsule_linux/src/linux/net/connect_dial.rs b/userland/capsule_linux/src/linux/net/connect_dial.rs new file mode 100644 index 000000000..6febedff5 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/connect_dial.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The two calls a stream outside the family makes to net.sockets: a +//! mixnet socket, and a connect to an address or to the name this capsule +//! invented it for. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::call::call; +use super::dns::host_for; +use super::ops::{DOMAIN, KIND_MIXNET, OP_CONNECT, OP_CONNECT_HOST, OP_SOCKET}; +use super::sock::Addr; + +pub fn open() -> Result { + let mut body = Vec::with_capacity(4); + body.extend_from_slice(&DOMAIN.to_le_bytes()); + body.extend_from_slice(&KIND_MIXNET.to_le_bytes()); + match call(OP_SOCKET, &body, 8) { + Some((0, out)) if out.len() >= 4 => { + Ok(u32::from_le_bytes([out[0], out[1], out[2], out[3]])) + } + Some(_) => Err(errno::fail(errno::ENOMEM)), + None => Err(errno::fail(errno::EIO)), + } +} + +/// The service's answer to a connect; None when it did not answer. +pub fn dial(guest: &Guest, handle: u32, to: Addr) -> Result)>, u64> { + /* An address this capsule invented for a name goes back to being the name. */ + if let Some(host) = host_for(guest, to.ip) { + let body = super::host_body::host_body(handle, to.port, &host) + .ok_or(errno::fail(errno::EINVAL))?; + return Ok(call(OP_CONNECT_HOST, &body, 0)); + } + let mut body = Vec::with_capacity(10); + body.extend_from_slice(&handle.to_le_bytes()); + body.extend_from_slice(&to.ip); + body.extend_from_slice(&to.port.to_le_bytes()); + Ok(call(OP_CONNECT, &body, 0)) +} diff --git a/userland/capsule_linux/src/linux/net/connect_lo.rs b/userland/capsule_linux/src/linux/net/connect_lo.rs new file mode 100644 index 000000000..53b1a0706 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/connect_lo.rs @@ -0,0 +1,69 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A stream connect to 127.0.0.0/8, which the family completes in the +//! caller's own call, as Linux's loopback does. + +use crate::linux::abi::errno; + +use super::sock::{self, Addr, Link}; + +pub fn loopback(id: u32, to: Addr, nonblock: bool) -> u64 { + sock::with(|t| { + let Some(s) = t.get_mut(id) else { + return errno::fail(errno::EBADF); + }; + if s.connecting { + return errno::fail(errno::EALREADY); + } + if s.connected || s.listening || s.svc.is_some() { + return errno::fail(errno::EISCONN); + } + s.error = 0; + let bound_here = s.local.is_none(); + if let Err(e) = t.autobind(id) { + return errno::fail(e); + } + let linked = t.link(id, to); + /* + * A full listener keeps a non-blocking connect until accept makes + * room, as Linux's SYN_SENT does. + */ + let from = t.get(id).and_then(|s| s.local).map_or(0, |a| a.port); + if let (Link::Full, true, Some(l)) = (&linked, nonblock, t.listener(to, from)) { + t.wait_room(id, l); + return errno::fail(errno::EINPROGRESS); + } + /* A connect that fails gives back the port it bound. */ + if !matches!(linked, Link::Done) && bound_here { + if let Some(s) = t.get_mut(id) { + s.local = None; + } + } + match linked { + Link::Done if nonblock => errno::fail(errno::EINPROGRESS), + Link::Done => errno::ok(0), + Link::Refused if nonblock => { + if let Some(s) = t.get_mut(id) { + s.error = errno::ECONNREFUSED; + } + errno::fail(errno::EINPROGRESS) + } + Link::Refused => errno::fail(errno::ECONNREFUSED), + Link::Full => errno::fail(errno::EAGAIN), + } + }) +} diff --git a/userland/capsule_linux/src/linux/net/connect_out.rs b/userland/capsule_linux/src/linux/net/connect_out.rs new file mode 100644 index 000000000..d6a73f635 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/connect_out.rs @@ -0,0 +1,60 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A stream to an address outside the family. It goes over the mixnet, +//! never the open network, and the guest holds no capability that could +//! name a socket: there is no second route to disable and no firewall rule +//! to remove. net.sockets holds the stream; the family's entry names it. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::connect_dial::{dial, open}; +use super::sock::{self, Addr}; + +pub fn connect(guest: &Guest, id: u32, to: Addr) -> u64 { + if sock::with(|t| t.get(id).is_some_and(|s| s.connected || s.listening || s.svc.is_some())) { + return errno::fail(errno::EISCONN); + } + let handle = match open() { + Ok(h) => h, + Err(e) => return e, + }; + let status = match dial(guest, handle, to) { + Ok(s) => s, + Err(e) => { + super::stream::close(handle); + return e; + } + }; + let answer = match status { + Some((0, _)) => errno::ok(0), + Some(_) => errno::fail(errno::ECONNREFUSED), + None => errno::fail(errno::EIO), + }; + if answer != 0 { + super::stream::close(handle); + return answer; + } + sock::with(|t| { + if let Some(s) = t.get_mut(id) { + s.svc = Some(handle); + s.remote = Some(to); + s.connected = true; + } + }); + answer +} diff --git a/userland/capsule_linux/src/linux/net/dest.rs b/userland/capsule_linux/src/linux/net/dest.rs new file mode 100644 index 000000000..6e5bfd9f0 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/dest.rs @@ -0,0 +1,43 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a send goes: nowhere but the peer for a stream, and for a datagram +//! the address or the Unix name it names, found now. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::peer_addr::To; +use super::sock::{Dest, Domain, Proto}; + +pub fn dest(guest: &Guest, proto: Proto, domain: Domain, to: Option) -> Result { + Ok(match (proto, to) { + (Proto::Stream, _) | (_, None) => Dest::Default, + (_, Some(To::Inet(a))) => Dest::Inet(a), + (_, Some(To::Unix(_))) if domain == Domain::Inet => { + return Err(errno::fail(errno::EAFNOSUPPORT)) + } + (_, Some(To::Unix(ua))) => { + let found = super::named::resolve(guest, &ua) + .ok_or(errno::EINVAL) + .and_then(|name| super::named::find(&name, Proto::Dgram)); + match found { + Ok(t) => Dest::Sock(t), + Err(e) => return Err(errno::fail(e)), + } + } + }) +} diff --git a/userland/capsule_linux/src/linux/net/dgram.rs b/userland/capsule_linux/src/linux/net/dgram.rs index dd3b31bd5..ab32f9bc2 100644 --- a/userland/capsule_linux/src/linux/net/dgram.rs +++ b/userland/capsule_linux/src/linux/net/dgram.rs @@ -14,39 +14,55 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `sendto` and `recvfrom`, which differ from write and read only in carrying -//! an address. +//! `sendto`, which differs from write only in carrying an address. -use crate::linux::call; -use crate::linux::guest::{Guest, Kind}; +use alloc::vec; -use super::addr::inet; -use super::dgram_addr::{encode, fill}; -use super::dns; +use crate::linux::guest::{Guest, Kind}; -pub fn sendto(guest: &mut Guest, fd: u64, buf: u64, len: u64, at: u64, alen: u64) -> u64 { - if !is_resolver(guest, fd) { - return call::write(guest, fd, buf, len); - } - /* - * A program with no `resolv.conf` asks the loopback address, and one with - * a configured nameserver asks that. - */ - let peer = inet(guest, at, alen).unwrap_or((53, [127, 0, 0, 1])); - dns::query(guest, fd, buf, len, encode(peer)) -} +use super::fd::sock_of; +use super::peer_addr::To; +use super::sock::{self, Domain, Proto}; +use super::sockaddr::is_loopback; -pub fn recvfrom(guest: &mut Guest, fd: u64, buf: u64, len: u64, at: u64, alen: u64) -> u64 { +pub fn sendto( + guest: &mut Guest, + fd: u64, + buf: u64, + len: u64, + flags: u64, + at: u64, + alen: u64, +) -> u64 { + let to = match super::peer_addr::address(guest, at, alen) { + Ok(to) => to, + Err(e) => return e, + }; if !is_resolver(guest, fd) { - return call::read(guest, fd, buf, len); - } - let (got, from) = dns::answer_out(guest, fd, buf, len); - match from { - Some(peer) if at != 0 => fill(guest, at, alen, peer, got), - _ => got, + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let inet_dgram = sock::with(|t| { + t.get(id).is_some_and(|s| s.proto == Proto::Dgram && s.domain == Domain::Inet) + }); + match to { + Some(To::Inet(a)) if inet_dgram && super::resolver::is_nameserver(a) => { + super::resolver::become_resolver(guest, fd) + } + Some(To::Inet(a)) if inet_dgram && !is_loopback(a.ip) => { + return super::policy::refuse_out("sendto", a) + } + to => return super::xfer_out::send(guest, id, &vec![(buf, len)], 0, flags, to), + } } + let to = match to { + Some(To::Inet(a)) => Some(a), + _ => None, + }; + super::resolver::query(guest, fd, buf, len, to) } -fn is_resolver(guest: &Guest, fd: u64) -> bool { +pub(super) fn is_resolver(guest: &Guest, fd: u64) -> bool { guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Resolver) } diff --git a/userland/capsule_linux/src/linux/net/dns/mod.rs b/userland/capsule_linux/src/linux/net/dns/mod.rs index 23b9adad3..c530ba42e 100644 --- a/userland/capsule_linux/src/linux/net/dns/mod.rs +++ b/userland/capsule_linux/src/linux/net/dns/mod.rs @@ -19,10 +19,8 @@ mod automap; mod decide; mod name; -mod open; mod reply; mod serve; pub use automap::host_for; -pub use open::open; pub use serve::{answer_out, query}; diff --git a/userland/capsule_linux/src/linux/net/fd.rs b/userland/capsule_linux/src/linux/net/fd.rs new file mode 100644 index 000000000..1da8a45e2 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/fd.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! From a descriptor to the family socket it names, and back. + +use crate::linux::abi::errno; +use crate::linux::guest::{Fd, Guest, Kind}; + +use super::sock; + +/// SOCK_NONBLOCK and SOCK_CLOEXEC, which socket, socketpair and accept4 +/// take with the type or as flags: O_NONBLOCK and O_CLOEXEC's values. +pub const SOCK_NONBLOCK: u64 = 0o4000; +pub const SOCK_CLOEXEC: u64 = 0o2_000_000; + +/// The socket `fd` names, or the errno Linux gives for a descriptor that is +/// not one: EBADF when nothing is open there, ENOTSOCK when a file is. +pub fn sock_of(guest: &Guest, fd: u64) -> Result { + match guest.fds.get(fd as usize) { + Some(f) if f.kind == Kind::Socket => Ok(f.handle), + Some(f) if f.is_open() => Err(errno::fail(errno::ENOTSOCK)), + _ => Err(errno::fail(errno::EBADF)), + } +} + +/// A descriptor for socket `id`, with the flags asked for. A socket with no +/// descriptor to name it is let go at once, since nobody could close it. +pub fn install(guest: &mut Guest, id: u32, flags: u64) -> u64 { + match crate::linux::file::install(guest, Fd::socket(id)) { + Some(n) => { + if let Some(f) = guest.fds.get_mut(n as usize) { + f.nonblock = flags & SOCK_NONBLOCK != 0; + f.cloexec = flags & SOCK_CLOEXEC != 0; + } + errno::ok(n) + } + None => { + sock::with(|t| t.release(id, guest.pid)); + errno::fail(errno::EMFILE) + } + } +} + +/// The socket `fd` names, if it names one. +pub fn sock_id(guest: &Guest, fd: u64) -> Option { + sock_of(guest, fd).ok() +} + +/// True for a stream socket. +pub fn is_stream(id: u32) -> bool { + sock::with(|t| t.get(id).is_some_and(|s| s.proto == sock::Proto::Stream)) +} + +/// True when `fd` is non-blocking. +pub fn nonblock(guest: &Guest, fd: u64) -> bool { + guest.fds.get(fd as usize).is_some_and(|f| f.nonblock) +} diff --git a/userland/capsule_linux/src/linux/net/flags.rs b/userland/capsule_linux/src/linux/net/flags.rs new file mode 100644 index 000000000..c077e41e7 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/flags.rs @@ -0,0 +1,23 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The flags the send and receive calls take, from include/linux/socket.h. + +pub const MSG_OOB: u64 = 0x1; +pub const MSG_PEEK: u64 = 0x2; +pub const MSG_TRUNC: u64 = 0x20; +pub const MSG_DONTWAIT: u64 = 0x40; +pub const MSG_WAITALL: u64 = 0x100; diff --git a/userland/capsule_linux/src/linux/net/iov.rs b/userland/capsule_linux/src/linux/net/iov.rs new file mode 100644 index 000000000..922dff54b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/iov.rs @@ -0,0 +1,75 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The byte vectors the message calls name: one buffer, or an iovec array. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +/// Linux's UIO_MAXIOV. +const IOV_MAX: u64 = 1024; +const IOVEC: usize = 16; + +pub type Iov = Vec<(u64, u64)>; + +/// The `count` entries of the iovec array at `at`. +pub fn read(guest: &Guest, at: u64, count: u64) -> Result { + if count > IOV_MAX { + return Err(errno::fail(errno::EINVAL)); + } + let raw = guest.read(at, count as usize * IOVEC).ok_or(errno::fail(errno::EFAULT))?; + let word = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + Ok((0..count as usize).map(|i| (word(i * IOVEC), word(i * IOVEC + 8))).collect()) +} + +pub fn total(iov: &Iov) -> usize { + iov.iter().map(|&(_, len)| len as usize).sum() +} + +/// The message's bytes from `skip` on, at most `cap` of them. +pub fn gather(guest: &Guest, iov: &Iov, skip: usize, cap: usize) -> Result, u64> { + let mut out = Vec::new(); + let mut pos = 0usize; + for &(base, len) in iov { + let len = len as usize; + let (from, to) = (skip.max(pos), (pos + len).min(skip + cap)); + if from < to { + let part = guest.read(base + (from - pos) as u64, to - from); + out.extend_from_slice(&part.ok_or(errno::fail(errno::EFAULT))?); + } + pos += len; + } + Ok(out) +} + +/// Put `bytes` into the message's buffers starting `skip` bytes in. +pub fn scatter(guest: &Guest, iov: &Iov, skip: usize, bytes: &[u8]) -> Result<(), u64> { + let mut pos = 0usize; + for &(base, len) in iov { + let len = len as usize; + let (from, to) = (skip.max(pos), (pos + len).min(skip + bytes.len())); + if from < to { + let part = &bytes[from - skip..to - skip]; + if guest.write(base + (from - pos) as u64, part) < part.len() as i64 { + return Err(errno::fail(errno::EFAULT)); + } + } + pos += len; + } + Ok(()) +} diff --git a/userland/capsule_linux/src/linux/net/listen.rs b/userland/capsule_linux/src/linux/net/listen.rs new file mode 100644 index 000000000..a22e3ce57 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/listen.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `listen`, on a socket bound to 127.0.0.1 (`policy`). + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::policy::not_loopback; +use super::sock::{self, Addr, Domain, Proto}; + +pub fn listen(guest: &mut Guest, fd: u64, backlog: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + sock::with(|t| { + let Some(s) = t.get(id) else { + return errno::fail(errno::EBADF); + }; + match (s.proto, s.domain, s.local) { + (Proto::Dgram, ..) => return errno::fail(errno::EOPNOTSUPP), + _ if s.connected || s.svc.is_some() => return errno::fail(errno::EINVAL), + /* A Unix socket must be bound first: Linux does not name it here. */ + (_, Domain::Unix, _) if s.uname.is_none() => return errno::fail(errno::EINVAL), + (_, Domain::Unix, _) => {} + /* Linux would bind 0.0.0.0 here, which is not the family's own. */ + (_, _, None) => return not_loopback("listen", Addr::default()), + /* Listeners share an address only when each set SO_REUSEPORT. */ + (_, _, Some(at)) + if !s.listening + && t.iter().any(|(_, o)| { + o.listening + && o.local == Some(at) + && !(o.opts.reuseport && s.opts.reuseport) + }) => + { + return errno::fail(errno::EADDRINUSE) + } + _ => {} + } + if let Some(s) = t.get_mut(id) { + s.listening = true; + /* An int, and somaxconn's 4096 is the most Linux keeps. */ + s.backlog = (backlog as i32).clamp(0, 4096) as usize; + } + errno::ok(0) + }) +} diff --git a/userland/capsule_linux/src/linux/net/mmsg.rs b/userland/capsule_linux/src/linux/net/mmsg.rs new file mode 100644 index 000000000..174adaa3c --- /dev/null +++ b/userland/capsule_linux/src/linux/net/mmsg.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sendmmsg` and `recvmmsg`: several messages in one call. A blocking +//! recvmmsg waits until all `vlen` have come, unless MSG_WAITFORONE; the +//! wait keeps its count between tries (`waits_sock`), so each try starts at +//! message `skip` and answers how many it moved. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::flags::MSG_DONTWAIT; +use super::mmsg_each::each; +use super::policy::refuse; + +/// Linux's UIO_MAXIOV caps vlen. +pub const MOST: u64 = 1024; +pub const MSG_WAITFORONE: u64 = 0x10000; + +/// `a` is sendmmsg's arguments: fd, vector, vlen, flags. +pub fn sendmmsg(guest: &mut Guest, a: [u64; 6], skip: usize) -> u64 { + each(guest, a, skip, |g, at, first| { + let f = if first { a[3] } else { a[3] | MSG_DONTWAIT }; + super::msg::sendmsg(g, a[0], at, f, 0) + }) +} + +/// `a` is recvmmsg's arguments: fd, vector, vlen, flags, timeout. +pub fn recvmmsg(guest: &mut Guest, a: [u64; 6], skip: usize) -> u64 { + if a[4] != 0 { + return refuse("recvmmsg timeout: SO_RCVTIMEO bounds the wait instead", errno::EINVAL); + } + let flags = a[3] & !MSG_WAITFORONE; + each(guest, a, skip, |g, at, first| { + let f = if first { flags } else { flags | MSG_DONTWAIT }; + super::msg_recv::recvmsg(g, a[0], at, f, 0) + }) +} diff --git a/userland/capsule_linux/src/linux/net/mmsg_each.rs b/userland/capsule_linux/src/linux/net/mmsg_each.rs new file mode 100644 index 000000000..a047e1447 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/mmsg_each.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The messages of one sendmmsg or recvmmsg, one after another. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::mmsg::MOST; + +/// struct mmsghdr on x86_64: a msghdr, then msg_len. +const MMSGHDR: u64 = 64; +const LEN_AT: u64 = 56; + +/// Run `one` on each message from `skip` until one fails: the count moved, +/// or the first failure's errno when none was. +pub fn each( + guest: &mut Guest, + a: [u64; 6], + skip: usize, + mut one: impl FnMut(&mut Guest, u64, bool) -> u64, +) -> u64 { + let vlen = a[2].min(MOST); + let mut moved = 0u64; + for i in skip as u64..vlen { + let at = a[1] + i * MMSGHDR; + let got = one(guest, at, moved == 0); + let Some(n) = errno::slot(got) else { + return if moved == 0 { got } else { errno::ok(moved) }; + }; + if guest.write(at + LEN_AT, &(n as u32).to_le_bytes()) < 4 { + return if moved == 0 { errno::fail(errno::EFAULT) } else { errno::ok(moved) }; + } + moved += 1; + } + errno::ok(moved) +} diff --git a/userland/capsule_linux/src/linux/net/mod.rs b/userland/capsule_linux/src/linux/net/mod.rs index c0155af09..78c0cc53b 100644 --- a/userland/capsule_linux/src/linux/net/mod.rs +++ b/userland/capsule_linux/src/linux/net/mod.rs @@ -14,30 +14,59 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Sockets, over the net.sockets service. +//! Sockets: the family's own, kept in `sock`, and a stream outside the +//! family, which net.sockets carries over the mixnet. -mod addr; +mod accept; +mod api; +mod bind; mod call; +mod call_kind; +mod cap; +mod close; mod connect; +mod connect_dgram; +mod connect_dial; +mod connect_lo; +mod connect_out; +mod dest; mod dgram; mod dgram_addr; -mod host_body; pub mod dns; +mod fd; +mod flags; +mod host_body; +mod iov; +mod listen; +mod mmsg; +mod mmsg_each; +mod msg; +mod msg_hdr; +mod msg_recv; +mod name; +mod named; mod ops; +mod opt; +mod pair; +mod peer_addr; +mod policy; mod poll; mod poll_set; mod poll_socket; pub mod raw; pub mod raw_io; +mod recvfrom; +mod resolver; pub mod route; mod select; +mod shutdown; +pub mod sock; +mod sockaddr; +mod sockaddr_out; mod socket; mod stream; +mod try_call; +mod xfer_in; +mod xfer_out; -pub use connect::connect; -pub use dgram::{recvfrom, sendto}; -pub use poll::{ready, POLLERR, POLLHUP}; -pub use poll_set::poll; -pub use select::{clear as select_clear, select}; -pub use socket::socket; -pub use stream::{close, recv, send}; +pub use api::*; diff --git a/userland/capsule_linux/src/linux/net/msg.rs b/userland/capsule_linux/src/linux/net/msg.rs new file mode 100644 index 000000000..091482db4 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/msg.rs @@ -0,0 +1,52 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sendmsg` on a family socket: an iovec and an address. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::msg_hdr::hdr; +use super::policy::refuse; + +/// `skip` bytes of the message went in an earlier try of this same call. +pub fn sendmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let h = match hdr(guest, msg) { + Ok(h) => h, + Err(e) => return e, + }; + if h.controllen != 0 { + return refuse( + "sendmsg control data on a family socket: SCM_RIGHTS is not carried yet", + errno::EINVAL, + ); + } + let to = match super::peer_addr::address(guest, h.name, h.namelen) { + Ok(to) => to, + Err(e) => return e, + }; + if let Some(super::peer_addr::To::Inet(a)) = &to { + if !super::sockaddr::is_loopback(a.ip) { + return super::policy::refuse_out("sendmsg", *a); + } + } + super::xfer_out::send(guest, id, &h.iov, skip, flags, to) +} diff --git a/userland/capsule_linux/src/linux/net/msg_hdr.rs b/userland/capsule_linux/src/linux/net/msg_hdr.rs new file mode 100644 index 000000000..e3f60a9c3 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/msg_hdr.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! struct msghdr, as sendmsg and recvmsg find it in guest memory. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::iov; + +/// struct msghdr on x86_64. +const MSGHDR: usize = 56; +pub const NAMELEN_AT: u64 = 8; +pub const CONTROLLEN_AT: u64 = 40; +pub const FLAGS_AT: u64 = 48; + +pub struct Hdr { + pub name: u64, + pub namelen: u64, + pub iov: iov::Iov, + pub controllen: u64, +} + +pub fn hdr(guest: &Guest, msg: u64) -> Result { + let raw = guest.read(msg, MSGHDR).ok_or(errno::fail(errno::EFAULT))?; + let word = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + Ok(Hdr { + name: word(0), + namelen: u64::from(u32::from_le_bytes([raw[8], raw[9], raw[10], raw[11]])), + iov: iov::read(guest, word(16), word(24))?, + controllen: word(40), + }) +} diff --git a/userland/capsule_linux/src/linux/net/msg_recv.rs b/userland/capsule_linux/src/linux/net/msg_recv.rs new file mode 100644 index 000000000..5efafbc97 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/msg_recv.rs @@ -0,0 +1,52 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `recvmsg` on a family socket: bytes into an iovec, and the sender's +//! address. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::flags::MSG_TRUNC; +use super::msg_hdr::{hdr, CONTROLLEN_AT, FLAGS_AT, NAMELEN_AT}; + +pub fn recvmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let h = match hdr(guest, msg) { + Ok(h) => h, + Err(e) => return e, + }; + let got = match super::xfer_in::recv(guest, id, &h.iov, skip, flags) { + Ok(got) => got, + Err(e) => return e, + }; + let cut = if got.whole > got.n { MSG_TRUNC as u32 } else { 0 }; + let (name, lenp) = if h.name != 0 { (h.name, msg + NAMELEN_AT) } else { (0, 0) }; + let value = super::peer_addr::finish(guest, got, flags, name, lenp); + if errno::slot(value).is_none() { + return value; + } + if guest.write(msg + CONTROLLEN_AT, &0u64.to_le_bytes()) < 8 + || guest.write(msg + FLAGS_AT, &cut.to_le_bytes()) < 4 + { + return errno::fail(errno::EFAULT); + } + value +} diff --git a/userland/capsule_linux/src/linux/net/name.rs b/userland/capsule_linux/src/linux/net/name.rs new file mode 100644 index 000000000..962cfaec1 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/name.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `getsockname` and `getpeername`. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::sock::{self, Domain, Peer}; + +pub fn getsockname(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + /* A socket not yet bound is 0.0.0.0 port 0, or an unnamed Unix socket. */ + let Some(me) = sock::with(|t| { + t.get(id).map(|s| match s.domain { + Domain::Inet => Peer::Inet(s.local.unwrap_or_default()), + Domain::Unix => Peer::Unix(s.uname.clone()), + }) + }) else { + return errno::fail(errno::EBADF); + }; + super::sockaddr_out::write(guest, at, lenp, &me) +} + +pub fn getpeername(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + /* + * A reset connection is closed, and has no peer; one whose peer only + * shut down, or left cleanly, still does. + */ + let peer: Option = sock::with(|t| { + let s = t.get(id)?; + let live = s.connected && !s.broken && s.error == 0; + live.then(|| match s.domain { + Domain::Inet => Peer::Inet(s.remote.unwrap_or_default()), + Domain::Unix => Peer::Unix(s.upeer.clone()), + }) + }); + match peer { + Some(p) => super::sockaddr_out::write(guest, at, lenp, &p), + None => errno::fail(errno::ENOTCONN), + } +} diff --git a/userland/capsule_linux/src/linux/net/named/addr.rs b/userland/capsule_linux/src/linux/net/named/addr.rs new file mode 100644 index 000000000..28fef072d --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/addr.rs @@ -0,0 +1,58 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sockaddr_un` as a guest names it: a path, an abstract name, or only the +//! family, which asks bind for a name of Linux's choosing. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; +use crate::linux::net::sockaddr::{self, AF_UNIX}; + +/// sun_family and the 108 bytes of sun_path. +const SOCKADDR_UN: u64 = 110; + +pub enum UAddr { + Auto, + Path(Vec), + Abstract(Vec), +} + +pub fn read(guest: &Guest, at: u64, len: u64) -> Result { + if !(2..=SOCKADDR_UN).contains(&len) { + return Err(errno::fail(errno::EINVAL)); + } + let raw = guest.read(at, len as usize).ok_or(errno::fail(errno::EFAULT))?; + let path = &raw[2..]; + Ok(match path.first() { + None => UAddr::Auto, + Some(0) => UAddr::Abstract(path[1..].to_vec()), + /* A path ends at its first NUL, however long the guest said it was. */ + Some(_) => { + let end = path.iter().position(|&b| b == 0).unwrap_or(path.len()); + UAddr::Path(path[..end].to_vec()) + } + }) +} + +/// The name a bind or connect gives: EINVAL for another family. +pub fn unix_addr(guest: &Guest, at: u64, len: u64) -> Result { + match sockaddr::read(guest, at, len)? { + (AF_UNIX, _) => read(guest, at, len), + _ => Err(errno::fail(errno::EINVAL)), + } +} diff --git a/userland/capsule_linux/src/linux/net/named/auto.rs b/userland/capsule_linux/src/linux/net/named/auto.rs new file mode 100644 index 000000000..6faec8b6c --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/auto.rs @@ -0,0 +1,32 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The name bind chooses when a Unix socket is given its family alone. + +use alloc::vec::Vec; + +use crate::linux::net::sock::{self, UName}; + +/// Linux's autobind: a NUL and five hex digits, the first unused. +pub fn fresh() -> Option { + (0u32..0x10_0000).find_map(|n| { + let mut key: Vec = alloc::vec![0]; + key.extend_from_slice(alloc::format!("{n:05x}").as_bytes()); + let name = UName { key: key.clone(), shown: key }; + let used = sock::with(|t| t.iter().any(|(_, s)| s.uname.as_ref() == Some(&name))); + (!used).then_some(name) + }) +} diff --git a/userland/capsule_linux/src/linux/net/named/bind.rs b/userland/capsule_linux/src/linux/net/named/bind.rs new file mode 100644 index 000000000..75b99e4ee --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/bind.rs @@ -0,0 +1,66 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `bind` on a Unix socket: a path, which becomes a file as on Linux, an +//! abstract name, or family alone, which asks for a name to be chosen. + +use crate::linux::abi::errno; +use crate::linux::file; +use crate::linux::guest::Guest; + +use super::addr::{unix_addr, UAddr}; +use super::auto::fresh; +use super::name::resolve; +use crate::linux::net::sock; + +pub fn bind(guest: &Guest, id: u32, at: u64, len: u64) -> u64 { + let ua = match unix_addr(guest, at, len) { + Ok(ua) => ua, + Err(e) => return e, + }; + if sock::with(|t| t.get(id).is_none_or(|s| s.uname.is_some())) { + return errno::fail(errno::EINVAL); + } + bind_name(guest, id, &ua) +} + +/// Bind Unix socket `id` to `ua`. A path must not exist yet, and becomes an +/// empty file; family alone chooses an abstract name of five hex digits. +fn bind_name(guest: &Guest, id: u32, ua: &UAddr) -> u64 { + let name = match resolve(guest, ua) { + Some(n) => n, + None => match fresh() { + Some(n) => n, + None => return errno::fail(errno::EADDRINUSE), + }, + }; + let taken = sock::with(|t| t.iter().any(|(_, s)| s.uname.as_ref() == Some(&name))); + if taken || (!name.is_abstract() && file::look(&name.key).is_some()) { + return errno::fail(errno::EADDRINUSE); + } + if !name.is_abstract() { + let key = file::key(&name.key); + if key.writable().is_err() { + return errno::fail(errno::EROFS); + } + /* The store refuses a file whose directory is missing. */ + if file::store_write(&key, &[]).is_err() { + return errno::fail(errno::ENOENT); + } + } + sock::with(|t| t.get_mut(id).map(|s| s.uname = Some(name))); + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/net/named/connect.rs b/userland/capsule_linux/src/linux/net/named/connect.rs new file mode 100644 index 000000000..e030d1abd --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/connect.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `connect` on a Unix socket. A connect to the display's path +//! turns the descriptor into the display connection this capsule serves +//! itself (`unix`); every other name is the family's own. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; +use crate::linux::net::sock::{self, Link, Proto}; +use crate::linux::net::sockaddr::{self, AF_UNSPEC}; +use crate::linux::unix; + +use super::addr::{unix_addr, UAddr}; +use super::display::display; +use super::name::{find, resolve}; + +pub fn connect(guest: &mut Guest, fd: u64, id: u32, proto: Proto, at: u64, len: u64) -> u64 { + if proto == Proto::Dgram && matches!(sockaddr::read(guest, at, len), Ok((AF_UNSPEC, _))) { + sock::with(|t| t.get_mut(id).map(|s| (s.peer, s.upeer, s.connected) = (None, None, false))); + return errno::ok(0); + } + let ua = match unix_addr(guest, at, len) { + Ok(ua) => ua, + Err(e) => return e, + }; + if let (Proto::Stream, UAddr::Path(p)) = (proto, &ua) { + if unix::is_display(p) { + return display(guest, fd, at, len); + } + } + let Some(name) = resolve(guest, &ua) else { + return errno::fail(errno::EINVAL); + }; + let target = match find(&name, proto) { + Ok(t) => t, + Err(e) => return errno::fail(e), + }; + sock::with(|t| { + let s = t.get_mut(id).ok_or(errno::EBADF)?; + match proto { + Proto::Stream if s.connected => Err(errno::EISCONN), + Proto::Stream if s.listening => Err(errno::EINVAL), + /* A Unix connect completes in the caller's call, blocking or not. */ + Proto::Stream => match t.join(id, target) { + Link::Done => Ok(()), + Link::Full => Err(errno::EAGAIN), + Link::Refused => Err(errno::ECONNREFUSED), + }, + Proto::Dgram => { + (s.peer, s.upeer, s.connected) = (Some(target), Some(name), true); + Ok(()) + } + } + }) + .map_or_else(errno::fail, |()| errno::ok(0)) +} diff --git a/userland/capsule_linux/src/linux/net/named/display.rs b/userland/capsule_linux/src/linux/net/named/display.rs new file mode 100644 index 000000000..ef6cc8755 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/display.rs @@ -0,0 +1,32 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The display's path: the one Unix name this capsule answers itself. + +use crate::linux::guest::{Fd, Guest}; +use crate::linux::unix; + +/// Let go of the family socket and make `fd` the display connection. +pub fn display(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { + crate::linux::net::close::close(guest, fd); + if let Some(f) = guest.fds.get_mut(fd as usize) { + let (cloexec, nonblock) = (f.cloexec, f.nonblock); + *f = Fd::unix(); + f.cloexec = cloexec; + f.nonblock = nonblock; + } + unix::connect(guest, fd, at, len) +} diff --git a/userland/capsule_linux/src/linux/net/named/mod.rs b/userland/capsule_linux/src/linux/net/named/mod.rs new file mode 100644 index 000000000..ec9aac5fa --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/mod.rs @@ -0,0 +1,30 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Unix sockets with names: the sockaddr_un a guest gives, the names a +//! family socket is bound to, and bind and connect on them. + +mod addr; +mod auto; +mod bind; +mod connect; +mod display; +mod name; + +pub use addr::{read as read_uaddr, UAddr}; +pub use bind::bind; +pub use connect::connect; +pub use name::{find, resolve}; diff --git a/userland/capsule_linux/src/linux/net/named/name.rs b/userland/capsule_linux/src/linux/net/named/name.rs new file mode 100644 index 000000000..121bcc83d --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/name.rs @@ -0,0 +1,65 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Unix names. A path is a file, as on Linux: bind makes it, it stays after +//! the socket closes until the guest unlinks it, and a connect to a path +//! with no listener is ECONNREFUSED if the file is there and ENOENT if not. +//! An abstract name has no file and goes with its socket. Neither is seen +//! outside the family: only the family's sockets are bound to them. + +use crate::linux::abi::errno; +use crate::linux::file; +use crate::linux::guest::Guest; + +use super::addr::UAddr; +use crate::linux::net::sock::{self, Domain, Proto, UName}; + +/// The name `ua` means for this guest; None asks for one to be chosen. +pub fn resolve(guest: &Guest, ua: &UAddr) -> Option { + match ua { + UAddr::Auto => None, + UAddr::Path(p) => Some(UName { key: file::visible(&guest.cwd, p), shown: p.clone() }), + UAddr::Abstract(n) => { + let mut key = alloc::vec![0u8]; + key.extend_from_slice(n); + Some(UName { key: key.clone(), shown: key }) + } + } +} + +/// The family socket of kind `proto` bound to `name`, the one a connect or +/// a send reaches: a stream's must listen. Otherwise Linux's answer. +pub fn find(name: &UName, proto: Proto) -> Result { + /* + * An accepted connection carries its listener's name, as on Linux; the + * listener is the one a connect reaches. + */ + let found = sock::with(|t| { + let named = + || t.iter().filter(|(_, s)| s.domain == Domain::Unix && s.uname.as_ref() == Some(name)); + named() + .find(|(_, s)| s.listening) + .or_else(|| named().next()) + .map(|(i, s)| (i, s.proto, s.listening)) + }); + match found { + Some((_, p, _)) if p != proto => Err(errno::EPROTOTYPE), + Some((i, Proto::Dgram, _)) | Some((i, _, true)) => Ok(i), + Some(_) => Err(errno::ECONNREFUSED), + None if name.is_abstract() || file::look(&name.key).is_some() => Err(errno::ECONNREFUSED), + None => Err(errno::ENOENT), + } +} diff --git a/userland/capsule_linux/src/linux/net/opt/apply.rs b/userland/capsule_linux/src/linux/net/opt/apply.rs new file mode 100644 index 000000000..dc3610b25 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/apply.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! One option set on one socket: kept as Linux keeps it, or refused. + +use crate::linux::abi::errno; +use crate::linux::net::sock::{Domain, Proto, Sock}; + +use super::ids::*; +use super::time::{keep, timeo}; + +/// `raw` holds up to sixteen bytes of the value; `len` is what the guest +/// said it gave, at least four. +pub fn apply(s: &mut Sock, (level, name): (u64, u64), raw: &[u8], len: u64) -> u64 { + let int = u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]); + let word = + |i: usize| raw.get(i..i + 8).map(|b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8]))); + let (proto, domain) = (s.proto, s.domain); + let o = &mut s.opts; + match (level, name) { + /* A Unix socket has only socket-level options on Linux. */ + (l, _) if domain == Domain::Unix && l != SOL_SOCKET => { + return errno::fail(errno::EOPNOTSUPP) + } + (l, n) if super::more::known(l, n) && !(l == IPPROTO_TCP && proto == Proto::Dgram) => { + return super::more::set(&mut o.more, proto, l, n, int) + } + (SOL_SOCKET, SO_REUSEADDR) => o.reuseaddr = int != 0, + (SOL_SOCKET, SO_REUSEPORT) => o.reuseport = int != 0, + (SOL_SOCKET, SO_KEEPALIVE) => o.keepalive = int != 0, + (SOL_SOCKET, SO_BROADCAST) => o.broadcast = int != 0, + (SOL_SOCKET, SO_RCVBUF) => o.rcvbuf = (int.min(BUF_MAX) * 2).max(RCVBUF_MIN), + (SOL_SOCKET, SO_SNDBUF) => o.sndbuf = (int.min(BUF_MAX) * 2).max(SNDBUF_MIN), + (SOL_SOCKET, SO_LINGER) if len >= 8 => { + o.linger = (u32::from(int != 0), u32::from_le_bytes([raw[4], raw[5], raw[6], raw[7]])) + } + (SOL_SOCKET, SO_LINGER) => return errno::fail(errno::EINVAL), + (SOL_SOCKET, SO_RCVTIMEO) => return timeo(&mut o.rcvtimeo, word(0), word(8)), + (SOL_SOCKET, SO_SNDTIMEO) => return timeo(&mut o.sndtimeo, word(0), word(8)), + (IPPROTO_TCP, _) if proto == Proto::Dgram => return errno::fail(errno::ENOPROTOOPT), + (IPPROTO_TCP, TCP_NODELAY) => o.nodelay = int != 0, + (IPPROTO_TCP, TCP_KEEPIDLE) => return keep(&mut o.keepidle, int, KEEP_MAX), + (IPPROTO_TCP, TCP_KEEPINTVL) => return keep(&mut o.keepintvl, int, KEEP_MAX), + (IPPROTO_TCP, TCP_KEEPCNT) => return keep(&mut o.keepcnt, int, KEEPCNT_MAX), + /* An AF_INET socket has no IPv6 options on Linux either. */ + (IPPROTO_IPV6, _) => return errno::fail(errno::ENOPROTOOPT), + _ => return super::get::unknown("setsockopt", level, name), + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/net/opt/get.rs b/userland/capsule_linux/src/linux/net/opt/get.rs new file mode 100644 index 000000000..67209a09b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/get.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `getsockopt`, for the options `opt_set` keeps and the ones a socket +//! reports about itself. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use crate::linux::net::fd::sock_of; +use crate::linux::net::sock; + +pub fn getsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, lenp: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let Some(raw) = guest.read(lenp, 4) else { + return errno::fail(errno::EFAULT); + }; + let room = i32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]); + if room < 0 { + return errno::fail(errno::EINVAL); + } + let value = sock::with(|t| t.get_mut(id).map(|s| super::value::value(s, level, name))); + let bytes = match value { + Some(Ok(bytes)) => bytes, + Some(Err(e)) => return e, + None => return errno::fail(errno::EBADF), + }; + let n = bytes.len().min(room as usize); + if guest.write(val, &bytes[..n]) < n as i64 || guest.write(lenp, &(n as u32).to_le_bytes()) < 4 + { + return errno::fail(errno::EFAULT); + } + errno::ok(0) +} + +/// ENOPROTOOPT for an option this capsule does not keep, said by name: a +/// program may depend on it, and Linux would have kept it. +pub fn unknown(call: &str, level: u64, name: u64) -> u64 { + let what = alloc::format!("{call} level {level} option {name}: not kept for a guest socket"); + crate::linux::net::policy::refuse(&what, errno::ENOPROTOOPT) +} diff --git a/userland/capsule_linux/src/linux/net/opt/ids.rs b/userland/capsule_linux/src/linux/net/opt/ids.rs new file mode 100644 index 000000000..bbeb753ee --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/ids.rs @@ -0,0 +1,60 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Option levels and names, from include/uapi/asm-generic/socket.h, +//! include/uapi/linux/in.h and include/uapi/linux/tcp.h. + +pub const SOL_SOCKET: u64 = 1; +pub const IPPROTO_IP: u64 = 0; +pub const IPPROTO_TCP: u64 = 6; +pub const IPPROTO_IPV6: u64 = 41; + +pub const SO_REUSEADDR: u64 = 2; +pub const SO_TYPE: u64 = 3; +pub const SO_ERROR: u64 = 4; +pub const SO_BROADCAST: u64 = 6; +pub const SO_SNDBUF: u64 = 7; +pub const SO_RCVBUF: u64 = 8; +pub const SO_KEEPALIVE: u64 = 9; +pub const SO_LINGER: u64 = 13; +pub const SO_REUSEPORT: u64 = 15; +pub const SO_RCVTIMEO: u64 = 20; +pub const SO_SNDTIMEO: u64 = 21; +pub const SO_ACCEPTCONN: u64 = 30; +pub const SO_PROTOCOL: u64 = 38; +pub const SO_DOMAIN: u64 = 39; + +pub const IP_TOS: u64 = 1; +pub const IP_TTL: u64 = 2; +pub const SO_PRIORITY: u64 = 12; + +pub const TCP_NODELAY: u64 = 1; +pub const TCP_KEEPIDLE: u64 = 4; +pub const TCP_KEEPINTVL: u64 = 5; +pub const TCP_KEEPCNT: u64 = 6; +pub const TCP_QUICKACK: u64 = 12; +pub const TCP_USER_TIMEOUT: u64 = 18; +pub const TCP_FASTOPEN: u64 = 23; + +/// net.core.rmem_max and wmem_max as Linux ships them: what SO_RCVBUF and +/// SO_SNDBUF are held to before they are doubled. +pub const BUF_MAX: u32 = 212_992; +/// The least Linux keeps: SOCK_MIN_RCVBUF and SOCK_MIN_SNDBUF. +pub const RCVBUF_MIN: u32 = 2304; +pub const SNDBUF_MIN: u32 = 4608; +/// MAX_TCP_KEEPIDLE, MAX_TCP_KEEPINTVL and MAX_TCP_KEEPCNT. +pub const KEEP_MAX: u32 = 32767; +pub const KEEPCNT_MAX: u32 = 127; diff --git a/userland/capsule_linux/src/linux/net/opt/mod.rs b/userland/capsule_linux/src/linux/net/opt/mod.rs new file mode 100644 index 000000000..bdeca38af --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/mod.rs @@ -0,0 +1,29 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Socket options: what setsockopt keeps and getsockopt reads back. + +mod apply; +mod get; +mod ids; +mod more; +mod set; +mod time; +mod value; + +pub use get::getsockopt; +pub use set::setsockopt; +pub use time::limit_ms; diff --git a/userland/capsule_linux/src/linux/net/opt/more.rs b/userland/capsule_linux/src/linux/net/opt/more.rs new file mode 100644 index 000000000..4e5368c65 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/more.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Options Linux keeps that change nothing a loopback connection can see: +//! the IP type of service and time to live, the priority a queueing +//! discipline would use, and TCP's user timeout, quick-ack and Fast Open +//! queue. Each is kept, checked as Linux checks it, and read back. + +use crate::linux::abi::errno; + +use super::ids::{ + IPPROTO_IP, IPPROTO_TCP, IP_TOS, IP_TTL, SOL_SOCKET, SO_PRIORITY, TCP_FASTOPEN, TCP_QUICKACK, + TCP_USER_TIMEOUT, +}; +use crate::linux::net::sock::{More, Proto}; + +/// Linux's net.ipv4.ip_default_ttl. +const DEFAULT_TTL: u32 = 64; +/// The ECN bits of the type of service, which a TCP socket does not keep. +const ECN_MASK: u32 = 3; + +pub fn known(level: u64, name: u64) -> bool { + matches!( + (level, name), + (IPPROTO_IP, IP_TOS | IP_TTL) + | (SOL_SOCKET, SO_PRIORITY) + | (IPPROTO_TCP, TCP_USER_TIMEOUT | TCP_QUICKACK | TCP_FASTOPEN) + ) +} + +pub fn set(m: &mut More, proto: Proto, level: u64, name: u64, v: u32) -> u64 { + match (level, name) { + (IPPROTO_IP, IP_TOS) if proto == Proto::Stream => m.tos = v & 0xff & !ECN_MASK, + (IPPROTO_IP, IP_TOS) => m.tos = v & 0xff, + (IPPROTO_IP, IP_TTL) if v as i32 == -1 => m.ttl = DEFAULT_TTL, + (IPPROTO_IP, IP_TTL) if (1..=255).contains(&v) => m.ttl = v, + (SOL_SOCKET, SO_PRIORITY) => m.priority = v, + (IPPROTO_TCP, TCP_USER_TIMEOUT) if (v as i32) >= 0 => m.user_timeout = v, + (IPPROTO_TCP, TCP_QUICKACK) => m.quickack = v != 0, + (IPPROTO_TCP, TCP_FASTOPEN) if (v as i32) >= 0 => m.fastopen = v, + _ => return errno::fail(errno::EINVAL), + } + errno::ok(0) +} + +pub fn get(m: &More, level: u64, name: u64) -> u32 { + match (level, name) { + (IPPROTO_IP, IP_TOS) => m.tos, + (IPPROTO_IP, IP_TTL) => m.ttl, + (SOL_SOCKET, SO_PRIORITY) => m.priority, + (IPPROTO_TCP, TCP_USER_TIMEOUT) => m.user_timeout, + (IPPROTO_TCP, TCP_QUICKACK) => u32::from(m.quickack), + _ => m.fastopen, + } +} diff --git a/userland/capsule_linux/src/linux/net/opt/set.rs b/userland/capsule_linux/src/linux/net/opt/set.rs new file mode 100644 index 000000000..3a432a200 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/set.rs @@ -0,0 +1,42 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `setsockopt`. Every option here is kept and reads back as Linux reads +//! it; one that would change nothing on the family's loopback is still kept, +//! since Linux keeps it too. One this capsule cannot honour is refused. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use crate::linux::net::fd::sock_of; +use crate::linux::net::sock; + +pub fn setsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, len: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let Some(raw) = guest.read(val, (len as usize).min(16)) else { + return errno::fail(errno::EFAULT); + }; + if len < 4 { + return errno::fail(errno::EINVAL); + } + sock::with(|t| match t.get_mut(id) { + Some(s) => super::apply::apply(s, (level, name), &raw, len), + None => errno::fail(errno::EBADF), + }) +} diff --git a/userland/capsule_linux/src/linux/net/opt/time.rs b/userland/capsule_linux/src/linux/net/opt/time.rs new file mode 100644 index 000000000..f68652bd6 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/time.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The options that hold a number of seconds: the keepalive times, and the +//! receive and send limits a wait keeps to. + +use crate::linux::abi::errno; + +use crate::linux::net::sock::{self, Opts}; + +/// A keepalive time or count, 1 up to Linux's most, else EINVAL. +pub fn keep(slot: &mut u32, v: u32, most: u32) -> u64 { + if v < 1 || v > most { + return errno::fail(errno::EINVAL); + } + *slot = v; + errno::ok(0) +} + +/// SO_RCVTIMEO or SO_SNDTIMEO from a struct timeval: EDOM for microseconds +/// out of range, and a negative time is no limit, as Linux treats both. +pub fn timeo(slot: &mut (u64, u64), sec: Option, usec: Option) -> u64 { + let (Some(sec), Some(usec)) = (sec, usec) else { + return errno::fail(errno::EINVAL); + }; + if usec as i64 >= 1_000_000 || (usec as i64) < 0 { + return errno::fail(errno::EDOM); + } + *slot = if (sec as i64) < 0 { (0, 0) } else { (sec, usec) }; + errno::ok(0) +} + +/// The limit a receive (`read`) or a send waits for, from its option. +pub fn limit_ms(id: u32, read: bool) -> Option { + sock::with(|t| t.get(id).map(|s| if read { s.opts.rcvtimeo } else { s.opts.sndtimeo })) + .and_then(Opts::limit_ms) +} diff --git a/userland/capsule_linux/src/linux/net/opt/value.rs b/userland/capsule_linux/src/linux/net/opt/value.rs new file mode 100644 index 000000000..684b4a795 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/value.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The value getsockopt reads for each option a socket has. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; + +use super::ids::*; +use crate::linux::net::sock::{Domain, Proto, Sock}; + +pub fn value(s: &mut Sock, level: u64, name: u64) -> Result, u64> { + let o = s.opts; + let int = |v: u32| Ok(v.to_le_bytes().to_vec()); + let pair = |a: u64, b: u64| Ok([a.to_le_bytes(), b.to_le_bytes()].concat()); + let stream = s.proto == Proto::Stream; + match (level, name) { + (SOL_SOCKET, SO_TYPE) => int(if stream { 1 } else { 2 }), + (SOL_SOCKET, SO_DOMAIN) => int(if s.domain == Domain::Unix { 1 } else { 2 }), + (SOL_SOCKET, SO_PROTOCOL) => match (s.domain, stream) { + (Domain::Unix, _) => int(0), + (_, true) => int(6), + (_, false) => int(17), + }, + (SOL_SOCKET, SO_ACCEPTCONN) => int(u32::from(s.listening)), + (SOL_SOCKET, SO_ERROR) => int(core::mem::take(&mut s.error) as u32), + (SOL_SOCKET, SO_REUSEADDR) => int(u32::from(o.reuseaddr)), + (SOL_SOCKET, SO_REUSEPORT) => int(u32::from(o.reuseport)), + (SOL_SOCKET, SO_KEEPALIVE) => int(u32::from(o.keepalive)), + (SOL_SOCKET, SO_BROADCAST) => int(u32::from(o.broadcast)), + (SOL_SOCKET, SO_RCVBUF) => int(o.rcvbuf), + (SOL_SOCKET, SO_SNDBUF) => int(o.sndbuf), + (SOL_SOCKET, SO_LINGER) => { + Ok([o.linger.0.to_le_bytes(), o.linger.1.to_le_bytes()].concat()) + } + (SOL_SOCKET, SO_RCVTIMEO) => pair(o.rcvtimeo.0, o.rcvtimeo.1), + (SOL_SOCKET, SO_SNDTIMEO) => pair(o.sndtimeo.0, o.sndtimeo.1), + (l, _) if s.domain == Domain::Unix && l != SOL_SOCKET => { + Err(errno::fail(errno::EOPNOTSUPP)) + } + (IPPROTO_TCP, _) if !stream && super::more::known(level, name) => { + Err(errno::fail(errno::EOPNOTSUPP)) + } + (l, n) if super::more::known(l, n) => int(super::more::get(&o.more, l, n)), + /* A datagram socket has no TCP options, and Linux says so this way. */ + (IPPROTO_TCP, _) if !stream => Err(errno::fail(errno::EOPNOTSUPP)), + (IPPROTO_TCP, TCP_NODELAY) => int(u32::from(o.nodelay)), + (IPPROTO_TCP, TCP_KEEPIDLE) => int(o.keepidle), + (IPPROTO_TCP, TCP_KEEPINTVL) => int(o.keepintvl), + (IPPROTO_TCP, TCP_KEEPCNT) => int(o.keepcnt), + (IPPROTO_IPV6, _) => Err(errno::fail(errno::EOPNOTSUPP)), + _ => Err(super::get::unknown("getsockopt", level, name)), + } +} diff --git a/userland/capsule_linux/src/linux/net/pair.rs b/userland/capsule_linux/src/linux/net/pair.rs new file mode 100644 index 000000000..e5b5cf3fd --- /dev/null +++ b/userland/capsule_linux/src/linux/net/pair.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `socketpair`: two connected Unix sockets, both ends in the family. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::{install, SOCK_CLOEXEC, SOCK_NONBLOCK}; +use super::sock::{self, Domain, Proto}; +use super::sockaddr::{AF_INET, AF_UNIX}; + +const SOCK_STREAM: u64 = 1; +const SOCK_DGRAM: u64 = 2; +const TYPE_MASK: u64 = 0xF; + +pub fn socketpair(guest: &mut Guest, family: u64, kind: u64, protocol: u64, out: u64) -> u64 { + let flags = kind & (SOCK_NONBLOCK | SOCK_CLOEXEC); + if kind & !(TYPE_MASK | flags) != 0 { + return errno::fail(errno::EINVAL); + } + match family { + f if f == u64::from(AF_UNIX) => {} + /* Linux has no connected pair for the internet families. */ + f if f == u64::from(AF_INET) => return errno::fail(errno::EOPNOTSUPP), + _ => return errno::fail(errno::EAFNOSUPPORT), + } + let proto = match kind & TYPE_MASK { + SOCK_STREAM => Proto::Stream, + SOCK_DGRAM => Proto::Dgram, + _ => return errno::fail(errno::ESOCKTNOSUPPORT), + }; + if protocol != 0 { + return errno::fail(errno::EPROTONOSUPPORT); + } + let (a, b) = sock::with(|t| t.pair(Domain::Unix, proto, guest.pid)); + let fa = install(guest, a, flags); + let Some(na) = errno::slot(fa) else { + sock::with(|t| t.release(b, guest.pid)); + return fa; + }; + let fb = install(guest, b, flags); + let Some(nb) = errno::slot(fb) else { + super::close::discard(guest, na as u64); + return fb; + }; + let mut pair = [0u8; 8]; + pair[..4].copy_from_slice(&(na as u32).to_le_bytes()); + pair[4..].copy_from_slice(&(nb as u32).to_le_bytes()); + /* Linux copies the pair out before it installs either descriptor. */ + if guest.write(out, &pair) < 8 { + super::close::discard(guest, na as u64); + super::close::discard(guest, nb as u64); + return errno::fail(errno::EFAULT); + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/net/peer_addr.rs b/userland/capsule_linux/src/linux/net/peer_addr.rs new file mode 100644 index 000000000..8c89528e3 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/peer_addr.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The addresses the send and receive calls carry: the one a send names, +//! and the sender a receive writes back. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::flags::MSG_TRUNC; +use super::named::UAddr; +use super::sock::Addr; +use super::sockaddr::{self, AF_INET, AF_UNIX}; +use super::xfer_in::In; + +/// The count a receive answers, with the sender written out. A stream has +/// no sender, and Linux says so with a length of zero. +pub fn finish(guest: &mut Guest, got: In, flags: u64, at: u64, alen: u64) -> u64 { + if at != 0 { + let wrote = match &got.from { + Some(from) => super::sockaddr_out::write(guest, at, alen, from), + None if alen != 0 && guest.write(alen, &0u32.to_le_bytes()) < 4 => { + errno::fail(errno::EFAULT) + } + None => errno::ok(0), + }; + if errno::slot(wrote).is_none() { + return wrote; + } + } + errno::ok(if flags & MSG_TRUNC != 0 { got.whole } else { got.n } as u64) +} + +/// Where a send names: an IPv4 address or a Unix name. +pub enum To { + Inet(Addr), + Unix(UAddr), +} + +/// The address a send names, if any. +pub fn address(guest: &Guest, at: u64, alen: u64) -> Result, u64> { + if at == 0 { + return Ok(None); + } + match sockaddr::read(guest, at, alen)? { + (AF_INET, a) => Ok(Some(To::Inet(a))), + (AF_UNIX, _) => Ok(Some(To::Unix(super::named::read_uaddr(guest, at, alen)?))), + _ => Err(errno::fail(errno::EAFNOSUPPORT)), + } +} diff --git a/userland/capsule_linux/src/linux/net/policy.rs b/userland/capsule_linux/src/linux/net/policy.rs new file mode 100644 index 000000000..46a1227e7 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/policy.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What this capsule lets a guest's sockets reach. A guest binds and +//! listens only on 127.0.0.0/8, which is the family's own: nothing outside +//! the capsule can connect to it. Anything else is EACCES, said by name. +//! Raw and packet sockets reach the link layer, below anything that could +//! confine them, and are EPERM. + +use alloc::format; + +use crate::linux::abi::errno; +use crate::linux::start::say; + +use super::sock::Addr; + +/// EACCES for a bind or listen outside loopback, with the address. +pub fn not_loopback(call: &str, at: Addr) -> u64 { + let [a, b, c, d] = at.ip; + let line = format!( + "[LINUX] refused {call} {a}.{b}.{c}.{d}:{}: a guest listens only on 127.0.0.0/8\n", + at.port + ); + say(line.as_bytes()); + errno::fail(errno::EACCES) +} + +/// ENETUNREACH for a datagram to anywhere outside the family: the mixnet +/// carries streams, and a guest's datagrams have no other way out. +pub fn refuse_out(call: &str, to: Addr) -> u64 { + let [a, b, c, d] = to.ip; + let line = format!( + "[LINUX] refused {call} {a}.{b}.{c}.{d}:{}: a guest's datagrams stay in the family\n", + to.port + ); + say(line.as_bytes()); + errno::fail(errno::ENETUNREACH) +} + +/// `errno` for a call this capsule declines, with why. +pub fn refuse(what: &str, errno: i64) -> u64 { + say(format!("[LINUX] refused {what}\n").as_bytes()); + errno::fail(errno) +} diff --git a/userland/capsule_linux/src/linux/net/poll_socket.rs b/userland/capsule_linux/src/linux/net/poll_socket.rs index 7d5fe6ef8..bec59ec99 100644 --- a/userland/capsule_linux/src/linux/net/poll_socket.rs +++ b/userland/capsule_linux/src/linux/net/poll_socket.rs @@ -14,15 +14,34 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Asking net.sockets whether one handle is ready. +//! A socket's readiness in poll's bits: the family's own sockets answer from +//! the table, a stream outside the family from net.sockets. use super::call::call; use super::ops::{OP_POLL, POLL_READABLE, POLL_WRITABLE}; +use super::sock; const POLLIN: u16 = 0x001; const POLLOUT: u16 = 0x004; +const POLLNVAL: u16 = 0x020; -pub(super) fn socket_bits(handle: u32) -> u16 { +pub(super) fn socket_bits(id: u32) -> u16 { + if let Some(bits) = sock::bits(id) { + return bits; + } + match sock::with(|t| t.get(id).and_then(|s| s.svc)) { + Some(handle) => service_bits(handle), + None => POLLNVAL, + } +} + +/// True when `id` is a stream net.sockets holds: nothing tells the family +/// when it changes, so a wait on it is looked at again on a tick. +pub fn outside(id: u32) -> bool { + sock::with(|t| t.get(id).is_some_and(|s| s.svc.is_some())) +} + +fn service_bits(handle: u32) -> u16 { let Some((0, out)) = call(OP_POLL, &handle.to_le_bytes(), 1) else { return 0; }; diff --git a/userland/capsule_linux/src/linux/net/addr.rs b/userland/capsule_linux/src/linux/net/recvfrom.rs similarity index 54% rename from userland/capsule_linux/src/linux/net/addr.rs rename to userland/capsule_linux/src/linux/net/recvfrom.rs index 330c461cf..f0179b9b7 100644 --- a/userland/capsule_linux/src/linux/net/addr.rs +++ b/userland/capsule_linux/src/linux/net/recvfrom.rs @@ -14,25 +14,33 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . +//! `recvfrom`: a read that says where the bytes came from. -//! `struct sockaddr_in` out of a guest. +use alloc::vec; use crate::linux::guest::Guest; -/// family(2) port(2) addr(4), and the rest of the sixteen bytes unused. -const SOCKADDR_IN: usize = 16; -const AF_INET: u16 = 2; +use super::dgram::is_resolver; +use super::fd::sock_of; -/// Port in network order and address in network order, which is the -/// order net.sockets wants as well, so neither is byte swapped here. -pub fn inet(guest: &Guest, at: u64, len: u64) -> Option<(u16, [u8; 4])> { - if len < SOCKADDR_IN as u64 { - return None; +pub fn recvfrom( + guest: &mut Guest, + fd: u64, + buf: u64, + len: u64, + flags: u64, + at: u64, + alen: u64, +) -> u64 { + if is_resolver(guest, fd) { + return super::resolver::answer(guest, fd, buf, len, at, alen); } - let raw = guest.read(at, SOCKADDR_IN)?; - if u16::from_le_bytes([raw[0], raw[1]]) != AF_INET { - return None; + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + match super::xfer_in::recv(guest, id, &vec![(buf, len)], 0, flags) { + Ok(got) => super::peer_addr::finish(guest, got, flags, at, alen), + Err(e) => e, } - let port = u16::from_be_bytes([raw[2], raw[3]]); - Some((port, [raw[4], raw[5], raw[6], raw[7]])) } diff --git a/userland/capsule_linux/src/linux/net/resolver.rs b/userland/capsule_linux/src/linux/net/resolver.rs new file mode 100644 index 000000000..9ca2e7bb3 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/resolver.rs @@ -0,0 +1,64 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A datagram socket that talks to a nameserver. This capsule answers name +//! queries itself (`dns`), so a query never leaves it: a socket that sends +//! to port 53 outside the family, or to a loopback port 53 no family socket +//! holds, becomes the resolver descriptor it always was before a guest +//! could bind a datagram socket of its own. + +use crate::linux::guest::{Fd, Guest}; + +use super::dgram_addr::{encode, fill}; +use super::dns; + +use super::sock::{self, Addr, Proto}; +use super::sockaddr::is_loopback; + +const NAMESERVER_PORT: u16 = 53; + +pub fn is_nameserver(to: Addr) -> bool { + to.port == NAMESERVER_PORT + && (!is_loopback(to.ip) || sock::with(|t| t.bound(Proto::Dgram, to).is_none())) +} + +/// Let go of the socket `fd` names and make `fd` the resolver, keeping its +/// descriptor flags. +pub fn become_resolver(guest: &mut Guest, fd: u64) { + super::close::close(guest, fd); + if let Some(f) = guest.fds.get_mut(fd as usize) { + let (cloexec, nonblock) = (f.cloexec, f.nonblock); + *f = Fd::resolver(); + f.cloexec = cloexec; + f.nonblock = nonblock; + } +} + +/// A query written to the resolver. A program with no `resolv.conf` asks +/// the loopback address, and one with a configured nameserver asks that. +pub fn query(guest: &mut Guest, fd: u64, buf: u64, len: u64, to: Option) -> u64 { + let peer = to.map_or((NAMESERVER_PORT, [127, 0, 0, 1]), |a| (a.port, a.ip)); + dns::query(guest, fd, buf, len, encode(peer)) +} + +/// An answer read from the resolver, with the nameserver it came from. +pub fn answer(guest: &mut Guest, fd: u64, buf: u64, len: u64, at: u64, alen: u64) -> u64 { + let (got, from) = dns::answer_out(guest, fd, buf, len); + match from { + Some(peer) if at != 0 => fill(guest, at, alen, peer, got), + _ => got, + } +} diff --git a/userland/capsule_linux/src/linux/net/shutdown.rs b/userland/capsule_linux/src/linux/net/shutdown.rs new file mode 100644 index 000000000..ce4cd7cb1 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/shutdown.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `shutdown`: one direction or both, on the socket rather than the +//! descriptor, which stays open. A dup'd or inherited descriptor sees it too. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::policy::refuse; +use super::sock::{self, Proto}; + +const SHUT_RD: u64 = 0; +const SHUT_WR: u64 = 1; +const SHUT_RDWR: u64 = 2; + +pub fn shutdown(guest: &Guest, fd: u64, how: u64) -> u64 { + if how > SHUT_RDWR { + return errno::fail(errno::EINVAL); + } + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let (rd, wr) = (how != SHUT_WR, how != SHUT_RD); + sock::with(|t| { + let Some(s) = t.get_mut(id) else { + return errno::fail(errno::EBADF); + }; + if s.svc.is_some() { + return refuse( + "shutdown of a stream outside the family: net.sockets has no half-close", + errno::EOPNOTSUPP, + ); + } + if s.listening { + /* Shutting a listener's reading side stops it listening. */ + if rd { + t.unlisten(id); + } + return errno::ok(0); + } + let connected = s.connected || (s.proto == Proto::Dgram && s.remote.is_some()); + if !connected { + return errno::fail(errno::ENOTCONN); + } + s.rd_shut |= rd; + s.wr_shut |= wr; + let peer = s.peer; + /* The peer reads end of file once it has what was already sent. */ + if let Some(p) = peer.filter(|_| wr).and_then(|p| t.get_mut(p)) { + p.eof = true; + } + errno::ok(0) + }) +} diff --git a/userland/capsule_linux/src/linux/net/sock/bind.rs b/userland/capsule_linux/src/linux/net/sock/bind.rs new file mode 100644 index 000000000..5b74f73f6 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/bind.rs @@ -0,0 +1,42 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A local address for a socket that sends or connects before it binds. + +use crate::linux::abi::errno::EADDRNOTAVAIL; + +use super::table::Socks; +use super::types::{Addr, Domain}; + +/// The loopback route's source address. +const LOOPBACK: [u8; 4] = [127, 0, 0, 1]; + +impl Socks { + /// Bind `id` to 127.0.0.1 and a free ephemeral port, as Linux's + /// autobind does, unless it is bound already or is a socketpair end, + /// which has no address. + pub fn autobind(&mut self, id: u32) -> Result<(), i64> { + let unbound = |s: &&super::types::Sock| s.local.is_none() && s.domain == Domain::Inet; + let Some(proto) = self.get(id).filter(unbound).map(|s| s.proto) else { + return Ok(()); + }; + let port = self.ephemeral(proto, LOOPBACK).ok_or(EADDRNOTAVAIL)?; + if let Some(s) = self.get_mut(id) { + s.local = Some(Addr { ip: LOOPBACK, port }); + } + Ok(()) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/cell.rs b/userland/capsule_linux/src/linux/net/sock/cell.rs new file mode 100644 index 000000000..47216b6e8 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/cell.rs @@ -0,0 +1,39 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where the table lives. This personality hosts one family and serves it +//! from one thread, so the table is a single process-wide value: net.sockets +//! keeps its own the same way. + +use core::cell::RefCell; + +use super::table::Socks; + +struct One(RefCell); + +/* + * SAFETY: the serve loop is the only thread in this capsule that reaches the + * table; guest threads run in their own processes and only trap into it. + */ +unsafe impl Sync for One {} + +static TABLE: One = One(RefCell::new(Socks::new())); + +/// Run `f` on the table. A call inside `f` that reaches the table again would +/// panic on the borrow, so no function here calls out while holding it. +pub fn with(f: impl FnOnce(&mut Socks) -> R) -> R { + f(&mut TABLE.0.borrow_mut()) +} diff --git a/userland/capsule_linux/src/linux/net/sock/deliver.rs b/userland/capsule_linux/src/linux/net/sock/deliver.rs new file mode 100644 index 000000000..a55a6bb70 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/deliver.rs @@ -0,0 +1,31 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes from one socket into the family: a stream's to its peer, a +//! datagram to where `dest` says, binding the sender first as Linux does. + +use super::gram_dest::Dest; +use super::table::Socks; +use super::types::Proto; + +impl Socks { + pub fn deliver(&mut self, id: u32, dest: Dest, bytes: &[u8]) -> Result { + match self.get(id).map(|s| s.proto) { + Some(Proto::Stream) => self.write(id, bytes), + _ => self.autobind(id).and_then(|()| self.send_gram(id, dest, bytes)), + } + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/free.rs b/userland/capsule_linux/src/linux/net/sock/free.rs new file mode 100644 index 000000000..3cbc55df6 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/free.rs @@ -0,0 +1,67 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Letting a socket go, and what its peer sees when it does. + +use crate::linux::abi::errno::ECONNRESET; + +use super::table::Socks; + +impl Socks { + /// `pid` no longer holds `id`. The socket goes when nobody does. + pub fn release(&mut self, id: u32, pid: u32) { + let Some(s) = self.get_mut(id) else { + return; + }; + let held = s.holders.len(); + s.holders.retain(|&p| p != pid); + if held != 0 && s.holders.is_empty() { + self.free(id, false); + } + } + + /// Close `id`. Its peer reads end of file, or ECONNRESET if this end + /// left bytes unread, set SO_LINGER to zero seconds, or `reset` is set, + /// which is when Linux sends a reset instead of a FIN. Connections still + /// queued on a listener are reset. + pub fn free(&mut self, id: u32, reset: bool) { + let Some(gone) = self.list.get_mut(id as usize).and_then(Option::take) else { + return; + }; + if let Some(h) = gone.svc { + super::super::stream::close(h); + } + let reset = reset || !gone.rx.is_empty() || gone.opts.linger == (1, 0); + /* + * A stream's peer points back; a connected Unix datagram socket + * points at this one alone. Either is told, and forgets the index. + */ + for p in self.list.iter_mut().flatten().filter(|p| p.peer == Some(id)) { + p.peer = None; + p.eof = true; + if reset && p.proto == super::types::Proto::Stream { + p.error = ECONNRESET; + } + } + for queued in gone.pending { + self.free(queued, true); + } + self.refuse_waiting(gone.syn.into_iter()); + for l in self.list.iter_mut().flatten() { + l.syn.retain(|&c| c != id); + } + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/gram.rs b/userland/capsule_linux/src/linux/net/sock/gram.rs new file mode 100644 index 000000000..8552f6de8 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/gram.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A datagram into the family: to whichever socket holds the port or the +//! Unix name it is sent to, or to a connected socket's peer. + +use crate::linux::abi::errno::{ECONNREFUSED, EPERM}; + +use super::gram_dest::Dest; +use super::name::{Gram, Peer}; +use super::table::Socks; +use super::types::Domain; + +impl Socks { + /// One datagram from `id`. One nobody holds the port of is dropped, as + /// on the wire. + pub fn send_gram(&mut self, id: u32, dest: Dest, bytes: &[u8]) -> Result { + let target = self.gram_target(id, dest, bytes.len())?; + let (Some(t), Some(from)) = (target, self.sender(id)) else { + return Ok(bytes.len()); + }; + let r = self.get_mut(t).ok_or(ECONNREFUSED)?; + /* A connected Unix socket takes only from its peer, and says so. */ + if r.domain == Domain::Unix && r.connected && r.peer.is_some_and(|p| p != id) { + return Err(EPERM); + } + let queued: usize = r.grams.iter().map(|g| g.bytes.len()).sum(); + let wanted = match (&from, r.remote) { + (Peer::Inet(a), Some(x)) => *a == x, + _ => true, + }; + if wanted && queued + bytes.len() <= r.opts.rcvbuf as usize { + r.grams.push_back(Gram { from, bytes: bytes.to_vec() }); + } + Ok(bytes.len()) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/gram_dest.rs b/userland/capsule_linux/src/linux/net/sock/gram_dest.rs new file mode 100644 index 000000000..812a081fa --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/gram_dest.rs @@ -0,0 +1,44 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a datagram goes, and which socket holds an IPv4 port. + +use crate::linux::abi::errno::ECONNREFUSED; + +use super::table::Socks; +use super::types::{Addr, Proto}; + +/// Where a datagram goes: where the socket connected, an IPv4 address, or +/// a Unix socket already found by its name (`unix_name::find`). +pub enum Dest { + Default, + Inet(Addr), + Sock(u32), +} + +impl Socks { + /// The socket bound to `to`; None drops the datagram, and a connected + /// sender is told on its next call. + pub(super) fn inet_target(&mut self, id: u32, to: Addr, connected: bool) -> Option { + let found = self.bound(Proto::Dgram, to); + if found.is_none() { + if let Some(s) = self.get_mut(id).filter(|_| connected) { + s.error = ECONNREFUSED; + } + } + found + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/gram_in.rs b/userland/capsule_linux/src/linux/net/sock/gram_in.rs new file mode 100644 index 000000000..328f2cc78 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/gram_in.rs @@ -0,0 +1,54 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A datagram out of the family's queue for one socket. + +use alloc::vec::Vec; +use core::mem; + +use crate::linux::abi::errno::{EAGAIN, EBADF}; + +use super::name::Peer; +use super::table::Socks; + +impl Socks { + /// The next datagram, cut to `want`, with its whole length and sender. + pub fn take_gram(&mut self, id: u32, want: usize, peek: bool) -> Result { + let s = self.get_mut(id).ok_or(EBADF)?; + if let Some(g) = s.grams.front() { + let (bytes, whole, from) = + (g.bytes[..want.min(g.bytes.len())].to_vec(), g.bytes.len(), g.from.clone()); + if !peek { + s.grams.pop_front(); + } + return Ok(Got { bytes, whole, from }); + } + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if s.rd_shut { + return Ok(Got { bytes: Vec::new(), whole: 0, from: Peer::Unix(None) }); + } + Err(EAGAIN) + } +} + +pub struct Got { + pub bytes: Vec, + /// The datagram's length before it was cut to fit. + pub whole: usize, + pub from: Peer, +} diff --git a/userland/capsule_linux/src/linux/net/sock/gram_target.rs b/userland/capsule_linux/src/linux/net/sock/gram_target.rs new file mode 100644 index 000000000..1fcdc62f5 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/gram_target.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which socket a datagram goes to, or why it goes nowhere. + +use core::mem; + +use crate::linux::abi::errno::{EBADF, ECONNREFUSED, EDESTADDRREQ, EINVAL, EMSGSIZE, ENOTCONN}; + +use super::gram_dest::Dest; +use super::name::Peer; +use super::table::Socks; +use super::types::Domain; + +/// The largest UDP payload IPv4 carries. +const MAX_GRAM: usize = 65507; + +impl Socks { + /// The socket a datagram from `id` goes to; None drops it unseen. + pub(super) fn gram_target( + &mut self, + id: u32, + dest: Dest, + len: usize, + ) -> Result, i64> { + let s = self.get_mut(id).ok_or(EBADF)?; + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if len > MAX_GRAM { + return Err(EMSGSIZE); + } + let (remote, peer, connected, domain) = (s.remote, s.peer, s.connected, s.domain); + match (dest, domain) { + (Dest::Sock(t), Domain::Unix) => Ok(Some(t)), + (Dest::Default, Domain::Unix) => match peer { + Some(p) => Ok(Some(p)), + None if connected => Err(ECONNREFUSED), + None => Err(ENOTCONN), + }, + (Dest::Inet(to), Domain::Inet) => Ok(self.inet_target(id, to, remote.is_some())), + (Dest::Default, Domain::Inet) => match remote { + Some(to) => Ok(self.inet_target(id, to, true)), + None => Err(EDESTADDRREQ), + }, + _ => Err(EINVAL), + } + } + + /// How a datagram from `id` names its sender. + pub(super) fn sender(&self, id: u32) -> Option { + self.get(id).map(|s| match s.domain { + Domain::Inet => Peer::Inet(s.local.unwrap_or_default()), + Domain::Unix => Peer::Unix(s.uname.clone()), + }) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/holders.rs b/userland/capsule_linux/src/linux/net/sock/holders.rs new file mode 100644 index 000000000..572e28b71 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/holders.rs @@ -0,0 +1,62 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which processes hold a socket. A descriptor copied by fork is held by the +//! child as well; the socket is let go when the last holder closes it or +//! ends. A process that ends without closing lets go of everything it held +//! when the family drops it, which is what `Holder` does. + +use alloc::vec::Vec; + +use super::cell::with; +use crate::linux::guest::{Fd, Kind}; + +/// Carried by each hosted process, as the pid its holdings are kept under. +pub struct Holder { + pid: u32, +} + +impl Holder { + pub fn new(pid: u32) -> Holder { + Holder { pid } + } + + /// A forked child holds every socket its parent's descriptors name. + pub fn fork(&self, child: u32, fds: &[Fd]) { + with(|t| { + for f in fds.iter().filter(|f| f.kind == Kind::Socket) { + if let Some(s) = t.get_mut(f.handle) { + if !s.holders.contains(&child) { + s.holders.push(child); + } + } + } + }); + } +} + +impl Drop for Holder { + fn drop(&mut self) { + let pid = self.pid; + with(|t| { + let held: Vec = + t.iter().filter(|(_, s)| s.holders.contains(&pid)).map(|(i, _)| i).collect(); + for id in held { + t.release(id, pid); + } + }); + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/kinds.rs b/userland/capsule_linux/src/linux/net/sock/kinds.rs new file mode 100644 index 000000000..e84a90ebd --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/kinds.rs @@ -0,0 +1,47 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What kind of socket, in which family, at which IPv4 address. + +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Proto { + Stream, + Dgram, +} + +/// AF_INET, or AF_UNIX for the two ends socketpair makes. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Domain { + Inet, + Unix, +} + +/// An IPv4 address and a port, the port in host order. +#[derive(Clone, Copy, PartialEq, Eq, Default)] +pub struct Addr { + pub ip: [u8; 4], + pub port: u16, +} + +/// How a stream connect to a listener went. +pub enum Link { + /// Connected; the listener has one more connection to accept. + Done, + /// Nothing listens there: Linux's loopback answers with a reset. + Refused, + /// The listener's queue is full. + Full, +} diff --git a/userland/capsule_linux/src/linux/net/sock/link.rs b/userland/capsule_linux/src/linux/net/sock/link.rs new file mode 100644 index 000000000..4a4a3d56b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/link.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Joining two stream ends: a connect on loopback or to a Unix name, which +//! Linux completes in the caller's own call, and socketpair. + +use super::kinds::Link; +use super::table::Socks; +use super::types::{Addr, Proto}; + +impl Socks { + /// Connect stream `id`, already bound, to the listener at `to`. + pub fn link(&mut self, id: u32, to: Addr) -> Link { + let from = self.get(id).and_then(|s| s.local).map_or(0, |a| a.port); + match self.listener(to, from) { + Some(l) => self.join(id, l), + None => Link::Refused, + } + } + + /// Connect stream `id` to listener `l`: a new server end, with the + /// listener's name and options, waits in its queue for accept. + pub fn join(&mut self, id: u32, l: u32) -> Link { + let Some(ls) = self.get(l) else { + return Link::Refused; + }; + /* Linux queues one more than the backlog it was given. */ + if ls.pending.len() > ls.backlog { + return Link::Full; + } + let (domain, local, uname, opts) = (ls.domain, ls.local, ls.uname.clone(), ls.opts); + let (from, from_name) = self.get(id).map_or((None, None), |c| (c.local, c.uname.clone())); + let server = self.open(domain, Proto::Stream, None); + if let Some(s) = self.get_mut(server) { + s.local = local; + s.remote = from; + s.uname = uname.clone(); + s.upeer = from_name; + s.peer = Some(id); + s.connected = true; + /* An accepted socket starts with its listener's options. */ + s.opts = opts; + } + if let Some(c) = self.get_mut(id) { + c.remote = local; + c.upeer = uname; + c.peer = Some(server); + c.connected = true; + } + if let Some(l) = self.get_mut(l) { + l.pending.push_back(server); + } + Link::Done + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/mod.rs b/userland/capsule_linux/src/linux/net/sock/mod.rs new file mode 100644 index 000000000..2c0193c71 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/mod.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The family's sockets: one table this personality keeps for every process +//! it hosts, so a connection between two of them, or two threads of one, +//! never leaves the capsule. A descriptor names an entry by index; the entry +//! records which processes hold it, and is let go when none does. +//! +//! A stream to 127.0.0.0/8 is a pair of entries, each holding what the other +//! wrote. A datagram to a bound loopback port is queued on that entry. A +//! stream to anywhere else is an entry backed by a net.sockets handle. + +mod bind; +mod cell; +mod deliver; +mod free; +mod gram; +mod gram_dest; +mod gram_in; +mod gram_target; +mod holders; +mod kinds; +mod link; +mod name; +mod new; +mod opts; +mod opts_more; +mod pair; +mod port; +mod progress; +mod put; +mod ready; +mod recv; +mod send; +mod syn; +mod table; +mod take; +mod types; +mod unlisten; + +pub use cell::with; +pub use gram_dest::Dest; +pub use holders::Holder; +pub use kinds::Link; +pub use name::{Peer, UName}; +pub use opts::Opts; +pub use opts_more::More; +pub use progress::{progress, set_progress}; +pub use ready::bits; +pub use types::{Addr, Domain, Proto, Sock}; diff --git a/userland/capsule_linux/src/linux/net/sock/name.rs b/userland/capsule_linux/src/linux/net/sock/name.rs new file mode 100644 index 000000000..665eac23d --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/name.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A Unix socket's name, who a message came from, and a datagram. + +use alloc::vec::Vec; + +use super::types::Addr; + +/// A bound Unix name. `key` is what two sockets must share to meet: the +/// resolved path, or a NUL then the abstract name. `shown` is the sun_path +/// the guest gave, which getsockname and accept report back. +#[derive(Clone, PartialEq, Eq)] +pub struct UName { + pub key: Vec, + pub shown: Vec, +} + +impl UName { + /// True for a name in the abstract namespace, which no file backs. + pub fn is_abstract(&self) -> bool { + self.key.first() == Some(&0) + } +} + +/// The far end of a message or a connection, as a sockaddr reports it: an +/// IPv4 address, or a Unix name, None for an unnamed socket. +#[derive(Clone)] +pub enum Peer { + Inet(Addr), + Unix(Option), +} + +/// A datagram waiting to be read, with its sender. +pub struct Gram { + pub from: Peer, + pub bytes: Vec, +} diff --git a/userland/capsule_linux/src/linux/net/sock/new.rs b/userland/capsule_linux/src/linux/net/sock/new.rs new file mode 100644 index 000000000..81749c62d --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/new.rs @@ -0,0 +1,54 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A socket as it starts: unbound, unconnected, Linux's default options. + +use alloc::collections::VecDeque; +use alloc::vec; +use alloc::vec::Vec; + +use super::opts::Opts; +use super::types::{Domain, Proto, Sock}; + +impl Sock { + pub fn new(domain: Domain, proto: Proto, pid: Option) -> Sock { + Sock { + domain, + proto, + local: None, + remote: None, + listening: false, + backlog: 0, + pending: VecDeque::new(), + syn: VecDeque::new(), + connecting: false, + peer: None, + connected: false, + rx: VecDeque::new(), + grams: VecDeque::new(), + eof: false, + wr_shut: false, + rd_shut: false, + broken: false, + error: 0, + opts: Opts::new(proto), + svc: None, + holders: pid.map_or_else(Vec::new, |p| vec![p]), + uname: None, + upeer: None, + } + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/opts.rs b/userland/capsule_linux/src/linux/net/sock/opts.rs new file mode 100644 index 000000000..4b6d268d8 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/opts.rs @@ -0,0 +1,74 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The options a socket keeps, with the values Linux starts a socket with +//! (read from a Linux 6.x host: tcp_rmem[1], tcp_wmem[1], rmem_default and +//! wmem_default, and the TCP keepalive defaults). + +use super::types::Proto; + +#[derive(Clone, Copy)] +pub struct Opts { + pub reuseaddr: bool, + pub reuseport: bool, + pub keepalive: bool, + pub broadcast: bool, + pub nodelay: bool, + /// As getsockopt reports them: Linux doubles what setsockopt was given. + pub rcvbuf: u32, + pub sndbuf: u32, + pub keepidle: u32, + pub keepintvl: u32, + pub keepcnt: u32, + /// SO_LINGER's l_onoff and l_linger. + pub linger: (u32, u32), + /// SO_RCVTIMEO and SO_SNDTIMEO as the guest wrote them, seconds and + /// microseconds, so they read back exactly. + pub rcvtimeo: (u64, u64), + pub sndtimeo: (u64, u64), + pub more: super::opts_more::More, +} + +impl Opts { + pub fn new(proto: Proto) -> Opts { + let (rcvbuf, sndbuf) = match proto { + Proto::Stream => (131_072, 16_384), + Proto::Dgram => (212_992, 212_992), + }; + Opts { + reuseaddr: false, + reuseport: false, + keepalive: false, + broadcast: false, + nodelay: false, + rcvbuf, + sndbuf, + keepidle: 7200, + keepintvl: 75, + keepcnt: 9, + linger: (0, 0), + rcvtimeo: (0, 0), + sndtimeo: (0, 0), + more: Default::default(), + } + } + + /// Milliseconds a receive or a send may wait, None for no limit. + pub fn limit_ms((sec, usec): (u64, u64)) -> Option { + let ms = sec.saturating_mul(1000).saturating_add(usec.div_ceil(1000)); + (ms != 0).then_some(ms) + } +} diff --git a/userland/capsule_linux/src/linux/net/dns/open.rs b/userland/capsule_linux/src/linux/net/sock/opts_more.rs similarity index 63% rename from userland/capsule_linux/src/linux/net/dns/open.rs rename to userland/capsule_linux/src/linux/net/sock/opts_more.rs index fb3bd77be..1e10aa6e0 100644 --- a/userland/capsule_linux/src/linux/net/dns/open.rs +++ b/userland/capsule_linux/src/linux/net/sock/opts_more.rs @@ -14,16 +14,20 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! The descriptor a program opens to reach its nameserver. +//! The options `opt_more` keeps, with Linux's starting values. -use crate::linux::abi::errno; -use crate::linux::guest::{Fd, Guest}; +#[derive(Clone, Copy)] +pub struct More { + pub tos: u32, + pub ttl: u32, + pub priority: u32, + pub user_timeout: u32, + pub quickack: bool, + pub fastopen: u32, +} -/// It looks like a datagram socket and holds no handle: there is -/// nothing on the other side of it, which is the point. -pub fn open(guest: &mut Guest) -> u64 { - match crate::linux::file::install(guest, Fd::resolver()) { - Some(n) => errno::ok(n), - None => errno::fail(errno::EMFILE), +impl Default for More { + fn default() -> More { + More { tos: 0, ttl: 64, priority: 0, user_timeout: 0, quickack: true, fastopen: 0 } } } diff --git a/userland/capsule_linux/src/linux/net/sock/pair.rs b/userland/capsule_linux/src/linux/net/sock/pair.rs new file mode 100644 index 000000000..04e294ed2 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/pair.rs @@ -0,0 +1,35 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! socketpair: two sockets connected from the start. + +use super::table::Socks; +use super::types::{Domain, Proto}; + +impl Socks { + /// Two connected sockets, both held by `pid`. + pub fn pair(&mut self, domain: Domain, proto: Proto, pid: u32) -> (u32, u32) { + let a = self.open(domain, proto, Some(pid)); + let b = self.open(domain, proto, Some(pid)); + for (me, other) in [(a, b), (b, a)] { + if let Some(s) = self.get_mut(me) { + s.peer = Some(other); + s.connected = true; + } + } + (a, b) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/port.rs b/userland/capsule_linux/src/linux/net/sock/port.rs new file mode 100644 index 000000000..e006c3dcb --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/port.rs @@ -0,0 +1,73 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Ports on the loopback address: who holds one, and a free one when a +//! program asks for port 0 or connects before it binds. + +use super::table::Socks; +use super::types::{Addr, Proto}; + +/// Linux's net.ipv4.ip_local_port_range default. +const FIRST: u16 = 32768; +const LAST: u16 = 60999; + +impl Socks { + /// The socket of this kind bound to exactly this address, if any. + pub fn bound(&self, proto: Proto, at: Addr) -> Option { + self.iter().find(|(_, s)| s.proto == proto && s.local == Some(at)).map(|(i, _)| i) + } + + /// The listener a connect to `at` from port `from` reaches. Listeners + /// that share a port with SO_REUSEPORT take connections by the + /// connecting port, as Linux spreads them by a hash of the connection. + pub fn listener(&self, at: Addr, from: u16) -> Option { + let group: alloc::vec::Vec = self + .iter() + .filter(|(_, s)| s.listening && s.proto == Proto::Stream && s.local == Some(at)) + .map(|(i, _)| i) + .collect(); + group.get(usize::from(from) % group.len().max(1)).copied() + } + + /// True when binding `id` to `at` takes a port another socket holds. + /// Two sockets share one when both set SO_REUSEPORT, or when both set + /// SO_REUSEADDR and the other does not listen, as Linux allows. + pub fn in_use(&self, id: u32, at: Addr) -> bool { + let Some(me) = self.get(id) else { + return true; + }; + self.iter().any(|(i, s)| { + i != id + && s.proto == me.proto + && s.local == Some(at) + && !(s.opts.reuseport && me.opts.reuseport) + && (s.listening || !(s.opts.reuseaddr && me.opts.reuseaddr)) + }) + } + + /// A port in the ephemeral range no socket of this kind holds on `ip`. + pub fn ephemeral(&mut self, proto: Proto, ip: [u8; 4]) -> Option { + let span = (LAST - FIRST + 1) as u32; + for step in 0..span { + let port = FIRST + ((u32::from(self.next_port) + step) % span) as u16; + if self.bound(proto, Addr { ip, port }).is_none() { + self.next_port = (port - FIRST + 1) % span as u16; + return Some(port); + } + } + None + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/progress.rs b/userland/capsule_linux/src/linux/net/sock/progress.rs new file mode 100644 index 000000000..a0d9cc555 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/progress.rs @@ -0,0 +1,35 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! How far a waiting call has got. A blocking send on Linux returns once +//! every byte is queued, and a receive with MSG_WAITALL once its buffer is +//! full; a call parked partway keeps its count here, by thread, until it is +//! answered. + +use super::cell::with; + +pub fn progress(tid: u32) -> usize { + with(|t| t.progress.iter().find(|p| p.0 == tid).map_or(0, |p| p.1)) +} + +pub fn set_progress(tid: u32, done: usize) { + with(|t| { + t.progress.retain(|p| p.0 != tid); + if done != 0 { + t.progress.push((tid, done)); + } + }); +} diff --git a/userland/capsule_linux/src/linux/net/sock/put.rs b/userland/capsule_linux/src/linux/net/sock/put.rs new file mode 100644 index 000000000..b7332f101 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/put.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes into a family stream's peer, as many as it has room for. + +use core::mem; + +use crate::linux::abi::errno::{EAGAIN, EBADF, EPIPE}; + +use super::table::Socks; +use super::types::Domain; + +impl Socks { + /// The bytes, and the peer that took them. + pub(super) fn put(&mut self, id: u32, bytes: &[u8]) -> Result<(usize, Option), i64> { + let s = self.get_mut(id).ok_or(EBADF)?; + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if !s.connected || s.wr_shut || s.broken { + return Err(EPIPE); + } + let Some(p) = s.peer else { + /* + * TCP takes the first write after the peer left, and the reset + * that answers it makes every later one EPIPE. A Unix socket + * knows at once. + */ + if s.domain == Domain::Unix { + return Err(EPIPE); + } + s.broken = true; + return Ok((bytes.len(), None)); + }; + let peer = self.get_mut(p).ok_or(EPIPE)?; + let room = (peer.opts.rcvbuf as usize).saturating_sub(peer.rx.len()); + if room == 0 && !bytes.is_empty() { + return Err(EAGAIN); + } + let n = room.min(bytes.len()); + peer.rx.extend(&bytes[..n]); + Ok((n, Some(p))) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/ready.rs b/userland/capsule_linux/src/linux/net/sock/ready.rs new file mode 100644 index 000000000..07a994043 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/ready.rs @@ -0,0 +1,71 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What a family socket can do now, in poll's bits, as Linux's tcp_poll, +//! udp_poll and unix_poll report it. + +use super::cell::with; +use super::types::{Proto, Sock}; + +const POLLIN: u16 = 0x001; +const POLLOUT: u16 = 0x004; +const POLLERR: u16 = 0x008; +const POLLHUP: u16 = 0x010; +const POLLRDHUP: u16 = 0x2000; + +/// The bits for `id`, or None when net.sockets holds its state. +pub fn bits(id: u32) -> Option { + with(|t| { + let s = t.get(id).filter(|s| s.svc.is_none())?; + /* A stream is writable while its peer has room, or once it is gone. */ + let room = + s.peer.and_then(|p| t.get(p)).is_none_or(|p| p.rx.len() < p.opts.rcvbuf as usize); + Some(of(s, room)) + }) +} + +fn of(s: &Sock, room: bool) -> u16 { + let err = if s.error != 0 { POLLERR } else { 0 }; + if s.proto == Proto::Dgram { + let rd = if s.grams.is_empty() { 0 } else { POLLIN }; + return err | rd | POLLOUT; + } + if s.listening { + return err | if s.pending.is_empty() { 0 } else { POLLIN }; + } + /* A connect waiting for room: SYN_SENT, neither readable nor writable. */ + if s.connecting { + return err; + } + /* Never connected, refused, or reset: Linux's TCP_CLOSE. */ + if !s.connected || s.broken { + let rd = if s.connected { POLLIN | POLLRDHUP } else { 0 }; + return err | rd | POLLOUT | POLLHUP; + } + let shut_rd = s.eof || s.rd_shut; + let mut set = err; + if !s.rx.is_empty() || shut_rd { + set |= POLLIN; + } + if shut_rd { + set |= POLLRDHUP; + } + if shut_rd && s.wr_shut { + set |= POLLHUP; + } + /* After SHUT_WR a write fails at once, so Linux reports it writable. */ + set | if room || s.wr_shut { POLLOUT } else { 0 } +} diff --git a/userland/capsule_linux/src/linux/net/sock/recv.rs b/userland/capsule_linux/src/linux/net/sock/recv.rs new file mode 100644 index 000000000..730901fce --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/recv.rs @@ -0,0 +1,49 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes out of a family stream. + +use alloc::vec::Vec; +use core::mem; + +use crate::linux::abi::errno::{EAGAIN, EBADF, ENOTCONN}; + +use super::table::Socks; + +impl Socks { + /// Up to `want` bytes of a stream; empty is end of file, EAGAIN is none + /// yet. Bytes come before a pending error, as Linux reads its queue first. + pub fn read(&mut self, id: u32, want: usize, peek: bool) -> Result, i64> { + let s = self.get_mut(id).ok_or(EBADF)?; + if !s.rx.is_empty() && want != 0 { + let n = want.min(s.rx.len()); + return Ok(match peek { + true => s.rx.iter().take(n).copied().collect(), + false => s.rx.drain(..n).collect(), + }); + } + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if s.listening || !s.connected { + return Err(ENOTCONN); + } + if want == 0 || s.eof || s.rd_shut { + return Ok(Vec::new()); + } + Err(EAGAIN) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/send.rs b/userland/capsule_linux/src/linux/net/sock/send.rs new file mode 100644 index 000000000..cdda1d092 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/send.rs @@ -0,0 +1,26 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes into a family stream: to its peer's queue. + +use super::table::Socks; + +impl Socks { + /// As many of `bytes` as the peer has room for, EAGAIN for none. + pub fn write(&mut self, id: u32, bytes: &[u8]) -> Result { + self.put(id, bytes).map(|(n, _)| n) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/syn.rs b/userland/capsule_linux/src/linux/net/sock/syn.rs new file mode 100644 index 000000000..0a29cb8ce --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/syn.rs @@ -0,0 +1,54 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Connects a full listener turned away. On Linux a non-blocking connect to +//! a listener whose queue is full answers EINPROGRESS and completes when an +//! accept makes room (its SYN is sent again); here it completes at that +//! accept. Until then the socket is not writable, and a second connect is +//! EALREADY. + +use super::table::Socks; + +impl Socks { + /// Queue uid=0(root) gid=0(root) groups=0(root)'s connect on listener `l`. + pub fn wait_room(&mut self, id: u32, l: u32) { + if let Some(c) = self.get_mut(id) { + c.connecting = true; + } + if let Some(ls) = self.get_mut(l) { + ls.syn.push_back(id); + } + } + + /// Complete the connects waiting on `l` while its queue has room. + pub fn make_room(&mut self, l: u32) { + loop { + let Some(ls) = self.get_mut(l) else { + return; + }; + if ls.pending.len() > ls.backlog { + return; + } + let Some(c) = ls.syn.pop_front() else { + return; + }; + if let Some(cs) = self.get_mut(c) { + cs.connecting = false; + self.join(c, l); + } + } + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/table.rs b/userland/capsule_linux/src/linux/net/sock/table.rs new file mode 100644 index 000000000..09287eebf --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/table.rs @@ -0,0 +1,64 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The table itself: entries by index, a freed index reused. + +use alloc::vec::Vec; + +use super::types::{Domain, Proto, Sock}; + +pub struct Socks { + pub(super) list: Vec>, + /// Where the next ephemeral port search starts. + pub(super) next_port: u16, + /// Waiting calls partway through, by thread (`progress`). + pub(super) progress: Vec<(u32, usize)>, +} + +impl Socks { + pub const fn new() -> Socks { + Socks { list: Vec::new(), next_port: 0, progress: Vec::new() } + } + + /// A fresh socket held by `pid`, or by nobody yet when `pid` is None (the + /// server end of a connection, until accept hands it out). + pub fn open(&mut self, domain: Domain, proto: Proto, pid: Option) -> u32 { + let sock = Sock::new(domain, proto, pid); + match self.list.iter().position(Option::is_none) { + Some(i) => { + self.list[i] = Some(sock); + i as u32 + } + None => { + self.list.push(Some(sock)); + (self.list.len() - 1) as u32 + } + } + } + + pub fn get(&self, id: u32) -> Option<&Sock> { + self.list.get(id as usize).and_then(Option::as_ref) + } + + pub fn get_mut(&mut self, id: u32) -> Option<&mut Sock> { + self.list.get_mut(id as usize).and_then(Option::as_mut) + } + + /// Every live socket, with its index. + pub fn iter(&self) -> impl Iterator { + self.list.iter().enumerate().filter_map(|(i, s)| s.as_ref().map(|s| (i as u32, s))) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/take.rs b/userland/capsule_linux/src/linux/net/sock/take.rs new file mode 100644 index 000000000..af6f3fbc9 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/take.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes out of one family socket: a stream's next bytes, or its next +//! datagram with the name of whoever sent it. + +use alloc::vec::Vec; + +use crate::linux::abi::errno::EBADF; + +use super::name::Peer; +use super::table::Socks; +use super::types::Proto; + +/// The bytes, a datagram's length before it was cut, and its sender. +pub type Taken = (Vec, usize, Option); + +impl Socks { + pub fn take(&mut self, id: u32, want: usize, peek: bool) -> Result { + match self.get(id).map(|s| s.proto) { + None => Err(EBADF), + Some(Proto::Stream) => { + let bytes = self.read(id, want, peek)?; + let n = bytes.len(); + Ok((bytes, n, None)) + } + Some(Proto::Dgram) => { + let got = self.take_gram(id, want, peek)?; + /* A socketpair's peer has no name, and Linux gives none. */ + let from = match got.from { + Peer::Unix(None) => None, + named => Some(named), + }; + Ok((got.bytes, got.whole, from)) + } + } + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/types.rs b/userland/capsule_linux/src/linux/net/sock/types.rs new file mode 100644 index 000000000..513b60d11 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/types.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! One socket, as the family sees it. + +use alloc::collections::VecDeque; +use alloc::vec::Vec; + +use super::name::{Gram, UName}; +use super::opts::Opts; + +pub use super::kinds::{Addr, Domain, Proto}; + +pub struct Sock { + pub domain: Domain, + pub proto: Proto, + pub local: Option, + pub remote: Option, + /// Set by listen, with the queue of connections accept has not taken. + pub listening: bool, + pub backlog: usize, + pub pending: VecDeque, + /// Connects the full queue turned away, oldest first (`syn`). + pub syn: VecDeque, + /// This end's connect waits in a listener's `syn` queue. + pub connecting: bool, + /// The other end of a stream, until it is let go. + pub peer: Option, + pub connected: bool, + /// Bytes the peer wrote that this end has not read. + pub rx: VecDeque, + /// Datagrams waiting to be read. + pub grams: VecDeque, + /// The peer will write nothing more: it shut its side or it is gone. + pub eof: bool, + pub wr_shut: bool, + pub rd_shut: bool, + /// Written to after the peer was gone: every later write is EPIPE. + pub broken: bool, + /// SO_ERROR: reported once, by the next call that looks. + pub error: i64, + pub opts: Opts, + /// The net.sockets handle, for a stream that reaches outside the family. + pub svc: Option, + /// The processes that hold a descriptor naming this socket. + pub holders: Vec, + /// A Unix socket's own name, and its connected peer's. + pub uname: Option, + pub upeer: Option, +} diff --git a/userland/capsule_linux/src/linux/net/sock/unlisten.rs b/userland/capsule_linux/src/linux/net/sock/unlisten.rs new file mode 100644 index 000000000..4a57a1a3b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/unlisten.rs @@ -0,0 +1,48 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A listener that stops: what it queued is reset, and the connects that +//! wait for its room are refused. + +use crate::linux::abi::errno::ECONNREFUSED; + +use super::table::Socks; + +impl Socks { + /// Listener `l` is gone: the connects waiting on it are refused. + pub fn refuse_waiting(&mut self, syn: impl Iterator) { + for c in syn { + if let Some(cs) = self.get_mut(c) { + cs.connecting = false; + cs.error = ECONNREFUSED; + } + } + } + + /// Listener `l` stops listening: what it queued is reset, and the + /// connects waiting for room are refused. + pub fn unlisten(&mut self, l: u32) { + let Some(s) = self.get_mut(l) else { + return; + }; + s.listening = false; + let (queued, waiting) = (core::mem::take(&mut s.pending), core::mem::take(&mut s.syn)); + for q in queued { + self.free(q, true); + } + self.refuse_waiting(waiting.into_iter()); + } +} diff --git a/userland/capsule_linux/src/linux/net/sockaddr.rs b/userland/capsule_linux/src/linux/net/sockaddr.rs new file mode 100644 index 000000000..6ca4fc158 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sockaddr.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sockaddr_in` and `sockaddr_un` between guest memory and `Addr`. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::sock::Addr; + +pub const AF_UNSPEC: u16 = 0; +pub const AF_UNIX: u16 = 1; +pub const AF_INET: u16 = 2; +pub const SOCKADDR_IN: usize = 16; + +/// The family and address a guest named; EINVAL for one too short to hold +/// its family, EFAULT for one it cannot read. +pub fn read(guest: &Guest, at: u64, len: u64) -> Result<(u16, Addr), u64> { + if len < 2 { + return Err(errno::fail(errno::EINVAL)); + } + let head = guest.read(at, 2).ok_or(errno::fail(errno::EFAULT))?; + let family = u16::from_le_bytes([head[0], head[1]]); + if family != AF_INET { + return Ok((family, Addr::default())); + } + if len < SOCKADDR_IN as u64 { + return Err(errno::fail(errno::EINVAL)); + } + let raw = guest.read(at, SOCKADDR_IN).ok_or(errno::fail(errno::EFAULT))?; + let port = u16::from_be_bytes([raw[2], raw[3]]); + Ok((family, Addr { ip: [raw[4], raw[5], raw[6], raw[7]], port })) +} + +/// True for 127.0.0.0/8, the only addresses the family keeps to itself. +pub fn is_loopback(ip: [u8; 4]) -> bool { + ip[0] == 127 +} diff --git a/userland/capsule_linux/src/linux/net/sockaddr_out.rs b/userland/capsule_linux/src/linux/net/sockaddr_out.rs new file mode 100644 index 000000000..5a20bd547 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sockaddr_out.rs @@ -0,0 +1,71 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A socket's address written into guest memory. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::sock::Peer; +use super::sockaddr::{AF_INET, AF_UNIX}; + +/// Write `peer` at `at`, cut to the length the guest offered at `lenp`, and +/// the whole length back at `lenp`, as Linux's move_addr_to_user does. An +/// unnamed Unix socket is the family alone; a path carries its NUL. +pub fn write(guest: &mut Guest, at: u64, lenp: u64, peer: &Peer) -> u64 { + if at == 0 || lenp == 0 { + return errno::ok(0); + } + let Some(raw) = guest.read(lenp, 4) else { + return errno::fail(errno::EFAULT); + }; + let room = u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]) as i32; + if room < 0 { + return errno::fail(errno::EINVAL); + } + let sa = encode(peer); + let n = sa.len().min(room as usize); + if guest.write(at, &sa[..n]) < n as i64 + || guest.write(lenp, &(sa.len() as u32).to_le_bytes()) < 4 + { + return errno::fail(errno::EFAULT); + } + errno::ok(0) +} + +fn encode(peer: &Peer) -> Vec { + let mut sa = Vec::with_capacity(16); + match peer { + Peer::Inet(addr) => { + sa.extend_from_slice(&AF_INET.to_le_bytes()); + sa.extend_from_slice(&addr.port.to_be_bytes()); + sa.extend_from_slice(&addr.ip); + sa.resize(16, 0); + } + Peer::Unix(name) => { + sa.extend_from_slice(&AF_UNIX.to_le_bytes()); + if let Some(n) = name { + sa.extend_from_slice(&n.shown); + if !n.is_abstract() { + sa.push(0); + } + } + } + } + sa +} diff --git a/userland/capsule_linux/src/linux/net/socket.rs b/userland/capsule_linux/src/linux/net/socket.rs index cdcbd5cbe..85d09936f 100644 --- a/userland/capsule_linux/src/linux/net/socket.rs +++ b/userland/capsule_linux/src/linux/net/socket.rs @@ -14,49 +14,62 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `socket` and `connect`, over net.sockets. - -use alloc::vec::Vec; - -use super::ops::{DOMAIN, KIND_MIXNET, OP_SOCKET}; +//! `socket` for AF_INET and AF_UNIX. The socket is the family's until it +//! connects outside 127.0.0.0/8 (then net.sockets holds the stream) or to +//! the display's path (then it is the display connection). use crate::linux::abi::errno; -use crate::linux::guest::{Fd, Guest}; +use crate::linux::guest::Guest; -use super::call::call; +use super::fd::{install, SOCK_CLOEXEC, SOCK_NONBLOCK}; +use super::policy::refuse; +use super::sock::{self, Domain, Proto}; +use super::sockaddr::{AF_INET, AF_UNIX}; -const AF_INET: u64 = 2; const SOCK_STREAM: u64 = 1; const SOCK_DGRAM: u64 = 2; -/// Linux ORs these into the type; neither changes what is opened here. -const TYPE_MASK: u64 = 0xFF; +const SOCK_RAW: u64 = 3; +const SOCK_SEQPACKET: u64 = 5; +const TYPE_MASK: u64 = 0xF; +const IPPROTO_TCP: u64 = 6; +const IPPROTO_UDP: u64 = 17; +/// The one protocol a Unix socket takes besides 0. +const PF_UNIX: u64 = 1; -pub fn socket(guest: &mut Guest, family: u64, kind: u64) -> u64 { - if family != AF_INET { - return errno::fail(errno::EAFNOSUPPORT); +pub fn socket(guest: &mut Guest, family: u64, kind: u64, protocol: u64) -> u64 { + let flags = kind & (SOCK_NONBLOCK | SOCK_CLOEXEC); + if kind & !(TYPE_MASK | flags) != 0 { + return errno::fail(errno::EINVAL); } - /* - * A guest's stream goes over the mixnet, never the open network, and it - * holds no capability that could name a socket: there is no second route - * to disable and no firewall rule to remove. - */ - let want = match kind & TYPE_MASK { - SOCK_STREAM => KIND_MIXNET, - SOCK_DGRAM => return super::dns::open(guest), - _ => return errno::fail(errno::ENOSYS), + let domain = match family { + f if f == u64::from(AF_INET) => Domain::Inet, + f if f == u64::from(AF_UNIX) => Domain::Unix, + _ => return errno::fail(errno::EAFNOSUPPORT), }; - let mut body = Vec::with_capacity(4); - body.extend_from_slice(&DOMAIN.to_le_bytes()); - body.extend_from_slice(&want.to_le_bytes()); - let Some((status, out)) = call(OP_SOCKET, &body, 8) else { - return errno::fail(errno::EIO); + let (proto, own) = match (kind & TYPE_MASK, domain) { + (SOCK_STREAM, Domain::Inet) => (Proto::Stream, IPPROTO_TCP), + (SOCK_DGRAM, Domain::Inet) => (Proto::Dgram, IPPROTO_UDP), + (SOCK_RAW, Domain::Inet) => { + return refuse("SOCK_RAW: raw sockets reach below any confinement", errno::EPERM) + } + (SOCK_STREAM, Domain::Unix) => (Proto::Stream, PF_UNIX), + /* Linux gives a raw Unix socket datagram semantics. */ + (SOCK_DGRAM | SOCK_RAW, Domain::Unix) => (Proto::Dgram, PF_UNIX), + (SOCK_SEQPACKET, Domain::Unix) => { + return refuse( + "SOCK_SEQPACKET: a Unix stream here does not keep message boundaries", + errno::ESOCKTNOSUPPORT, + ) + } + _ => return errno::fail(errno::ESOCKTNOSUPPORT), }; - if status != 0 || out.len() < 4 { - return errno::fail(errno::ENOMEM); - } - let handle = u32::from_le_bytes([out[0], out[1], out[2], out[3]]); - match crate::linux::file::install(guest, Fd::socket(handle)) { - Some(n) => errno::ok(n), - None => errno::fail(errno::EMFILE), + /* + * Anything but the type's own protocol, MPTCP included, is one this + * stack does not have; Go falls back to TCP on this answer. + */ + if protocol != 0 && protocol != own { + return errno::fail(errno::EPROTONOSUPPORT); } + let id = sock::with(|t| t.open(domain, proto, Some(guest.pid))); + install(guest, id, flags) } diff --git a/userland/capsule_linux/src/linux/net/stream.rs b/userland/capsule_linux/src/linux/net/stream.rs index 94c6effd5..0123a5c07 100644 --- a/userland/capsule_linux/src/linux/net/stream.rs +++ b/userland/capsule_linux/src/linux/net/stream.rs @@ -14,51 +14,41 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - -//! Bytes on and off a connected socket. +//! Bytes on a stream that reaches outside the family, through net.sockets. use alloc::vec::Vec; use crate::linux::abi::errno; -use crate::linux::guest::Guest; use super::call::call; use super::ops::{OP_CLOSE, OP_RECV, OP_SEND}; -/// One transfer. The server's reply buffer is the ceiling, not this. -const MAX_IO: u64 = 32 << 10; +/// The most net.sockets carries in one call. +pub const MAX_IO: usize = 32 << 10; -pub fn send(guest: &Guest, handle: u32, buf: u64, len: u64) -> u64 { - let take = len.min(MAX_IO); - let Some(bytes) = guest.read(buf, take as usize) else { - return errno::fail(errno::EFAULT); - }; +pub fn send_bytes(handle: u32, bytes: &[u8]) -> u64 { let mut body = Vec::with_capacity(4 + bytes.len()); body.extend_from_slice(&handle.to_le_bytes()); - body.extend_from_slice(&bytes); + body.extend_from_slice(bytes); match call(OP_SEND, &body, 0) { - Some((0, _)) => errno::ok(take), + Some((0, _)) => errno::ok(bytes.len() as u64), Some(_) => errno::fail(errno::EPIPE), None => errno::fail(errno::EIO), } } -pub fn recv(guest: &Guest, handle: u32, buf: u64, len: u64) -> u64 { - let want = len.min(MAX_IO) as usize; - let Some((status, bytes)) = call(OP_RECV, &handle.to_le_bytes(), want) else { - return errno::fail(errno::EIO); +/// Up to `want` bytes. net.sockets answers the same for a quiet stream, a +/// closed one and a reset one, so any refusal reads as ECONNRESET. +pub fn recv_bytes(handle: u32, want: usize) -> Result, u64> { + let want = want.min(MAX_IO); + let Some((status, mut bytes)) = call(OP_RECV, &handle.to_le_bytes(), want) else { + return Err(errno::fail(errno::EIO)); }; if status != 0 { - return errno::fail(errno::ECONNRESET); - } - if bytes.is_empty() { - return errno::ok(0); - } - let n = bytes.len().min(want); - if guest.write(buf, &bytes[..n]) < n as i64 { - return errno::fail(errno::EFAULT); + return Err(errno::fail(errno::ECONNRESET)); } - errno::ok(n as u64) + bytes.truncate(want); + Ok(bytes) } pub fn close(handle: u32) { diff --git a/userland/capsule_linux/src/linux/net/try_call.rs b/userland/capsule_linux/src/linux/net/try_call.rs new file mode 100644 index 000000000..90203c3e4 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/try_call.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! One try at a socket call that may wait. The serve loop's `waits_sock` +//! parks a blocking one that cannot finish and tries it again; `done` is +//! how much of it an earlier try moved (bytes, or messages for the mmsg +//! calls), and a try goes on from there. + +use alloc::vec; + +use crate::linux::abi::{errno, nr, nr_path as np}; +use crate::linux::guest::Guest; + +use super::call_kind::{bytes_in, msg_len}; +use super::{iov, mmsg}; + +/// The answer, and the whole amount the call asks to move. +pub fn try_call(guest: &mut Guest, n: u64, a: [u64; 6], done: usize) -> (u64, usize) { + let fd = a[0]; + let Ok(id) = super::fd::sock_of(guest, fd) else { + return (errno::fail(errno::EBADF), 0); + }; + let one = |buf: u64, len: u64| vec![(buf, len)]; + match n { + nr::READ => (bytes_in(guest, id, &one(a[1], a[2]), done, 0), a[2] as usize), + nr::WRITE => { + (super::xfer_out::send(guest, id, &one(a[1], a[2]), done, 0, None), a[2] as usize) + } + np::READV | nr::WRITEV => match iov::read(guest, a[1], a[2]) { + Ok(v) if n == nr::WRITEV => { + (super::xfer_out::send(guest, id, &v, done, 0, None), iov::total(&v)) + } + Ok(v) => (bytes_in(guest, id, &v, done, 0), iov::total(&v)), + Err(e) => (e, 0), + }, + nr::RECVFROM if done == 0 => { + (super::recvfrom(guest, fd, a[1], a[2], a[3], a[4], a[5]), a[2] as usize) + } + nr::RECVFROM => (bytes_in(guest, id, &one(a[1], a[2]), done, a[3]), a[2] as usize), + nr::SENDTO if done == 0 => { + (super::sendto(guest, fd, a[1], a[2], a[3], a[4], a[5]), a[2] as usize) + } + nr::SENDTO => { + (super::xfer_out::send(guest, id, &one(a[1], a[2]), done, a[3], None), a[2] as usize) + } + nr::SENDMSG => (super::msg::sendmsg(guest, fd, a[1], a[2], done), msg_len(guest, a[1])), + nr::RECVMSG => { + (super::msg_recv::recvmsg(guest, fd, a[1], a[2], done), msg_len(guest, a[1])) + } + nr::SENDMMSG => (mmsg::sendmmsg(guest, a, done), a[2].min(mmsg::MOST) as usize), + nr::RECVMMSG => (mmsg::recvmmsg(guest, a, done), a[2].min(mmsg::MOST) as usize), + nr::ACCEPT => (super::accept::accept4(guest, fd, a[1], a[2], 0), 0), + nr::ACCEPT4 => (super::accept::accept4(guest, fd, a[1], a[2], a[3]), 0), + nr::CONNECT => (super::connect(guest, fd, a[1], a[2]), 0), + _ => (errno::fail(errno::ENOSYS), 0), + } +} diff --git a/userland/capsule_linux/src/linux/net/xfer_in.rs b/userland/capsule_linux/src/linux/net/xfer_in.rs new file mode 100644 index 000000000..ddfbdb409 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/xfer_in.rs @@ -0,0 +1,67 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes into a guest from a family socket or a stream net.sockets holds. + +use alloc::vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::flags::{MSG_OOB, MSG_PEEK, MSG_TRUNC}; +use super::iov::{self, Iov}; +use super::sock::{self, Peer, Proto}; + +pub struct In { + /// Bytes put in the guest's buffers. + pub n: usize, + /// A datagram's length before it was cut to fit. + pub whole: usize, + /// Where a datagram came from; a stream, and an unnamed sender, say + /// nothing. + pub from: Option, +} + +/// Receive into `iov` from socket `id`, `skip` bytes of it filled already. +pub fn recv(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64) -> Result { + if flags & MSG_OOB != 0 { + return Err(errno::fail(errno::EINVAL)); + } + let Some((proto, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.svc))) else { + return Err(errno::fail(errno::EBADF)); + }; + let want = iov::total(iov).saturating_sub(skip); + let peek = flags & MSG_PEEK != 0; + if let Some(h) = svc { + let bytes = super::stream::recv_bytes(h, want)?; + iov::scatter(guest, iov, skip, &bytes)?; + return Ok(In { n: bytes.len(), whole: bytes.len(), from: None }); + } + let (bytes, whole, from) = sock::with(|t| t.take(id, want, peek)).map_err(errno::fail)?; + /* A stream's MSG_TRUNC discards what it would have read. */ + if proto != Proto::Stream || flags & MSG_TRUNC == 0 { + iov::scatter(guest, iov, skip, &bytes)?; + } + Ok(In { n: bytes.len(), whole, from }) +} + +/// `read` on a socket. +pub fn read(guest: &Guest, id: u32, buf: u64, len: u64) -> u64 { + match recv(guest, id, &vec![(buf, len)], 0, 0) { + Ok(got) => errno::ok(got.n as u64), + Err(e) => e, + } +} diff --git a/userland/capsule_linux/src/linux/net/xfer_out.rs b/userland/capsule_linux/src/linux/net/xfer_out.rs new file mode 100644 index 000000000..d2ab7fb63 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/xfer_out.rs @@ -0,0 +1,62 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes out of a socket: to a stream's peer, a datagram to the port or the +//! Unix name it is sent to, or a stream outside the family through +//! net.sockets. + +use alloc::vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::flags::MSG_OOB; +use super::iov::{self, Iov}; +use super::peer_addr::To; +use super::sock::{self, Proto}; + +/// Send the message in `iov` from socket `id`, `skip` bytes of it having +/// gone already: the count sent now, or an errno. +pub fn send(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64, to: Option) -> u64 { + if flags & MSG_OOB != 0 { + return errno::fail(errno::EOPNOTSUPP); + } + let Some((proto, domain, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.domain, s.svc))) + else { + return errno::fail(errno::EBADF); + }; + let cap = super::cap::cap(svc.is_some(), proto == Proto::Stream); + let bytes = match iov::gather(guest, iov, skip, cap) { + Ok(b) => b, + Err(e) => return e, + }; + if let Some(h) = svc { + return super::stream::send_bytes(h, &bytes); + } + let dest = match super::dest::dest(guest, proto, domain, to) { + Ok(d) => d, + Err(e) => return e, + }; + match sock::with(|t| t.deliver(id, dest, &bytes)) { + Ok(n) => errno::ok(n as u64), + Err(e) => errno::fail(e), + } +} + +/// `write` on a socket: a send with no flags and no address. +pub fn write(guest: &Guest, id: u32, buf: u64, len: u64) -> u64 { + send(guest, id, &vec![(buf, len)], 0, 0, None) +} diff --git a/userland/capsule_linux/src/linux/serve/dispatch.rs b/userland/capsule_linux/src/linux/serve/dispatch.rs index e292a5f27..4523e707f 100644 --- a/userland/capsule_linux/src/linux/serve/dispatch.rs +++ b/userland/capsule_linux/src/linux/serve/dispatch.rs @@ -21,6 +21,7 @@ use nonos_libc::ForeignFrame; use super::answer::Answer; use super::table::plain; +use super::waits_sock; use crate::linux::abi::{nr, nr_path as np}; use crate::linux::call::{clone, exit_thread, futex}; use crate::linux::guest::Guest; @@ -58,6 +59,7 @@ fn route(guest: &mut Guest, frame: &ForeignFrame) -> Answer { np::CLOCK_NANOSLEEP => { crate::linux::call::clock_nanosleep(guest, frame.pid, a[0], a[1], a[2]) } + n if waits_sock::takes(guest, n, a[0]) => waits_sock::io(guest, frame.pid, n, a), nr::READ | nr::WRITE if super::waits::may_wait(guest, frame.nr, a[0]) => { super::waits::io(guest, frame.pid, frame.nr, a) } diff --git a/userland/capsule_linux/src/linux/serve/family_waits.rs b/userland/capsule_linux/src/linux/serve/family_waits.rs index 735f19a44..6f50f98fb 100644 --- a/userland/capsule_linux/src/linux/serve/family_waits.rs +++ b/userland/capsule_linux/src/linux/serve/family_waits.rs @@ -25,10 +25,11 @@ use super::waits::{attempt, expire}; use super::waits_fds::watched; use crate::linux::call::now_ms; use crate::linux::guest::Kind; +use crate::linux::net::outside; const CLOCK_MONOTONIC: u64 = 1; -/// How often a wait on a socket is looked at again: its readiness changes -/// with no call for the family to answer. A timer is looked at when it fires. +/// How often a wait on a stream net.sockets holds is looked at again; a family +/// socket changes only in an answer. A timer is looked at when it fires. const TICK_MS: u64 = 10; impl Family { @@ -73,9 +74,10 @@ impl Family { if let Some(d) = wait.deadline { keep(d.saturating_sub(now)); } + super::waits_sock::ticks(g, wait).then(|| keep(TICK_MS)); for fd in watched(g, wait) { match g.fds.get(fd as usize) { - Some(f) if f.kind == Kind::Socket => keep(TICK_MS), + Some(f) if f.kind == Kind::Socket && outside(f.handle) => keep(TICK_MS), Some(f) if f.kind == Kind::Timer => { // One that has already fired was seen by the last look. let due = self.timers.get(f.handle as usize).map_or(0, |t| t.due); diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index 700f5b2dd..12527be3b 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -28,8 +28,8 @@ mod family_waits; mod loop_impl; mod pid_map; mod pid_ns; -mod refused; mod pid_out; +mod refused; mod table; mod table_file; mod table_link; @@ -40,6 +40,8 @@ mod tally; mod unserved; mod waits; mod waits_fds; +mod waits_sock; +mod waits_sock_kind; mod waits_time; pub use answer::Answer; diff --git a/userland/capsule_linux/src/linux/serve/table_net.rs b/userland/capsule_linux/src/linux/serve/table_net.rs index cf5902775..119cd7a7f 100644 --- a/userland/capsule_linux/src/linux/serve/table_net.rs +++ b/userland/capsule_linux/src/linux/serve/table_net.rs @@ -17,23 +17,37 @@ //! Calls that name a socket. use crate::linux::abi::nr; -use crate::linux::call; use crate::linux::guest::Guest; use crate::linux::net; use crate::linux::unix::{self, is_unix}; +/// Every socket call. The ones that can wait reach here only when they +/// cannot: a socket's own calls are routed to `waits_sock` first, and the +/// rest are a resolver's, a display socket's, or not a socket at all. pub fn net_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option { let _ = tid; Some(match nr { - nr::SOCKET if a[0] == 1 => unix::socket(guest, a[1]), - nr::SOCKET => net::socket(guest, a[0], a[1]), + nr::SOCKET => net::socket(guest, a[0], a[1], a[2]), + nr::SOCKETPAIR => net::socketpair(guest, a[0], a[1], a[2], a[3]), nr::CONNECT if is_unix(guest, a[0]) => unix::connect(guest, a[0], a[1], a[2]), nr::CONNECT => net::connect(guest, a[0], a[1], a[2]), - nr::SENDTO => net::sendto(guest, a[0], a[1], a[2], a[4], a[5]), - nr::RECVFROM => net::recvfrom(guest, a[0], a[1], a[2], a[4], a[5]), - nr::SHUTDOWN => call::close(guest, a[0]), - nr::SENDMSG => unix::sendmsg(guest, a[0], a[1]), - nr::RECVMSG => unix::recvmsg(guest, a[0], a[1]), + nr::BIND => net::bind(guest, a[0], a[1], a[2]), + nr::LISTEN => net::listen(guest, a[0], a[1]), + nr::ACCEPT => net::accept4(guest, a[0], a[1], a[2], 0), + nr::ACCEPT4 => net::accept4(guest, a[0], a[1], a[2], a[3]), + nr::GETSOCKNAME => net::getsockname(guest, a[0], a[1], a[2]), + nr::GETPEERNAME => net::getpeername(guest, a[0], a[1], a[2]), + nr::SETSOCKOPT => net::setsockopt(guest, a[0], a[1], a[2], a[3], a[4]), + nr::GETSOCKOPT => net::getsockopt(guest, a[0], a[1], a[2], a[3], a[4]), + nr::SENDTO => net::sendto(guest, a[0], a[1], a[2], a[3], a[4], a[5]), + nr::RECVFROM => net::recvfrom(guest, a[0], a[1], a[2], a[3], a[4], a[5]), + nr::SHUTDOWN => net::shutdown(guest, a[0], a[1]), + nr::SENDMSG if is_unix(guest, a[0]) => unix::sendmsg(guest, a[0], a[1]), + nr::RECVMSG if is_unix(guest, a[0]) => unix::recvmsg(guest, a[0], a[1]), + nr::SENDMSG => net::sendmsg(guest, a[0], a[1], a[2], 0), + nr::RECVMSG => net::recvmsg(guest, a[0], a[1], a[2], 0), + nr::SENDMMSG => net::sendmmsg(guest, a, 0), + nr::RECVMMSG => net::recvmmsg(guest, a, 0), _ => return None, }) } diff --git a/userland/capsule_linux/src/linux/serve/waits.rs b/userland/capsule_linux/src/linux/serve/waits.rs index 6de3c6d7a..0361528ce 100644 --- a/userland/capsule_linux/src/linux/serve/waits.rs +++ b/userland/capsule_linux/src/linux/serve/waits.rs @@ -79,6 +79,7 @@ pub fn attempt(guest: &mut Guest, wait: &Blocked) -> Option { let a = wait.args; let again = errno::fail(errno::EAGAIN); match wait.nr { + n if super::waits_sock::takes(guest, n, a[0]) => super::waits_sock::attempt(guest, wait), nr::READ => Some(call::read(guest, a[0], a[1], a[2])).filter(|&v| v != again), nr::WRITE => Some(call::write(guest, a[0], a[1], a[2])).filter(|&v| v != again), nr::POLL | np::PPOLL => Some(net::poll(guest, a[0], a[1])).filter(|&v| v != 0), diff --git a/userland/capsule_linux/src/linux/serve/waits_sock.rs b/userland/capsule_linux/src/linux/serve/waits_sock.rs new file mode 100644 index 000000000..9a7fede40 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/waits_sock.rs @@ -0,0 +1,72 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Socket calls that wait. Each is tried at once; a blocking one that cannot +//! finish is parked and tried again after every answer (`family_waits`). +//! SO_RCVTIMEO and SO_SNDTIMEO bound the wait, which then answers EAGAIN, or +//! the count moved so far, as Linux does. + +use crate::linux::abi::errno; +use crate::linux::call::now_ms; +use crate::linux::guest::{Blocked, Guest}; +use crate::linux::net::{self, sock}; + +use super::answer::Answer; +use super::waits_sock_kind::deadline; +pub use super::waits_sock_kind::{takes, ticks}; + +const CLOCK_MONOTONIC: u64 = 1; +const MSG_DONTWAIT: u64 = 0x40; + +pub fn io(guest: &mut Guest, tid: u32, n: u64, a: [u64; 6]) -> Answer { + let wait = Blocked { tid, nr: n, args: a, deadline: deadline(guest, n, a[0]) }; + match attempt(guest, &wait) { + Some(v) => Answer::value(v), + None => { + guest.blocked.push(wait); + Answer::Park + } + } +} + +/// The call's answer if it can give one now, None to go on waiting. +pub fn attempt(guest: &mut Guest, wait: &Blocked) -> Option { + let (n, a, tid) = (wait.nr, wait.args, wait.tid); + let done = sock::progress(tid); + let (value, whole) = net::try_call(guest, n, a, done); + let flags = net::call_flags(n, a); + let blocking = + flags & MSG_DONTWAIT == 0 && !guest.fds.get(a[0] as usize).is_some_and(|f| f.nonblock); + let late = wait.deadline.is_some_and(|d| now_ms(CLOCK_MONOTONIC).is_some_and(|now| d <= now)); + let answer = match errno::slot(value) { + Some(moved) => { + let total = done + moved; + let stream = net::sock_id(guest, a[0]).is_some_and(net::is_stream); + if blocking && moved != 0 && total < whole && !late && net::wants_all(stream, n, flags) + { + sock::set_progress(tid, total); + return None; + } + total as u64 + } + None if value == errno::fail(errno::EAGAIN) && blocking && !late => return None, + /* What moved before an error or the time limit is what Linux answers. */ + None if done != 0 => done as u64, + None => value, + }; + sock::set_progress(tid, 0); + Some(answer) +} diff --git a/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs b/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs new file mode 100644 index 000000000..31f2bf7d0 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs @@ -0,0 +1,62 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which calls `waits_sock` takes, which of its waits need a tick, and +//! when a wait's time runs out. + +use crate::linux::abi::{nr, nr_path as np}; +use crate::linux::call::now_ms; +use crate::linux::guest::{Blocked, Guest, Kind}; +use crate::linux::net; + +const CLOCK_MONOTONIC: u64 = 1; + +/// True for a call on a socket descriptor that may have to wait. +pub fn takes(guest: &Guest, n: u64, fd: u64) -> bool { + matches!( + n, + nr::READ + | nr::WRITE + | np::READV + | nr::WRITEV + | nr::ACCEPT + | nr::ACCEPT4 + | nr::CONNECT + | nr::SENDTO + | nr::RECVFROM + | nr::SENDMSG + | nr::RECVMSG + | nr::SENDMMSG + | nr::RECVMMSG + ) && guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Socket) +} + +/// True when a parked call waits on a stream net.sockets holds, which +/// nothing but a look tells the family has changed. +pub fn ticks(guest: &Guest, wait: &Blocked) -> bool { + takes(guest, wait.nr, wait.args[0]) + && net::sock_id(guest, wait.args[0]).is_some_and(net::outside) +} + +/// When a call's SO_RCVTIMEO or SO_SNDTIMEO runs out, if it has one. +pub fn deadline(guest: &Guest, n: u64, fd: u64) -> Option { + let reads = !matches!( + n, + nr::WRITE | nr::WRITEV | nr::SENDTO | nr::SENDMSG | nr::SENDMMSG | nr::CONNECT + ); + let limit = net::sock_id(guest, fd).and_then(|id| net::limit_ms(id, reads))?; + Some(now_ms(CLOCK_MONOTONIC).unwrap_or(0).saturating_add(limit)) +} diff --git a/userland/capsule_linux/src/linux/unix/mod.rs b/userland/capsule_linux/src/linux/unix/mod.rs index b03ee7949..b84a27e89 100644 --- a/userland/capsule_linux/src/linux/unix/mod.rs +++ b/userland/capsule_linux/src/linux/unix/mod.rs @@ -14,8 +14,8 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - -//! Unix domain sockets, both ends inside this capsule. +//! The display connection: a Unix socket whose far end is this capsule. +//! Every other Unix socket is the family's own (`net::sock`). mod conn; mod give; @@ -29,8 +29,9 @@ mod sock; mod sock_io; pub use conn::Conn; -pub use sock::is_unix; +pub use path::is_display; pub use recvmsg::recvmsg; pub use sendmsg::sendmsg; -pub use sock::{connect, socket}; +pub use sock::connect; +pub use sock::is_unix; pub use sock_io::{recv, send}; diff --git a/userland/capsule_linux/src/linux/unix/sock.rs b/userland/capsule_linux/src/linux/unix/sock.rs index 2e290cd5e..d6d16a1fb 100644 --- a/userland/capsule_linux/src/linux/unix/sock.rs +++ b/userland/capsule_linux/src/linux/unix/sock.rs @@ -14,35 +14,21 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - -//! The four calls a client makes on a display socket. +//! Connecting to the display, and telling a display socket apart. use crate::linux::abi::errno; -use crate::linux::guest::{Fd, Guest, Kind}; +use crate::linux::guest::{Guest, Kind}; use super::path::{is_display, sun_path}; -const SOCK_STREAM: u64 = 1; -const TYPE_MASK: u64 = 0xFF; - -pub fn socket(guest: &mut Guest, kind: u64) -> u64 { - if kind & TYPE_MASK != SOCK_STREAM { - return errno::fail(errno::ENOSYS); - } - match crate::linux::file::install(guest, Fd::unix()) { - Some(n) => errno::ok(n), - None => errno::fail(errno::EMFILE), - } -} - pub fn connect(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { let Some(path) = sun_path(guest, at, len) else { return errno::fail(errno::EINVAL); }; if !is_display(&path) { /* - * Nothing else listens in here, and a client that reaches a socket - * which silently accepts would block forever on a reply. + * Only the display is served here; every other name is a family + * socket's (net::unix_calls), which never reaches this call. */ return errno::fail(errno::ECONNREFUSED); } diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index c42c8e873..900254bac 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -34,6 +34,12 @@ $(NONOS_BAKED_TRUST_DIR)/keys/guest_%_publisher_mldsa65.pub: | $(CAPSULE_SIGN_BI # name, service port, reply port[, prebuilt ELF[, guest path]]. The enrolled # copy is named guest_, so its certificate and trailer cannot collide # with a capsule's. The guest path defaults to /bin/. +# Every guest is built, signed and proven, but the store the vfs loads holds +# at most 16 MiB, which the whole set outgrows. LINUX_GUEST_STORE_ONLY, when +# set, names the guests the store carries, for example +# make LINUX_GUEST_STORE_ONLY="busybox csock" LINUX_GUEST_BOOT_ARGS=... \ +# target/qemu-virtio-blk.img.store.stamp +# and unset it carries them all, as before. define LINUX_GUEST CAPSULE_SLUG := linux-guest-$(1) CAPSULE_HANDLE := linux.guest.$(1) @@ -54,10 +60,12 @@ nonos-mk-check-linux-guest-$(1)-keys: \ $(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_ed25519.pub \ $(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_mldsa65.pub LINUX_GUEST_STORE_DEPS += $$(linux-guest-$(1)_ARTIFACTS) $$(linux-guest-$(1)_ATTESTATION) +ifneq ($(if $(LINUX_GUEST_STORE_ONLY),$(filter $(1),$(LINUX_GUEST_STORE_ONLY)),all),) LINUX_GUEST_STORE_ENTRIES += --entry /linux$(or $(5),/bin/$(1))=$$(linux-guest-$(1)_BIN) \ --entry /linux$(or $(5),/bin/$(1)).nonos_id_cert.bin=$$(linux-guest-$(1)_CERT) \ --entry /linux$(or $(5),/bin/$(1)).manifest.bin=$$(linux-guest-$(1)_MANIFEST) \ --entry /linux$(or $(5),/bin/$(1)).zk_trailer.bin=$$(linux-guest-$(1)_ATTESTATION) +endif endef $(eval $(call LINUX_GUEST,suite,4950,4951)) @@ -99,6 +107,9 @@ $(eval $(call LINUX_GUEST,gopoll,4976,4977,$(GO_OUT)/poll)) # A goroutine spinning with no call, which only a signal to its running # thread can move off the one CPU the guest has. $(eval $(call LINUX_GUEST,gopreempt,4944,4945,$(GO_OUT)/preempt)) +# net/http inside one guest: a server on 127.0.0.1:0 and its client, twenty +# GETs over one kept-alive connection. +$(eval $(call LINUX_GUEST,gohttp,5002,5003,$(GO_OUT)/http)) # A C guest that faults in a worker thread while main joins: it proves the # whole process ends, as on Linux, and that musl threads run. Static, so no @@ -119,6 +130,36 @@ $(LINUX_GUESTS_C)/cwait: $(LINUX_GUESTS_DIR)/c/cwait.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait)) +# Sockets as Linux has them, on the family's own loopback: socketpair, a +# listener with accept4's flags, a non-blocking connect, a refused port, +# half-close, epoll on a listener, end of file, EAGAIN, MSG_PEEK, EPIPE, an +# accept and a receive that wait, the options a server sets, and fork. +$(LINUX_GUESTS_C)/csock: $(LINUX_GUESTS_DIR)/c/csock.c $(wildcard $(LINUX_GUESTS_DIR)/c/csock_*.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,csock,5000,5001,$(LINUX_GUESTS_C)/csock)) + +# Datagrams on the family's loopback: an echo, a connected socket, MSG_TRUNC, +# a refused port, sendmmsg and recvmmsg, and no peer at all. +$(LINUX_GUESTS_C)/cudp: $(LINUX_GUESTS_DIR)/c/cudp.c $(wildcard $(LINUX_GUESTS_DIR)/c/cudp_*.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,cudp,5004,5005,$(LINUX_GUESTS_C)/cudp)) + +# What a guest's sockets may reach, and what a descriptor number alone gets. +$(LINUX_GUESTS_C)/cpolicy: $(LINUX_GUESTS_DIR)/c/cpolicy.c $(wildcard $(LINUX_GUESTS_DIR)/c/cpolicy_*.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,cpolicy,5006,5007,$(LINUX_GUESTS_C)/cpolicy)) + +# A guest blocked in accept with nothing happening, for the loop's wakeups. +$(LINUX_GUESTS_C)/cidle: $(LINUX_GUESTS_DIR)/c/cidle.c $(wildcard $(LINUX_GUESTS_DIR)/c/cidle_*.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,cidle,5008,5009,$(LINUX_GUESTS_C)/cidle)) + +# Unix sockets with names: a path, what it leaves behind, abstract names, +# a connected datagram socket, autobind, and a connection across fork. +$(LINUX_GUESTS_C)/cunix: $(LINUX_GUESTS_DIR)/c/cunix.c $(wildcard $(LINUX_GUESTS_DIR)/c/cunix_*.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,cunix,5010,5011,$(LINUX_GUESTS_C)/cunix)) + # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the # normal image, which does not set NONOS_LINUX_GUESTS, still ships them. diff --git a/userland/linux_guests/c/cidle.c b/userland/linux_guests/c/cidle.c new file mode 100644 index 000000000..d94d14006 --- /dev/null +++ b/userland/linux_guests/c/cidle.c @@ -0,0 +1,41 @@ +/* + * An idle guest: main blocks in accept on 127.0.0.1 while nothing happens + * for the number of seconds given (default 10), then a thread connects. It + * is what the serve loop's wakeups are measured against: a family socket + * changes only in an answer, so the loop has nothing to look at meanwhile. + */ + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "cidle_1.h" + +int main(int argc, char **argv) { + if (argc > 1) { + idle_s = atoi(argv[1]); + } + int l = socket(AF_INET, SOCK_STREAM, 0); + memset(&addr, 0, sizeof addr); + addr.sin_family = AF_INET; + addr.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + socklen_t len = sizeof addr; + bind(l, (void *)&addr, sizeof addr); + getsockname(l, (void *)&addr, &len); + listen(l, 1); + pthread_t t; + long t0 = now_ms(); + pthread_create(&t, 0, late, 0); + int s = accept(l, 0, 0); + long waited = now_ms() - t0; + pthread_join(t, 0); + printf("[C] cidle %s: accept blocked %ld ms for a connect after %d s\n", + s >= 0 && waited >= idle_s * 1000 ? "PASS" : "FAIL", waited, idle_s); + fflush(stdout); + return s >= 0 ? 0 : 1; +} diff --git a/userland/linux_guests/c/cidle_1.h b/userland/linux_guests/c/cidle_1.h new file mode 100644 index 000000000..e624b6e19 --- /dev/null +++ b/userland/linux_guests/c/cidle_1.h @@ -0,0 +1,20 @@ +/* cidle, part 1 of 1: included once, by cidle.c. */ + +static struct sockaddr_in addr; + +static int idle_s = 10; + +static long now_ms(void) { + struct timespec ts; + clock_gettime(CLOCK_MONOTONIC, &ts); + return ts.tv_sec * 1000 + ts.tv_nsec / 1000000; +} + +static void *late(void *arg) { + (void)arg; + struct timespec ts = {idle_s, 0}; + nanosleep(&ts, 0); + int c = socket(AF_INET, SOCK_STREAM, 0); + connect(c, (void *)&addr, sizeof addr); + return (void *)(long)c; +} diff --git a/userland/linux_guests/c/cpolicy.c b/userland/linux_guests/c/cpolicy.c new file mode 100644 index 000000000..0ff76e759 --- /dev/null +++ b/userland/linux_guests/c/cpolicy.c @@ -0,0 +1,61 @@ +/* + * What a guest's sockets may reach, and what a descriptor number alone gets. + * Each part prints the errno it got; the NONOS answer is the capsule's policy + * and differs from an unconfined Linux by design, so the host's line records + * what Linux itself would allow. Parts: + * bind-any bind 0.0.0.0:0 NONOS EACCES, Linux 0 + * bind-out bind 10.0.2.15:0 NONOS EACCES, Linux EADDRNOTAVAIL + * listen-unbound listen with no bind NONOS EACCES, Linux 0 + * udp-out sendto 192.0.2.1:9 NONOS ENETUNREACH, Linux 1 + * raw socket(SOCK_RAW) NONOS EPERM, Linux EPERM unless root + * forged recv/close on numbers the guest never opened, and on a + * pipe: EBADF, EBADF, ENOTSOCK on both + */ + +#include +#include +#include +#include +#include +#include + +#include "cpolicy_1.h" + +int main(void) { + int s = socket(AF_INET, SOCK_STREAM, 0); + struct sockaddr_in any = at(0, 0), out = at(0x0a00020f, 0), far = at(0xc0000201, 9); + int bind_any = err(bind(s, (void *)&any, sizeof any)); + close(s); + s = socket(AF_INET, SOCK_STREAM, 0); + int bind_out = err(bind(s, (void *)&out, sizeof out)); + close(s); + s = socket(AF_INET, SOCK_STREAM, 0); + int listen_unbound = err(listen(s, 1)); + close(s); + int u = socket(AF_INET, SOCK_DGRAM, 0); + int udp_out = sendto(u, "x", 1, 0, (void *)&far, sizeof far) == 1 ? 0 : errno; + close(u); + int raw = err(socket(AF_INET, SOCK_RAW, IPPROTO_ICMP)); + char buf[4]; + int pipe_fds[2]; + if (pipe(pipe_fds)) { + printf("[C] cpolicy FAIL: pipe errno %d\n", errno); + return 1; + } + int forged_recv = err(recv(777, buf, 4, MSG_DONTWAIT)); + int forged_close = err(close(778)); + int pipe_recv = err(recv(pipe_fds[0], buf, 4, MSG_DONTWAIT)); + int pipe_name = err(getsockname(pipe_fds[0], (void *)&any, &(socklen_t){sizeof any})); + printf("[C] cpolicy bind-any %d bind-out %d listen-unbound %d udp-out %d raw %d " + "forged %d %d pipe %d %d\n", + bind_any, bind_out, listen_unbound, udp_out, raw, forged_recv, forged_close, + pipe_recv, pipe_name); + int confined = bind_any == EACCES && bind_out == EACCES && listen_unbound == EACCES && + udp_out == ENETUNREACH && raw == EPERM; + int forged = forged_recv == EBADF && forged_close == EBADF && pipe_recv == ENOTSOCK && + pipe_name == ENOTSOCK; + printf("[C] cpolicy %s: confined %d, forged numbers refused %d\n", + confined && forged ? "PASS" : "FAIL", confined, forged); + fflush(stdout); + return confined && forged ? 0 : 1; +} diff --git a/userland/linux_guests/c/cpolicy_1.h b/userland/linux_guests/c/cpolicy_1.h new file mode 100644 index 000000000..6127f8e45 --- /dev/null +++ b/userland/linux_guests/c/cpolicy_1.h @@ -0,0 +1,14 @@ +/* cpolicy, part 1 of 1: included once, by cpolicy.c. */ + +static struct sockaddr_in at(unsigned ip, int port) { + struct sockaddr_in sa; + memset(&sa, 0, sizeof sa); + sa.sin_family = AF_INET; + sa.sin_port = htons(port); + sa.sin_addr.s_addr = htonl(ip); + return sa; +} + +static int err(int rc) { + return rc < 0 ? errno : 0; +} diff --git a/userland/linux_guests/c/csock.c b/userland/linux_guests/c/csock.c new file mode 100644 index 000000000..481813ffe --- /dev/null +++ b/userland/linux_guests/c/csock.c @@ -0,0 +1,64 @@ +/* + * Sockets, as Linux has them: a socketpair each way, a listener on 127.0.0.1 + * with its name and accept4's flags, a non-blocking connect that answers + * EINPROGRESS and then SO_ERROR 0, a refused port, shutdown(SHUT_WR) giving + * the peer end of file while the other way still flows, epoll readiness on a + * listener, end of file, EAGAIN on an empty non-blocking receive, MSG_PEEK + * and MSG_DONTWAIT, EPIPE after the peer is gone, an accept and a receive + * that wait for another thread, the options Go and a C server set, a + * socket a forked child shares, a connect a full listener holds, the + * options a loopback connection cannot tell apart, and listeners that share + * a port with SO_REUSEPORT. Each part prints as it passes and every part + * runs, so one run names each part that fails. + */ + +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "csock_1.h" +#include "csock_2.h" +#include "csock_3.h" +#include "csock_4.h" +#include "csock_5.h" +#include "csock_6.h" +#include "csock_7.h" +#include "csock_8.h" +#include "csock_9.h" +#include "csock_10.h" +#include "csock_11.h" + +int main(void) { + signal(SIGPIPE, SIG_IGN); + int (*const part[])(void) = { + pair_stream, pair_dgram, listen_accept, nb_connect, refused, half_close, + epoll_listener, eof, empty_recv, peek, epipe, blocking_accept, + blocking_recv, options, fork_share, backlog, quiet_options, reuseport, + }; + const int count = sizeof part / sizeof part[0]; + long t0 = now_ms(); + int failed = 0; + for (int i = 0; i < count; i++) { + failed += part[i](); + } + if (failed) { + printf("[C] csock FAIL: %d parts failed, %d passed\n", failed, parts); + fflush(stdout); + return 1; + } + printf("[C] csock PASS: %d parts in %ld ms\n", parts, now_ms() - t0); + fflush(stdout); + return 0; +} diff --git a/userland/linux_guests/c/csock_1.h b/userland/linux_guests/c/csock_1.h new file mode 100644 index 000000000..807f0aea0 --- /dev/null +++ b/userland/linux_guests/c/csock_1.h @@ -0,0 +1,67 @@ +/* csock, part 1 of 11: included once, by csock.c. */ + +static int parts; + +static long now_ms(void) { + struct timespec ts; + clock_gettime(CLOCK_MONOTONIC, &ts); + return ts.tv_sec * 1000 + ts.tv_nsec / 1000000; +} + +static void nap_ms(long ms) { + struct timespec ts = {ms / 1000, (ms % 1000) * 1000000}; + nanosleep(&ts, 0); +} + +static int fail(const char *what, long a, long b) { + printf("[C] csock FAIL: %s (%ld, %ld)\n", what, a, b); + fflush(stdout); + return 1; +} + +static void ok(const char *part, const char *detail, long n) { + parts++; + printf("[C] csock %s ok: %s %ld\n", part, detail, n); + fflush(stdout); +} + +static struct sockaddr_in loopback(int port) { + struct sockaddr_in sa; + memset(&sa, 0, sizeof sa); + sa.sin_family = AF_INET; + sa.sin_port = htons(port); + sa.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + return sa; +} + +/* A listener on 127.0.0.1 at a port the kernel picks; its port through *port. */ +static int listener(int *port, int flags) { + int s = socket(AF_INET, SOCK_STREAM | flags, 0); + if (s < 0) { + return -errno; + } + int one = 1; + setsockopt(s, SOL_SOCKET, SO_REUSEADDR, &one, sizeof one); + struct sockaddr_in sa = loopback(0); + socklen_t len = sizeof sa; + if (bind(s, (void *)&sa, sizeof sa) || getsockname(s, (void *)&sa, &len) || listen(s, 8)) { + int e = errno; + close(s); + return -e; + } + *port = ntohs(sa.sin_port); + return s; +} + +static int dial(int port) { + int c = socket(AF_INET, SOCK_STREAM, 0); + struct sockaddr_in sa = loopback(port); + if (c < 0 || connect(c, (void *)&sa, sizeof sa)) { + int e = errno; + if (c >= 0) { + close(c); + } + return -e; + } + return c; +} diff --git a/userland/linux_guests/c/csock_10.h b/userland/linux_guests/c/csock_10.h new file mode 100644 index 000000000..b459408e0 --- /dev/null +++ b/userland/linux_guests/c/csock_10.h @@ -0,0 +1,33 @@ +/* csock, part 10 of 11: included once, by csock.c. */ + +/* + * The options Linux keeps that a loopback connection cannot tell apart, + * read back as Linux reads them, and the ones a socket's kind refuses. + */ +static int quiet_options(void) { + int t = socket(AF_INET, SOCK_STREAM, 0), u = socket(AF_INET, SOCK_DGRAM, 0); + int x = socket(AF_UNIX, SOCK_STREAM, 0); + int tos = 0x13, ttl = 32, zero = 0, ut = 5000, prio = 6; + setsockopt(t, IPPROTO_IP, IP_TOS, &tos, sizeof tos); + setsockopt(t, IPPROTO_IP, IP_TTL, &ttl, sizeof ttl); + int bad_ttl = setsockopt(t, IPPROTO_IP, IP_TTL, &zero, sizeof zero) ? errno : 0; + setsockopt(t, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut); + setsockopt(t, SOL_SOCKET, SO_PRIORITY, &prio, sizeof prio); + int got_tos = get_int(t, IPPROTO_IP, IP_TOS), got_ttl = get_int(t, IPPROTO_IP, IP_TTL); + int got_ut = get_int(t, IPPROTO_TCP, TCP_USER_TIMEOUT); + int got_qa = get_int(t, IPPROTO_TCP, TCP_QUICKACK), got_prio = get_int(t, SOL_SOCKET, SO_PRIORITY); + int udp_tcp = setsockopt(u, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut) ? errno : 0; + int unix_tcp = setsockopt(x, IPPROTO_TCP, TCP_NODELAY, &prio, sizeof prio) ? errno : 0; + close(t); + close(u); + close(x); + /* A TCP socket drops the two ECN bits of the type of service. */ + if (got_tos != 0x10 || got_ttl != 32 || bad_ttl != EINVAL || got_ut != 5000 || got_qa != 1 || + got_prio != 6 || udp_tcp != ENOPROTOOPT || unix_tcp != EOPNOTSUPP) { + printf("[C] csock quiet_options got %d %d %d %d %d %d %d %d\n", got_tos, got_ttl, bad_ttl, + got_ut, got_qa, got_prio, udp_tcp, unix_tcp); + return fail("quiet_options: kept and read back as Linux does", got_tos, got_ttl); + } + ok("quiet_options", "6 kept and read back; refusals", unix_tcp); + return 0; +} diff --git a/userland/linux_guests/c/csock_11.h b/userland/linux_guests/c/csock_11.h new file mode 100644 index 000000000..4bf6fb89c --- /dev/null +++ b/userland/linux_guests/c/csock_11.h @@ -0,0 +1,52 @@ +/* csock, part 11 of 11: included once, by csock.c. */ + +/* + * Listeners that set SO_REUSEPORT share a port and between them take every + * connection; a socket without it cannot bind there, and when one listener + * closes the other takes what comes next. + */ +static int reuseport(void) { + struct sockaddr_in sa = loopback(0); + socklen_t len = sizeof sa; + int one = 1, l[2], taken[2] = {0, 0}; + for (int i = 0; i < 2; i++) { + l[i] = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + setsockopt(l[i], SOL_SOCKET, SO_REUSEPORT, &one, sizeof one); + if (bind(l[i], (void *)&sa, sizeof sa) || listen(l[i], 16)) { + return fail("reuseport: two listeners on one port", i, errno); + } + getsockname(l[i], (void *)&sa, &len); + } + int plain = socket(AF_INET, SOCK_STREAM, 0); + int refused = bind(plain, (void *)&sa, sizeof sa) ? errno : 0; + close(plain); + int c[16]; + for (int i = 0; i < 16; i++) { + c[i] = dial(ntohs(sa.sin_port)); + } + for (int i = 0; i < 2; i++) { + int a; + while ((a = accept(l[i], 0, 0)) >= 0) { + taken[i]++; + close(a); + } + } + close(l[0]); + int late = dial(ntohs(sa.sin_port)); + struct pollfd p = {l[1], POLLIN, 0}; + int ready = poll(&p, 1, 3000); + int last = accept(l[1], 0, 0); + for (int i = 0; i < 16; i++) { + close(c[i]); + } + close(late); + close(last); + close(l[1]); + if (refused != EADDRINUSE || taken[0] + taken[1] != 16 || !taken[0] || !taken[1] || + ready != 1 || last < 0) { + printf("[C] csock reuseport got %d %d+%d %d %d\n", refused, taken[0], taken[1], ready, last); + return fail("reuseport: shared, spread, and the survivor takes the rest", taken[0], taken[1]); + } + ok("reuseport", "16 connections spread over 2 listeners; without it errno", refused); + return 0; +} diff --git a/userland/linux_guests/c/csock_2.h b/userland/linux_guests/c/csock_2.h new file mode 100644 index 000000000..97feafea4 --- /dev/null +++ b/userland/linux_guests/c/csock_2.h @@ -0,0 +1,57 @@ +/* csock, part 2 of 11: included once, by csock.c. */ + +/* A connected pair over 127.0.0.1: *a is the client, *b the accepted end. */ +static int tcp_pair(int *a, int *b) { + int port, l = listener(&port, 0); + if (l < 0) { + return l; + } + *a = dial(port); + *b = *a < 0 ? -1 : accept(l, 0, 0); + close(l); + return *a < 0 ? *a : *b < 0 ? -errno : 0; +} + +/* The parts of csock, one Linux behaviour each. Included once, by csock.c. */ + +static int pair_stream(void) { + int sv[2]; + char buf[8] = {0}; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv)) { + return fail("pair_stream: socketpair", -1, errno); + } + if (write(sv[0], "ping", 4) != 4 || read(sv[1], buf, 8) != 4 || memcmp(buf, "ping", 4)) { + return fail("pair_stream: a to b", 0, errno); + } + if (write(sv[1], "pong!", 5) != 5 || read(sv[0], buf, 8) != 5 || memcmp(buf, "pong!", 5)) { + return fail("pair_stream: b to a", 0, errno); + } + close(sv[0]); + long got = read(sv[1], buf, 8); + close(sv[1]); + if (got != 0) { + return fail("pair_stream: end of file after close", got, errno); + } + ok("pair_stream", "bytes each way, then eof", 9); + return 0; +} + +static int pair_dgram(void) { + int sv[2]; + char buf[16]; + if (socketpair(AF_UNIX, SOCK_DGRAM, 0, sv)) { + return fail("pair_dgram: socketpair", -1, errno); + } + if (write(sv[0], "one", 3) != 3 || write(sv[0], "second", 6) != 6) { + return fail("pair_dgram: write", -1, errno); + } + long a = read(sv[1], buf, sizeof buf); + long b = read(sv[1], buf, sizeof buf); + close(sv[0]); + close(sv[1]); + if (a != 3 || b != 6) { + return fail("pair_dgram: boundaries kept", a, b); + } + ok("pair_dgram", "two datagrams, sizes 3 and", b); + return 0; +} diff --git a/userland/linux_guests/c/csock_3.h b/userland/linux_guests/c/csock_3.h new file mode 100644 index 000000000..0d2beb006 --- /dev/null +++ b/userland/linux_guests/c/csock_3.h @@ -0,0 +1,47 @@ +/* csock, part 3 of 11: included once, by csock.c. */ + +static int listen_accept(void) { + int port, l = listener(&port, SOCK_NONBLOCK); + if (l < 0) { + return fail("listen_accept: listener", l, 0); + } + if (port == 0) { + return fail("listen_accept: getsockname gave port 0", 0, 0); + } + if (accept4(l, 0, 0, 0) != -1 || errno != EAGAIN) { + return fail("listen_accept: empty non-blocking accept is EAGAIN", errno, EAGAIN); + } + int c = dial(port); + if (c < 0) { + return fail("listen_accept: connect", c, 0); + } + struct sockaddr_in peer, mine; + socklen_t plen = sizeof peer, mlen = sizeof mine; + int s = accept4(l, (void *)&peer, &plen, SOCK_NONBLOCK | SOCK_CLOEXEC); + if (s < 0) { + return fail("listen_accept: accept4", -1, errno); + } + if (!(fcntl(s, F_GETFL) & O_NONBLOCK) || !(fcntl(s, F_GETFD) & FD_CLOEXEC)) { + return fail("listen_accept: accept4 flags", fcntl(s, F_GETFL), fcntl(s, F_GETFD)); + } + getsockname(c, (void *)&mine, &mlen); + if (peer.sin_port != mine.sin_port || peer.sin_addr.s_addr != htonl(INADDR_LOOPBACK)) { + return fail("listen_accept: accept's address is the client's", ntohs(peer.sin_port), + ntohs(mine.sin_port)); + } + struct sockaddr_in back; + socklen_t blen = sizeof back; + getpeername(c, (void *)&back, &blen); + if (ntohs(back.sin_port) != port || blen != sizeof back) { + return fail("listen_accept: getpeername", ntohs(back.sin_port), port); + } + char buf[8]; + if (write(c, "hi", 2) != 2 || read(s, buf, 8) != 2) { + return fail("listen_accept: bytes", 0, errno); + } + close(c); + close(s); + close(l); + ok("listen_accept", "accept4 NONBLOCK|CLOEXEC on port", port > 0); + return 0; +} diff --git a/userland/linux_guests/c/csock_4.h b/userland/linux_guests/c/csock_4.h new file mode 100644 index 000000000..c1b1ec64b --- /dev/null +++ b/userland/linux_guests/c/csock_4.h @@ -0,0 +1,52 @@ +/* csock, part 4 of 11: included once, by csock.c. */ + +/* csock: connecting, and closing one way or both. */ + +static int nb_connect(void) { + int port, l = listener(&port, 0); + int c = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + struct sockaddr_in sa = loopback(port); + int rc = connect(c, (void *)&sa, sizeof sa); + int e = errno; + if (rc != -1 || e != EINPROGRESS) { + return fail("nb_connect: EINPROGRESS", rc, e); + } + struct pollfd p = {c, POLLOUT, 0}; + if (poll(&p, 1, 5000) != 1 || !(p.revents & POLLOUT)) { + return fail("nb_connect: POLLOUT", p.revents, 0); + } + int err = -1; + socklen_t len = sizeof err; + if (getsockopt(c, SOL_SOCKET, SO_ERROR, &err, &len) || err != 0) { + return fail("nb_connect: SO_ERROR", err, errno); + } + close(c); + close(l); + ok("nb_connect", "EINPROGRESS, POLLOUT, SO_ERROR", err); + return 0; +} + +/* A port with no listener: a listener is opened and closed to find one. */ +static int refused(void) { + int port, l = listener(&port, 0); + close(l); + int c = dial(port); + if (c != -ECONNREFUSED) { + return fail("refused: blocking connect", c, -ECONNREFUSED); + } + int n = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + struct sockaddr_in sa = loopback(port); + int rc = connect(n, (void *)&sa, sizeof sa); + int e = errno; + struct pollfd p = {n, POLLOUT, 0}; + poll(&p, 1, 5000); + int err = 0; + socklen_t len = sizeof err; + getsockopt(n, SOL_SOCKET, SO_ERROR, &err, &len); + close(n); + if (rc != -1 || e != EINPROGRESS || err != ECONNREFUSED || !(p.revents & POLLERR)) { + return fail("refused: non-blocking connect then SO_ERROR", e, err); + } + ok("refused", "ECONNREFUSED both ways, errno", ECONNREFUSED); + return 0; +} diff --git a/userland/linux_guests/c/csock_5.h b/userland/linux_guests/c/csock_5.h new file mode 100644 index 000000000..0fd77ed48 --- /dev/null +++ b/userland/linux_guests/c/csock_5.h @@ -0,0 +1,69 @@ +/* csock, part 5 of 11: included once, by csock.c. */ + +static int half_close(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("half_close: pair", 0, errno); + } + if (shutdown(a, SHUT_WR)) { + return fail("half_close: shutdown", -1, errno); + } + long got = read(b, buf, 8); + if (got != 0) { + return fail("half_close: peer reads end of file", got, errno); + } + if (write(b, "back", 4) != 4 || read(a, buf, 8) != 4) { + return fail("half_close: the other way still flows", 0, errno); + } + if (write(a, "x", 1) != -1 || errno != EPIPE) { + return fail("half_close: write after SHUT_WR is EPIPE", errno, EPIPE); + } + if (fcntl(a, F_GETFD) < 0) { + return fail("half_close: shutdown closed the descriptor", errno, 0); + } + close(a); + close(b); + ok("half_close", "eof one way, 4 bytes back", 4); + return 0; +} + +static int epoll_listener(void) { + int port, l = listener(&port, SOCK_NONBLOCK); + int ep = epoll_create1(0); + struct epoll_event ev = {.events = EPOLLIN, .data.fd = l}, out; + epoll_ctl(ep, EPOLL_CTL_ADD, l, &ev); + int before = epoll_wait(ep, &out, 1, 0); + int c = dial(port); + int after = epoll_wait(ep, &out, 1, 5000); + int s = accept(l, 0, 0); + int drained = epoll_wait(ep, &out, 1, 0); + close(s); + close(c); + close(ep); + close(l); + if (before != 0 || after != 1 || !(out.events & EPOLLIN) || s < 0 || drained != 0) { + return fail("epoll_listener: EPOLLIN only while pending", before * 10 + after, drained); + } + ok("epoll_listener", "EPOLLIN once pending, then", drained); + return 0; +} + +static int eof(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("eof: pair", 0, errno); + } + if (write(a, "last", 4) != 4) { + return fail("eof: write", -1, errno); + } + close(a); + long first = read(b, buf, 8), then = read(b, buf, 8); + close(b); + if (first != 4 || then != 0) { + return fail("eof: data then end of file", first, then); + } + ok("eof", "4 bytes then", then); + return 0; +} diff --git a/userland/linux_guests/c/csock_6.h b/userland/linux_guests/c/csock_6.h new file mode 100644 index 000000000..fc011214a --- /dev/null +++ b/userland/linux_guests/c/csock_6.h @@ -0,0 +1,67 @@ +/* csock, part 6 of 11: included once, by csock.c. */ + +/* csock: receiving without waiting, waiting, options, and fork. */ + +static int empty_recv(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("empty_recv: pair", 0, errno); + } + if (recv(b, buf, 8, MSG_DONTWAIT) != -1 || errno != EAGAIN) { + return fail("empty_recv: MSG_DONTWAIT", errno, EAGAIN); + } + fcntl(b, F_SETFL, O_NONBLOCK); + if (read(b, buf, 8) != -1 || errno != EAGAIN) { + return fail("empty_recv: O_NONBLOCK read", errno, EAGAIN); + } + close(a); + close(b); + ok("empty_recv", "EAGAIN, errno", EAGAIN); + return 0; +} + +static int peek(void) { + int a, b; + char buf[8] = {0}; + if (tcp_pair(&a, &b)) { + return fail("peek: pair", 0, errno); + } + if (write(a, "abc", 3) != 3) { + return fail("peek: write", -1, errno); + } + long p = recv(b, buf, 8, MSG_PEEK); + long r = recv(b, buf, 8, 0); + close(a); + close(b); + if (p != 3 || r != 3 || memcmp(buf, "abc", 3)) { + return fail("peek: data stays", p, r); + } + ok("peek", "peeked then read", r); + return 0; +} + +static int epipe(void) { + int a, b; + if (tcp_pair(&a, &b)) { + return fail("epipe: pair", 0, errno); + } + close(b); + long first = send(a, "x", 1, MSG_NOSIGNAL); + long second = send(a, "x", 1, MSG_NOSIGNAL); + int e = errno; + close(a); + if (first != 1 || second != -1 || e != EPIPE) { + return fail("epipe: first send taken, then EPIPE", first, e); + } + ok("epipe", "second send errno", e); + return 0; +} + +static int late_port; + +static void *late_dial(void *arg) { + (void)arg; + nap_ms(100); + return (void *)(long)dial(late_port); +} diff --git a/userland/linux_guests/c/csock_7.h b/userland/linux_guests/c/csock_7.h new file mode 100644 index 000000000..872f2ee6b --- /dev/null +++ b/userland/linux_guests/c/csock_7.h @@ -0,0 +1,61 @@ +/* csock, part 7 of 11: included once, by csock.c. */ + +static int blocking_accept(void) { + int l = listener(&late_port, 0); + pthread_t t; + long t0 = now_ms(); + pthread_create(&t, 0, late_dial, 0); + int s = accept(l, 0, 0); + long waited = now_ms() - t0; + void *c; + pthread_join(t, &c); + close((int)(long)c); + close(s); + close(l); + if (s < 0 || waited < 80) { + return fail("blocking_accept: waited for the connect", s, waited); + } + ok("blocking_accept", "waited ms", waited >= 80); + return 0; +} + +static int late_fd; + +static void *late_write(void *arg) { + (void)arg; + nap_ms(100); + if (write(late_fd, "late", 4) != 4) { + printf("[C] csock late_write: errno %d\n", errno); + } + return 0; +} + +static int blocking_recv(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("blocking_recv: pair", 0, errno); + } + late_fd = a; + pthread_t t; + long t0 = now_ms(); + pthread_create(&t, 0, late_write, 0); + long got = recv(b, buf, 8, 0); + long waited = now_ms() - t0; + pthread_join(t, 0); + close(a); + close(b); + if (got != 4 || waited < 80) { + return fail("blocking_recv: waited for the bytes", got, waited); + } + ok("blocking_recv", "4 bytes after waiting", waited >= 80); + return 0; +} + +/* csock: the options a server sets, and a socket a forked child shares. */ + +static int get_int(int s, int level, int opt) { + int v = -1; + socklen_t len = sizeof v; + return getsockopt(s, level, opt, &v, &len) ? -errno : v; +} diff --git a/userland/linux_guests/c/csock_8.h b/userland/linux_guests/c/csock_8.h new file mode 100644 index 000000000..3a3f7f366 --- /dev/null +++ b/userland/linux_guests/c/csock_8.h @@ -0,0 +1,54 @@ +/* csock, part 8 of 11: included once, by csock.c. */ + +static int options(void) { + int port, l = listener(&port, 0); + int one = 1; + if (get_int(l, SOL_SOCKET, SO_TYPE) != SOCK_STREAM || + get_int(l, SOL_SOCKET, SO_DOMAIN) != AF_INET || + get_int(l, SOL_SOCKET, SO_ACCEPTCONN) != 1 || + get_int(l, SOL_SOCKET, SO_REUSEADDR) != 1) { + return fail("options: type, domain, acceptconn, reuseaddr", + get_int(l, SOL_SOCKET, SO_TYPE), get_int(l, SOL_SOCKET, SO_ACCEPTCONN)); + } + int c = dial(port); + int sets[][2] = { + {SOL_SOCKET, SO_KEEPALIVE}, {IPPROTO_TCP, TCP_NODELAY}, {SOL_SOCKET, SO_REUSEPORT}, + {SOL_SOCKET, SO_BROADCAST}, + }; + for (unsigned i = 0; i < sizeof sets / sizeof sets[0]; i++) { + if (setsockopt(c, sets[i][0], sets[i][1], &one, sizeof one) || + get_int(c, sets[i][0], sets[i][1]) != 1) { + return fail("options: set then read back", sets[i][1], errno); + } + } + int idle = 15; + if (setsockopt(c, IPPROTO_TCP, TCP_KEEPIDLE, &idle, sizeof idle) || + get_int(c, IPPROTO_TCP, TCP_KEEPIDLE) != 15) { + return fail("options: TCP_KEEPIDLE", get_int(c, IPPROTO_TCP, TCP_KEEPIDLE), errno); + } + int buf = 65536; + if (setsockopt(c, SOL_SOCKET, SO_RCVBUF, &buf, sizeof buf) || + get_int(c, SOL_SOCKET, SO_RCVBUF) != 2 * buf) { + return fail("options: SO_RCVBUF doubled", get_int(c, SOL_SOCKET, SO_RCVBUF), 2 * buf); + } + struct linger lg = {1, 5}, back = {0, 0}; + socklen_t len = sizeof back; + setsockopt(c, SOL_SOCKET, SO_LINGER, &lg, sizeof lg); + getsockopt(c, SOL_SOCKET, SO_LINGER, &back, &len); + struct timeval tv = {2, 500000}, tb = {0, 0}; + len = sizeof tb; + setsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof tv); + getsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tb, &len); + int bogus = setsockopt(c, SOL_SOCKET, 9999, &one, sizeof one); + int e = errno; + close(c); + close(l); + if (back.l_onoff != 1 || back.l_linger != 5 || tb.tv_sec != 2 || tb.tv_usec != 500000) { + return fail("options: SO_LINGER and SO_RCVTIMEO read back", back.l_linger, tb.tv_usec); + } + if (bogus != -1 || e != ENOPROTOOPT) { + return fail("options: an unknown option is ENOPROTOOPT", bogus, e); + } + ok("options", "11 options, unknown errno", e); + return 0; +} diff --git a/userland/linux_guests/c/csock_9.h b/userland/linux_guests/c/csock_9.h new file mode 100644 index 000000000..8286cdd30 --- /dev/null +++ b/userland/linux_guests/c/csock_9.h @@ -0,0 +1,72 @@ +/* csock, part 9 of 11: included once, by csock.c. */ + +/* + * The child holds the accepted end after the parent closes its own: the + * client still reads the child's bytes, then end of file when the child exits. + */ +static int fork_share(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("fork_share: pair", 0, errno); + } + pid_t kid = fork(); + if (kid == 0) { + close(a); + nap_ms(50); + _exit(write(b, "kid", 3) == 3 ? 0 : 1); + } + close(b); + long got = read(a, buf, 8); + int status = 0; + waitpid(kid, &status, 0); + long then = read(a, buf, 8); + close(a); + if (got != 3 || then != 0 || status != 0) { + return fail("fork_share: child's bytes, then eof", got, then); + } + ok("fork_share", "3 bytes from the child, then", then); + return 0; +} + +/* + * A listener with a backlog of 0 queues one connect; the next non-blocking + * one answers EINPROGRESS, is not writable, and a second connect on it is + * EALREADY, until an accept makes room and it completes. + */ +static int backlog(void) { + int s = socket(AF_INET, SOCK_STREAM, 0); + struct sockaddr_in sa = loopback(0); + socklen_t len = sizeof sa; + bind(s, (void *)&sa, sizeof sa); + getsockname(s, (void *)&sa, &len); + listen(s, 0); + int c0 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + int c1 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + connect(c0, (void *)&sa, sizeof sa); + int r1 = connect(c1, (void *)&sa, sizeof sa); + int e1 = errno; + struct pollfd p = {c1, POLLOUT, 0}; + int early = poll(&p, 1, 100); + int again = connect(c1, (void *)&sa, sizeof sa) ? errno : 0; + int a = accept(s, 0, 0); + p.revents = 0; + int later = poll(&p, 1, 3000); + int err = -1; + socklen_t el = sizeof err; + getsockopt(c1, SOL_SOCKET, SO_ERROR, &err, &el); + struct pollfd q = {s, POLLIN, 0}; + int b = poll(&q, 1, 3000) == 1 ? accept(s, 0, 0) : -1; + close(a); + close(b); + close(c0); + close(c1); + close(s); + if (r1 != -1 || e1 != EINPROGRESS || early != 0 || again != EALREADY || later != 1 || + err != 0 || b < 0) { + printf("[C] csock backlog got %d %d %d %d %d %d %d\n", r1, e1, early, again, later, err, b); + return fail("backlog: EINPROGRESS, waits, EALREADY, then connected", e1, again); + } + ok("backlog", "a full queue's connect completed after accept; EALREADY", again); + return 0; +} diff --git a/userland/linux_guests/c/cudp.c b/userland/linux_guests/c/cudp.c new file mode 100644 index 000000000..f379124f5 --- /dev/null +++ b/userland/linux_guests/c/cudp.c @@ -0,0 +1,39 @@ +/* + * Datagrams on the loopback address, as Linux has them: an echo between an + * unbound client and a bound server, each told the other's address; a + * connected socket that sends without naming a peer and keeps only its + * peer's datagrams; a datagram cut to the buffer with MSG_TRUNC, and + * recvmsg's MSG_TRUNC flag; ECONNREFUSED on a connected socket whose peer + * port is closed; sendmmsg and recvmmsg; and EDESTADDRREQ with no peer at + * all. Each part prints as it passes and every part runs. + */ + +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include + +#include "cudp_1.h" +#include "cudp_2.h" +#include "cudp_3.h" + +int main(void) { + int (*const part[])(void) = {echo, connected, cut, refused, mmsg, nameless}; + const int count = sizeof part / sizeof part[0]; + int failed = 0; + for (int i = 0; i < count; i++) { + failed += part[i](); + } + if (failed) { + printf("[C] cudp FAIL: %d parts failed, %d passed\n", failed, parts); + fflush(stdout); + return 1; + } + printf("[C] cudp PASS: %d parts\n", parts); + fflush(stdout); + return 0; +} diff --git a/userland/linux_guests/c/cudp_1.h b/userland/linux_guests/c/cudp_1.h new file mode 100644 index 000000000..02ce6c122 --- /dev/null +++ b/userland/linux_guests/c/cudp_1.h @@ -0,0 +1,51 @@ +/* cudp, part 1 of 3: included once, by cudp.c. */ + +static int parts; + +static int fail(const char *what, long a, long b) { + printf("[C] cudp FAIL: %s (%ld, %ld)\n", what, a, b); + fflush(stdout); + return 1; +} + +static void ok(const char *part, const char *detail, long n) { + parts++; + printf("[C] cudp %s ok: %s %ld\n", part, detail, n); + fflush(stdout); +} + +/* A datagram socket bound to 127.0.0.1 at a port the kernel picks. */ +static int bound(struct sockaddr_in *at) { + int s = socket(AF_INET, SOCK_DGRAM, 0); + memset(at, 0, sizeof *at); + at->sin_family = AF_INET; + at->sin_addr.s_addr = htonl(INADDR_LOOPBACK); + socklen_t len = sizeof *at; + if (s < 0 || bind(s, (void *)at, sizeof *at) || getsockname(s, (void *)at, &len)) { + return -1; + } + return s; +} + +static int echo(void) { + struct sockaddr_in srv, from, back; + int s = bound(&srv), c = socket(AF_INET, SOCK_DGRAM, 0); + char buf[32]; + socklen_t flen = sizeof from, blen = sizeof back; + if (s < 0 || sendto(c, "ping", 4, 0, (void *)&srv, sizeof srv) != 4) { + return fail("echo: send", s, errno); + } + long got = recvfrom(s, buf, sizeof buf, 0, (void *)&from, &flen); + if (got != 4 || from.sin_port == 0 || flen != sizeof from) { + return fail("echo: server heard the client and its address", got, ntohs(from.sin_port)); + } + sendto(s, "pong!", 5, 0, (void *)&from, flen); + got = recvfrom(c, buf, sizeof buf, 0, (void *)&back, &blen); + close(s); + close(c); + if (got != 5 || back.sin_port != srv.sin_port || memcmp(buf, "pong!", 5)) { + return fail("echo: client heard the server", got, ntohs(back.sin_port)); + } + ok("echo", "4 bytes there, back from the server's port", ntohs(back.sin_port) > 0); + return 0; +} diff --git a/userland/linux_guests/c/cudp_2.h b/userland/linux_guests/c/cudp_2.h new file mode 100644 index 000000000..96d477852 --- /dev/null +++ b/userland/linux_guests/c/cudp_2.h @@ -0,0 +1,68 @@ +/* cudp, part 2 of 3: included once, by cudp.c. */ + +static int connected(void) { + struct sockaddr_in a, b, x; + int sa = bound(&a), sb = bound(&b), sx = bound(&x); + char buf[8]; + connect(sb, (void *)&a, sizeof a); + if (send(sb, "to-a", 4, 0) != 4 || recv(sa, buf, 8, 0) != 4) { + return fail("connected: send without an address", 0, errno); + } + /* sb keeps only a's datagrams: x's is dropped, a's arrives. */ + sendto(sx, "noise", 5, 0, (void *)&b, sizeof b); + sendto(sa, "real", 4, 0, (void *)&b, sizeof b); + long got = recv(sb, buf, 8, 0); + long more = recv(sb, buf, 8, MSG_DONTWAIT); + int e = errno; + close(sa); + close(sb); + close(sx); + if (got != 4 || more != -1 || e != EAGAIN) { + return fail("connected: only the peer's datagrams kept", got, more); + } + ok("connected", "a stranger's datagram dropped, peer's bytes", got); + return 0; +} + +/* cudp: truncation, a refused port, several messages at once, and no peer. */ + +static int cut(void) { + struct sockaddr_in a; + int s = bound(&a), c = socket(AF_INET, SOCK_DGRAM, 0); + char big[100], buf[4]; + memset(big, 'z', sizeof big); + sendto(c, big, sizeof big, 0, (void *)&a, sizeof a); + long whole = recv(s, buf, sizeof buf, MSG_TRUNC); + long rest = recv(s, buf, sizeof buf, MSG_DONTWAIT); + sendto(c, big, sizeof big, 0, (void *)&a, sizeof a); + struct iovec iov = {buf, sizeof buf}; + struct msghdr m = {0}; + m.msg_iov = &iov; + m.msg_iovlen = 1; + long cut = recvmsg(s, &m, 0); + close(s); + close(c); + if (whole != 100 || rest != -1 || cut != 4 || !(m.msg_flags & MSG_TRUNC)) { + return fail("trunc: cut to 4, whole length 100, flag set", whole, cut); + } + ok("trunc", "MSG_TRUNC gave the whole length", whole); + return 0; +} + +/* A port with no socket: one is bound and closed to find it. */ +static int refused(void) { + struct sockaddr_in gone; + close(bound(&gone)); + int c = socket(AF_INET, SOCK_DGRAM, 0); + char buf[4]; + connect(c, (void *)&gone, sizeof gone); + long sent = send(c, "x", 1, 0); + long got = recv(c, buf, sizeof buf, MSG_DONTWAIT); + int e = errno; + close(c); + if (sent != 1 || got != -1 || e != ECONNREFUSED) { + return fail("refused: send taken, then ECONNREFUSED", sent, e); + } + ok("refused", "connected to a closed port, errno", e); + return 0; +} diff --git a/userland/linux_guests/c/cudp_3.h b/userland/linux_guests/c/cudp_3.h new file mode 100644 index 000000000..686e9ea28 --- /dev/null +++ b/userland/linux_guests/c/cudp_3.h @@ -0,0 +1,41 @@ +/* cudp, part 3 of 3: included once, by cudp.c. */ + +static int mmsg(void) { + struct sockaddr_in a; + int s = bound(&a), c = socket(AF_INET, SOCK_DGRAM, 0); + connect(c, (void *)&a, sizeof a); + char out[3][4] = {"one", "two", "six"}, in[3][8]; + struct iovec oi[3], ii[3]; + struct mmsghdr om[3], im[3]; + memset(om, 0, sizeof om); + memset(im, 0, sizeof im); + for (int i = 0; i < 3; i++) { + oi[i] = (struct iovec){out[i], 3}; + ii[i] = (struct iovec){in[i], 8}; + om[i].msg_hdr.msg_iov = &oi[i]; + om[i].msg_hdr.msg_iovlen = 1; + im[i].msg_hdr.msg_iov = &ii[i]; + im[i].msg_hdr.msg_iovlen = 1; + } + int sent = sendmmsg(c, om, 3, 0); + int got = recvmmsg(s, im, 3, MSG_WAITFORONE, 0); + close(s); + close(c); + if (sent != 3 || got != 3 || im[2].msg_len != 3 || memcmp(in[2], "six", 3)) { + return fail("mmsg: three out, three in", sent, got); + } + ok("mmsg", "sendmmsg and recvmmsg moved", got); + return 0; +} + +static int nameless(void) { + int c = socket(AF_INET, SOCK_DGRAM, 0); + long sent = send(c, "x", 1, 0); + int e = errno; + close(c); + if (sent != -1 || e != EDESTADDRREQ) { + return fail("nameless: no peer is EDESTADDRREQ", sent, e); + } + ok("nameless", "errno", e); + return 0; +} diff --git a/userland/linux_guests/c/cunix.c b/userland/linux_guests/c/cunix.c new file mode 100644 index 000000000..c333e06ab --- /dev/null +++ b/userland/linux_guests/c/cunix.c @@ -0,0 +1,44 @@ +/* + * Unix sockets with names, as Linux has them: a listener on a path, its + * name and its client's, a connect that completes at once; ENOENT for a path + * with nothing there and ECONNREFUSED for one with no listener; a path that + * stays after its socket closes until it is unlinked; abstract names and the + * sender a datagram reports; a connected datagram socket that refuses + * strangers; a name bind chooses; and a connection across fork. Each part + * prints as it passes and every part runs. + */ + +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include +#define PATH "/tmp/cunix.sock" +#define GONE "/tmp/cunix.none" + +#include "cunix_1.h" +#include "cunix_2.h" +#include "cunix_3.h" +#include "cunix_4.h" + +int main(void) { + int (*const part[])(void) = {path_stream, left_behind, abstract_dgram, + connected_dgram, autobind, across_fork}; + const int count = sizeof part / sizeof part[0]; + int failed = 0; + for (int i = 0; i < count; i++) { + failed += part[i](); + } + if (failed) { + printf("[C] cunix FAIL: %d parts failed, %d passed\n", failed, parts); + fflush(stdout); + return 1; + } + printf("[C] cunix PASS: %d parts\n", parts); + fflush(stdout); + return 0; +} diff --git a/userland/linux_guests/c/cunix_1.h b/userland/linux_guests/c/cunix_1.h new file mode 100644 index 000000000..bf03cf563 --- /dev/null +++ b/userland/linux_guests/c/cunix_1.h @@ -0,0 +1,27 @@ +/* cunix, part 1 of 4: included once, by cunix.c. */ + +static int parts; + +static int fail(const char *what, long a, long b) { + printf("[C] cunix FAIL: %s (%ld, %ld)\n", what, a, b); + fflush(stdout); + return 1; +} + +static void ok(const char *part, const char *detail, long n) { + parts++; + printf("[C] cunix %s ok: %s %ld\n", part, detail, n); + fflush(stdout); +} + +/* A sockaddr_un for a path, or for an abstract name when `abs` is set. */ +static socklen_t name(struct sockaddr_un *a, const char *p, int abs) { + memset(a, 0, sizeof *a); + a->sun_family = AF_UNIX; + if (abs) { + memcpy(a->sun_path + 1, p, strlen(p)); + return offsetof(struct sockaddr_un, sun_path) + 1 + strlen(p); + } + strcpy(a->sun_path, p); + return offsetof(struct sockaddr_un, sun_path) + strlen(p) + 1; +} diff --git a/userland/linux_guests/c/cunix_2.h b/userland/linux_guests/c/cunix_2.h new file mode 100644 index 000000000..4f6d31a9f --- /dev/null +++ b/userland/linux_guests/c/cunix_2.h @@ -0,0 +1,50 @@ +/* cunix, part 2 of 4: included once, by cunix.c. */ + +static int path_stream(void) { + struct sockaddr_un a, b; + socklen_t l = name(&a, PATH, 0), bl = sizeof b; + unlink(PATH); + int s = socket(AF_UNIX, SOCK_STREAM, 0); + if (listen(s, 1) != -1 || errno != EINVAL) { + return fail("path_stream: listen before bind is EINVAL", errno, EINVAL); + } + if (bind(s, (void *)&a, l) || listen(s, 4)) { + return fail("path_stream: bind and listen", -1, errno); + } + int twice = socket(AF_UNIX, SOCK_STREAM, 0); + if (bind(twice, (void *)&a, l) != -1 || errno != EADDRINUSE) { + return fail("path_stream: a second bind is EADDRINUSE", errno, EADDRINUSE); + } + close(twice); + int c = socket(AF_UNIX, SOCK_STREAM | SOCK_NONBLOCK, 0); + if (connect(c, (void *)&a, l)) { + return fail("path_stream: a non-blocking connect completes at once", -1, errno); + } + int t = accept(s, (void *)&b, &bl); + if (t < 0 || bl != 2) { + return fail("path_stream: accept names an unnamed client with 2 bytes", t, bl); + } + bl = sizeof b; + getsockname(s, (void *)&b, &bl); + if (bl != l || strcmp(b.sun_path, PATH)) { + return fail("path_stream: getsockname", bl, l); + } + bl = sizeof b; + getpeername(c, (void *)&b, &bl); + if (bl != l || strcmp(b.sun_path, PATH)) { + return fail("path_stream: the client's peer is the path", bl, l); + } + char buf[8]; + if (write(c, "unix", 4) != 4 || read(t, buf, 8) != 4) { + return fail("path_stream: bytes", 0, errno); + } + close(c); + long eof = read(t, buf, 8); + close(t); + close(s); + if (eof != 0) { + return fail("path_stream: end of file", eof, 0); + } + ok("path_stream", "bound, named, connected, 4 bytes, eof; name length", l); + return 0; +} diff --git a/userland/linux_guests/c/cunix_3.h b/userland/linux_guests/c/cunix_3.h new file mode 100644 index 000000000..4ea494bc5 --- /dev/null +++ b/userland/linux_guests/c/cunix_3.h @@ -0,0 +1,66 @@ +/* cunix, part 3 of 4: included once, by cunix.c. */ + +/* cunix: what a path leaves behind, abstract datagrams, and fork. */ + +static int left_behind(void) { + struct sockaddr_un a, g; + socklen_t l = name(&a, PATH, 0), gl = name(&g, GONE, 0); + unlink(GONE); + int c = socket(AF_UNIX, SOCK_STREAM, 0); + int missing = connect(c, (void *)&g, gl) ? errno : 0; + int s = socket(AF_UNIX, SOCK_STREAM, 0); + bind(s, (void *)&a, l); + int unheard = connect(c, (void *)&a, l) ? errno : 0; + close(s); + int s2 = socket(AF_UNIX, SOCK_STREAM, 0); + int rebind = bind(s2, (void *)&a, l) ? errno : 0; + int closed = connect(c, (void *)&a, l) ? errno : 0; + unlink(PATH); + int after = bind(s2, (void *)&a, l) ? errno : 0; + close(s2); + close(c); + unlink(PATH); + if (missing != ENOENT || unheard != ECONNREFUSED || rebind != EADDRINUSE || + closed != ECONNREFUSED || after != 0) { + printf("[C] cunix left_behind got %d %d %d %d %d\n", missing, unheard, rebind, closed, + after); + return fail("left_behind: ENOENT, ECONNREFUSED, EADDRINUSE, ECONNREFUSED, 0", 0, 0); + } + ok("left_behind", "the path stays until unlink; rebind after it", after); + return 0; +} + +static int abstract_dgram(void) { + struct sockaddr_un x, y, b, g; + socklen_t xl = name(&x, "cunix-x", 1), yl = name(&y, "cunix-y", 1), gl = name(&g, GONE, 0); + int rx = socket(AF_UNIX, SOCK_DGRAM, 0), tx = socket(AF_UNIX, SOCK_DGRAM, 0); + char buf[8]; + socklen_t bl = sizeof b; + if (bind(rx, (void *)&x, xl) || getsockname(rx, (void *)&b, &bl) || bl != xl) { + return fail("abstract_dgram: bind and name", bl, xl); + } + sendto(tx, "a", 1, 0, (void *)&x, xl); + bl = sizeof b; + long got = recvfrom(rx, buf, 8, 0, (void *)&b, &bl); + if (got != 1 || bl != 0) { + return fail("abstract_dgram: an unnamed sender reports length 0", got, bl); + } + bind(tx, (void *)&y, yl); + sendto(tx, "b", 1, 0, (void *)&x, xl); + bl = sizeof b; + recvfrom(rx, buf, 8, 0, (void *)&b, &bl); + if (bl != yl || memcmp(b.sun_path, y.sun_path, yl - 2)) { + return fail("abstract_dgram: a named sender reports its name", bl, yl); + } + int missing = sendto(tx, "c", 1, 0, (void *)&g, gl) < 0 ? errno : 0; + int lone = socket(AF_UNIX, SOCK_DGRAM, 0); + int nopeer = send(lone, "d", 1, 0) < 0 ? errno : 0; + close(lone); + close(rx); + close(tx); + if (missing != ENOENT || nopeer != ENOTCONN) { + return fail("abstract_dgram: ENOENT, then ENOTCONN", missing, nopeer); + } + ok("abstract_dgram", "names reported, no peer errno", nopeer); + return 0; +} diff --git a/userland/linux_guests/c/cunix_4.h b/userland/linux_guests/c/cunix_4.h new file mode 100644 index 000000000..c68289a96 --- /dev/null +++ b/userland/linux_guests/c/cunix_4.h @@ -0,0 +1,70 @@ +/* cunix, part 4 of 4: included once, by cunix.c. */ + +/* cunix: a connected datagram socket, a chosen name, and fork. */ + +static int connected_dgram(void) { + struct sockaddr_un r, a; + socklen_t rl = name(&r, "cunix-r", 1), al = name(&a, "cunix-a", 1); + int rs = socket(AF_UNIX, SOCK_DGRAM, 0), as = socket(AF_UNIX, SOCK_DGRAM, 0); + int stranger = socket(AF_UNIX, SOCK_DGRAM, 0); + bind(rs, (void *)&r, rl); + bind(as, (void *)&a, al); + /* r talks only to a: a stranger's datagram to r is refused. */ + connect(rs, (void *)&a, al); + int refused = sendto(stranger, "s", 1, 0, (void *)&r, rl) < 0 ? errno : 0; + long sent = send(rs, "to-a", 4, 0); + char buf[8]; + long got = recv(as, buf, 8, 0); + close(rs); + close(as); + close(stranger); + if (refused != EPERM || sent != 4 || got != 4) { + return fail("connected_dgram: EPERM for a stranger, 4 bytes to the peer", refused, got); + } + ok("connected_dgram", "a stranger refused, errno", refused); + return 0; +} + +static int autobind(void) { + struct sockaddr_un a, b; + int s = socket(AF_UNIX, SOCK_DGRAM, 0); + memset(&a, 0, sizeof a); + a.sun_family = AF_UNIX; + socklen_t bl = sizeof b; + int rc = bind(s, (void *)&a, sizeof(sa_family_t)); + getsockname(s, (void *)&b, &bl); + close(s); + /* Linux chooses a NUL and five hex digits. */ + if (rc || bl != 8 || b.sun_path[0] != 0) { + return fail("autobind: a NUL and five hex digits", rc, bl); + } + ok("autobind", "a chosen abstract name, length", bl); + return 0; +} + +static int across_fork(void) { + struct sockaddr_un a; + socklen_t l = name(&a, PATH, 0); + unlink(PATH); + int s = socket(AF_UNIX, SOCK_STREAM, 0); + bind(s, (void *)&a, l); + listen(s, 1); + pid_t kid = fork(); + if (kid == 0) { + int c = socket(AF_UNIX, SOCK_STREAM, 0); + _exit(connect(c, (void *)&a, l) || write(c, "kid", 3) != 3); + } + int t = accept(s, 0, 0); + char buf[8]; + long got = read(t, buf, 8); + int status = -1; + waitpid(kid, &status, 0); + close(t); + close(s); + unlink(PATH); + if (got != 3 || status != 0) { + return fail("across_fork: the child's connection and bytes", got, status); + } + ok("across_fork", "bytes from the child", got); + return 0; +} diff --git a/userland/linux_guests/go/http/go.mod b/userland/linux_guests/go/http/go.mod new file mode 100644 index 000000000..0fe31ea2f --- /dev/null +++ b/userland/linux_guests/go/http/go.mod @@ -0,0 +1,3 @@ +module nonos/guest/http + +go 1.24 diff --git a/userland/linux_guests/go/http/main.go b/userland/linux_guests/go/http/main.go new file mode 100644 index 000000000..ea13e86b2 --- /dev/null +++ b/userland/linux_guests/go/http/main.go @@ -0,0 +1,56 @@ +/* + * net/http as Linux runs it, inside one guest: a server on 127.0.0.1:0 and a + * client of it. Twenty GETs, each answered 200 with the path it asked for, + * over one kept-alive connection, which the server's own count of new + * connections shows. + */ +package main + +import ( + "fmt" + "io" + "net" + "net/http" + "os" + "sync/atomic" +) + +func main() { + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + fmt.Println("[GO] gohttp FAIL: listen:", err) + os.Exit(1) + } + var conns atomic.Int32 + srv := &http.Server{ + Handler: http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + io.WriteString(w, "hello "+r.URL.Path) + }), + ConnState: func(_ net.Conn, s http.ConnState) { + if s == http.StateNew { + conns.Add(1) + } + }, + } + go srv.Serve(ln) + client := &http.Client{} + const gets = 20 + for i := 0; i < gets; i++ { + r, err := client.Get(fmt.Sprintf("http://%s/%d", ln.Addr(), i)) + if err != nil { + fmt.Println("[GO] gohttp FAIL: get", i, err) + os.Exit(1) + } + body, err := io.ReadAll(r.Body) + r.Body.Close() + if err != nil || r.StatusCode != 200 || string(body) != fmt.Sprintf("hello /%d", i) { + fmt.Println("[GO] gohttp FAIL: reply", i, r.StatusCode, string(body), err) + os.Exit(1) + } + } + if n := conns.Load(); n != 1 { + fmt.Println("[GO] gohttp FAIL: keep-alive: connections", n) + os.Exit(1) + } + fmt.Printf("[GO] gohttp PASS: %d GETs answered 200 over %d connection\n", gets, conns.Load()) +}