diff --git a/userland/capsule_linux/src/linux/abi/errno.rs b/userland/capsule_linux/src/linux/abi/errno.rs
index 448c49fa7..188d8ca83 100644
--- a/userland/capsule_linux/src/linux/abi/errno.rs
+++ b/userland/capsule_linux/src/linux/abi/errno.rs
@@ -16,6 +16,8 @@
//! Linux errno values, and the convention for returning them.
+pub use super::errno_sock::*;
+
pub const EPERM: i64 = 1;
pub const ENOENT: i64 = 2;
pub const EINTR: i64 = 4;
diff --git a/userland/capsule_linux/src/linux/abi/errno_sock.rs b/userland/capsule_linux/src/linux/abi/errno_sock.rs
new file mode 100644
index 000000000..d06e632ef
--- /dev/null
+++ b/userland/capsule_linux/src/linux/abi/errno_sock.rs
@@ -0,0 +1,33 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Linux errno values the socket calls answer with, from
+//! include/uapi/asm-generic/errno-base.h and errno.h.
+
+pub const EDOM: i64 = 33;
+pub const EDESTADDRREQ: i64 = 89;
+pub const EMSGSIZE: i64 = 90;
+pub const EPROTOTYPE: i64 = 91;
+pub const ENOPROTOOPT: i64 = 92;
+pub const EPROTONOSUPPORT: i64 = 93;
+pub const ESOCKTNOSUPPORT: i64 = 94;
+pub const EOPNOTSUPP: i64 = 95;
+pub const EADDRINUSE: i64 = 98;
+pub const EADDRNOTAVAIL: i64 = 99;
+pub const ENETUNREACH: i64 = 101;
+pub const EISCONN: i64 = 106;
+pub const ECONNABORTED: i64 = 103;
+pub const EALREADY: i64 = 114;
diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs
index 68e04090a..319e970d7 100644
--- a/userland/capsule_linux/src/linux/abi/mod.rs
+++ b/userland/capsule_linux/src/linux/abi/mod.rs
@@ -19,8 +19,10 @@
#![allow(dead_code)]
pub mod errno;
+pub mod errno_sock;
pub mod name;
pub mod nr;
-pub mod nr_path;
pub mod nr_high;
+pub mod nr_path;
pub mod nr_sched;
+pub mod nr_sock;
diff --git a/userland/capsule_linux/src/linux/abi/nr.rs b/userland/capsule_linux/src/linux/abi/nr.rs
index 1e800243b..b489413e7 100644
--- a/userland/capsule_linux/src/linux/abi/nr.rs
+++ b/userland/capsule_linux/src/linux/abi/nr.rs
@@ -14,11 +14,11 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-
//! Linux x86_64 syscall numbers, by family.
pub use super::nr_high::*;
pub use super::nr_sched::*;
+pub use super::nr_sock::*;
pub const READ: u64 = 0;
pub const WRITE: u64 = 1;
diff --git a/userland/capsule_linux/src/linux/abi/nr_sock.rs b/userland/capsule_linux/src/linux/abi/nr_sock.rs
new file mode 100644
index 000000000..3a8a13fba
--- /dev/null
+++ b/userland/capsule_linux/src/linux/abi/nr_sock.rs
@@ -0,0 +1,28 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Linux x86_64 syscall numbers for sockets, from
+//! arch/x86/entry/syscalls/syscall_64.tbl. Same contract as `nr`.
+
+pub const BIND: u64 = 49;
+pub const LISTEN: u64 = 50;
+pub const GETSOCKNAME: u64 = 51;
+pub const GETPEERNAME: u64 = 52;
+pub const SOCKETPAIR: u64 = 53;
+pub const SETSOCKOPT: u64 = 54;
+pub const GETSOCKOPT: u64 = 55;
+pub const RECVMMSG: u64 = 299;
+pub const SENDMMSG: u64 = 307;
diff --git a/userland/capsule_linux/src/linux/call/io.rs b/userland/capsule_linux/src/linux/call/io.rs
index f5145519b..07d8a2725 100644
--- a/userland/capsule_linux/src/linux/call/io.rs
+++ b/userland/capsule_linux/src/linux/call/io.rs
@@ -59,8 +59,6 @@ pub fn read(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 {
}
pub fn close(guest: &mut Guest, fd: u64) -> u64 {
- if let Some(h) = guest.socket_handle(fd) {
- net::close(h);
- }
+ net::close(guest, fd);
file::close(guest, fd)
}
diff --git a/userland/capsule_linux/src/linux/guest/fork_state.rs b/userland/capsule_linux/src/linux/guest/fork_state.rs
index 54b26502a..39074ba25 100644
--- a/userland/capsule_linux/src/linux/guest/fork_state.rs
+++ b/userland/capsule_linux/src/linux/guest/fork_state.rs
@@ -46,6 +46,7 @@ impl Guest {
g.sid = self.sid;
g.umask = self.umask;
g.links = self.links.clone();
+ self.sockets.fork(child, &self.fds);
g
}
}
diff --git a/userland/capsule_linux/src/linux/guest/handle.rs b/userland/capsule_linux/src/linux/guest/handle.rs
index 1f12ca784..8aaedd7c8 100644
--- a/userland/capsule_linux/src/linux/guest/handle.rs
+++ b/userland/capsule_linux/src/linux/guest/handle.rs
@@ -86,4 +86,6 @@ pub struct Guest {
pub blocked: Vec,
/// The image's symbolic links, read once and shared by the family.
pub links: alloc::rc::Rc,
+ /// Lets go of this process's family sockets when it is dropped (net::sock).
+ pub sockets: crate::linux::net::sock::Holder,
}
diff --git a/userland/capsule_linux/src/linux/guest/handle_new.rs b/userland/capsule_linux/src/linux/guest/handle_new.rs
index 0d8b798db..103c11717 100644
--- a/userland/capsule_linux/src/linux/guest/handle_new.rs
+++ b/userland/capsule_linux/src/linux/guest/handle_new.rs
@@ -61,6 +61,7 @@ impl Guest {
sleepers: Vec::new(),
blocked: Vec::new(),
links: Default::default(),
+ sockets: crate::linux::net::sock::Holder::new(pid),
}
}
}
diff --git a/userland/capsule_linux/src/linux/heap.rs b/userland/capsule_linux/src/linux/heap.rs
index b669aa944..e61776739 100644
--- a/userland/capsule_linux/src/linux/heap.rs
+++ b/userland/capsule_linux/src/linux/heap.rs
@@ -21,9 +21,14 @@ use nonos_libc::{heap_init, heap_init_sized, mk_args};
/// An install holds a distribution's index while it resolves a closure.
/// Kali's main is 21 MB fetched and 85 MB inflated, parsed into records
-/// beside it; Alpine's is a few. A run takes the default.
+/// beside it; Alpine's is a few.
const INSTALL_HEAP: usize = 320 << 20;
+/// A run holds the program it loads, read whole from the store, beside the
+/// family's own state. A 6 MB Go program outgrew the 16 MiB default while it
+/// was read; this is the most a program may be (`source::MAX_IMAGE`).
+const RUN_HEAP: usize = 64 << 20;
+
pub fn init() {
let mut buf = [0u8; 256];
let n = mk_args(buf.as_mut_ptr(), buf.len());
@@ -35,6 +40,8 @@ pub fn init() {
// it is read, with that reason, instead of here without one.
let line = b"[LINUX] no room for a large index, installing in the default heap\n";
let _ = nonos_libc::mk_debug(line.as_ptr(), line.len());
+ } else if heap_init_sized(RUN_HEAP).is_ok() {
+ return;
}
let _ = heap_init();
}
diff --git a/userland/capsule_linux/src/linux/net/accept.rs b/userland/capsule_linux/src/linux/net/accept.rs
new file mode 100644
index 000000000..bbf6f926c
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/accept.rs
@@ -0,0 +1,68 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `accept` and `accept4`: the oldest connection a listener has queued, as
+//! a new descriptor held by the caller alone.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::close::discard;
+use super::fd::{install, sock_of, SOCK_CLOEXEC, SOCK_NONBLOCK};
+use super::sock::{self, Domain, Peer, Proto};
+
+pub fn accept4(guest: &mut Guest, fd: u64, at: u64, lenp: u64, flags: u64) -> u64 {
+ if flags & !(SOCK_NONBLOCK | SOCK_CLOEXEC) != 0 {
+ return errno::fail(errno::EINVAL);
+ }
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ let pid = guest.pid;
+ let taken = sock::with(|t| {
+ let s = t.get_mut(id).ok_or(errno::EBADF)?;
+ if s.proto == Proto::Dgram {
+ return Err(errno::EOPNOTSUPP);
+ }
+ if !s.listening {
+ return Err(errno::EINVAL);
+ }
+ let child = s.pending.pop_front().ok_or(errno::EAGAIN)?;
+ t.make_room(id);
+ let c = t.get_mut(child).ok_or(errno::ECONNABORTED)?;
+ c.holders.push(pid);
+ let from = match c.domain {
+ Domain::Inet => Peer::Inet(c.remote.unwrap_or_default()),
+ Domain::Unix => Peer::Unix(c.upeer.clone()),
+ };
+ Ok((child, from))
+ });
+ let (child, from) = match taken {
+ Ok(v) => v,
+ Err(e) => return errno::fail(e),
+ };
+ let n = install(guest, child, flags);
+ let Some(slot) = errno::slot(n) else {
+ return n;
+ };
+ let wrote = super::sockaddr_out::write(guest, at, lenp, &from);
+ if errno::slot(wrote).is_none() {
+ discard(guest, slot as u64);
+ return wrote;
+ }
+ n
+}
diff --git a/userland/capsule_linux/src/linux/net/api.rs b/userland/capsule_linux/src/linux/net/api.rs
new file mode 100644
index 000000000..455637135
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/api.rs
@@ -0,0 +1,42 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! What the rest of the personality calls on sockets.
+
+pub use super::accept::accept4;
+pub use super::bind::bind;
+pub use super::call_kind::{flags as call_flags, wants_all};
+pub use super::close::close;
+pub use super::connect::connect;
+pub use super::dgram::sendto;
+pub use super::fd::{is_stream, sock_id};
+pub use super::listen::listen;
+pub use super::mmsg::{recvmmsg, sendmmsg};
+pub use super::msg::sendmsg;
+pub use super::msg_recv::recvmsg;
+pub use super::name::{getpeername, getsockname};
+pub use super::opt::{getsockopt, limit_ms, setsockopt};
+pub use super::pair::socketpair;
+pub use super::poll::{ready, POLLERR, POLLHUP};
+pub use super::poll_set::poll;
+pub use super::poll_socket::outside;
+pub use super::recvfrom::recvfrom;
+pub use super::select::{clear as select_clear, select};
+pub use super::shutdown::shutdown;
+pub use super::socket::socket;
+pub use super::try_call::try_call;
+pub use super::xfer_in::read as recv;
+pub use super::xfer_out::write as send;
diff --git a/userland/capsule_linux/src/linux/net/bind.rs b/userland/capsule_linux/src/linux/net/bind.rs
new file mode 100644
index 000000000..b8e002b26
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/bind.rs
@@ -0,0 +1,65 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `bind` and `listen`, on 127.0.0.0/8 only (`policy`).
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::fd::sock_of;
+use super::policy::not_loopback;
+use super::sock::{self, Domain};
+use super::sockaddr::{self, is_loopback, AF_INET};
+
+pub fn bind(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 {
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ if sock::with(|t| t.get(id).is_some_and(|s| s.domain == Domain::Unix)) {
+ return super::named::bind(guest, id, at, len);
+ }
+ let (family, mut want) = match sockaddr::read(guest, at, len) {
+ Ok(v) => v,
+ Err(e) => return e,
+ };
+ if family != AF_INET {
+ return errno::fail(errno::EAFNOSUPPORT);
+ }
+ if !is_loopback(want.ip) {
+ return not_loopback("bind", want);
+ }
+ sock::with(|t| {
+ let Some(s) = t.get(id) else {
+ return errno::fail(errno::EBADF);
+ };
+ if s.domain != Domain::Inet || s.local.is_some() || s.svc.is_some() {
+ return errno::fail(errno::EINVAL);
+ }
+ if want.port == 0 {
+ match t.ephemeral(s.proto, want.ip) {
+ Some(port) => want.port = port,
+ None => return errno::fail(errno::EADDRINUSE),
+ }
+ } else if t.in_use(id, want) {
+ return errno::fail(errno::EADDRINUSE);
+ }
+ if let Some(s) = t.get_mut(id) {
+ s.local = Some(want);
+ }
+ errno::ok(0)
+ })
+}
diff --git a/userland/capsule_linux/src/linux/net/call_kind.rs b/userland/capsule_linux/src/linux/net/call_kind.rs
new file mode 100644
index 000000000..6a28e0a56
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/call_kind.rs
@@ -0,0 +1,61 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! What kind of wait a socket call makes: its flags, and whether a
+//! blocking one waits until it has moved everything.
+
+use crate::linux::abi::{errno, nr};
+use crate::linux::guest::Guest;
+
+use super::flags::{MSG_DONTWAIT, MSG_WAITALL};
+use super::{iov, mmsg};
+
+/// The call's flags, where it has them.
+pub fn flags(n: u64, a: [u64; 6]) -> u64 {
+ match n {
+ nr::RECVFROM | nr::SENDTO | nr::RECVMMSG | nr::SENDMMSG | nr::ACCEPT4 => a[3],
+ nr::RECVMSG | nr::SENDMSG => a[2],
+ _ => 0,
+ }
+}
+
+/// True when a blocking call waits until it has moved everything it asked
+/// for: a send on a stream, a receive with MSG_WAITALL, and recvmmsg
+/// without MSG_WAITFORONE.
+pub fn wants_all(stream: bool, n: u64, flags: u64) -> bool {
+ match n {
+ nr::WRITE | nr::WRITEV | nr::SENDTO | nr::SENDMSG => stream,
+ nr::RECVFROM | nr::RECVMSG => stream && flags & MSG_WAITALL != 0,
+ nr::RECVMMSG => flags & (mmsg::MSG_WAITFORONE | MSG_DONTWAIT) == 0,
+ _ => false,
+ }
+}
+
+/// A receive's answer: the count, or the errno.
+pub(super) fn bytes_in(guest: &Guest, id: u32, v: &iov::Iov, done: usize, flags: u64) -> u64 {
+ match super::xfer_in::recv(guest, id, v, done, flags) {
+ Ok(got) => errno::ok(got.n as u64),
+ Err(e) => e,
+ }
+}
+
+/// The bytes a msghdr's iovecs ask for.
+pub(super) fn msg_len(guest: &Guest, msg: u64) -> usize {
+ let word = |at: u64| {
+ guest.read(at, 8).map_or(0, |b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8])))
+ };
+ iov::read(guest, word(msg + 16), word(msg + 24)).map_or(0, |v| iov::total(&v))
+}
diff --git a/userland/capsule_linux/src/linux/net/cap.rs b/userland/capsule_linux/src/linux/net/cap.rs
new file mode 100644
index 000000000..e51aa5d03
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/cap.rs
@@ -0,0 +1,33 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! How many bytes one send takes from the guest.
+
+/// What one send takes from the guest at most: the default receive buffer
+/// of a stream's peer, so a single call can fill it.
+const STREAM_CAP: usize = 128 << 10;
+/// One byte past the largest datagram, so a larger one is seen and refused.
+const GRAM_CAP: usize = 65508;
+
+/// The most one send gathers: what net.sockets carries in a call, a
+/// stream peer's default queue, or one byte past the largest datagram.
+pub fn cap(outside: bool, stream: bool) -> usize {
+ match (outside, stream) {
+ (true, _) => super::stream::MAX_IO,
+ (false, true) => STREAM_CAP,
+ (false, false) => GRAM_CAP,
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/close.rs b/userland/capsule_linux/src/linux/net/close.rs
new file mode 100644
index 000000000..0dbf62ee9
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/close.rs
@@ -0,0 +1,45 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Closing a socket descriptor. The socket goes with the last descriptor
+//! naming it in the last process holding it.
+
+use crate::linux::guest::{Guest, Kind};
+
+use super::fd::sock_of;
+use super::sock;
+
+/// Called by close before it clears `fd`: this process lets go of the
+/// socket unless another of its descriptors still names it.
+pub fn close(guest: &mut Guest, fd: u64) {
+ let Ok(id) = sock_of(guest, fd) else {
+ return;
+ };
+ let named_again = guest
+ .fds
+ .iter()
+ .enumerate()
+ .any(|(i, f)| i as u64 != fd && f.kind == Kind::Socket && f.handle == id);
+ if !named_again {
+ sock::with(|t| t.release(id, guest.pid));
+ }
+}
+
+/// Close a descriptor this module opened and cannot hand out after all.
+pub(super) fn discard(guest: &mut Guest, fd: u64) {
+ close(guest, fd);
+ let _ = crate::linux::file::close(guest, fd);
+}
diff --git a/userland/capsule_linux/src/linux/net/connect.rs b/userland/capsule_linux/src/linux/net/connect.rs
index 51f727956..8d630bc56 100644
--- a/userland/capsule_linux/src/linux/net/connect.rs
+++ b/userland/capsule_linux/src/linux/net/connect.rs
@@ -14,54 +14,38 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-//! `connect`, for a socket the guest opened here.
-
-use alloc::vec::Vec;
+//! `connect`. On 127.0.0.0/8 the family links the two ends in the caller's
+//! own call, as Linux's loopback does, and a non-blocking socket answers
+//! EINPROGRESS all the same; anywhere else a stream goes over the mixnet.
use crate::linux::abi::errno;
use crate::linux::guest::{Guest, Kind};
-use super::addr::inet;
-use super::call::call;
-use super::dns::host_for;
-use super::ops::{OP_CONNECT, OP_CONNECT_HOST};
+use super::fd::{nonblock, sock_of};
+use super::sock::{self, Domain, Proto};
+use super::sockaddr::{self, is_loopback, AF_INET};
pub fn connect(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 {
- /*
- * A program may connect its nameserver socket before writing to
- * it. There is nothing to reach: this capsule is the nameserver.
- */
+ /* This capsule is the nameserver, so its socket has nothing to reach. */
if guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Resolver) {
return errno::ok(0);
}
- let Some(handle) = guest.socket_handle(fd) else {
- return errno::fail(errno::ENOTSOCK);
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
};
- let Some((port, ip)) = inet(guest, at, len) else {
- return errno::fail(errno::EAFNOSUPPORT);
+ let (family, to) = match sockaddr::read(guest, at, len) {
+ Ok(v) => v,
+ Err(e) => return e,
};
- // An address this capsule invented for a name goes back to being the name.
- if let Some(host) = host_for(guest, ip) {
- return by_host(handle, &host, port);
- }
- let mut body = Vec::with_capacity(10);
- body.extend_from_slice(&handle.to_le_bytes());
- body.extend_from_slice(&ip);
- body.extend_from_slice(&port.to_le_bytes());
- match call(OP_CONNECT, &body, 0) {
- Some((0, _)) => errno::ok(0),
- Some(_) => errno::fail(errno::ECONNREFUSED),
- None => errno::fail(errno::EIO),
- }
-}
-
-fn by_host(handle: u32, host: &[u8], port: u16) -> u64 {
- let Some(body) = super::host_body::host_body(handle, port, host) else {
- return errno::fail(errno::EINVAL);
+ let Some((proto, domain)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.domain))) else {
+ return errno::fail(errno::EBADF);
};
- match call(OP_CONNECT_HOST, &body, 0) {
- Some((0, _)) => errno::ok(0),
- Some(_) => errno::fail(errno::ECONNREFUSED),
- None => errno::fail(errno::EIO),
+ match (proto, domain) {
+ (_, Domain::Unix) => super::named::connect(guest, fd, id, proto, at, len),
+ (Proto::Dgram, _) => super::connect_dgram::connect(guest, fd, id, family, to),
+ _ if family != AF_INET => errno::fail(errno::EAFNOSUPPORT),
+ _ if is_loopback(to.ip) => super::connect_lo::loopback(id, to, nonblock(guest, fd)),
+ _ => super::connect_out::connect(guest, id, to),
}
}
diff --git a/userland/capsule_linux/src/linux/net/connect_dgram.rs b/userland/capsule_linux/src/linux/net/connect_dgram.rs
new file mode 100644
index 000000000..ee74eb7f9
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/connect_dgram.rs
@@ -0,0 +1,51 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `connect` on a datagram socket: it only names where sends go and whose
+//! datagrams are kept. AF_UNSPEC forgets it again.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::policy::refuse_out;
+use super::sock::{self, Addr};
+use super::sockaddr::{is_loopback, AF_INET, AF_UNSPEC};
+
+pub fn connect(guest: &mut Guest, fd: u64, id: u32, family: u16, to: Addr) -> u64 {
+ if family == AF_UNSPEC {
+ sock::with(|t| t.get_mut(id).map(|s| s.remote = None));
+ return errno::ok(0);
+ }
+ if family != AF_INET {
+ return errno::fail(errno::EAFNOSUPPORT);
+ }
+ if super::resolver::is_nameserver(to) {
+ super::resolver::become_resolver(guest, fd);
+ return errno::ok(0);
+ }
+ if !is_loopback(to.ip) {
+ return refuse_out("connect", to);
+ }
+ sock::with(|t| {
+ t.autobind(id)?;
+ if let Some(s) = t.get_mut(id) {
+ s.remote = Some(to);
+ s.error = 0;
+ }
+ Ok(())
+ })
+ .map_or_else(errno::fail, |()| errno::ok(0))
+}
diff --git a/userland/capsule_linux/src/linux/net/connect_dial.rs b/userland/capsule_linux/src/linux/net/connect_dial.rs
new file mode 100644
index 000000000..6febedff5
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/connect_dial.rs
@@ -0,0 +1,57 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The two calls a stream outside the family makes to net.sockets: a
+//! mixnet socket, and a connect to an address or to the name this capsule
+//! invented it for.
+
+use alloc::vec::Vec;
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::call::call;
+use super::dns::host_for;
+use super::ops::{DOMAIN, KIND_MIXNET, OP_CONNECT, OP_CONNECT_HOST, OP_SOCKET};
+use super::sock::Addr;
+
+pub fn open() -> Result {
+ let mut body = Vec::with_capacity(4);
+ body.extend_from_slice(&DOMAIN.to_le_bytes());
+ body.extend_from_slice(&KIND_MIXNET.to_le_bytes());
+ match call(OP_SOCKET, &body, 8) {
+ Some((0, out)) if out.len() >= 4 => {
+ Ok(u32::from_le_bytes([out[0], out[1], out[2], out[3]]))
+ }
+ Some(_) => Err(errno::fail(errno::ENOMEM)),
+ None => Err(errno::fail(errno::EIO)),
+ }
+}
+
+/// The service's answer to a connect; None when it did not answer.
+pub fn dial(guest: &Guest, handle: u32, to: Addr) -> Result)>, u64> {
+ /* An address this capsule invented for a name goes back to being the name. */
+ if let Some(host) = host_for(guest, to.ip) {
+ let body = super::host_body::host_body(handle, to.port, &host)
+ .ok_or(errno::fail(errno::EINVAL))?;
+ return Ok(call(OP_CONNECT_HOST, &body, 0));
+ }
+ let mut body = Vec::with_capacity(10);
+ body.extend_from_slice(&handle.to_le_bytes());
+ body.extend_from_slice(&to.ip);
+ body.extend_from_slice(&to.port.to_le_bytes());
+ Ok(call(OP_CONNECT, &body, 0))
+}
diff --git a/userland/capsule_linux/src/linux/net/connect_lo.rs b/userland/capsule_linux/src/linux/net/connect_lo.rs
new file mode 100644
index 000000000..53b1a0706
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/connect_lo.rs
@@ -0,0 +1,69 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A stream connect to 127.0.0.0/8, which the family completes in the
+//! caller's own call, as Linux's loopback does.
+
+use crate::linux::abi::errno;
+
+use super::sock::{self, Addr, Link};
+
+pub fn loopback(id: u32, to: Addr, nonblock: bool) -> u64 {
+ sock::with(|t| {
+ let Some(s) = t.get_mut(id) else {
+ return errno::fail(errno::EBADF);
+ };
+ if s.connecting {
+ return errno::fail(errno::EALREADY);
+ }
+ if s.connected || s.listening || s.svc.is_some() {
+ return errno::fail(errno::EISCONN);
+ }
+ s.error = 0;
+ let bound_here = s.local.is_none();
+ if let Err(e) = t.autobind(id) {
+ return errno::fail(e);
+ }
+ let linked = t.link(id, to);
+ /*
+ * A full listener keeps a non-blocking connect until accept makes
+ * room, as Linux's SYN_SENT does.
+ */
+ let from = t.get(id).and_then(|s| s.local).map_or(0, |a| a.port);
+ if let (Link::Full, true, Some(l)) = (&linked, nonblock, t.listener(to, from)) {
+ t.wait_room(id, l);
+ return errno::fail(errno::EINPROGRESS);
+ }
+ /* A connect that fails gives back the port it bound. */
+ if !matches!(linked, Link::Done) && bound_here {
+ if let Some(s) = t.get_mut(id) {
+ s.local = None;
+ }
+ }
+ match linked {
+ Link::Done if nonblock => errno::fail(errno::EINPROGRESS),
+ Link::Done => errno::ok(0),
+ Link::Refused if nonblock => {
+ if let Some(s) = t.get_mut(id) {
+ s.error = errno::ECONNREFUSED;
+ }
+ errno::fail(errno::EINPROGRESS)
+ }
+ Link::Refused => errno::fail(errno::ECONNREFUSED),
+ Link::Full => errno::fail(errno::EAGAIN),
+ }
+ })
+}
diff --git a/userland/capsule_linux/src/linux/net/connect_out.rs b/userland/capsule_linux/src/linux/net/connect_out.rs
new file mode 100644
index 000000000..d6a73f635
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/connect_out.rs
@@ -0,0 +1,60 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A stream to an address outside the family. It goes over the mixnet,
+//! never the open network, and the guest holds no capability that could
+//! name a socket: there is no second route to disable and no firewall rule
+//! to remove. net.sockets holds the stream; the family's entry names it.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::connect_dial::{dial, open};
+use super::sock::{self, Addr};
+
+pub fn connect(guest: &Guest, id: u32, to: Addr) -> u64 {
+ if sock::with(|t| t.get(id).is_some_and(|s| s.connected || s.listening || s.svc.is_some())) {
+ return errno::fail(errno::EISCONN);
+ }
+ let handle = match open() {
+ Ok(h) => h,
+ Err(e) => return e,
+ };
+ let status = match dial(guest, handle, to) {
+ Ok(s) => s,
+ Err(e) => {
+ super::stream::close(handle);
+ return e;
+ }
+ };
+ let answer = match status {
+ Some((0, _)) => errno::ok(0),
+ Some(_) => errno::fail(errno::ECONNREFUSED),
+ None => errno::fail(errno::EIO),
+ };
+ if answer != 0 {
+ super::stream::close(handle);
+ return answer;
+ }
+ sock::with(|t| {
+ if let Some(s) = t.get_mut(id) {
+ s.svc = Some(handle);
+ s.remote = Some(to);
+ s.connected = true;
+ }
+ });
+ answer
+}
diff --git a/userland/capsule_linux/src/linux/net/dest.rs b/userland/capsule_linux/src/linux/net/dest.rs
new file mode 100644
index 000000000..6e5bfd9f0
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/dest.rs
@@ -0,0 +1,43 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Where a send goes: nowhere but the peer for a stream, and for a datagram
+//! the address or the Unix name it names, found now.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::peer_addr::To;
+use super::sock::{Dest, Domain, Proto};
+
+pub fn dest(guest: &Guest, proto: Proto, domain: Domain, to: Option) -> Result {
+ Ok(match (proto, to) {
+ (Proto::Stream, _) | (_, None) => Dest::Default,
+ (_, Some(To::Inet(a))) => Dest::Inet(a),
+ (_, Some(To::Unix(_))) if domain == Domain::Inet => {
+ return Err(errno::fail(errno::EAFNOSUPPORT))
+ }
+ (_, Some(To::Unix(ua))) => {
+ let found = super::named::resolve(guest, &ua)
+ .ok_or(errno::EINVAL)
+ .and_then(|name| super::named::find(&name, Proto::Dgram));
+ match found {
+ Ok(t) => Dest::Sock(t),
+ Err(e) => return Err(errno::fail(e)),
+ }
+ }
+ })
+}
diff --git a/userland/capsule_linux/src/linux/net/dgram.rs b/userland/capsule_linux/src/linux/net/dgram.rs
index dd3b31bd5..ab32f9bc2 100644
--- a/userland/capsule_linux/src/linux/net/dgram.rs
+++ b/userland/capsule_linux/src/linux/net/dgram.rs
@@ -14,39 +14,55 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-//! `sendto` and `recvfrom`, which differ from write and read only in carrying
-//! an address.
+//! `sendto`, which differs from write only in carrying an address.
-use crate::linux::call;
-use crate::linux::guest::{Guest, Kind};
+use alloc::vec;
-use super::addr::inet;
-use super::dgram_addr::{encode, fill};
-use super::dns;
+use crate::linux::guest::{Guest, Kind};
-pub fn sendto(guest: &mut Guest, fd: u64, buf: u64, len: u64, at: u64, alen: u64) -> u64 {
- if !is_resolver(guest, fd) {
- return call::write(guest, fd, buf, len);
- }
- /*
- * A program with no `resolv.conf` asks the loopback address, and one with
- * a configured nameserver asks that.
- */
- let peer = inet(guest, at, alen).unwrap_or((53, [127, 0, 0, 1]));
- dns::query(guest, fd, buf, len, encode(peer))
-}
+use super::fd::sock_of;
+use super::peer_addr::To;
+use super::sock::{self, Domain, Proto};
+use super::sockaddr::is_loopback;
-pub fn recvfrom(guest: &mut Guest, fd: u64, buf: u64, len: u64, at: u64, alen: u64) -> u64 {
+pub fn sendto(
+ guest: &mut Guest,
+ fd: u64,
+ buf: u64,
+ len: u64,
+ flags: u64,
+ at: u64,
+ alen: u64,
+) -> u64 {
+ let to = match super::peer_addr::address(guest, at, alen) {
+ Ok(to) => to,
+ Err(e) => return e,
+ };
if !is_resolver(guest, fd) {
- return call::read(guest, fd, buf, len);
- }
- let (got, from) = dns::answer_out(guest, fd, buf, len);
- match from {
- Some(peer) if at != 0 => fill(guest, at, alen, peer, got),
- _ => got,
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ let inet_dgram = sock::with(|t| {
+ t.get(id).is_some_and(|s| s.proto == Proto::Dgram && s.domain == Domain::Inet)
+ });
+ match to {
+ Some(To::Inet(a)) if inet_dgram && super::resolver::is_nameserver(a) => {
+ super::resolver::become_resolver(guest, fd)
+ }
+ Some(To::Inet(a)) if inet_dgram && !is_loopback(a.ip) => {
+ return super::policy::refuse_out("sendto", a)
+ }
+ to => return super::xfer_out::send(guest, id, &vec![(buf, len)], 0, flags, to),
+ }
}
+ let to = match to {
+ Some(To::Inet(a)) => Some(a),
+ _ => None,
+ };
+ super::resolver::query(guest, fd, buf, len, to)
}
-fn is_resolver(guest: &Guest, fd: u64) -> bool {
+pub(super) fn is_resolver(guest: &Guest, fd: u64) -> bool {
guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Resolver)
}
diff --git a/userland/capsule_linux/src/linux/net/dns/mod.rs b/userland/capsule_linux/src/linux/net/dns/mod.rs
index 23b9adad3..c530ba42e 100644
--- a/userland/capsule_linux/src/linux/net/dns/mod.rs
+++ b/userland/capsule_linux/src/linux/net/dns/mod.rs
@@ -19,10 +19,8 @@
mod automap;
mod decide;
mod name;
-mod open;
mod reply;
mod serve;
pub use automap::host_for;
-pub use open::open;
pub use serve::{answer_out, query};
diff --git a/userland/capsule_linux/src/linux/net/fd.rs b/userland/capsule_linux/src/linux/net/fd.rs
new file mode 100644
index 000000000..1da8a45e2
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/fd.rs
@@ -0,0 +1,70 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! From a descriptor to the family socket it names, and back.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::{Fd, Guest, Kind};
+
+use super::sock;
+
+/// SOCK_NONBLOCK and SOCK_CLOEXEC, which socket, socketpair and accept4
+/// take with the type or as flags: O_NONBLOCK and O_CLOEXEC's values.
+pub const SOCK_NONBLOCK: u64 = 0o4000;
+pub const SOCK_CLOEXEC: u64 = 0o2_000_000;
+
+/// The socket `fd` names, or the errno Linux gives for a descriptor that is
+/// not one: EBADF when nothing is open there, ENOTSOCK when a file is.
+pub fn sock_of(guest: &Guest, fd: u64) -> Result {
+ match guest.fds.get(fd as usize) {
+ Some(f) if f.kind == Kind::Socket => Ok(f.handle),
+ Some(f) if f.is_open() => Err(errno::fail(errno::ENOTSOCK)),
+ _ => Err(errno::fail(errno::EBADF)),
+ }
+}
+
+/// A descriptor for socket `id`, with the flags asked for. A socket with no
+/// descriptor to name it is let go at once, since nobody could close it.
+pub fn install(guest: &mut Guest, id: u32, flags: u64) -> u64 {
+ match crate::linux::file::install(guest, Fd::socket(id)) {
+ Some(n) => {
+ if let Some(f) = guest.fds.get_mut(n as usize) {
+ f.nonblock = flags & SOCK_NONBLOCK != 0;
+ f.cloexec = flags & SOCK_CLOEXEC != 0;
+ }
+ errno::ok(n)
+ }
+ None => {
+ sock::with(|t| t.release(id, guest.pid));
+ errno::fail(errno::EMFILE)
+ }
+ }
+}
+
+/// The socket `fd` names, if it names one.
+pub fn sock_id(guest: &Guest, fd: u64) -> Option {
+ sock_of(guest, fd).ok()
+}
+
+/// True for a stream socket.
+pub fn is_stream(id: u32) -> bool {
+ sock::with(|t| t.get(id).is_some_and(|s| s.proto == sock::Proto::Stream))
+}
+
+/// True when `fd` is non-blocking.
+pub fn nonblock(guest: &Guest, fd: u64) -> bool {
+ guest.fds.get(fd as usize).is_some_and(|f| f.nonblock)
+}
diff --git a/userland/capsule_linux/src/linux/net/flags.rs b/userland/capsule_linux/src/linux/net/flags.rs
new file mode 100644
index 000000000..c077e41e7
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/flags.rs
@@ -0,0 +1,23 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The flags the send and receive calls take, from include/linux/socket.h.
+
+pub const MSG_OOB: u64 = 0x1;
+pub const MSG_PEEK: u64 = 0x2;
+pub const MSG_TRUNC: u64 = 0x20;
+pub const MSG_DONTWAIT: u64 = 0x40;
+pub const MSG_WAITALL: u64 = 0x100;
diff --git a/userland/capsule_linux/src/linux/net/iov.rs b/userland/capsule_linux/src/linux/net/iov.rs
new file mode 100644
index 000000000..922dff54b
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/iov.rs
@@ -0,0 +1,75 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The byte vectors the message calls name: one buffer, or an iovec array.
+
+use alloc::vec::Vec;
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+/// Linux's UIO_MAXIOV.
+const IOV_MAX: u64 = 1024;
+const IOVEC: usize = 16;
+
+pub type Iov = Vec<(u64, u64)>;
+
+/// The `count` entries of the iovec array at `at`.
+pub fn read(guest: &Guest, at: u64, count: u64) -> Result {
+ if count > IOV_MAX {
+ return Err(errno::fail(errno::EINVAL));
+ }
+ let raw = guest.read(at, count as usize * IOVEC).ok_or(errno::fail(errno::EFAULT))?;
+ let word = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8]));
+ Ok((0..count as usize).map(|i| (word(i * IOVEC), word(i * IOVEC + 8))).collect())
+}
+
+pub fn total(iov: &Iov) -> usize {
+ iov.iter().map(|&(_, len)| len as usize).sum()
+}
+
+/// The message's bytes from `skip` on, at most `cap` of them.
+pub fn gather(guest: &Guest, iov: &Iov, skip: usize, cap: usize) -> Result, u64> {
+ let mut out = Vec::new();
+ let mut pos = 0usize;
+ for &(base, len) in iov {
+ let len = len as usize;
+ let (from, to) = (skip.max(pos), (pos + len).min(skip + cap));
+ if from < to {
+ let part = guest.read(base + (from - pos) as u64, to - from);
+ out.extend_from_slice(&part.ok_or(errno::fail(errno::EFAULT))?);
+ }
+ pos += len;
+ }
+ Ok(out)
+}
+
+/// Put `bytes` into the message's buffers starting `skip` bytes in.
+pub fn scatter(guest: &Guest, iov: &Iov, skip: usize, bytes: &[u8]) -> Result<(), u64> {
+ let mut pos = 0usize;
+ for &(base, len) in iov {
+ let len = len as usize;
+ let (from, to) = (skip.max(pos), (pos + len).min(skip + bytes.len()));
+ if from < to {
+ let part = &bytes[from - skip..to - skip];
+ if guest.write(base + (from - pos) as u64, part) < part.len() as i64 {
+ return Err(errno::fail(errno::EFAULT));
+ }
+ }
+ pos += len;
+ }
+ Ok(())
+}
diff --git a/userland/capsule_linux/src/linux/net/listen.rs b/userland/capsule_linux/src/linux/net/listen.rs
new file mode 100644
index 000000000..a22e3ce57
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/listen.rs
@@ -0,0 +1,63 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `listen`, on a socket bound to 127.0.0.1 (`policy`).
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::fd::sock_of;
+use super::policy::not_loopback;
+use super::sock::{self, Addr, Domain, Proto};
+
+pub fn listen(guest: &mut Guest, fd: u64, backlog: u64) -> u64 {
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ sock::with(|t| {
+ let Some(s) = t.get(id) else {
+ return errno::fail(errno::EBADF);
+ };
+ match (s.proto, s.domain, s.local) {
+ (Proto::Dgram, ..) => return errno::fail(errno::EOPNOTSUPP),
+ _ if s.connected || s.svc.is_some() => return errno::fail(errno::EINVAL),
+ /* A Unix socket must be bound first: Linux does not name it here. */
+ (_, Domain::Unix, _) if s.uname.is_none() => return errno::fail(errno::EINVAL),
+ (_, Domain::Unix, _) => {}
+ /* Linux would bind 0.0.0.0 here, which is not the family's own. */
+ (_, _, None) => return not_loopback("listen", Addr::default()),
+ /* Listeners share an address only when each set SO_REUSEPORT. */
+ (_, _, Some(at))
+ if !s.listening
+ && t.iter().any(|(_, o)| {
+ o.listening
+ && o.local == Some(at)
+ && !(o.opts.reuseport && s.opts.reuseport)
+ }) =>
+ {
+ return errno::fail(errno::EADDRINUSE)
+ }
+ _ => {}
+ }
+ if let Some(s) = t.get_mut(id) {
+ s.listening = true;
+ /* An int, and somaxconn's 4096 is the most Linux keeps. */
+ s.backlog = (backlog as i32).clamp(0, 4096) as usize;
+ }
+ errno::ok(0)
+ })
+}
diff --git a/userland/capsule_linux/src/linux/net/mmsg.rs b/userland/capsule_linux/src/linux/net/mmsg.rs
new file mode 100644
index 000000000..174adaa3c
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/mmsg.rs
@@ -0,0 +1,51 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `sendmmsg` and `recvmmsg`: several messages in one call. A blocking
+//! recvmmsg waits until all `vlen` have come, unless MSG_WAITFORONE; the
+//! wait keeps its count between tries (`waits_sock`), so each try starts at
+//! message `skip` and answers how many it moved.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::flags::MSG_DONTWAIT;
+use super::mmsg_each::each;
+use super::policy::refuse;
+
+/// Linux's UIO_MAXIOV caps vlen.
+pub const MOST: u64 = 1024;
+pub const MSG_WAITFORONE: u64 = 0x10000;
+
+/// `a` is sendmmsg's arguments: fd, vector, vlen, flags.
+pub fn sendmmsg(guest: &mut Guest, a: [u64; 6], skip: usize) -> u64 {
+ each(guest, a, skip, |g, at, first| {
+ let f = if first { a[3] } else { a[3] | MSG_DONTWAIT };
+ super::msg::sendmsg(g, a[0], at, f, 0)
+ })
+}
+
+/// `a` is recvmmsg's arguments: fd, vector, vlen, flags, timeout.
+pub fn recvmmsg(guest: &mut Guest, a: [u64; 6], skip: usize) -> u64 {
+ if a[4] != 0 {
+ return refuse("recvmmsg timeout: SO_RCVTIMEO bounds the wait instead", errno::EINVAL);
+ }
+ let flags = a[3] & !MSG_WAITFORONE;
+ each(guest, a, skip, |g, at, first| {
+ let f = if first { flags } else { flags | MSG_DONTWAIT };
+ super::msg_recv::recvmsg(g, a[0], at, f, 0)
+ })
+}
diff --git a/userland/capsule_linux/src/linux/net/mmsg_each.rs b/userland/capsule_linux/src/linux/net/mmsg_each.rs
new file mode 100644
index 000000000..a047e1447
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/mmsg_each.rs
@@ -0,0 +1,50 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The messages of one sendmmsg or recvmmsg, one after another.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::mmsg::MOST;
+
+/// struct mmsghdr on x86_64: a msghdr, then msg_len.
+const MMSGHDR: u64 = 64;
+const LEN_AT: u64 = 56;
+
+/// Run `one` on each message from `skip` until one fails: the count moved,
+/// or the first failure's errno when none was.
+pub fn each(
+ guest: &mut Guest,
+ a: [u64; 6],
+ skip: usize,
+ mut one: impl FnMut(&mut Guest, u64, bool) -> u64,
+) -> u64 {
+ let vlen = a[2].min(MOST);
+ let mut moved = 0u64;
+ for i in skip as u64..vlen {
+ let at = a[1] + i * MMSGHDR;
+ let got = one(guest, at, moved == 0);
+ let Some(n) = errno::slot(got) else {
+ return if moved == 0 { got } else { errno::ok(moved) };
+ };
+ if guest.write(at + LEN_AT, &(n as u32).to_le_bytes()) < 4 {
+ return if moved == 0 { errno::fail(errno::EFAULT) } else { errno::ok(moved) };
+ }
+ moved += 1;
+ }
+ errno::ok(moved)
+}
diff --git a/userland/capsule_linux/src/linux/net/mod.rs b/userland/capsule_linux/src/linux/net/mod.rs
index c0155af09..78c0cc53b 100644
--- a/userland/capsule_linux/src/linux/net/mod.rs
+++ b/userland/capsule_linux/src/linux/net/mod.rs
@@ -14,30 +14,59 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-//! Sockets, over the net.sockets service.
+//! Sockets: the family's own, kept in `sock`, and a stream outside the
+//! family, which net.sockets carries over the mixnet.
-mod addr;
+mod accept;
+mod api;
+mod bind;
mod call;
+mod call_kind;
+mod cap;
+mod close;
mod connect;
+mod connect_dgram;
+mod connect_dial;
+mod connect_lo;
+mod connect_out;
+mod dest;
mod dgram;
mod dgram_addr;
-mod host_body;
pub mod dns;
+mod fd;
+mod flags;
+mod host_body;
+mod iov;
+mod listen;
+mod mmsg;
+mod mmsg_each;
+mod msg;
+mod msg_hdr;
+mod msg_recv;
+mod name;
+mod named;
mod ops;
+mod opt;
+mod pair;
+mod peer_addr;
+mod policy;
mod poll;
mod poll_set;
mod poll_socket;
pub mod raw;
pub mod raw_io;
+mod recvfrom;
+mod resolver;
pub mod route;
mod select;
+mod shutdown;
+pub mod sock;
+mod sockaddr;
+mod sockaddr_out;
mod socket;
mod stream;
+mod try_call;
+mod xfer_in;
+mod xfer_out;
-pub use connect::connect;
-pub use dgram::{recvfrom, sendto};
-pub use poll::{ready, POLLERR, POLLHUP};
-pub use poll_set::poll;
-pub use select::{clear as select_clear, select};
-pub use socket::socket;
-pub use stream::{close, recv, send};
+pub use api::*;
diff --git a/userland/capsule_linux/src/linux/net/msg.rs b/userland/capsule_linux/src/linux/net/msg.rs
new file mode 100644
index 000000000..091482db4
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/msg.rs
@@ -0,0 +1,52 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `sendmsg` on a family socket: an iovec and an address.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::fd::sock_of;
+use super::msg_hdr::hdr;
+use super::policy::refuse;
+
+/// `skip` bytes of the message went in an earlier try of this same call.
+pub fn sendmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> u64 {
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ let h = match hdr(guest, msg) {
+ Ok(h) => h,
+ Err(e) => return e,
+ };
+ if h.controllen != 0 {
+ return refuse(
+ "sendmsg control data on a family socket: SCM_RIGHTS is not carried yet",
+ errno::EINVAL,
+ );
+ }
+ let to = match super::peer_addr::address(guest, h.name, h.namelen) {
+ Ok(to) => to,
+ Err(e) => return e,
+ };
+ if let Some(super::peer_addr::To::Inet(a)) = &to {
+ if !super::sockaddr::is_loopback(a.ip) {
+ return super::policy::refuse_out("sendmsg", *a);
+ }
+ }
+ super::xfer_out::send(guest, id, &h.iov, skip, flags, to)
+}
diff --git a/userland/capsule_linux/src/linux/net/msg_hdr.rs b/userland/capsule_linux/src/linux/net/msg_hdr.rs
new file mode 100644
index 000000000..e3f60a9c3
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/msg_hdr.rs
@@ -0,0 +1,46 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! struct msghdr, as sendmsg and recvmsg find it in guest memory.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::iov;
+
+/// struct msghdr on x86_64.
+const MSGHDR: usize = 56;
+pub const NAMELEN_AT: u64 = 8;
+pub const CONTROLLEN_AT: u64 = 40;
+pub const FLAGS_AT: u64 = 48;
+
+pub struct Hdr {
+ pub name: u64,
+ pub namelen: u64,
+ pub iov: iov::Iov,
+ pub controllen: u64,
+}
+
+pub fn hdr(guest: &Guest, msg: u64) -> Result {
+ let raw = guest.read(msg, MSGHDR).ok_or(errno::fail(errno::EFAULT))?;
+ let word = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8]));
+ Ok(Hdr {
+ name: word(0),
+ namelen: u64::from(u32::from_le_bytes([raw[8], raw[9], raw[10], raw[11]])),
+ iov: iov::read(guest, word(16), word(24))?,
+ controllen: word(40),
+ })
+}
diff --git a/userland/capsule_linux/src/linux/net/msg_recv.rs b/userland/capsule_linux/src/linux/net/msg_recv.rs
new file mode 100644
index 000000000..5efafbc97
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/msg_recv.rs
@@ -0,0 +1,52 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `recvmsg` on a family socket: bytes into an iovec, and the sender's
+//! address.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::fd::sock_of;
+use super::flags::MSG_TRUNC;
+use super::msg_hdr::{hdr, CONTROLLEN_AT, FLAGS_AT, NAMELEN_AT};
+
+pub fn recvmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> u64 {
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ let h = match hdr(guest, msg) {
+ Ok(h) => h,
+ Err(e) => return e,
+ };
+ let got = match super::xfer_in::recv(guest, id, &h.iov, skip, flags) {
+ Ok(got) => got,
+ Err(e) => return e,
+ };
+ let cut = if got.whole > got.n { MSG_TRUNC as u32 } else { 0 };
+ let (name, lenp) = if h.name != 0 { (h.name, msg + NAMELEN_AT) } else { (0, 0) };
+ let value = super::peer_addr::finish(guest, got, flags, name, lenp);
+ if errno::slot(value).is_none() {
+ return value;
+ }
+ if guest.write(msg + CONTROLLEN_AT, &0u64.to_le_bytes()) < 8
+ || guest.write(msg + FLAGS_AT, &cut.to_le_bytes()) < 4
+ {
+ return errno::fail(errno::EFAULT);
+ }
+ value
+}
diff --git a/userland/capsule_linux/src/linux/net/name.rs b/userland/capsule_linux/src/linux/net/name.rs
new file mode 100644
index 000000000..962cfaec1
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/name.rs
@@ -0,0 +1,63 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `getsockname` and `getpeername`.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::fd::sock_of;
+use super::sock::{self, Domain, Peer};
+
+pub fn getsockname(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 {
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ /* A socket not yet bound is 0.0.0.0 port 0, or an unnamed Unix socket. */
+ let Some(me) = sock::with(|t| {
+ t.get(id).map(|s| match s.domain {
+ Domain::Inet => Peer::Inet(s.local.unwrap_or_default()),
+ Domain::Unix => Peer::Unix(s.uname.clone()),
+ })
+ }) else {
+ return errno::fail(errno::EBADF);
+ };
+ super::sockaddr_out::write(guest, at, lenp, &me)
+}
+
+pub fn getpeername(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 {
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ /*
+ * A reset connection is closed, and has no peer; one whose peer only
+ * shut down, or left cleanly, still does.
+ */
+ let peer: Option = sock::with(|t| {
+ let s = t.get(id)?;
+ let live = s.connected && !s.broken && s.error == 0;
+ live.then(|| match s.domain {
+ Domain::Inet => Peer::Inet(s.remote.unwrap_or_default()),
+ Domain::Unix => Peer::Unix(s.upeer.clone()),
+ })
+ });
+ match peer {
+ Some(p) => super::sockaddr_out::write(guest, at, lenp, &p),
+ None => errno::fail(errno::ENOTCONN),
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/named/addr.rs b/userland/capsule_linux/src/linux/net/named/addr.rs
new file mode 100644
index 000000000..28fef072d
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/named/addr.rs
@@ -0,0 +1,58 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `sockaddr_un` as a guest names it: a path, an abstract name, or only the
+//! family, which asks bind for a name of Linux's choosing.
+
+use alloc::vec::Vec;
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+use crate::linux::net::sockaddr::{self, AF_UNIX};
+
+/// sun_family and the 108 bytes of sun_path.
+const SOCKADDR_UN: u64 = 110;
+
+pub enum UAddr {
+ Auto,
+ Path(Vec),
+ Abstract(Vec),
+}
+
+pub fn read(guest: &Guest, at: u64, len: u64) -> Result {
+ if !(2..=SOCKADDR_UN).contains(&len) {
+ return Err(errno::fail(errno::EINVAL));
+ }
+ let raw = guest.read(at, len as usize).ok_or(errno::fail(errno::EFAULT))?;
+ let path = &raw[2..];
+ Ok(match path.first() {
+ None => UAddr::Auto,
+ Some(0) => UAddr::Abstract(path[1..].to_vec()),
+ /* A path ends at its first NUL, however long the guest said it was. */
+ Some(_) => {
+ let end = path.iter().position(|&b| b == 0).unwrap_or(path.len());
+ UAddr::Path(path[..end].to_vec())
+ }
+ })
+}
+
+/// The name a bind or connect gives: EINVAL for another family.
+pub fn unix_addr(guest: &Guest, at: u64, len: u64) -> Result {
+ match sockaddr::read(guest, at, len)? {
+ (AF_UNIX, _) => read(guest, at, len),
+ _ => Err(errno::fail(errno::EINVAL)),
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/named/auto.rs b/userland/capsule_linux/src/linux/net/named/auto.rs
new file mode 100644
index 000000000..6faec8b6c
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/named/auto.rs
@@ -0,0 +1,32 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The name bind chooses when a Unix socket is given its family alone.
+
+use alloc::vec::Vec;
+
+use crate::linux::net::sock::{self, UName};
+
+/// Linux's autobind: a NUL and five hex digits, the first unused.
+pub fn fresh() -> Option {
+ (0u32..0x10_0000).find_map(|n| {
+ let mut key: Vec = alloc::vec![0];
+ key.extend_from_slice(alloc::format!("{n:05x}").as_bytes());
+ let name = UName { key: key.clone(), shown: key };
+ let used = sock::with(|t| t.iter().any(|(_, s)| s.uname.as_ref() == Some(&name)));
+ (!used).then_some(name)
+ })
+}
diff --git a/userland/capsule_linux/src/linux/net/named/bind.rs b/userland/capsule_linux/src/linux/net/named/bind.rs
new file mode 100644
index 000000000..75b99e4ee
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/named/bind.rs
@@ -0,0 +1,66 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `bind` on a Unix socket: a path, which becomes a file as on Linux, an
+//! abstract name, or family alone, which asks for a name to be chosen.
+
+use crate::linux::abi::errno;
+use crate::linux::file;
+use crate::linux::guest::Guest;
+
+use super::addr::{unix_addr, UAddr};
+use super::auto::fresh;
+use super::name::resolve;
+use crate::linux::net::sock;
+
+pub fn bind(guest: &Guest, id: u32, at: u64, len: u64) -> u64 {
+ let ua = match unix_addr(guest, at, len) {
+ Ok(ua) => ua,
+ Err(e) => return e,
+ };
+ if sock::with(|t| t.get(id).is_none_or(|s| s.uname.is_some())) {
+ return errno::fail(errno::EINVAL);
+ }
+ bind_name(guest, id, &ua)
+}
+
+/// Bind Unix socket `id` to `ua`. A path must not exist yet, and becomes an
+/// empty file; family alone chooses an abstract name of five hex digits.
+fn bind_name(guest: &Guest, id: u32, ua: &UAddr) -> u64 {
+ let name = match resolve(guest, ua) {
+ Some(n) => n,
+ None => match fresh() {
+ Some(n) => n,
+ None => return errno::fail(errno::EADDRINUSE),
+ },
+ };
+ let taken = sock::with(|t| t.iter().any(|(_, s)| s.uname.as_ref() == Some(&name)));
+ if taken || (!name.is_abstract() && file::look(&name.key).is_some()) {
+ return errno::fail(errno::EADDRINUSE);
+ }
+ if !name.is_abstract() {
+ let key = file::key(&name.key);
+ if key.writable().is_err() {
+ return errno::fail(errno::EROFS);
+ }
+ /* The store refuses a file whose directory is missing. */
+ if file::store_write(&key, &[]).is_err() {
+ return errno::fail(errno::ENOENT);
+ }
+ }
+ sock::with(|t| t.get_mut(id).map(|s| s.uname = Some(name)));
+ errno::ok(0)
+}
diff --git a/userland/capsule_linux/src/linux/net/named/connect.rs b/userland/capsule_linux/src/linux/net/named/connect.rs
new file mode 100644
index 000000000..e030d1abd
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/named/connect.rs
@@ -0,0 +1,70 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `connect` on a Unix socket. A connect to the display's path
+//! turns the descriptor into the display connection this capsule serves
+//! itself (`unix`); every other name is the family's own.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+use crate::linux::net::sock::{self, Link, Proto};
+use crate::linux::net::sockaddr::{self, AF_UNSPEC};
+use crate::linux::unix;
+
+use super::addr::{unix_addr, UAddr};
+use super::display::display;
+use super::name::{find, resolve};
+
+pub fn connect(guest: &mut Guest, fd: u64, id: u32, proto: Proto, at: u64, len: u64) -> u64 {
+ if proto == Proto::Dgram && matches!(sockaddr::read(guest, at, len), Ok((AF_UNSPEC, _))) {
+ sock::with(|t| t.get_mut(id).map(|s| (s.peer, s.upeer, s.connected) = (None, None, false)));
+ return errno::ok(0);
+ }
+ let ua = match unix_addr(guest, at, len) {
+ Ok(ua) => ua,
+ Err(e) => return e,
+ };
+ if let (Proto::Stream, UAddr::Path(p)) = (proto, &ua) {
+ if unix::is_display(p) {
+ return display(guest, fd, at, len);
+ }
+ }
+ let Some(name) = resolve(guest, &ua) else {
+ return errno::fail(errno::EINVAL);
+ };
+ let target = match find(&name, proto) {
+ Ok(t) => t,
+ Err(e) => return errno::fail(e),
+ };
+ sock::with(|t| {
+ let s = t.get_mut(id).ok_or(errno::EBADF)?;
+ match proto {
+ Proto::Stream if s.connected => Err(errno::EISCONN),
+ Proto::Stream if s.listening => Err(errno::EINVAL),
+ /* A Unix connect completes in the caller's call, blocking or not. */
+ Proto::Stream => match t.join(id, target) {
+ Link::Done => Ok(()),
+ Link::Full => Err(errno::EAGAIN),
+ Link::Refused => Err(errno::ECONNREFUSED),
+ },
+ Proto::Dgram => {
+ (s.peer, s.upeer, s.connected) = (Some(target), Some(name), true);
+ Ok(())
+ }
+ }
+ })
+ .map_or_else(errno::fail, |()| errno::ok(0))
+}
diff --git a/userland/capsule_linux/src/linux/net/named/display.rs b/userland/capsule_linux/src/linux/net/named/display.rs
new file mode 100644
index 000000000..ef6cc8755
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/named/display.rs
@@ -0,0 +1,32 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The display's path: the one Unix name this capsule answers itself.
+
+use crate::linux::guest::{Fd, Guest};
+use crate::linux::unix;
+
+/// Let go of the family socket and make `fd` the display connection.
+pub fn display(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 {
+ crate::linux::net::close::close(guest, fd);
+ if let Some(f) = guest.fds.get_mut(fd as usize) {
+ let (cloexec, nonblock) = (f.cloexec, f.nonblock);
+ *f = Fd::unix();
+ f.cloexec = cloexec;
+ f.nonblock = nonblock;
+ }
+ unix::connect(guest, fd, at, len)
+}
diff --git a/userland/capsule_linux/src/linux/net/named/mod.rs b/userland/capsule_linux/src/linux/net/named/mod.rs
new file mode 100644
index 000000000..ec9aac5fa
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/named/mod.rs
@@ -0,0 +1,30 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Unix sockets with names: the sockaddr_un a guest gives, the names a
+//! family socket is bound to, and bind and connect on them.
+
+mod addr;
+mod auto;
+mod bind;
+mod connect;
+mod display;
+mod name;
+
+pub use addr::{read as read_uaddr, UAddr};
+pub use bind::bind;
+pub use connect::connect;
+pub use name::{find, resolve};
diff --git a/userland/capsule_linux/src/linux/net/named/name.rs b/userland/capsule_linux/src/linux/net/named/name.rs
new file mode 100644
index 000000000..121bcc83d
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/named/name.rs
@@ -0,0 +1,65 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Unix names. A path is a file, as on Linux: bind makes it, it stays after
+//! the socket closes until the guest unlinks it, and a connect to a path
+//! with no listener is ECONNREFUSED if the file is there and ENOENT if not.
+//! An abstract name has no file and goes with its socket. Neither is seen
+//! outside the family: only the family's sockets are bound to them.
+
+use crate::linux::abi::errno;
+use crate::linux::file;
+use crate::linux::guest::Guest;
+
+use super::addr::UAddr;
+use crate::linux::net::sock::{self, Domain, Proto, UName};
+
+/// The name `ua` means for this guest; None asks for one to be chosen.
+pub fn resolve(guest: &Guest, ua: &UAddr) -> Option {
+ match ua {
+ UAddr::Auto => None,
+ UAddr::Path(p) => Some(UName { key: file::visible(&guest.cwd, p), shown: p.clone() }),
+ UAddr::Abstract(n) => {
+ let mut key = alloc::vec![0u8];
+ key.extend_from_slice(n);
+ Some(UName { key: key.clone(), shown: key })
+ }
+ }
+}
+
+/// The family socket of kind `proto` bound to `name`, the one a connect or
+/// a send reaches: a stream's must listen. Otherwise Linux's answer.
+pub fn find(name: &UName, proto: Proto) -> Result {
+ /*
+ * An accepted connection carries its listener's name, as on Linux; the
+ * listener is the one a connect reaches.
+ */
+ let found = sock::with(|t| {
+ let named =
+ || t.iter().filter(|(_, s)| s.domain == Domain::Unix && s.uname.as_ref() == Some(name));
+ named()
+ .find(|(_, s)| s.listening)
+ .or_else(|| named().next())
+ .map(|(i, s)| (i, s.proto, s.listening))
+ });
+ match found {
+ Some((_, p, _)) if p != proto => Err(errno::EPROTOTYPE),
+ Some((i, Proto::Dgram, _)) | Some((i, _, true)) => Ok(i),
+ Some(_) => Err(errno::ECONNREFUSED),
+ None if name.is_abstract() || file::look(&name.key).is_some() => Err(errno::ECONNREFUSED),
+ None => Err(errno::ENOENT),
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/opt/apply.rs b/userland/capsule_linux/src/linux/net/opt/apply.rs
new file mode 100644
index 000000000..dc3610b25
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/opt/apply.rs
@@ -0,0 +1,63 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! One option set on one socket: kept as Linux keeps it, or refused.
+
+use crate::linux::abi::errno;
+use crate::linux::net::sock::{Domain, Proto, Sock};
+
+use super::ids::*;
+use super::time::{keep, timeo};
+
+/// `raw` holds up to sixteen bytes of the value; `len` is what the guest
+/// said it gave, at least four.
+pub fn apply(s: &mut Sock, (level, name): (u64, u64), raw: &[u8], len: u64) -> u64 {
+ let int = u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]);
+ let word =
+ |i: usize| raw.get(i..i + 8).map(|b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8])));
+ let (proto, domain) = (s.proto, s.domain);
+ let o = &mut s.opts;
+ match (level, name) {
+ /* A Unix socket has only socket-level options on Linux. */
+ (l, _) if domain == Domain::Unix && l != SOL_SOCKET => {
+ return errno::fail(errno::EOPNOTSUPP)
+ }
+ (l, n) if super::more::known(l, n) && !(l == IPPROTO_TCP && proto == Proto::Dgram) => {
+ return super::more::set(&mut o.more, proto, l, n, int)
+ }
+ (SOL_SOCKET, SO_REUSEADDR) => o.reuseaddr = int != 0,
+ (SOL_SOCKET, SO_REUSEPORT) => o.reuseport = int != 0,
+ (SOL_SOCKET, SO_KEEPALIVE) => o.keepalive = int != 0,
+ (SOL_SOCKET, SO_BROADCAST) => o.broadcast = int != 0,
+ (SOL_SOCKET, SO_RCVBUF) => o.rcvbuf = (int.min(BUF_MAX) * 2).max(RCVBUF_MIN),
+ (SOL_SOCKET, SO_SNDBUF) => o.sndbuf = (int.min(BUF_MAX) * 2).max(SNDBUF_MIN),
+ (SOL_SOCKET, SO_LINGER) if len >= 8 => {
+ o.linger = (u32::from(int != 0), u32::from_le_bytes([raw[4], raw[5], raw[6], raw[7]]))
+ }
+ (SOL_SOCKET, SO_LINGER) => return errno::fail(errno::EINVAL),
+ (SOL_SOCKET, SO_RCVTIMEO) => return timeo(&mut o.rcvtimeo, word(0), word(8)),
+ (SOL_SOCKET, SO_SNDTIMEO) => return timeo(&mut o.sndtimeo, word(0), word(8)),
+ (IPPROTO_TCP, _) if proto == Proto::Dgram => return errno::fail(errno::ENOPROTOOPT),
+ (IPPROTO_TCP, TCP_NODELAY) => o.nodelay = int != 0,
+ (IPPROTO_TCP, TCP_KEEPIDLE) => return keep(&mut o.keepidle, int, KEEP_MAX),
+ (IPPROTO_TCP, TCP_KEEPINTVL) => return keep(&mut o.keepintvl, int, KEEP_MAX),
+ (IPPROTO_TCP, TCP_KEEPCNT) => return keep(&mut o.keepcnt, int, KEEPCNT_MAX),
+ /* An AF_INET socket has no IPv6 options on Linux either. */
+ (IPPROTO_IPV6, _) => return errno::fail(errno::ENOPROTOOPT),
+ _ => return super::get::unknown("setsockopt", level, name),
+ }
+ errno::ok(0)
+}
diff --git a/userland/capsule_linux/src/linux/net/opt/get.rs b/userland/capsule_linux/src/linux/net/opt/get.rs
new file mode 100644
index 000000000..67209a09b
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/opt/get.rs
@@ -0,0 +1,57 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `getsockopt`, for the options `opt_set` keeps and the ones a socket
+//! reports about itself.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use crate::linux::net::fd::sock_of;
+use crate::linux::net::sock;
+
+pub fn getsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, lenp: u64) -> u64 {
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ let Some(raw) = guest.read(lenp, 4) else {
+ return errno::fail(errno::EFAULT);
+ };
+ let room = i32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]);
+ if room < 0 {
+ return errno::fail(errno::EINVAL);
+ }
+ let value = sock::with(|t| t.get_mut(id).map(|s| super::value::value(s, level, name)));
+ let bytes = match value {
+ Some(Ok(bytes)) => bytes,
+ Some(Err(e)) => return e,
+ None => return errno::fail(errno::EBADF),
+ };
+ let n = bytes.len().min(room as usize);
+ if guest.write(val, &bytes[..n]) < n as i64 || guest.write(lenp, &(n as u32).to_le_bytes()) < 4
+ {
+ return errno::fail(errno::EFAULT);
+ }
+ errno::ok(0)
+}
+
+/// ENOPROTOOPT for an option this capsule does not keep, said by name: a
+/// program may depend on it, and Linux would have kept it.
+pub fn unknown(call: &str, level: u64, name: u64) -> u64 {
+ let what = alloc::format!("{call} level {level} option {name}: not kept for a guest socket");
+ crate::linux::net::policy::refuse(&what, errno::ENOPROTOOPT)
+}
diff --git a/userland/capsule_linux/src/linux/net/opt/ids.rs b/userland/capsule_linux/src/linux/net/opt/ids.rs
new file mode 100644
index 000000000..bbeb753ee
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/opt/ids.rs
@@ -0,0 +1,60 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Option levels and names, from include/uapi/asm-generic/socket.h,
+//! include/uapi/linux/in.h and include/uapi/linux/tcp.h.
+
+pub const SOL_SOCKET: u64 = 1;
+pub const IPPROTO_IP: u64 = 0;
+pub const IPPROTO_TCP: u64 = 6;
+pub const IPPROTO_IPV6: u64 = 41;
+
+pub const SO_REUSEADDR: u64 = 2;
+pub const SO_TYPE: u64 = 3;
+pub const SO_ERROR: u64 = 4;
+pub const SO_BROADCAST: u64 = 6;
+pub const SO_SNDBUF: u64 = 7;
+pub const SO_RCVBUF: u64 = 8;
+pub const SO_KEEPALIVE: u64 = 9;
+pub const SO_LINGER: u64 = 13;
+pub const SO_REUSEPORT: u64 = 15;
+pub const SO_RCVTIMEO: u64 = 20;
+pub const SO_SNDTIMEO: u64 = 21;
+pub const SO_ACCEPTCONN: u64 = 30;
+pub const SO_PROTOCOL: u64 = 38;
+pub const SO_DOMAIN: u64 = 39;
+
+pub const IP_TOS: u64 = 1;
+pub const IP_TTL: u64 = 2;
+pub const SO_PRIORITY: u64 = 12;
+
+pub const TCP_NODELAY: u64 = 1;
+pub const TCP_KEEPIDLE: u64 = 4;
+pub const TCP_KEEPINTVL: u64 = 5;
+pub const TCP_KEEPCNT: u64 = 6;
+pub const TCP_QUICKACK: u64 = 12;
+pub const TCP_USER_TIMEOUT: u64 = 18;
+pub const TCP_FASTOPEN: u64 = 23;
+
+/// net.core.rmem_max and wmem_max as Linux ships them: what SO_RCVBUF and
+/// SO_SNDBUF are held to before they are doubled.
+pub const BUF_MAX: u32 = 212_992;
+/// The least Linux keeps: SOCK_MIN_RCVBUF and SOCK_MIN_SNDBUF.
+pub const RCVBUF_MIN: u32 = 2304;
+pub const SNDBUF_MIN: u32 = 4608;
+/// MAX_TCP_KEEPIDLE, MAX_TCP_KEEPINTVL and MAX_TCP_KEEPCNT.
+pub const KEEP_MAX: u32 = 32767;
+pub const KEEPCNT_MAX: u32 = 127;
diff --git a/userland/capsule_linux/src/linux/net/opt/mod.rs b/userland/capsule_linux/src/linux/net/opt/mod.rs
new file mode 100644
index 000000000..bdeca38af
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/opt/mod.rs
@@ -0,0 +1,29 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Socket options: what setsockopt keeps and getsockopt reads back.
+
+mod apply;
+mod get;
+mod ids;
+mod more;
+mod set;
+mod time;
+mod value;
+
+pub use get::getsockopt;
+pub use set::setsockopt;
+pub use time::limit_ms;
diff --git a/userland/capsule_linux/src/linux/net/opt/more.rs b/userland/capsule_linux/src/linux/net/opt/more.rs
new file mode 100644
index 000000000..4e5368c65
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/opt/more.rs
@@ -0,0 +1,68 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Options Linux keeps that change nothing a loopback connection can see:
+//! the IP type of service and time to live, the priority a queueing
+//! discipline would use, and TCP's user timeout, quick-ack and Fast Open
+//! queue. Each is kept, checked as Linux checks it, and read back.
+
+use crate::linux::abi::errno;
+
+use super::ids::{
+ IPPROTO_IP, IPPROTO_TCP, IP_TOS, IP_TTL, SOL_SOCKET, SO_PRIORITY, TCP_FASTOPEN, TCP_QUICKACK,
+ TCP_USER_TIMEOUT,
+};
+use crate::linux::net::sock::{More, Proto};
+
+/// Linux's net.ipv4.ip_default_ttl.
+const DEFAULT_TTL: u32 = 64;
+/// The ECN bits of the type of service, which a TCP socket does not keep.
+const ECN_MASK: u32 = 3;
+
+pub fn known(level: u64, name: u64) -> bool {
+ matches!(
+ (level, name),
+ (IPPROTO_IP, IP_TOS | IP_TTL)
+ | (SOL_SOCKET, SO_PRIORITY)
+ | (IPPROTO_TCP, TCP_USER_TIMEOUT | TCP_QUICKACK | TCP_FASTOPEN)
+ )
+}
+
+pub fn set(m: &mut More, proto: Proto, level: u64, name: u64, v: u32) -> u64 {
+ match (level, name) {
+ (IPPROTO_IP, IP_TOS) if proto == Proto::Stream => m.tos = v & 0xff & !ECN_MASK,
+ (IPPROTO_IP, IP_TOS) => m.tos = v & 0xff,
+ (IPPROTO_IP, IP_TTL) if v as i32 == -1 => m.ttl = DEFAULT_TTL,
+ (IPPROTO_IP, IP_TTL) if (1..=255).contains(&v) => m.ttl = v,
+ (SOL_SOCKET, SO_PRIORITY) => m.priority = v,
+ (IPPROTO_TCP, TCP_USER_TIMEOUT) if (v as i32) >= 0 => m.user_timeout = v,
+ (IPPROTO_TCP, TCP_QUICKACK) => m.quickack = v != 0,
+ (IPPROTO_TCP, TCP_FASTOPEN) if (v as i32) >= 0 => m.fastopen = v,
+ _ => return errno::fail(errno::EINVAL),
+ }
+ errno::ok(0)
+}
+
+pub fn get(m: &More, level: u64, name: u64) -> u32 {
+ match (level, name) {
+ (IPPROTO_IP, IP_TOS) => m.tos,
+ (IPPROTO_IP, IP_TTL) => m.ttl,
+ (SOL_SOCKET, SO_PRIORITY) => m.priority,
+ (IPPROTO_TCP, TCP_USER_TIMEOUT) => m.user_timeout,
+ (IPPROTO_TCP, TCP_QUICKACK) => u32::from(m.quickack),
+ _ => m.fastopen,
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/opt/set.rs b/userland/capsule_linux/src/linux/net/opt/set.rs
new file mode 100644
index 000000000..3a432a200
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/opt/set.rs
@@ -0,0 +1,42 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `setsockopt`. Every option here is kept and reads back as Linux reads
+//! it; one that would change nothing on the family's loopback is still kept,
+//! since Linux keeps it too. One this capsule cannot honour is refused.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use crate::linux::net::fd::sock_of;
+use crate::linux::net::sock;
+
+pub fn setsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, len: u64) -> u64 {
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ let Some(raw) = guest.read(val, (len as usize).min(16)) else {
+ return errno::fail(errno::EFAULT);
+ };
+ if len < 4 {
+ return errno::fail(errno::EINVAL);
+ }
+ sock::with(|t| match t.get_mut(id) {
+ Some(s) => super::apply::apply(s, (level, name), &raw, len),
+ None => errno::fail(errno::EBADF),
+ })
+}
diff --git a/userland/capsule_linux/src/linux/net/opt/time.rs b/userland/capsule_linux/src/linux/net/opt/time.rs
new file mode 100644
index 000000000..f68652bd6
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/opt/time.rs
@@ -0,0 +1,50 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The options that hold a number of seconds: the keepalive times, and the
+//! receive and send limits a wait keeps to.
+
+use crate::linux::abi::errno;
+
+use crate::linux::net::sock::{self, Opts};
+
+/// A keepalive time or count, 1 up to Linux's most, else EINVAL.
+pub fn keep(slot: &mut u32, v: u32, most: u32) -> u64 {
+ if v < 1 || v > most {
+ return errno::fail(errno::EINVAL);
+ }
+ *slot = v;
+ errno::ok(0)
+}
+
+/// SO_RCVTIMEO or SO_SNDTIMEO from a struct timeval: EDOM for microseconds
+/// out of range, and a negative time is no limit, as Linux treats both.
+pub fn timeo(slot: &mut (u64, u64), sec: Option, usec: Option) -> u64 {
+ let (Some(sec), Some(usec)) = (sec, usec) else {
+ return errno::fail(errno::EINVAL);
+ };
+ if usec as i64 >= 1_000_000 || (usec as i64) < 0 {
+ return errno::fail(errno::EDOM);
+ }
+ *slot = if (sec as i64) < 0 { (0, 0) } else { (sec, usec) };
+ errno::ok(0)
+}
+
+/// The limit a receive (`read`) or a send waits for, from its option.
+pub fn limit_ms(id: u32, read: bool) -> Option {
+ sock::with(|t| t.get(id).map(|s| if read { s.opts.rcvtimeo } else { s.opts.sndtimeo }))
+ .and_then(Opts::limit_ms)
+}
diff --git a/userland/capsule_linux/src/linux/net/opt/value.rs b/userland/capsule_linux/src/linux/net/opt/value.rs
new file mode 100644
index 000000000..684b4a795
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/opt/value.rs
@@ -0,0 +1,68 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The value getsockopt reads for each option a socket has.
+
+use alloc::vec::Vec;
+
+use crate::linux::abi::errno;
+
+use super::ids::*;
+use crate::linux::net::sock::{Domain, Proto, Sock};
+
+pub fn value(s: &mut Sock, level: u64, name: u64) -> Result, u64> {
+ let o = s.opts;
+ let int = |v: u32| Ok(v.to_le_bytes().to_vec());
+ let pair = |a: u64, b: u64| Ok([a.to_le_bytes(), b.to_le_bytes()].concat());
+ let stream = s.proto == Proto::Stream;
+ match (level, name) {
+ (SOL_SOCKET, SO_TYPE) => int(if stream { 1 } else { 2 }),
+ (SOL_SOCKET, SO_DOMAIN) => int(if s.domain == Domain::Unix { 1 } else { 2 }),
+ (SOL_SOCKET, SO_PROTOCOL) => match (s.domain, stream) {
+ (Domain::Unix, _) => int(0),
+ (_, true) => int(6),
+ (_, false) => int(17),
+ },
+ (SOL_SOCKET, SO_ACCEPTCONN) => int(u32::from(s.listening)),
+ (SOL_SOCKET, SO_ERROR) => int(core::mem::take(&mut s.error) as u32),
+ (SOL_SOCKET, SO_REUSEADDR) => int(u32::from(o.reuseaddr)),
+ (SOL_SOCKET, SO_REUSEPORT) => int(u32::from(o.reuseport)),
+ (SOL_SOCKET, SO_KEEPALIVE) => int(u32::from(o.keepalive)),
+ (SOL_SOCKET, SO_BROADCAST) => int(u32::from(o.broadcast)),
+ (SOL_SOCKET, SO_RCVBUF) => int(o.rcvbuf),
+ (SOL_SOCKET, SO_SNDBUF) => int(o.sndbuf),
+ (SOL_SOCKET, SO_LINGER) => {
+ Ok([o.linger.0.to_le_bytes(), o.linger.1.to_le_bytes()].concat())
+ }
+ (SOL_SOCKET, SO_RCVTIMEO) => pair(o.rcvtimeo.0, o.rcvtimeo.1),
+ (SOL_SOCKET, SO_SNDTIMEO) => pair(o.sndtimeo.0, o.sndtimeo.1),
+ (l, _) if s.domain == Domain::Unix && l != SOL_SOCKET => {
+ Err(errno::fail(errno::EOPNOTSUPP))
+ }
+ (IPPROTO_TCP, _) if !stream && super::more::known(level, name) => {
+ Err(errno::fail(errno::EOPNOTSUPP))
+ }
+ (l, n) if super::more::known(l, n) => int(super::more::get(&o.more, l, n)),
+ /* A datagram socket has no TCP options, and Linux says so this way. */
+ (IPPROTO_TCP, _) if !stream => Err(errno::fail(errno::EOPNOTSUPP)),
+ (IPPROTO_TCP, TCP_NODELAY) => int(u32::from(o.nodelay)),
+ (IPPROTO_TCP, TCP_KEEPIDLE) => int(o.keepidle),
+ (IPPROTO_TCP, TCP_KEEPINTVL) => int(o.keepintvl),
+ (IPPROTO_TCP, TCP_KEEPCNT) => int(o.keepcnt),
+ (IPPROTO_IPV6, _) => Err(errno::fail(errno::EOPNOTSUPP)),
+ _ => Err(super::get::unknown("getsockopt", level, name)),
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/pair.rs b/userland/capsule_linux/src/linux/net/pair.rs
new file mode 100644
index 000000000..e5b5cf3fd
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/pair.rs
@@ -0,0 +1,70 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `socketpair`: two connected Unix sockets, both ends in the family.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::fd::{install, SOCK_CLOEXEC, SOCK_NONBLOCK};
+use super::sock::{self, Domain, Proto};
+use super::sockaddr::{AF_INET, AF_UNIX};
+
+const SOCK_STREAM: u64 = 1;
+const SOCK_DGRAM: u64 = 2;
+const TYPE_MASK: u64 = 0xF;
+
+pub fn socketpair(guest: &mut Guest, family: u64, kind: u64, protocol: u64, out: u64) -> u64 {
+ let flags = kind & (SOCK_NONBLOCK | SOCK_CLOEXEC);
+ if kind & !(TYPE_MASK | flags) != 0 {
+ return errno::fail(errno::EINVAL);
+ }
+ match family {
+ f if f == u64::from(AF_UNIX) => {}
+ /* Linux has no connected pair for the internet families. */
+ f if f == u64::from(AF_INET) => return errno::fail(errno::EOPNOTSUPP),
+ _ => return errno::fail(errno::EAFNOSUPPORT),
+ }
+ let proto = match kind & TYPE_MASK {
+ SOCK_STREAM => Proto::Stream,
+ SOCK_DGRAM => Proto::Dgram,
+ _ => return errno::fail(errno::ESOCKTNOSUPPORT),
+ };
+ if protocol != 0 {
+ return errno::fail(errno::EPROTONOSUPPORT);
+ }
+ let (a, b) = sock::with(|t| t.pair(Domain::Unix, proto, guest.pid));
+ let fa = install(guest, a, flags);
+ let Some(na) = errno::slot(fa) else {
+ sock::with(|t| t.release(b, guest.pid));
+ return fa;
+ };
+ let fb = install(guest, b, flags);
+ let Some(nb) = errno::slot(fb) else {
+ super::close::discard(guest, na as u64);
+ return fb;
+ };
+ let mut pair = [0u8; 8];
+ pair[..4].copy_from_slice(&(na as u32).to_le_bytes());
+ pair[4..].copy_from_slice(&(nb as u32).to_le_bytes());
+ /* Linux copies the pair out before it installs either descriptor. */
+ if guest.write(out, &pair) < 8 {
+ super::close::discard(guest, na as u64);
+ super::close::discard(guest, nb as u64);
+ return errno::fail(errno::EFAULT);
+ }
+ errno::ok(0)
+}
diff --git a/userland/capsule_linux/src/linux/net/peer_addr.rs b/userland/capsule_linux/src/linux/net/peer_addr.rs
new file mode 100644
index 000000000..8c89528e3
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/peer_addr.rs
@@ -0,0 +1,63 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The addresses the send and receive calls carry: the one a send names,
+//! and the sender a receive writes back.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::flags::MSG_TRUNC;
+use super::named::UAddr;
+use super::sock::Addr;
+use super::sockaddr::{self, AF_INET, AF_UNIX};
+use super::xfer_in::In;
+
+/// The count a receive answers, with the sender written out. A stream has
+/// no sender, and Linux says so with a length of zero.
+pub fn finish(guest: &mut Guest, got: In, flags: u64, at: u64, alen: u64) -> u64 {
+ if at != 0 {
+ let wrote = match &got.from {
+ Some(from) => super::sockaddr_out::write(guest, at, alen, from),
+ None if alen != 0 && guest.write(alen, &0u32.to_le_bytes()) < 4 => {
+ errno::fail(errno::EFAULT)
+ }
+ None => errno::ok(0),
+ };
+ if errno::slot(wrote).is_none() {
+ return wrote;
+ }
+ }
+ errno::ok(if flags & MSG_TRUNC != 0 { got.whole } else { got.n } as u64)
+}
+
+/// Where a send names: an IPv4 address or a Unix name.
+pub enum To {
+ Inet(Addr),
+ Unix(UAddr),
+}
+
+/// The address a send names, if any.
+pub fn address(guest: &Guest, at: u64, alen: u64) -> Result, u64> {
+ if at == 0 {
+ return Ok(None);
+ }
+ match sockaddr::read(guest, at, alen)? {
+ (AF_INET, a) => Ok(Some(To::Inet(a))),
+ (AF_UNIX, _) => Ok(Some(To::Unix(super::named::read_uaddr(guest, at, alen)?))),
+ _ => Err(errno::fail(errno::EAFNOSUPPORT)),
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/policy.rs b/userland/capsule_linux/src/linux/net/policy.rs
new file mode 100644
index 000000000..46a1227e7
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/policy.rs
@@ -0,0 +1,57 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! What this capsule lets a guest's sockets reach. A guest binds and
+//! listens only on 127.0.0.0/8, which is the family's own: nothing outside
+//! the capsule can connect to it. Anything else is EACCES, said by name.
+//! Raw and packet sockets reach the link layer, below anything that could
+//! confine them, and are EPERM.
+
+use alloc::format;
+
+use crate::linux::abi::errno;
+use crate::linux::start::say;
+
+use super::sock::Addr;
+
+/// EACCES for a bind or listen outside loopback, with the address.
+pub fn not_loopback(call: &str, at: Addr) -> u64 {
+ let [a, b, c, d] = at.ip;
+ let line = format!(
+ "[LINUX] refused {call} {a}.{b}.{c}.{d}:{}: a guest listens only on 127.0.0.0/8\n",
+ at.port
+ );
+ say(line.as_bytes());
+ errno::fail(errno::EACCES)
+}
+
+/// ENETUNREACH for a datagram to anywhere outside the family: the mixnet
+/// carries streams, and a guest's datagrams have no other way out.
+pub fn refuse_out(call: &str, to: Addr) -> u64 {
+ let [a, b, c, d] = to.ip;
+ let line = format!(
+ "[LINUX] refused {call} {a}.{b}.{c}.{d}:{}: a guest's datagrams stay in the family\n",
+ to.port
+ );
+ say(line.as_bytes());
+ errno::fail(errno::ENETUNREACH)
+}
+
+/// `errno` for a call this capsule declines, with why.
+pub fn refuse(what: &str, errno: i64) -> u64 {
+ say(format!("[LINUX] refused {what}\n").as_bytes());
+ errno::fail(errno)
+}
diff --git a/userland/capsule_linux/src/linux/net/poll_socket.rs b/userland/capsule_linux/src/linux/net/poll_socket.rs
index 7d5fe6ef8..bec59ec99 100644
--- a/userland/capsule_linux/src/linux/net/poll_socket.rs
+++ b/userland/capsule_linux/src/linux/net/poll_socket.rs
@@ -14,15 +14,34 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-//! Asking net.sockets whether one handle is ready.
+//! A socket's readiness in poll's bits: the family's own sockets answer from
+//! the table, a stream outside the family from net.sockets.
use super::call::call;
use super::ops::{OP_POLL, POLL_READABLE, POLL_WRITABLE};
+use super::sock;
const POLLIN: u16 = 0x001;
const POLLOUT: u16 = 0x004;
+const POLLNVAL: u16 = 0x020;
-pub(super) fn socket_bits(handle: u32) -> u16 {
+pub(super) fn socket_bits(id: u32) -> u16 {
+ if let Some(bits) = sock::bits(id) {
+ return bits;
+ }
+ match sock::with(|t| t.get(id).and_then(|s| s.svc)) {
+ Some(handle) => service_bits(handle),
+ None => POLLNVAL,
+ }
+}
+
+/// True when `id` is a stream net.sockets holds: nothing tells the family
+/// when it changes, so a wait on it is looked at again on a tick.
+pub fn outside(id: u32) -> bool {
+ sock::with(|t| t.get(id).is_some_and(|s| s.svc.is_some()))
+}
+
+fn service_bits(handle: u32) -> u16 {
let Some((0, out)) = call(OP_POLL, &handle.to_le_bytes(), 1) else {
return 0;
};
diff --git a/userland/capsule_linux/src/linux/net/addr.rs b/userland/capsule_linux/src/linux/net/recvfrom.rs
similarity index 54%
rename from userland/capsule_linux/src/linux/net/addr.rs
rename to userland/capsule_linux/src/linux/net/recvfrom.rs
index 330c461cf..f0179b9b7 100644
--- a/userland/capsule_linux/src/linux/net/addr.rs
+++ b/userland/capsule_linux/src/linux/net/recvfrom.rs
@@ -14,25 +14,33 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
+//! `recvfrom`: a read that says where the bytes came from.
-//! `struct sockaddr_in` out of a guest.
+use alloc::vec;
use crate::linux::guest::Guest;
-/// family(2) port(2) addr(4), and the rest of the sixteen bytes unused.
-const SOCKADDR_IN: usize = 16;
-const AF_INET: u16 = 2;
+use super::dgram::is_resolver;
+use super::fd::sock_of;
-/// Port in network order and address in network order, which is the
-/// order net.sockets wants as well, so neither is byte swapped here.
-pub fn inet(guest: &Guest, at: u64, len: u64) -> Option<(u16, [u8; 4])> {
- if len < SOCKADDR_IN as u64 {
- return None;
+pub fn recvfrom(
+ guest: &mut Guest,
+ fd: u64,
+ buf: u64,
+ len: u64,
+ flags: u64,
+ at: u64,
+ alen: u64,
+) -> u64 {
+ if is_resolver(guest, fd) {
+ return super::resolver::answer(guest, fd, buf, len, at, alen);
}
- let raw = guest.read(at, SOCKADDR_IN)?;
- if u16::from_le_bytes([raw[0], raw[1]]) != AF_INET {
- return None;
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ match super::xfer_in::recv(guest, id, &vec![(buf, len)], 0, flags) {
+ Ok(got) => super::peer_addr::finish(guest, got, flags, at, alen),
+ Err(e) => e,
}
- let port = u16::from_be_bytes([raw[2], raw[3]]);
- Some((port, [raw[4], raw[5], raw[6], raw[7]]))
}
diff --git a/userland/capsule_linux/src/linux/net/resolver.rs b/userland/capsule_linux/src/linux/net/resolver.rs
new file mode 100644
index 000000000..9ca2e7bb3
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/resolver.rs
@@ -0,0 +1,64 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A datagram socket that talks to a nameserver. This capsule answers name
+//! queries itself (`dns`), so a query never leaves it: a socket that sends
+//! to port 53 outside the family, or to a loopback port 53 no family socket
+//! holds, becomes the resolver descriptor it always was before a guest
+//! could bind a datagram socket of its own.
+
+use crate::linux::guest::{Fd, Guest};
+
+use super::dgram_addr::{encode, fill};
+use super::dns;
+
+use super::sock::{self, Addr, Proto};
+use super::sockaddr::is_loopback;
+
+const NAMESERVER_PORT: u16 = 53;
+
+pub fn is_nameserver(to: Addr) -> bool {
+ to.port == NAMESERVER_PORT
+ && (!is_loopback(to.ip) || sock::with(|t| t.bound(Proto::Dgram, to).is_none()))
+}
+
+/// Let go of the socket `fd` names and make `fd` the resolver, keeping its
+/// descriptor flags.
+pub fn become_resolver(guest: &mut Guest, fd: u64) {
+ super::close::close(guest, fd);
+ if let Some(f) = guest.fds.get_mut(fd as usize) {
+ let (cloexec, nonblock) = (f.cloexec, f.nonblock);
+ *f = Fd::resolver();
+ f.cloexec = cloexec;
+ f.nonblock = nonblock;
+ }
+}
+
+/// A query written to the resolver. A program with no `resolv.conf` asks
+/// the loopback address, and one with a configured nameserver asks that.
+pub fn query(guest: &mut Guest, fd: u64, buf: u64, len: u64, to: Option) -> u64 {
+ let peer = to.map_or((NAMESERVER_PORT, [127, 0, 0, 1]), |a| (a.port, a.ip));
+ dns::query(guest, fd, buf, len, encode(peer))
+}
+
+/// An answer read from the resolver, with the nameserver it came from.
+pub fn answer(guest: &mut Guest, fd: u64, buf: u64, len: u64, at: u64, alen: u64) -> u64 {
+ let (got, from) = dns::answer_out(guest, fd, buf, len);
+ match from {
+ Some(peer) if at != 0 => fill(guest, at, alen, peer, got),
+ _ => got,
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/shutdown.rs b/userland/capsule_linux/src/linux/net/shutdown.rs
new file mode 100644
index 000000000..ce4cd7cb1
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/shutdown.rs
@@ -0,0 +1,70 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! `shutdown`: one direction or both, on the socket rather than the
+//! descriptor, which stays open. A dup'd or inherited descriptor sees it too.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::fd::sock_of;
+use super::policy::refuse;
+use super::sock::{self, Proto};
+
+const SHUT_RD: u64 = 0;
+const SHUT_WR: u64 = 1;
+const SHUT_RDWR: u64 = 2;
+
+pub fn shutdown(guest: &Guest, fd: u64, how: u64) -> u64 {
+ if how > SHUT_RDWR {
+ return errno::fail(errno::EINVAL);
+ }
+ let id = match sock_of(guest, fd) {
+ Ok(id) => id,
+ Err(e) => return e,
+ };
+ let (rd, wr) = (how != SHUT_WR, how != SHUT_RD);
+ sock::with(|t| {
+ let Some(s) = t.get_mut(id) else {
+ return errno::fail(errno::EBADF);
+ };
+ if s.svc.is_some() {
+ return refuse(
+ "shutdown of a stream outside the family: net.sockets has no half-close",
+ errno::EOPNOTSUPP,
+ );
+ }
+ if s.listening {
+ /* Shutting a listener's reading side stops it listening. */
+ if rd {
+ t.unlisten(id);
+ }
+ return errno::ok(0);
+ }
+ let connected = s.connected || (s.proto == Proto::Dgram && s.remote.is_some());
+ if !connected {
+ return errno::fail(errno::ENOTCONN);
+ }
+ s.rd_shut |= rd;
+ s.wr_shut |= wr;
+ let peer = s.peer;
+ /* The peer reads end of file once it has what was already sent. */
+ if let Some(p) = peer.filter(|_| wr).and_then(|p| t.get_mut(p)) {
+ p.eof = true;
+ }
+ errno::ok(0)
+ })
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/bind.rs b/userland/capsule_linux/src/linux/net/sock/bind.rs
new file mode 100644
index 000000000..5b74f73f6
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/bind.rs
@@ -0,0 +1,42 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A local address for a socket that sends or connects before it binds.
+
+use crate::linux::abi::errno::EADDRNOTAVAIL;
+
+use super::table::Socks;
+use super::types::{Addr, Domain};
+
+/// The loopback route's source address.
+const LOOPBACK: [u8; 4] = [127, 0, 0, 1];
+
+impl Socks {
+ /// Bind `id` to 127.0.0.1 and a free ephemeral port, as Linux's
+ /// autobind does, unless it is bound already or is a socketpair end,
+ /// which has no address.
+ pub fn autobind(&mut self, id: u32) -> Result<(), i64> {
+ let unbound = |s: &&super::types::Sock| s.local.is_none() && s.domain == Domain::Inet;
+ let Some(proto) = self.get(id).filter(unbound).map(|s| s.proto) else {
+ return Ok(());
+ };
+ let port = self.ephemeral(proto, LOOPBACK).ok_or(EADDRNOTAVAIL)?;
+ if let Some(s) = self.get_mut(id) {
+ s.local = Some(Addr { ip: LOOPBACK, port });
+ }
+ Ok(())
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/cell.rs b/userland/capsule_linux/src/linux/net/sock/cell.rs
new file mode 100644
index 000000000..47216b6e8
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/cell.rs
@@ -0,0 +1,39 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Where the table lives. This personality hosts one family and serves it
+//! from one thread, so the table is a single process-wide value: net.sockets
+//! keeps its own the same way.
+
+use core::cell::RefCell;
+
+use super::table::Socks;
+
+struct One(RefCell);
+
+/*
+ * SAFETY: the serve loop is the only thread in this capsule that reaches the
+ * table; guest threads run in their own processes and only trap into it.
+ */
+unsafe impl Sync for One {}
+
+static TABLE: One = One(RefCell::new(Socks::new()));
+
+/// Run `f` on the table. A call inside `f` that reaches the table again would
+/// panic on the borrow, so no function here calls out while holding it.
+pub fn with(f: impl FnOnce(&mut Socks) -> R) -> R {
+ f(&mut TABLE.0.borrow_mut())
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/deliver.rs b/userland/capsule_linux/src/linux/net/sock/deliver.rs
new file mode 100644
index 000000000..a55a6bb70
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/deliver.rs
@@ -0,0 +1,31 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Bytes from one socket into the family: a stream's to its peer, a
+//! datagram to where `dest` says, binding the sender first as Linux does.
+
+use super::gram_dest::Dest;
+use super::table::Socks;
+use super::types::Proto;
+
+impl Socks {
+ pub fn deliver(&mut self, id: u32, dest: Dest, bytes: &[u8]) -> Result {
+ match self.get(id).map(|s| s.proto) {
+ Some(Proto::Stream) => self.write(id, bytes),
+ _ => self.autobind(id).and_then(|()| self.send_gram(id, dest, bytes)),
+ }
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/free.rs b/userland/capsule_linux/src/linux/net/sock/free.rs
new file mode 100644
index 000000000..3cbc55df6
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/free.rs
@@ -0,0 +1,67 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Letting a socket go, and what its peer sees when it does.
+
+use crate::linux::abi::errno::ECONNRESET;
+
+use super::table::Socks;
+
+impl Socks {
+ /// `pid` no longer holds `id`. The socket goes when nobody does.
+ pub fn release(&mut self, id: u32, pid: u32) {
+ let Some(s) = self.get_mut(id) else {
+ return;
+ };
+ let held = s.holders.len();
+ s.holders.retain(|&p| p != pid);
+ if held != 0 && s.holders.is_empty() {
+ self.free(id, false);
+ }
+ }
+
+ /// Close `id`. Its peer reads end of file, or ECONNRESET if this end
+ /// left bytes unread, set SO_LINGER to zero seconds, or `reset` is set,
+ /// which is when Linux sends a reset instead of a FIN. Connections still
+ /// queued on a listener are reset.
+ pub fn free(&mut self, id: u32, reset: bool) {
+ let Some(gone) = self.list.get_mut(id as usize).and_then(Option::take) else {
+ return;
+ };
+ if let Some(h) = gone.svc {
+ super::super::stream::close(h);
+ }
+ let reset = reset || !gone.rx.is_empty() || gone.opts.linger == (1, 0);
+ /*
+ * A stream's peer points back; a connected Unix datagram socket
+ * points at this one alone. Either is told, and forgets the index.
+ */
+ for p in self.list.iter_mut().flatten().filter(|p| p.peer == Some(id)) {
+ p.peer = None;
+ p.eof = true;
+ if reset && p.proto == super::types::Proto::Stream {
+ p.error = ECONNRESET;
+ }
+ }
+ for queued in gone.pending {
+ self.free(queued, true);
+ }
+ self.refuse_waiting(gone.syn.into_iter());
+ for l in self.list.iter_mut().flatten() {
+ l.syn.retain(|&c| c != id);
+ }
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/gram.rs b/userland/capsule_linux/src/linux/net/sock/gram.rs
new file mode 100644
index 000000000..8552f6de8
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/gram.rs
@@ -0,0 +1,50 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A datagram into the family: to whichever socket holds the port or the
+//! Unix name it is sent to, or to a connected socket's peer.
+
+use crate::linux::abi::errno::{ECONNREFUSED, EPERM};
+
+use super::gram_dest::Dest;
+use super::name::{Gram, Peer};
+use super::table::Socks;
+use super::types::Domain;
+
+impl Socks {
+ /// One datagram from `id`. One nobody holds the port of is dropped, as
+ /// on the wire.
+ pub fn send_gram(&mut self, id: u32, dest: Dest, bytes: &[u8]) -> Result {
+ let target = self.gram_target(id, dest, bytes.len())?;
+ let (Some(t), Some(from)) = (target, self.sender(id)) else {
+ return Ok(bytes.len());
+ };
+ let r = self.get_mut(t).ok_or(ECONNREFUSED)?;
+ /* A connected Unix socket takes only from its peer, and says so. */
+ if r.domain == Domain::Unix && r.connected && r.peer.is_some_and(|p| p != id) {
+ return Err(EPERM);
+ }
+ let queued: usize = r.grams.iter().map(|g| g.bytes.len()).sum();
+ let wanted = match (&from, r.remote) {
+ (Peer::Inet(a), Some(x)) => *a == x,
+ _ => true,
+ };
+ if wanted && queued + bytes.len() <= r.opts.rcvbuf as usize {
+ r.grams.push_back(Gram { from, bytes: bytes.to_vec() });
+ }
+ Ok(bytes.len())
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/gram_dest.rs b/userland/capsule_linux/src/linux/net/sock/gram_dest.rs
new file mode 100644
index 000000000..812a081fa
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/gram_dest.rs
@@ -0,0 +1,44 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Where a datagram goes, and which socket holds an IPv4 port.
+
+use crate::linux::abi::errno::ECONNREFUSED;
+
+use super::table::Socks;
+use super::types::{Addr, Proto};
+
+/// Where a datagram goes: where the socket connected, an IPv4 address, or
+/// a Unix socket already found by its name (`unix_name::find`).
+pub enum Dest {
+ Default,
+ Inet(Addr),
+ Sock(u32),
+}
+
+impl Socks {
+ /// The socket bound to `to`; None drops the datagram, and a connected
+ /// sender is told on its next call.
+ pub(super) fn inet_target(&mut self, id: u32, to: Addr, connected: bool) -> Option {
+ let found = self.bound(Proto::Dgram, to);
+ if found.is_none() {
+ if let Some(s) = self.get_mut(id).filter(|_| connected) {
+ s.error = ECONNREFUSED;
+ }
+ }
+ found
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/gram_in.rs b/userland/capsule_linux/src/linux/net/sock/gram_in.rs
new file mode 100644
index 000000000..328f2cc78
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/gram_in.rs
@@ -0,0 +1,54 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A datagram out of the family's queue for one socket.
+
+use alloc::vec::Vec;
+use core::mem;
+
+use crate::linux::abi::errno::{EAGAIN, EBADF};
+
+use super::name::Peer;
+use super::table::Socks;
+
+impl Socks {
+ /// The next datagram, cut to `want`, with its whole length and sender.
+ pub fn take_gram(&mut self, id: u32, want: usize, peek: bool) -> Result {
+ let s = self.get_mut(id).ok_or(EBADF)?;
+ if let Some(g) = s.grams.front() {
+ let (bytes, whole, from) =
+ (g.bytes[..want.min(g.bytes.len())].to_vec(), g.bytes.len(), g.from.clone());
+ if !peek {
+ s.grams.pop_front();
+ }
+ return Ok(Got { bytes, whole, from });
+ }
+ if s.error != 0 {
+ return Err(mem::take(&mut s.error));
+ }
+ if s.rd_shut {
+ return Ok(Got { bytes: Vec::new(), whole: 0, from: Peer::Unix(None) });
+ }
+ Err(EAGAIN)
+ }
+}
+
+pub struct Got {
+ pub bytes: Vec,
+ /// The datagram's length before it was cut to fit.
+ pub whole: usize,
+ pub from: Peer,
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/gram_target.rs b/userland/capsule_linux/src/linux/net/sock/gram_target.rs
new file mode 100644
index 000000000..1fcdc62f5
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/gram_target.rs
@@ -0,0 +1,70 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Which socket a datagram goes to, or why it goes nowhere.
+
+use core::mem;
+
+use crate::linux::abi::errno::{EBADF, ECONNREFUSED, EDESTADDRREQ, EINVAL, EMSGSIZE, ENOTCONN};
+
+use super::gram_dest::Dest;
+use super::name::Peer;
+use super::table::Socks;
+use super::types::Domain;
+
+/// The largest UDP payload IPv4 carries.
+const MAX_GRAM: usize = 65507;
+
+impl Socks {
+ /// The socket a datagram from `id` goes to; None drops it unseen.
+ pub(super) fn gram_target(
+ &mut self,
+ id: u32,
+ dest: Dest,
+ len: usize,
+ ) -> Result, i64> {
+ let s = self.get_mut(id).ok_or(EBADF)?;
+ if s.error != 0 {
+ return Err(mem::take(&mut s.error));
+ }
+ if len > MAX_GRAM {
+ return Err(EMSGSIZE);
+ }
+ let (remote, peer, connected, domain) = (s.remote, s.peer, s.connected, s.domain);
+ match (dest, domain) {
+ (Dest::Sock(t), Domain::Unix) => Ok(Some(t)),
+ (Dest::Default, Domain::Unix) => match peer {
+ Some(p) => Ok(Some(p)),
+ None if connected => Err(ECONNREFUSED),
+ None => Err(ENOTCONN),
+ },
+ (Dest::Inet(to), Domain::Inet) => Ok(self.inet_target(id, to, remote.is_some())),
+ (Dest::Default, Domain::Inet) => match remote {
+ Some(to) => Ok(self.inet_target(id, to, true)),
+ None => Err(EDESTADDRREQ),
+ },
+ _ => Err(EINVAL),
+ }
+ }
+
+ /// How a datagram from `id` names its sender.
+ pub(super) fn sender(&self, id: u32) -> Option {
+ self.get(id).map(|s| match s.domain {
+ Domain::Inet => Peer::Inet(s.local.unwrap_or_default()),
+ Domain::Unix => Peer::Unix(s.uname.clone()),
+ })
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/holders.rs b/userland/capsule_linux/src/linux/net/sock/holders.rs
new file mode 100644
index 000000000..572e28b71
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/holders.rs
@@ -0,0 +1,62 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Which processes hold a socket. A descriptor copied by fork is held by the
+//! child as well; the socket is let go when the last holder closes it or
+//! ends. A process that ends without closing lets go of everything it held
+//! when the family drops it, which is what `Holder` does.
+
+use alloc::vec::Vec;
+
+use super::cell::with;
+use crate::linux::guest::{Fd, Kind};
+
+/// Carried by each hosted process, as the pid its holdings are kept under.
+pub struct Holder {
+ pid: u32,
+}
+
+impl Holder {
+ pub fn new(pid: u32) -> Holder {
+ Holder { pid }
+ }
+
+ /// A forked child holds every socket its parent's descriptors name.
+ pub fn fork(&self, child: u32, fds: &[Fd]) {
+ with(|t| {
+ for f in fds.iter().filter(|f| f.kind == Kind::Socket) {
+ if let Some(s) = t.get_mut(f.handle) {
+ if !s.holders.contains(&child) {
+ s.holders.push(child);
+ }
+ }
+ }
+ });
+ }
+}
+
+impl Drop for Holder {
+ fn drop(&mut self) {
+ let pid = self.pid;
+ with(|t| {
+ let held: Vec =
+ t.iter().filter(|(_, s)| s.holders.contains(&pid)).map(|(i, _)| i).collect();
+ for id in held {
+ t.release(id, pid);
+ }
+ });
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/kinds.rs b/userland/capsule_linux/src/linux/net/sock/kinds.rs
new file mode 100644
index 000000000..e84a90ebd
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/kinds.rs
@@ -0,0 +1,47 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! What kind of socket, in which family, at which IPv4 address.
+
+#[derive(Clone, Copy, PartialEq, Eq)]
+pub enum Proto {
+ Stream,
+ Dgram,
+}
+
+/// AF_INET, or AF_UNIX for the two ends socketpair makes.
+#[derive(Clone, Copy, PartialEq, Eq)]
+pub enum Domain {
+ Inet,
+ Unix,
+}
+
+/// An IPv4 address and a port, the port in host order.
+#[derive(Clone, Copy, PartialEq, Eq, Default)]
+pub struct Addr {
+ pub ip: [u8; 4],
+ pub port: u16,
+}
+
+/// How a stream connect to a listener went.
+pub enum Link {
+ /// Connected; the listener has one more connection to accept.
+ Done,
+ /// Nothing listens there: Linux's loopback answers with a reset.
+ Refused,
+ /// The listener's queue is full.
+ Full,
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/link.rs b/userland/capsule_linux/src/linux/net/sock/link.rs
new file mode 100644
index 000000000..4a4a3d56b
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/link.rs
@@ -0,0 +1,68 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Joining two stream ends: a connect on loopback or to a Unix name, which
+//! Linux completes in the caller's own call, and socketpair.
+
+use super::kinds::Link;
+use super::table::Socks;
+use super::types::{Addr, Proto};
+
+impl Socks {
+ /// Connect stream `id`, already bound, to the listener at `to`.
+ pub fn link(&mut self, id: u32, to: Addr) -> Link {
+ let from = self.get(id).and_then(|s| s.local).map_or(0, |a| a.port);
+ match self.listener(to, from) {
+ Some(l) => self.join(id, l),
+ None => Link::Refused,
+ }
+ }
+
+ /// Connect stream `id` to listener `l`: a new server end, with the
+ /// listener's name and options, waits in its queue for accept.
+ pub fn join(&mut self, id: u32, l: u32) -> Link {
+ let Some(ls) = self.get(l) else {
+ return Link::Refused;
+ };
+ /* Linux queues one more than the backlog it was given. */
+ if ls.pending.len() > ls.backlog {
+ return Link::Full;
+ }
+ let (domain, local, uname, opts) = (ls.domain, ls.local, ls.uname.clone(), ls.opts);
+ let (from, from_name) = self.get(id).map_or((None, None), |c| (c.local, c.uname.clone()));
+ let server = self.open(domain, Proto::Stream, None);
+ if let Some(s) = self.get_mut(server) {
+ s.local = local;
+ s.remote = from;
+ s.uname = uname.clone();
+ s.upeer = from_name;
+ s.peer = Some(id);
+ s.connected = true;
+ /* An accepted socket starts with its listener's options. */
+ s.opts = opts;
+ }
+ if let Some(c) = self.get_mut(id) {
+ c.remote = local;
+ c.upeer = uname;
+ c.peer = Some(server);
+ c.connected = true;
+ }
+ if let Some(l) = self.get_mut(l) {
+ l.pending.push_back(server);
+ }
+ Link::Done
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/mod.rs b/userland/capsule_linux/src/linux/net/sock/mod.rs
new file mode 100644
index 000000000..2c0193c71
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/mod.rs
@@ -0,0 +1,63 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The family's sockets: one table this personality keeps for every process
+//! it hosts, so a connection between two of them, or two threads of one,
+//! never leaves the capsule. A descriptor names an entry by index; the entry
+//! records which processes hold it, and is let go when none does.
+//!
+//! A stream to 127.0.0.0/8 is a pair of entries, each holding what the other
+//! wrote. A datagram to a bound loopback port is queued on that entry. A
+//! stream to anywhere else is an entry backed by a net.sockets handle.
+
+mod bind;
+mod cell;
+mod deliver;
+mod free;
+mod gram;
+mod gram_dest;
+mod gram_in;
+mod gram_target;
+mod holders;
+mod kinds;
+mod link;
+mod name;
+mod new;
+mod opts;
+mod opts_more;
+mod pair;
+mod port;
+mod progress;
+mod put;
+mod ready;
+mod recv;
+mod send;
+mod syn;
+mod table;
+mod take;
+mod types;
+mod unlisten;
+
+pub use cell::with;
+pub use gram_dest::Dest;
+pub use holders::Holder;
+pub use kinds::Link;
+pub use name::{Peer, UName};
+pub use opts::Opts;
+pub use opts_more::More;
+pub use progress::{progress, set_progress};
+pub use ready::bits;
+pub use types::{Addr, Domain, Proto, Sock};
diff --git a/userland/capsule_linux/src/linux/net/sock/name.rs b/userland/capsule_linux/src/linux/net/sock/name.rs
new file mode 100644
index 000000000..665eac23d
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/name.rs
@@ -0,0 +1,51 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A Unix socket's name, who a message came from, and a datagram.
+
+use alloc::vec::Vec;
+
+use super::types::Addr;
+
+/// A bound Unix name. `key` is what two sockets must share to meet: the
+/// resolved path, or a NUL then the abstract name. `shown` is the sun_path
+/// the guest gave, which getsockname and accept report back.
+#[derive(Clone, PartialEq, Eq)]
+pub struct UName {
+ pub key: Vec,
+ pub shown: Vec,
+}
+
+impl UName {
+ /// True for a name in the abstract namespace, which no file backs.
+ pub fn is_abstract(&self) -> bool {
+ self.key.first() == Some(&0)
+ }
+}
+
+/// The far end of a message or a connection, as a sockaddr reports it: an
+/// IPv4 address, or a Unix name, None for an unnamed socket.
+#[derive(Clone)]
+pub enum Peer {
+ Inet(Addr),
+ Unix(Option),
+}
+
+/// A datagram waiting to be read, with its sender.
+pub struct Gram {
+ pub from: Peer,
+ pub bytes: Vec,
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/new.rs b/userland/capsule_linux/src/linux/net/sock/new.rs
new file mode 100644
index 000000000..81749c62d
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/new.rs
@@ -0,0 +1,54 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A socket as it starts: unbound, unconnected, Linux's default options.
+
+use alloc::collections::VecDeque;
+use alloc::vec;
+use alloc::vec::Vec;
+
+use super::opts::Opts;
+use super::types::{Domain, Proto, Sock};
+
+impl Sock {
+ pub fn new(domain: Domain, proto: Proto, pid: Option) -> Sock {
+ Sock {
+ domain,
+ proto,
+ local: None,
+ remote: None,
+ listening: false,
+ backlog: 0,
+ pending: VecDeque::new(),
+ syn: VecDeque::new(),
+ connecting: false,
+ peer: None,
+ connected: false,
+ rx: VecDeque::new(),
+ grams: VecDeque::new(),
+ eof: false,
+ wr_shut: false,
+ rd_shut: false,
+ broken: false,
+ error: 0,
+ opts: Opts::new(proto),
+ svc: None,
+ holders: pid.map_or_else(Vec::new, |p| vec![p]),
+ uname: None,
+ upeer: None,
+ }
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/opts.rs b/userland/capsule_linux/src/linux/net/sock/opts.rs
new file mode 100644
index 000000000..4b6d268d8
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/opts.rs
@@ -0,0 +1,74 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The options a socket keeps, with the values Linux starts a socket with
+//! (read from a Linux 6.x host: tcp_rmem[1], tcp_wmem[1], rmem_default and
+//! wmem_default, and the TCP keepalive defaults).
+
+use super::types::Proto;
+
+#[derive(Clone, Copy)]
+pub struct Opts {
+ pub reuseaddr: bool,
+ pub reuseport: bool,
+ pub keepalive: bool,
+ pub broadcast: bool,
+ pub nodelay: bool,
+ /// As getsockopt reports them: Linux doubles what setsockopt was given.
+ pub rcvbuf: u32,
+ pub sndbuf: u32,
+ pub keepidle: u32,
+ pub keepintvl: u32,
+ pub keepcnt: u32,
+ /// SO_LINGER's l_onoff and l_linger.
+ pub linger: (u32, u32),
+ /// SO_RCVTIMEO and SO_SNDTIMEO as the guest wrote them, seconds and
+ /// microseconds, so they read back exactly.
+ pub rcvtimeo: (u64, u64),
+ pub sndtimeo: (u64, u64),
+ pub more: super::opts_more::More,
+}
+
+impl Opts {
+ pub fn new(proto: Proto) -> Opts {
+ let (rcvbuf, sndbuf) = match proto {
+ Proto::Stream => (131_072, 16_384),
+ Proto::Dgram => (212_992, 212_992),
+ };
+ Opts {
+ reuseaddr: false,
+ reuseport: false,
+ keepalive: false,
+ broadcast: false,
+ nodelay: false,
+ rcvbuf,
+ sndbuf,
+ keepidle: 7200,
+ keepintvl: 75,
+ keepcnt: 9,
+ linger: (0, 0),
+ rcvtimeo: (0, 0),
+ sndtimeo: (0, 0),
+ more: Default::default(),
+ }
+ }
+
+ /// Milliseconds a receive or a send may wait, None for no limit.
+ pub fn limit_ms((sec, usec): (u64, u64)) -> Option {
+ let ms = sec.saturating_mul(1000).saturating_add(usec.div_ceil(1000));
+ (ms != 0).then_some(ms)
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/dns/open.rs b/userland/capsule_linux/src/linux/net/sock/opts_more.rs
similarity index 63%
rename from userland/capsule_linux/src/linux/net/dns/open.rs
rename to userland/capsule_linux/src/linux/net/sock/opts_more.rs
index fb3bd77be..1e10aa6e0 100644
--- a/userland/capsule_linux/src/linux/net/dns/open.rs
+++ b/userland/capsule_linux/src/linux/net/sock/opts_more.rs
@@ -14,16 +14,20 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-//! The descriptor a program opens to reach its nameserver.
+//! The options `opt_more` keeps, with Linux's starting values.
-use crate::linux::abi::errno;
-use crate::linux::guest::{Fd, Guest};
+#[derive(Clone, Copy)]
+pub struct More {
+ pub tos: u32,
+ pub ttl: u32,
+ pub priority: u32,
+ pub user_timeout: u32,
+ pub quickack: bool,
+ pub fastopen: u32,
+}
-/// It looks like a datagram socket and holds no handle: there is
-/// nothing on the other side of it, which is the point.
-pub fn open(guest: &mut Guest) -> u64 {
- match crate::linux::file::install(guest, Fd::resolver()) {
- Some(n) => errno::ok(n),
- None => errno::fail(errno::EMFILE),
+impl Default for More {
+ fn default() -> More {
+ More { tos: 0, ttl: 64, priority: 0, user_timeout: 0, quickack: true, fastopen: 0 }
}
}
diff --git a/userland/capsule_linux/src/linux/net/sock/pair.rs b/userland/capsule_linux/src/linux/net/sock/pair.rs
new file mode 100644
index 000000000..04e294ed2
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/pair.rs
@@ -0,0 +1,35 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! socketpair: two sockets connected from the start.
+
+use super::table::Socks;
+use super::types::{Domain, Proto};
+
+impl Socks {
+ /// Two connected sockets, both held by `pid`.
+ pub fn pair(&mut self, domain: Domain, proto: Proto, pid: u32) -> (u32, u32) {
+ let a = self.open(domain, proto, Some(pid));
+ let b = self.open(domain, proto, Some(pid));
+ for (me, other) in [(a, b), (b, a)] {
+ if let Some(s) = self.get_mut(me) {
+ s.peer = Some(other);
+ s.connected = true;
+ }
+ }
+ (a, b)
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/port.rs b/userland/capsule_linux/src/linux/net/sock/port.rs
new file mode 100644
index 000000000..e006c3dcb
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/port.rs
@@ -0,0 +1,73 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Ports on the loopback address: who holds one, and a free one when a
+//! program asks for port 0 or connects before it binds.
+
+use super::table::Socks;
+use super::types::{Addr, Proto};
+
+/// Linux's net.ipv4.ip_local_port_range default.
+const FIRST: u16 = 32768;
+const LAST: u16 = 60999;
+
+impl Socks {
+ /// The socket of this kind bound to exactly this address, if any.
+ pub fn bound(&self, proto: Proto, at: Addr) -> Option {
+ self.iter().find(|(_, s)| s.proto == proto && s.local == Some(at)).map(|(i, _)| i)
+ }
+
+ /// The listener a connect to `at` from port `from` reaches. Listeners
+ /// that share a port with SO_REUSEPORT take connections by the
+ /// connecting port, as Linux spreads them by a hash of the connection.
+ pub fn listener(&self, at: Addr, from: u16) -> Option {
+ let group: alloc::vec::Vec = self
+ .iter()
+ .filter(|(_, s)| s.listening && s.proto == Proto::Stream && s.local == Some(at))
+ .map(|(i, _)| i)
+ .collect();
+ group.get(usize::from(from) % group.len().max(1)).copied()
+ }
+
+ /// True when binding `id` to `at` takes a port another socket holds.
+ /// Two sockets share one when both set SO_REUSEPORT, or when both set
+ /// SO_REUSEADDR and the other does not listen, as Linux allows.
+ pub fn in_use(&self, id: u32, at: Addr) -> bool {
+ let Some(me) = self.get(id) else {
+ return true;
+ };
+ self.iter().any(|(i, s)| {
+ i != id
+ && s.proto == me.proto
+ && s.local == Some(at)
+ && !(s.opts.reuseport && me.opts.reuseport)
+ && (s.listening || !(s.opts.reuseaddr && me.opts.reuseaddr))
+ })
+ }
+
+ /// A port in the ephemeral range no socket of this kind holds on `ip`.
+ pub fn ephemeral(&mut self, proto: Proto, ip: [u8; 4]) -> Option {
+ let span = (LAST - FIRST + 1) as u32;
+ for step in 0..span {
+ let port = FIRST + ((u32::from(self.next_port) + step) % span) as u16;
+ if self.bound(proto, Addr { ip, port }).is_none() {
+ self.next_port = (port - FIRST + 1) % span as u16;
+ return Some(port);
+ }
+ }
+ None
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/progress.rs b/userland/capsule_linux/src/linux/net/sock/progress.rs
new file mode 100644
index 000000000..a0d9cc555
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/progress.rs
@@ -0,0 +1,35 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! How far a waiting call has got. A blocking send on Linux returns once
+//! every byte is queued, and a receive with MSG_WAITALL once its buffer is
+//! full; a call parked partway keeps its count here, by thread, until it is
+//! answered.
+
+use super::cell::with;
+
+pub fn progress(tid: u32) -> usize {
+ with(|t| t.progress.iter().find(|p| p.0 == tid).map_or(0, |p| p.1))
+}
+
+pub fn set_progress(tid: u32, done: usize) {
+ with(|t| {
+ t.progress.retain(|p| p.0 != tid);
+ if done != 0 {
+ t.progress.push((tid, done));
+ }
+ });
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/put.rs b/userland/capsule_linux/src/linux/net/sock/put.rs
new file mode 100644
index 000000000..b7332f101
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/put.rs
@@ -0,0 +1,57 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Bytes into a family stream's peer, as many as it has room for.
+
+use core::mem;
+
+use crate::linux::abi::errno::{EAGAIN, EBADF, EPIPE};
+
+use super::table::Socks;
+use super::types::Domain;
+
+impl Socks {
+ /// The bytes, and the peer that took them.
+ pub(super) fn put(&mut self, id: u32, bytes: &[u8]) -> Result<(usize, Option), i64> {
+ let s = self.get_mut(id).ok_or(EBADF)?;
+ if s.error != 0 {
+ return Err(mem::take(&mut s.error));
+ }
+ if !s.connected || s.wr_shut || s.broken {
+ return Err(EPIPE);
+ }
+ let Some(p) = s.peer else {
+ /*
+ * TCP takes the first write after the peer left, and the reset
+ * that answers it makes every later one EPIPE. A Unix socket
+ * knows at once.
+ */
+ if s.domain == Domain::Unix {
+ return Err(EPIPE);
+ }
+ s.broken = true;
+ return Ok((bytes.len(), None));
+ };
+ let peer = self.get_mut(p).ok_or(EPIPE)?;
+ let room = (peer.opts.rcvbuf as usize).saturating_sub(peer.rx.len());
+ if room == 0 && !bytes.is_empty() {
+ return Err(EAGAIN);
+ }
+ let n = room.min(bytes.len());
+ peer.rx.extend(&bytes[..n]);
+ Ok((n, Some(p)))
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/ready.rs b/userland/capsule_linux/src/linux/net/sock/ready.rs
new file mode 100644
index 000000000..07a994043
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/ready.rs
@@ -0,0 +1,71 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! What a family socket can do now, in poll's bits, as Linux's tcp_poll,
+//! udp_poll and unix_poll report it.
+
+use super::cell::with;
+use super::types::{Proto, Sock};
+
+const POLLIN: u16 = 0x001;
+const POLLOUT: u16 = 0x004;
+const POLLERR: u16 = 0x008;
+const POLLHUP: u16 = 0x010;
+const POLLRDHUP: u16 = 0x2000;
+
+/// The bits for `id`, or None when net.sockets holds its state.
+pub fn bits(id: u32) -> Option {
+ with(|t| {
+ let s = t.get(id).filter(|s| s.svc.is_none())?;
+ /* A stream is writable while its peer has room, or once it is gone. */
+ let room =
+ s.peer.and_then(|p| t.get(p)).is_none_or(|p| p.rx.len() < p.opts.rcvbuf as usize);
+ Some(of(s, room))
+ })
+}
+
+fn of(s: &Sock, room: bool) -> u16 {
+ let err = if s.error != 0 { POLLERR } else { 0 };
+ if s.proto == Proto::Dgram {
+ let rd = if s.grams.is_empty() { 0 } else { POLLIN };
+ return err | rd | POLLOUT;
+ }
+ if s.listening {
+ return err | if s.pending.is_empty() { 0 } else { POLLIN };
+ }
+ /* A connect waiting for room: SYN_SENT, neither readable nor writable. */
+ if s.connecting {
+ return err;
+ }
+ /* Never connected, refused, or reset: Linux's TCP_CLOSE. */
+ if !s.connected || s.broken {
+ let rd = if s.connected { POLLIN | POLLRDHUP } else { 0 };
+ return err | rd | POLLOUT | POLLHUP;
+ }
+ let shut_rd = s.eof || s.rd_shut;
+ let mut set = err;
+ if !s.rx.is_empty() || shut_rd {
+ set |= POLLIN;
+ }
+ if shut_rd {
+ set |= POLLRDHUP;
+ }
+ if shut_rd && s.wr_shut {
+ set |= POLLHUP;
+ }
+ /* After SHUT_WR a write fails at once, so Linux reports it writable. */
+ set | if room || s.wr_shut { POLLOUT } else { 0 }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/recv.rs b/userland/capsule_linux/src/linux/net/sock/recv.rs
new file mode 100644
index 000000000..730901fce
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/recv.rs
@@ -0,0 +1,49 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Bytes out of a family stream.
+
+use alloc::vec::Vec;
+use core::mem;
+
+use crate::linux::abi::errno::{EAGAIN, EBADF, ENOTCONN};
+
+use super::table::Socks;
+
+impl Socks {
+ /// Up to `want` bytes of a stream; empty is end of file, EAGAIN is none
+ /// yet. Bytes come before a pending error, as Linux reads its queue first.
+ pub fn read(&mut self, id: u32, want: usize, peek: bool) -> Result, i64> {
+ let s = self.get_mut(id).ok_or(EBADF)?;
+ if !s.rx.is_empty() && want != 0 {
+ let n = want.min(s.rx.len());
+ return Ok(match peek {
+ true => s.rx.iter().take(n).copied().collect(),
+ false => s.rx.drain(..n).collect(),
+ });
+ }
+ if s.error != 0 {
+ return Err(mem::take(&mut s.error));
+ }
+ if s.listening || !s.connected {
+ return Err(ENOTCONN);
+ }
+ if want == 0 || s.eof || s.rd_shut {
+ return Ok(Vec::new());
+ }
+ Err(EAGAIN)
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/send.rs b/userland/capsule_linux/src/linux/net/sock/send.rs
new file mode 100644
index 000000000..cdda1d092
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/send.rs
@@ -0,0 +1,26 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Bytes into a family stream: to its peer's queue.
+
+use super::table::Socks;
+
+impl Socks {
+ /// As many of `bytes` as the peer has room for, EAGAIN for none.
+ pub fn write(&mut self, id: u32, bytes: &[u8]) -> Result {
+ self.put(id, bytes).map(|(n, _)| n)
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/syn.rs b/userland/capsule_linux/src/linux/net/sock/syn.rs
new file mode 100644
index 000000000..0a29cb8ce
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/syn.rs
@@ -0,0 +1,54 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Connects a full listener turned away. On Linux a non-blocking connect to
+//! a listener whose queue is full answers EINPROGRESS and completes when an
+//! accept makes room (its SYN is sent again); here it completes at that
+//! accept. Until then the socket is not writable, and a second connect is
+//! EALREADY.
+
+use super::table::Socks;
+
+impl Socks {
+ /// Queue uid=0(root) gid=0(root) groups=0(root)'s connect on listener `l`.
+ pub fn wait_room(&mut self, id: u32, l: u32) {
+ if let Some(c) = self.get_mut(id) {
+ c.connecting = true;
+ }
+ if let Some(ls) = self.get_mut(l) {
+ ls.syn.push_back(id);
+ }
+ }
+
+ /// Complete the connects waiting on `l` while its queue has room.
+ pub fn make_room(&mut self, l: u32) {
+ loop {
+ let Some(ls) = self.get_mut(l) else {
+ return;
+ };
+ if ls.pending.len() > ls.backlog {
+ return;
+ }
+ let Some(c) = ls.syn.pop_front() else {
+ return;
+ };
+ if let Some(cs) = self.get_mut(c) {
+ cs.connecting = false;
+ self.join(c, l);
+ }
+ }
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/table.rs b/userland/capsule_linux/src/linux/net/sock/table.rs
new file mode 100644
index 000000000..09287eebf
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/table.rs
@@ -0,0 +1,64 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! The table itself: entries by index, a freed index reused.
+
+use alloc::vec::Vec;
+
+use super::types::{Domain, Proto, Sock};
+
+pub struct Socks {
+ pub(super) list: Vec>,
+ /// Where the next ephemeral port search starts.
+ pub(super) next_port: u16,
+ /// Waiting calls partway through, by thread (`progress`).
+ pub(super) progress: Vec<(u32, usize)>,
+}
+
+impl Socks {
+ pub const fn new() -> Socks {
+ Socks { list: Vec::new(), next_port: 0, progress: Vec::new() }
+ }
+
+ /// A fresh socket held by `pid`, or by nobody yet when `pid` is None (the
+ /// server end of a connection, until accept hands it out).
+ pub fn open(&mut self, domain: Domain, proto: Proto, pid: Option) -> u32 {
+ let sock = Sock::new(domain, proto, pid);
+ match self.list.iter().position(Option::is_none) {
+ Some(i) => {
+ self.list[i] = Some(sock);
+ i as u32
+ }
+ None => {
+ self.list.push(Some(sock));
+ (self.list.len() - 1) as u32
+ }
+ }
+ }
+
+ pub fn get(&self, id: u32) -> Option<&Sock> {
+ self.list.get(id as usize).and_then(Option::as_ref)
+ }
+
+ pub fn get_mut(&mut self, id: u32) -> Option<&mut Sock> {
+ self.list.get_mut(id as usize).and_then(Option::as_mut)
+ }
+
+ /// Every live socket, with its index.
+ pub fn iter(&self) -> impl Iterator- {
+ self.list.iter().enumerate().filter_map(|(i, s)| s.as_ref().map(|s| (i as u32, s)))
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/take.rs b/userland/capsule_linux/src/linux/net/sock/take.rs
new file mode 100644
index 000000000..af6f3fbc9
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/take.rs
@@ -0,0 +1,51 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see
.
+
+//! Bytes out of one family socket: a stream's next bytes, or its next
+//! datagram with the name of whoever sent it.
+
+use alloc::vec::Vec;
+
+use crate::linux::abi::errno::EBADF;
+
+use super::name::Peer;
+use super::table::Socks;
+use super::types::Proto;
+
+/// The bytes, a datagram's length before it was cut, and its sender.
+pub type Taken = (Vec, usize, Option);
+
+impl Socks {
+ pub fn take(&mut self, id: u32, want: usize, peek: bool) -> Result {
+ match self.get(id).map(|s| s.proto) {
+ None => Err(EBADF),
+ Some(Proto::Stream) => {
+ let bytes = self.read(id, want, peek)?;
+ let n = bytes.len();
+ Ok((bytes, n, None))
+ }
+ Some(Proto::Dgram) => {
+ let got = self.take_gram(id, want, peek)?;
+ /* A socketpair's peer has no name, and Linux gives none. */
+ let from = match got.from {
+ Peer::Unix(None) => None,
+ named => Some(named),
+ };
+ Ok((got.bytes, got.whole, from))
+ }
+ }
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/types.rs b/userland/capsule_linux/src/linux/net/sock/types.rs
new file mode 100644
index 000000000..513b60d11
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/types.rs
@@ -0,0 +1,63 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! One socket, as the family sees it.
+
+use alloc::collections::VecDeque;
+use alloc::vec::Vec;
+
+use super::name::{Gram, UName};
+use super::opts::Opts;
+
+pub use super::kinds::{Addr, Domain, Proto};
+
+pub struct Sock {
+ pub domain: Domain,
+ pub proto: Proto,
+ pub local: Option,
+ pub remote: Option,
+ /// Set by listen, with the queue of connections accept has not taken.
+ pub listening: bool,
+ pub backlog: usize,
+ pub pending: VecDeque,
+ /// Connects the full queue turned away, oldest first (`syn`).
+ pub syn: VecDeque,
+ /// This end's connect waits in a listener's `syn` queue.
+ pub connecting: bool,
+ /// The other end of a stream, until it is let go.
+ pub peer: Option,
+ pub connected: bool,
+ /// Bytes the peer wrote that this end has not read.
+ pub rx: VecDeque,
+ /// Datagrams waiting to be read.
+ pub grams: VecDeque,
+ /// The peer will write nothing more: it shut its side or it is gone.
+ pub eof: bool,
+ pub wr_shut: bool,
+ pub rd_shut: bool,
+ /// Written to after the peer was gone: every later write is EPIPE.
+ pub broken: bool,
+ /// SO_ERROR: reported once, by the next call that looks.
+ pub error: i64,
+ pub opts: Opts,
+ /// The net.sockets handle, for a stream that reaches outside the family.
+ pub svc: Option,
+ /// The processes that hold a descriptor naming this socket.
+ pub holders: Vec,
+ /// A Unix socket's own name, and its connected peer's.
+ pub uname: Option,
+ pub upeer: Option,
+}
diff --git a/userland/capsule_linux/src/linux/net/sock/unlisten.rs b/userland/capsule_linux/src/linux/net/sock/unlisten.rs
new file mode 100644
index 000000000..4a57a1a3b
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sock/unlisten.rs
@@ -0,0 +1,48 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A listener that stops: what it queued is reset, and the connects that
+//! wait for its room are refused.
+
+use crate::linux::abi::errno::ECONNREFUSED;
+
+use super::table::Socks;
+
+impl Socks {
+ /// Listener `l` is gone: the connects waiting on it are refused.
+ pub fn refuse_waiting(&mut self, syn: impl Iterator- ) {
+ for c in syn {
+ if let Some(cs) = self.get_mut(c) {
+ cs.connecting = false;
+ cs.error = ECONNREFUSED;
+ }
+ }
+ }
+
+ /// Listener `l` stops listening: what it queued is reset, and the
+ /// connects waiting for room are refused.
+ pub fn unlisten(&mut self, l: u32) {
+ let Some(s) = self.get_mut(l) else {
+ return;
+ };
+ s.listening = false;
+ let (queued, waiting) = (core::mem::take(&mut s.pending), core::mem::take(&mut s.syn));
+ for q in queued {
+ self.free(q, true);
+ }
+ self.refuse_waiting(waiting.into_iter());
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/sockaddr.rs b/userland/capsule_linux/src/linux/net/sockaddr.rs
new file mode 100644
index 000000000..6ca4fc158
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sockaddr.rs
@@ -0,0 +1,51 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see
.
+
+//! `sockaddr_in` and `sockaddr_un` between guest memory and `Addr`.
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::sock::Addr;
+
+pub const AF_UNSPEC: u16 = 0;
+pub const AF_UNIX: u16 = 1;
+pub const AF_INET: u16 = 2;
+pub const SOCKADDR_IN: usize = 16;
+
+/// The family and address a guest named; EINVAL for one too short to hold
+/// its family, EFAULT for one it cannot read.
+pub fn read(guest: &Guest, at: u64, len: u64) -> Result<(u16, Addr), u64> {
+ if len < 2 {
+ return Err(errno::fail(errno::EINVAL));
+ }
+ let head = guest.read(at, 2).ok_or(errno::fail(errno::EFAULT))?;
+ let family = u16::from_le_bytes([head[0], head[1]]);
+ if family != AF_INET {
+ return Ok((family, Addr::default()));
+ }
+ if len < SOCKADDR_IN as u64 {
+ return Err(errno::fail(errno::EINVAL));
+ }
+ let raw = guest.read(at, SOCKADDR_IN).ok_or(errno::fail(errno::EFAULT))?;
+ let port = u16::from_be_bytes([raw[2], raw[3]]);
+ Ok((family, Addr { ip: [raw[4], raw[5], raw[6], raw[7]], port }))
+}
+
+/// True for 127.0.0.0/8, the only addresses the family keeps to itself.
+pub fn is_loopback(ip: [u8; 4]) -> bool {
+ ip[0] == 127
+}
diff --git a/userland/capsule_linux/src/linux/net/sockaddr_out.rs b/userland/capsule_linux/src/linux/net/sockaddr_out.rs
new file mode 100644
index 000000000..5a20bd547
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/sockaddr_out.rs
@@ -0,0 +1,71 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! A socket's address written into guest memory.
+
+use alloc::vec::Vec;
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::sock::Peer;
+use super::sockaddr::{AF_INET, AF_UNIX};
+
+/// Write `peer` at `at`, cut to the length the guest offered at `lenp`, and
+/// the whole length back at `lenp`, as Linux's move_addr_to_user does. An
+/// unnamed Unix socket is the family alone; a path carries its NUL.
+pub fn write(guest: &mut Guest, at: u64, lenp: u64, peer: &Peer) -> u64 {
+ if at == 0 || lenp == 0 {
+ return errno::ok(0);
+ }
+ let Some(raw) = guest.read(lenp, 4) else {
+ return errno::fail(errno::EFAULT);
+ };
+ let room = u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]) as i32;
+ if room < 0 {
+ return errno::fail(errno::EINVAL);
+ }
+ let sa = encode(peer);
+ let n = sa.len().min(room as usize);
+ if guest.write(at, &sa[..n]) < n as i64
+ || guest.write(lenp, &(sa.len() as u32).to_le_bytes()) < 4
+ {
+ return errno::fail(errno::EFAULT);
+ }
+ errno::ok(0)
+}
+
+fn encode(peer: &Peer) -> Vec {
+ let mut sa = Vec::with_capacity(16);
+ match peer {
+ Peer::Inet(addr) => {
+ sa.extend_from_slice(&AF_INET.to_le_bytes());
+ sa.extend_from_slice(&addr.port.to_be_bytes());
+ sa.extend_from_slice(&addr.ip);
+ sa.resize(16, 0);
+ }
+ Peer::Unix(name) => {
+ sa.extend_from_slice(&AF_UNIX.to_le_bytes());
+ if let Some(n) = name {
+ sa.extend_from_slice(&n.shown);
+ if !n.is_abstract() {
+ sa.push(0);
+ }
+ }
+ }
+ }
+ sa
+}
diff --git a/userland/capsule_linux/src/linux/net/socket.rs b/userland/capsule_linux/src/linux/net/socket.rs
index cdcbd5cbe..85d09936f 100644
--- a/userland/capsule_linux/src/linux/net/socket.rs
+++ b/userland/capsule_linux/src/linux/net/socket.rs
@@ -14,49 +14,62 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-//! `socket` and `connect`, over net.sockets.
-
-use alloc::vec::Vec;
-
-use super::ops::{DOMAIN, KIND_MIXNET, OP_SOCKET};
+//! `socket` for AF_INET and AF_UNIX. The socket is the family's until it
+//! connects outside 127.0.0.0/8 (then net.sockets holds the stream) or to
+//! the display's path (then it is the display connection).
use crate::linux::abi::errno;
-use crate::linux::guest::{Fd, Guest};
+use crate::linux::guest::Guest;
-use super::call::call;
+use super::fd::{install, SOCK_CLOEXEC, SOCK_NONBLOCK};
+use super::policy::refuse;
+use super::sock::{self, Domain, Proto};
+use super::sockaddr::{AF_INET, AF_UNIX};
-const AF_INET: u64 = 2;
const SOCK_STREAM: u64 = 1;
const SOCK_DGRAM: u64 = 2;
-/// Linux ORs these into the type; neither changes what is opened here.
-const TYPE_MASK: u64 = 0xFF;
+const SOCK_RAW: u64 = 3;
+const SOCK_SEQPACKET: u64 = 5;
+const TYPE_MASK: u64 = 0xF;
+const IPPROTO_TCP: u64 = 6;
+const IPPROTO_UDP: u64 = 17;
+/// The one protocol a Unix socket takes besides 0.
+const PF_UNIX: u64 = 1;
-pub fn socket(guest: &mut Guest, family: u64, kind: u64) -> u64 {
- if family != AF_INET {
- return errno::fail(errno::EAFNOSUPPORT);
+pub fn socket(guest: &mut Guest, family: u64, kind: u64, protocol: u64) -> u64 {
+ let flags = kind & (SOCK_NONBLOCK | SOCK_CLOEXEC);
+ if kind & !(TYPE_MASK | flags) != 0 {
+ return errno::fail(errno::EINVAL);
}
- /*
- * A guest's stream goes over the mixnet, never the open network, and it
- * holds no capability that could name a socket: there is no second route
- * to disable and no firewall rule to remove.
- */
- let want = match kind & TYPE_MASK {
- SOCK_STREAM => KIND_MIXNET,
- SOCK_DGRAM => return super::dns::open(guest),
- _ => return errno::fail(errno::ENOSYS),
+ let domain = match family {
+ f if f == u64::from(AF_INET) => Domain::Inet,
+ f if f == u64::from(AF_UNIX) => Domain::Unix,
+ _ => return errno::fail(errno::EAFNOSUPPORT),
};
- let mut body = Vec::with_capacity(4);
- body.extend_from_slice(&DOMAIN.to_le_bytes());
- body.extend_from_slice(&want.to_le_bytes());
- let Some((status, out)) = call(OP_SOCKET, &body, 8) else {
- return errno::fail(errno::EIO);
+ let (proto, own) = match (kind & TYPE_MASK, domain) {
+ (SOCK_STREAM, Domain::Inet) => (Proto::Stream, IPPROTO_TCP),
+ (SOCK_DGRAM, Domain::Inet) => (Proto::Dgram, IPPROTO_UDP),
+ (SOCK_RAW, Domain::Inet) => {
+ return refuse("SOCK_RAW: raw sockets reach below any confinement", errno::EPERM)
+ }
+ (SOCK_STREAM, Domain::Unix) => (Proto::Stream, PF_UNIX),
+ /* Linux gives a raw Unix socket datagram semantics. */
+ (SOCK_DGRAM | SOCK_RAW, Domain::Unix) => (Proto::Dgram, PF_UNIX),
+ (SOCK_SEQPACKET, Domain::Unix) => {
+ return refuse(
+ "SOCK_SEQPACKET: a Unix stream here does not keep message boundaries",
+ errno::ESOCKTNOSUPPORT,
+ )
+ }
+ _ => return errno::fail(errno::ESOCKTNOSUPPORT),
};
- if status != 0 || out.len() < 4 {
- return errno::fail(errno::ENOMEM);
- }
- let handle = u32::from_le_bytes([out[0], out[1], out[2], out[3]]);
- match crate::linux::file::install(guest, Fd::socket(handle)) {
- Some(n) => errno::ok(n),
- None => errno::fail(errno::EMFILE),
+ /*
+ * Anything but the type's own protocol, MPTCP included, is one this
+ * stack does not have; Go falls back to TCP on this answer.
+ */
+ if protocol != 0 && protocol != own {
+ return errno::fail(errno::EPROTONOSUPPORT);
}
+ let id = sock::with(|t| t.open(domain, proto, Some(guest.pid)));
+ install(guest, id, flags)
}
diff --git a/userland/capsule_linux/src/linux/net/stream.rs b/userland/capsule_linux/src/linux/net/stream.rs
index 94c6effd5..0123a5c07 100644
--- a/userland/capsule_linux/src/linux/net/stream.rs
+++ b/userland/capsule_linux/src/linux/net/stream.rs
@@ -14,51 +14,41 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-
-//! Bytes on and off a connected socket.
+//! Bytes on a stream that reaches outside the family, through net.sockets.
use alloc::vec::Vec;
use crate::linux::abi::errno;
-use crate::linux::guest::Guest;
use super::call::call;
use super::ops::{OP_CLOSE, OP_RECV, OP_SEND};
-/// One transfer. The server's reply buffer is the ceiling, not this.
-const MAX_IO: u64 = 32 << 10;
+/// The most net.sockets carries in one call.
+pub const MAX_IO: usize = 32 << 10;
-pub fn send(guest: &Guest, handle: u32, buf: u64, len: u64) -> u64 {
- let take = len.min(MAX_IO);
- let Some(bytes) = guest.read(buf, take as usize) else {
- return errno::fail(errno::EFAULT);
- };
+pub fn send_bytes(handle: u32, bytes: &[u8]) -> u64 {
let mut body = Vec::with_capacity(4 + bytes.len());
body.extend_from_slice(&handle.to_le_bytes());
- body.extend_from_slice(&bytes);
+ body.extend_from_slice(bytes);
match call(OP_SEND, &body, 0) {
- Some((0, _)) => errno::ok(take),
+ Some((0, _)) => errno::ok(bytes.len() as u64),
Some(_) => errno::fail(errno::EPIPE),
None => errno::fail(errno::EIO),
}
}
-pub fn recv(guest: &Guest, handle: u32, buf: u64, len: u64) -> u64 {
- let want = len.min(MAX_IO) as usize;
- let Some((status, bytes)) = call(OP_RECV, &handle.to_le_bytes(), want) else {
- return errno::fail(errno::EIO);
+/// Up to `want` bytes. net.sockets answers the same for a quiet stream, a
+/// closed one and a reset one, so any refusal reads as ECONNRESET.
+pub fn recv_bytes(handle: u32, want: usize) -> Result, u64> {
+ let want = want.min(MAX_IO);
+ let Some((status, mut bytes)) = call(OP_RECV, &handle.to_le_bytes(), want) else {
+ return Err(errno::fail(errno::EIO));
};
if status != 0 {
- return errno::fail(errno::ECONNRESET);
- }
- if bytes.is_empty() {
- return errno::ok(0);
- }
- let n = bytes.len().min(want);
- if guest.write(buf, &bytes[..n]) < n as i64 {
- return errno::fail(errno::EFAULT);
+ return Err(errno::fail(errno::ECONNRESET));
}
- errno::ok(n as u64)
+ bytes.truncate(want);
+ Ok(bytes)
}
pub fn close(handle: u32) {
diff --git a/userland/capsule_linux/src/linux/net/try_call.rs b/userland/capsule_linux/src/linux/net/try_call.rs
new file mode 100644
index 000000000..90203c3e4
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/try_call.rs
@@ -0,0 +1,70 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! One try at a socket call that may wait. The serve loop's `waits_sock`
+//! parks a blocking one that cannot finish and tries it again; `done` is
+//! how much of it an earlier try moved (bytes, or messages for the mmsg
+//! calls), and a try goes on from there.
+
+use alloc::vec;
+
+use crate::linux::abi::{errno, nr, nr_path as np};
+use crate::linux::guest::Guest;
+
+use super::call_kind::{bytes_in, msg_len};
+use super::{iov, mmsg};
+
+/// The answer, and the whole amount the call asks to move.
+pub fn try_call(guest: &mut Guest, n: u64, a: [u64; 6], done: usize) -> (u64, usize) {
+ let fd = a[0];
+ let Ok(id) = super::fd::sock_of(guest, fd) else {
+ return (errno::fail(errno::EBADF), 0);
+ };
+ let one = |buf: u64, len: u64| vec![(buf, len)];
+ match n {
+ nr::READ => (bytes_in(guest, id, &one(a[1], a[2]), done, 0), a[2] as usize),
+ nr::WRITE => {
+ (super::xfer_out::send(guest, id, &one(a[1], a[2]), done, 0, None), a[2] as usize)
+ }
+ np::READV | nr::WRITEV => match iov::read(guest, a[1], a[2]) {
+ Ok(v) if n == nr::WRITEV => {
+ (super::xfer_out::send(guest, id, &v, done, 0, None), iov::total(&v))
+ }
+ Ok(v) => (bytes_in(guest, id, &v, done, 0), iov::total(&v)),
+ Err(e) => (e, 0),
+ },
+ nr::RECVFROM if done == 0 => {
+ (super::recvfrom(guest, fd, a[1], a[2], a[3], a[4], a[5]), a[2] as usize)
+ }
+ nr::RECVFROM => (bytes_in(guest, id, &one(a[1], a[2]), done, a[3]), a[2] as usize),
+ nr::SENDTO if done == 0 => {
+ (super::sendto(guest, fd, a[1], a[2], a[3], a[4], a[5]), a[2] as usize)
+ }
+ nr::SENDTO => {
+ (super::xfer_out::send(guest, id, &one(a[1], a[2]), done, a[3], None), a[2] as usize)
+ }
+ nr::SENDMSG => (super::msg::sendmsg(guest, fd, a[1], a[2], done), msg_len(guest, a[1])),
+ nr::RECVMSG => {
+ (super::msg_recv::recvmsg(guest, fd, a[1], a[2], done), msg_len(guest, a[1]))
+ }
+ nr::SENDMMSG => (mmsg::sendmmsg(guest, a, done), a[2].min(mmsg::MOST) as usize),
+ nr::RECVMMSG => (mmsg::recvmmsg(guest, a, done), a[2].min(mmsg::MOST) as usize),
+ nr::ACCEPT => (super::accept::accept4(guest, fd, a[1], a[2], 0), 0),
+ nr::ACCEPT4 => (super::accept::accept4(guest, fd, a[1], a[2], a[3]), 0),
+ nr::CONNECT => (super::connect(guest, fd, a[1], a[2]), 0),
+ _ => (errno::fail(errno::ENOSYS), 0),
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/xfer_in.rs b/userland/capsule_linux/src/linux/net/xfer_in.rs
new file mode 100644
index 000000000..ddfbdb409
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/xfer_in.rs
@@ -0,0 +1,67 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Bytes into a guest from a family socket or a stream net.sockets holds.
+
+use alloc::vec;
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::flags::{MSG_OOB, MSG_PEEK, MSG_TRUNC};
+use super::iov::{self, Iov};
+use super::sock::{self, Peer, Proto};
+
+pub struct In {
+ /// Bytes put in the guest's buffers.
+ pub n: usize,
+ /// A datagram's length before it was cut to fit.
+ pub whole: usize,
+ /// Where a datagram came from; a stream, and an unnamed sender, say
+ /// nothing.
+ pub from: Option,
+}
+
+/// Receive into `iov` from socket `id`, `skip` bytes of it filled already.
+pub fn recv(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64) -> Result {
+ if flags & MSG_OOB != 0 {
+ return Err(errno::fail(errno::EINVAL));
+ }
+ let Some((proto, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.svc))) else {
+ return Err(errno::fail(errno::EBADF));
+ };
+ let want = iov::total(iov).saturating_sub(skip);
+ let peek = flags & MSG_PEEK != 0;
+ if let Some(h) = svc {
+ let bytes = super::stream::recv_bytes(h, want)?;
+ iov::scatter(guest, iov, skip, &bytes)?;
+ return Ok(In { n: bytes.len(), whole: bytes.len(), from: None });
+ }
+ let (bytes, whole, from) = sock::with(|t| t.take(id, want, peek)).map_err(errno::fail)?;
+ /* A stream's MSG_TRUNC discards what it would have read. */
+ if proto != Proto::Stream || flags & MSG_TRUNC == 0 {
+ iov::scatter(guest, iov, skip, &bytes)?;
+ }
+ Ok(In { n: bytes.len(), whole, from })
+}
+
+/// `read` on a socket.
+pub fn read(guest: &Guest, id: u32, buf: u64, len: u64) -> u64 {
+ match recv(guest, id, &vec![(buf, len)], 0, 0) {
+ Ok(got) => errno::ok(got.n as u64),
+ Err(e) => e,
+ }
+}
diff --git a/userland/capsule_linux/src/linux/net/xfer_out.rs b/userland/capsule_linux/src/linux/net/xfer_out.rs
new file mode 100644
index 000000000..d2ab7fb63
--- /dev/null
+++ b/userland/capsule_linux/src/linux/net/xfer_out.rs
@@ -0,0 +1,62 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Bytes out of a socket: to a stream's peer, a datagram to the port or the
+//! Unix name it is sent to, or a stream outside the family through
+//! net.sockets.
+
+use alloc::vec;
+
+use crate::linux::abi::errno;
+use crate::linux::guest::Guest;
+
+use super::flags::MSG_OOB;
+use super::iov::{self, Iov};
+use super::peer_addr::To;
+use super::sock::{self, Proto};
+
+/// Send the message in `iov` from socket `id`, `skip` bytes of it having
+/// gone already: the count sent now, or an errno.
+pub fn send(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64, to: Option) -> u64 {
+ if flags & MSG_OOB != 0 {
+ return errno::fail(errno::EOPNOTSUPP);
+ }
+ let Some((proto, domain, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.domain, s.svc)))
+ else {
+ return errno::fail(errno::EBADF);
+ };
+ let cap = super::cap::cap(svc.is_some(), proto == Proto::Stream);
+ let bytes = match iov::gather(guest, iov, skip, cap) {
+ Ok(b) => b,
+ Err(e) => return e,
+ };
+ if let Some(h) = svc {
+ return super::stream::send_bytes(h, &bytes);
+ }
+ let dest = match super::dest::dest(guest, proto, domain, to) {
+ Ok(d) => d,
+ Err(e) => return e,
+ };
+ match sock::with(|t| t.deliver(id, dest, &bytes)) {
+ Ok(n) => errno::ok(n as u64),
+ Err(e) => errno::fail(e),
+ }
+}
+
+/// `write` on a socket: a send with no flags and no address.
+pub fn write(guest: &Guest, id: u32, buf: u64, len: u64) -> u64 {
+ send(guest, id, &vec![(buf, len)], 0, 0, None)
+}
diff --git a/userland/capsule_linux/src/linux/serve/dispatch.rs b/userland/capsule_linux/src/linux/serve/dispatch.rs
index e292a5f27..4523e707f 100644
--- a/userland/capsule_linux/src/linux/serve/dispatch.rs
+++ b/userland/capsule_linux/src/linux/serve/dispatch.rs
@@ -21,6 +21,7 @@ use nonos_libc::ForeignFrame;
use super::answer::Answer;
use super::table::plain;
+use super::waits_sock;
use crate::linux::abi::{nr, nr_path as np};
use crate::linux::call::{clone, exit_thread, futex};
use crate::linux::guest::Guest;
@@ -58,6 +59,7 @@ fn route(guest: &mut Guest, frame: &ForeignFrame) -> Answer {
np::CLOCK_NANOSLEEP => {
crate::linux::call::clock_nanosleep(guest, frame.pid, a[0], a[1], a[2])
}
+ n if waits_sock::takes(guest, n, a[0]) => waits_sock::io(guest, frame.pid, n, a),
nr::READ | nr::WRITE if super::waits::may_wait(guest, frame.nr, a[0]) => {
super::waits::io(guest, frame.pid, frame.nr, a)
}
diff --git a/userland/capsule_linux/src/linux/serve/family_waits.rs b/userland/capsule_linux/src/linux/serve/family_waits.rs
index 735f19a44..6f50f98fb 100644
--- a/userland/capsule_linux/src/linux/serve/family_waits.rs
+++ b/userland/capsule_linux/src/linux/serve/family_waits.rs
@@ -25,10 +25,11 @@ use super::waits::{attempt, expire};
use super::waits_fds::watched;
use crate::linux::call::now_ms;
use crate::linux::guest::Kind;
+use crate::linux::net::outside;
const CLOCK_MONOTONIC: u64 = 1;
-/// How often a wait on a socket is looked at again: its readiness changes
-/// with no call for the family to answer. A timer is looked at when it fires.
+/// How often a wait on a stream net.sockets holds is looked at again; a family
+/// socket changes only in an answer. A timer is looked at when it fires.
const TICK_MS: u64 = 10;
impl Family {
@@ -73,9 +74,10 @@ impl Family {
if let Some(d) = wait.deadline {
keep(d.saturating_sub(now));
}
+ super::waits_sock::ticks(g, wait).then(|| keep(TICK_MS));
for fd in watched(g, wait) {
match g.fds.get(fd as usize) {
- Some(f) if f.kind == Kind::Socket => keep(TICK_MS),
+ Some(f) if f.kind == Kind::Socket && outside(f.handle) => keep(TICK_MS),
Some(f) if f.kind == Kind::Timer => {
// One that has already fired was seen by the last look.
let due = self.timers.get(f.handle as usize).map_or(0, |t| t.due);
diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs
index 700f5b2dd..12527be3b 100644
--- a/userland/capsule_linux/src/linux/serve/mod.rs
+++ b/userland/capsule_linux/src/linux/serve/mod.rs
@@ -28,8 +28,8 @@ mod family_waits;
mod loop_impl;
mod pid_map;
mod pid_ns;
-mod refused;
mod pid_out;
+mod refused;
mod table;
mod table_file;
mod table_link;
@@ -40,6 +40,8 @@ mod tally;
mod unserved;
mod waits;
mod waits_fds;
+mod waits_sock;
+mod waits_sock_kind;
mod waits_time;
pub use answer::Answer;
diff --git a/userland/capsule_linux/src/linux/serve/table_net.rs b/userland/capsule_linux/src/linux/serve/table_net.rs
index cf5902775..119cd7a7f 100644
--- a/userland/capsule_linux/src/linux/serve/table_net.rs
+++ b/userland/capsule_linux/src/linux/serve/table_net.rs
@@ -17,23 +17,37 @@
//! Calls that name a socket.
use crate::linux::abi::nr;
-use crate::linux::call;
use crate::linux::guest::Guest;
use crate::linux::net;
use crate::linux::unix::{self, is_unix};
+/// Every socket call. The ones that can wait reach here only when they
+/// cannot: a socket's own calls are routed to `waits_sock` first, and the
+/// rest are a resolver's, a display socket's, or not a socket at all.
pub fn net_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option {
let _ = tid;
Some(match nr {
- nr::SOCKET if a[0] == 1 => unix::socket(guest, a[1]),
- nr::SOCKET => net::socket(guest, a[0], a[1]),
+ nr::SOCKET => net::socket(guest, a[0], a[1], a[2]),
+ nr::SOCKETPAIR => net::socketpair(guest, a[0], a[1], a[2], a[3]),
nr::CONNECT if is_unix(guest, a[0]) => unix::connect(guest, a[0], a[1], a[2]),
nr::CONNECT => net::connect(guest, a[0], a[1], a[2]),
- nr::SENDTO => net::sendto(guest, a[0], a[1], a[2], a[4], a[5]),
- nr::RECVFROM => net::recvfrom(guest, a[0], a[1], a[2], a[4], a[5]),
- nr::SHUTDOWN => call::close(guest, a[0]),
- nr::SENDMSG => unix::sendmsg(guest, a[0], a[1]),
- nr::RECVMSG => unix::recvmsg(guest, a[0], a[1]),
+ nr::BIND => net::bind(guest, a[0], a[1], a[2]),
+ nr::LISTEN => net::listen(guest, a[0], a[1]),
+ nr::ACCEPT => net::accept4(guest, a[0], a[1], a[2], 0),
+ nr::ACCEPT4 => net::accept4(guest, a[0], a[1], a[2], a[3]),
+ nr::GETSOCKNAME => net::getsockname(guest, a[0], a[1], a[2]),
+ nr::GETPEERNAME => net::getpeername(guest, a[0], a[1], a[2]),
+ nr::SETSOCKOPT => net::setsockopt(guest, a[0], a[1], a[2], a[3], a[4]),
+ nr::GETSOCKOPT => net::getsockopt(guest, a[0], a[1], a[2], a[3], a[4]),
+ nr::SENDTO => net::sendto(guest, a[0], a[1], a[2], a[3], a[4], a[5]),
+ nr::RECVFROM => net::recvfrom(guest, a[0], a[1], a[2], a[3], a[4], a[5]),
+ nr::SHUTDOWN => net::shutdown(guest, a[0], a[1]),
+ nr::SENDMSG if is_unix(guest, a[0]) => unix::sendmsg(guest, a[0], a[1]),
+ nr::RECVMSG if is_unix(guest, a[0]) => unix::recvmsg(guest, a[0], a[1]),
+ nr::SENDMSG => net::sendmsg(guest, a[0], a[1], a[2], 0),
+ nr::RECVMSG => net::recvmsg(guest, a[0], a[1], a[2], 0),
+ nr::SENDMMSG => net::sendmmsg(guest, a, 0),
+ nr::RECVMMSG => net::recvmmsg(guest, a, 0),
_ => return None,
})
}
diff --git a/userland/capsule_linux/src/linux/serve/waits.rs b/userland/capsule_linux/src/linux/serve/waits.rs
index 6de3c6d7a..0361528ce 100644
--- a/userland/capsule_linux/src/linux/serve/waits.rs
+++ b/userland/capsule_linux/src/linux/serve/waits.rs
@@ -79,6 +79,7 @@ pub fn attempt(guest: &mut Guest, wait: &Blocked) -> Option {
let a = wait.args;
let again = errno::fail(errno::EAGAIN);
match wait.nr {
+ n if super::waits_sock::takes(guest, n, a[0]) => super::waits_sock::attempt(guest, wait),
nr::READ => Some(call::read(guest, a[0], a[1], a[2])).filter(|&v| v != again),
nr::WRITE => Some(call::write(guest, a[0], a[1], a[2])).filter(|&v| v != again),
nr::POLL | np::PPOLL => Some(net::poll(guest, a[0], a[1])).filter(|&v| v != 0),
diff --git a/userland/capsule_linux/src/linux/serve/waits_sock.rs b/userland/capsule_linux/src/linux/serve/waits_sock.rs
new file mode 100644
index 000000000..9a7fede40
--- /dev/null
+++ b/userland/capsule_linux/src/linux/serve/waits_sock.rs
@@ -0,0 +1,72 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Socket calls that wait. Each is tried at once; a blocking one that cannot
+//! finish is parked and tried again after every answer (`family_waits`).
+//! SO_RCVTIMEO and SO_SNDTIMEO bound the wait, which then answers EAGAIN, or
+//! the count moved so far, as Linux does.
+
+use crate::linux::abi::errno;
+use crate::linux::call::now_ms;
+use crate::linux::guest::{Blocked, Guest};
+use crate::linux::net::{self, sock};
+
+use super::answer::Answer;
+use super::waits_sock_kind::deadline;
+pub use super::waits_sock_kind::{takes, ticks};
+
+const CLOCK_MONOTONIC: u64 = 1;
+const MSG_DONTWAIT: u64 = 0x40;
+
+pub fn io(guest: &mut Guest, tid: u32, n: u64, a: [u64; 6]) -> Answer {
+ let wait = Blocked { tid, nr: n, args: a, deadline: deadline(guest, n, a[0]) };
+ match attempt(guest, &wait) {
+ Some(v) => Answer::value(v),
+ None => {
+ guest.blocked.push(wait);
+ Answer::Park
+ }
+ }
+}
+
+/// The call's answer if it can give one now, None to go on waiting.
+pub fn attempt(guest: &mut Guest, wait: &Blocked) -> Option {
+ let (n, a, tid) = (wait.nr, wait.args, wait.tid);
+ let done = sock::progress(tid);
+ let (value, whole) = net::try_call(guest, n, a, done);
+ let flags = net::call_flags(n, a);
+ let blocking =
+ flags & MSG_DONTWAIT == 0 && !guest.fds.get(a[0] as usize).is_some_and(|f| f.nonblock);
+ let late = wait.deadline.is_some_and(|d| now_ms(CLOCK_MONOTONIC).is_some_and(|now| d <= now));
+ let answer = match errno::slot(value) {
+ Some(moved) => {
+ let total = done + moved;
+ let stream = net::sock_id(guest, a[0]).is_some_and(net::is_stream);
+ if blocking && moved != 0 && total < whole && !late && net::wants_all(stream, n, flags)
+ {
+ sock::set_progress(tid, total);
+ return None;
+ }
+ total as u64
+ }
+ None if value == errno::fail(errno::EAGAIN) && blocking && !late => return None,
+ /* What moved before an error or the time limit is what Linux answers. */
+ None if done != 0 => done as u64,
+ None => value,
+ };
+ sock::set_progress(tid, 0);
+ Some(answer)
+}
diff --git a/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs b/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs
new file mode 100644
index 000000000..31f2bf7d0
--- /dev/null
+++ b/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs
@@ -0,0 +1,62 @@
+// NONOS Operating System
+// Copyright (C) 2026 NONOS Contributors
+//
+// This program is free software: you can redistribute it and/or modify
+// it under the terms of the GNU Affero General Public License as published by
+// the Free Software Foundation, either version 3 of the License, or
+// (at your option) any later version.
+//
+// This program is distributed in the hope that it will be useful,
+// but WITHOUT ANY WARRANTY; without even the implied warranty of
+// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+// GNU Affero General Public License for more details.
+//
+// You should have received a copy of the GNU Affero General Public License
+// along with this program. If not, see .
+
+//! Which calls `waits_sock` takes, which of its waits need a tick, and
+//! when a wait's time runs out.
+
+use crate::linux::abi::{nr, nr_path as np};
+use crate::linux::call::now_ms;
+use crate::linux::guest::{Blocked, Guest, Kind};
+use crate::linux::net;
+
+const CLOCK_MONOTONIC: u64 = 1;
+
+/// True for a call on a socket descriptor that may have to wait.
+pub fn takes(guest: &Guest, n: u64, fd: u64) -> bool {
+ matches!(
+ n,
+ nr::READ
+ | nr::WRITE
+ | np::READV
+ | nr::WRITEV
+ | nr::ACCEPT
+ | nr::ACCEPT4
+ | nr::CONNECT
+ | nr::SENDTO
+ | nr::RECVFROM
+ | nr::SENDMSG
+ | nr::RECVMSG
+ | nr::SENDMMSG
+ | nr::RECVMMSG
+ ) && guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Socket)
+}
+
+/// True when a parked call waits on a stream net.sockets holds, which
+/// nothing but a look tells the family has changed.
+pub fn ticks(guest: &Guest, wait: &Blocked) -> bool {
+ takes(guest, wait.nr, wait.args[0])
+ && net::sock_id(guest, wait.args[0]).is_some_and(net::outside)
+}
+
+/// When a call's SO_RCVTIMEO or SO_SNDTIMEO runs out, if it has one.
+pub fn deadline(guest: &Guest, n: u64, fd: u64) -> Option {
+ let reads = !matches!(
+ n,
+ nr::WRITE | nr::WRITEV | nr::SENDTO | nr::SENDMSG | nr::SENDMMSG | nr::CONNECT
+ );
+ let limit = net::sock_id(guest, fd).and_then(|id| net::limit_ms(id, reads))?;
+ Some(now_ms(CLOCK_MONOTONIC).unwrap_or(0).saturating_add(limit))
+}
diff --git a/userland/capsule_linux/src/linux/unix/mod.rs b/userland/capsule_linux/src/linux/unix/mod.rs
index b03ee7949..b84a27e89 100644
--- a/userland/capsule_linux/src/linux/unix/mod.rs
+++ b/userland/capsule_linux/src/linux/unix/mod.rs
@@ -14,8 +14,8 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-
-//! Unix domain sockets, both ends inside this capsule.
+//! The display connection: a Unix socket whose far end is this capsule.
+//! Every other Unix socket is the family's own (`net::sock`).
mod conn;
mod give;
@@ -29,8 +29,9 @@ mod sock;
mod sock_io;
pub use conn::Conn;
-pub use sock::is_unix;
+pub use path::is_display;
pub use recvmsg::recvmsg;
pub use sendmsg::sendmsg;
-pub use sock::{connect, socket};
+pub use sock::connect;
+pub use sock::is_unix;
pub use sock_io::{recv, send};
diff --git a/userland/capsule_linux/src/linux/unix/sock.rs b/userland/capsule_linux/src/linux/unix/sock.rs
index 2e290cd5e..d6d16a1fb 100644
--- a/userland/capsule_linux/src/linux/unix/sock.rs
+++ b/userland/capsule_linux/src/linux/unix/sock.rs
@@ -14,35 +14,21 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-
-//! The four calls a client makes on a display socket.
+//! Connecting to the display, and telling a display socket apart.
use crate::linux::abi::errno;
-use crate::linux::guest::{Fd, Guest, Kind};
+use crate::linux::guest::{Guest, Kind};
use super::path::{is_display, sun_path};
-const SOCK_STREAM: u64 = 1;
-const TYPE_MASK: u64 = 0xFF;
-
-pub fn socket(guest: &mut Guest, kind: u64) -> u64 {
- if kind & TYPE_MASK != SOCK_STREAM {
- return errno::fail(errno::ENOSYS);
- }
- match crate::linux::file::install(guest, Fd::unix()) {
- Some(n) => errno::ok(n),
- None => errno::fail(errno::EMFILE),
- }
-}
-
pub fn connect(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 {
let Some(path) = sun_path(guest, at, len) else {
return errno::fail(errno::EINVAL);
};
if !is_display(&path) {
/*
- * Nothing else listens in here, and a client that reaches a socket
- * which silently accepts would block forever on a reply.
+ * Only the display is served here; every other name is a family
+ * socket's (net::unix_calls), which never reaches this call.
*/
return errno::fail(errno::ECONNREFUSED);
}
diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk
index c42c8e873..900254bac 100644
--- a/userland/linux_guests/Guests.mk
+++ b/userland/linux_guests/Guests.mk
@@ -34,6 +34,12 @@ $(NONOS_BAKED_TRUST_DIR)/keys/guest_%_publisher_mldsa65.pub: | $(CAPSULE_SIGN_BI
# name, service port, reply port[, prebuilt ELF[, guest path]]. The enrolled
# copy is named guest_, so its certificate and trailer cannot collide
# with a capsule's. The guest path defaults to /bin/.
+# Every guest is built, signed and proven, but the store the vfs loads holds
+# at most 16 MiB, which the whole set outgrows. LINUX_GUEST_STORE_ONLY, when
+# set, names the guests the store carries, for example
+# make LINUX_GUEST_STORE_ONLY="busybox csock" LINUX_GUEST_BOOT_ARGS=... \
+# target/qemu-virtio-blk.img.store.stamp
+# and unset it carries them all, as before.
define LINUX_GUEST
CAPSULE_SLUG := linux-guest-$(1)
CAPSULE_HANDLE := linux.guest.$(1)
@@ -54,10 +60,12 @@ nonos-mk-check-linux-guest-$(1)-keys: \
$(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_ed25519.pub \
$(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_mldsa65.pub
LINUX_GUEST_STORE_DEPS += $$(linux-guest-$(1)_ARTIFACTS) $$(linux-guest-$(1)_ATTESTATION)
+ifneq ($(if $(LINUX_GUEST_STORE_ONLY),$(filter $(1),$(LINUX_GUEST_STORE_ONLY)),all),)
LINUX_GUEST_STORE_ENTRIES += --entry /linux$(or $(5),/bin/$(1))=$$(linux-guest-$(1)_BIN) \
--entry /linux$(or $(5),/bin/$(1)).nonos_id_cert.bin=$$(linux-guest-$(1)_CERT) \
--entry /linux$(or $(5),/bin/$(1)).manifest.bin=$$(linux-guest-$(1)_MANIFEST) \
--entry /linux$(or $(5),/bin/$(1)).zk_trailer.bin=$$(linux-guest-$(1)_ATTESTATION)
+endif
endef
$(eval $(call LINUX_GUEST,suite,4950,4951))
@@ -99,6 +107,9 @@ $(eval $(call LINUX_GUEST,gopoll,4976,4977,$(GO_OUT)/poll))
# A goroutine spinning with no call, which only a signal to its running
# thread can move off the one CPU the guest has.
$(eval $(call LINUX_GUEST,gopreempt,4944,4945,$(GO_OUT)/preempt))
+# net/http inside one guest: a server on 127.0.0.1:0 and its client, twenty
+# GETs over one kept-alive connection.
+$(eval $(call LINUX_GUEST,gohttp,5002,5003,$(GO_OUT)/http))
# A C guest that faults in a worker thread while main joins: it proves the
# whole process ends, as on Linux, and that musl threads run. Static, so no
@@ -119,6 +130,36 @@ $(LINUX_GUESTS_C)/cwait: $(LINUX_GUESTS_DIR)/c/cwait.c
@mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $<
$(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait))
+# Sockets as Linux has them, on the family's own loopback: socketpair, a
+# listener with accept4's flags, a non-blocking connect, a refused port,
+# half-close, epoll on a listener, end of file, EAGAIN, MSG_PEEK, EPIPE, an
+# accept and a receive that wait, the options a server sets, and fork.
+$(LINUX_GUESTS_C)/csock: $(LINUX_GUESTS_DIR)/c/csock.c $(wildcard $(LINUX_GUESTS_DIR)/c/csock_*.h)
+ @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $<
+$(eval $(call LINUX_GUEST,csock,5000,5001,$(LINUX_GUESTS_C)/csock))
+
+# Datagrams on the family's loopback: an echo, a connected socket, MSG_TRUNC,
+# a refused port, sendmmsg and recvmmsg, and no peer at all.
+$(LINUX_GUESTS_C)/cudp: $(LINUX_GUESTS_DIR)/c/cudp.c $(wildcard $(LINUX_GUESTS_DIR)/c/cudp_*.h)
+ @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $<
+$(eval $(call LINUX_GUEST,cudp,5004,5005,$(LINUX_GUESTS_C)/cudp))
+
+# What a guest's sockets may reach, and what a descriptor number alone gets.
+$(LINUX_GUESTS_C)/cpolicy: $(LINUX_GUESTS_DIR)/c/cpolicy.c $(wildcard $(LINUX_GUESTS_DIR)/c/cpolicy_*.h)
+ @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $<
+$(eval $(call LINUX_GUEST,cpolicy,5006,5007,$(LINUX_GUESTS_C)/cpolicy))
+
+# A guest blocked in accept with nothing happening, for the loop's wakeups.
+$(LINUX_GUESTS_C)/cidle: $(LINUX_GUESTS_DIR)/c/cidle.c $(wildcard $(LINUX_GUESTS_DIR)/c/cidle_*.h)
+ @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $<
+$(eval $(call LINUX_GUEST,cidle,5008,5009,$(LINUX_GUESTS_C)/cidle))
+
+# Unix sockets with names: a path, what it leaves behind, abstract names,
+# a connected datagram socket, autobind, and a connection across fork.
+$(LINUX_GUESTS_C)/cunix: $(LINUX_GUESTS_DIR)/c/cunix.c $(wildcard $(LINUX_GUESTS_DIR)/c/cunix_*.h)
+ @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $<
+$(eval $(call LINUX_GUEST,cunix,5010,5011,$(LINUX_GUESTS_C)/cunix))
+
# The Linux-guest test store is about guests, not the desktop's media and demo
# capsules. Drop both so the signed guest set fits the vfs load budget; the
# normal image, which does not set NONOS_LINUX_GUESTS, still ships them.
diff --git a/userland/linux_guests/c/cidle.c b/userland/linux_guests/c/cidle.c
new file mode 100644
index 000000000..d94d14006
--- /dev/null
+++ b/userland/linux_guests/c/cidle.c
@@ -0,0 +1,41 @@
+/*
+ * An idle guest: main blocks in accept on 127.0.0.1 while nothing happens
+ * for the number of seconds given (default 10), then a thread connects. It
+ * is what the serve loop's wakeups are measured against: a family socket
+ * changes only in an answer, so the loop has nothing to look at meanwhile.
+ */
+
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+
+#include "cidle_1.h"
+
+int main(int argc, char **argv) {
+ if (argc > 1) {
+ idle_s = atoi(argv[1]);
+ }
+ int l = socket(AF_INET, SOCK_STREAM, 0);
+ memset(&addr, 0, sizeof addr);
+ addr.sin_family = AF_INET;
+ addr.sin_addr.s_addr = htonl(INADDR_LOOPBACK);
+ socklen_t len = sizeof addr;
+ bind(l, (void *)&addr, sizeof addr);
+ getsockname(l, (void *)&addr, &len);
+ listen(l, 1);
+ pthread_t t;
+ long t0 = now_ms();
+ pthread_create(&t, 0, late, 0);
+ int s = accept(l, 0, 0);
+ long waited = now_ms() - t0;
+ pthread_join(t, 0);
+ printf("[C] cidle %s: accept blocked %ld ms for a connect after %d s\n",
+ s >= 0 && waited >= idle_s * 1000 ? "PASS" : "FAIL", waited, idle_s);
+ fflush(stdout);
+ return s >= 0 ? 0 : 1;
+}
diff --git a/userland/linux_guests/c/cidle_1.h b/userland/linux_guests/c/cidle_1.h
new file mode 100644
index 000000000..e624b6e19
--- /dev/null
+++ b/userland/linux_guests/c/cidle_1.h
@@ -0,0 +1,20 @@
+/* cidle, part 1 of 1: included once, by cidle.c. */
+
+static struct sockaddr_in addr;
+
+static int idle_s = 10;
+
+static long now_ms(void) {
+ struct timespec ts;
+ clock_gettime(CLOCK_MONOTONIC, &ts);
+ return ts.tv_sec * 1000 + ts.tv_nsec / 1000000;
+}
+
+static void *late(void *arg) {
+ (void)arg;
+ struct timespec ts = {idle_s, 0};
+ nanosleep(&ts, 0);
+ int c = socket(AF_INET, SOCK_STREAM, 0);
+ connect(c, (void *)&addr, sizeof addr);
+ return (void *)(long)c;
+}
diff --git a/userland/linux_guests/c/cpolicy.c b/userland/linux_guests/c/cpolicy.c
new file mode 100644
index 000000000..0ff76e759
--- /dev/null
+++ b/userland/linux_guests/c/cpolicy.c
@@ -0,0 +1,61 @@
+/*
+ * What a guest's sockets may reach, and what a descriptor number alone gets.
+ * Each part prints the errno it got; the NONOS answer is the capsule's policy
+ * and differs from an unconfined Linux by design, so the host's line records
+ * what Linux itself would allow. Parts:
+ * bind-any bind 0.0.0.0:0 NONOS EACCES, Linux 0
+ * bind-out bind 10.0.2.15:0 NONOS EACCES, Linux EADDRNOTAVAIL
+ * listen-unbound listen with no bind NONOS EACCES, Linux 0
+ * udp-out sendto 192.0.2.1:9 NONOS ENETUNREACH, Linux 1
+ * raw socket(SOCK_RAW) NONOS EPERM, Linux EPERM unless root
+ * forged recv/close on numbers the guest never opened, and on a
+ * pipe: EBADF, EBADF, ENOTSOCK on both
+ */
+
+#include
+#include
+#include
+#include
+#include
+#include
+
+#include "cpolicy_1.h"
+
+int main(void) {
+ int s = socket(AF_INET, SOCK_STREAM, 0);
+ struct sockaddr_in any = at(0, 0), out = at(0x0a00020f, 0), far = at(0xc0000201, 9);
+ int bind_any = err(bind(s, (void *)&any, sizeof any));
+ close(s);
+ s = socket(AF_INET, SOCK_STREAM, 0);
+ int bind_out = err(bind(s, (void *)&out, sizeof out));
+ close(s);
+ s = socket(AF_INET, SOCK_STREAM, 0);
+ int listen_unbound = err(listen(s, 1));
+ close(s);
+ int u = socket(AF_INET, SOCK_DGRAM, 0);
+ int udp_out = sendto(u, "x", 1, 0, (void *)&far, sizeof far) == 1 ? 0 : errno;
+ close(u);
+ int raw = err(socket(AF_INET, SOCK_RAW, IPPROTO_ICMP));
+ char buf[4];
+ int pipe_fds[2];
+ if (pipe(pipe_fds)) {
+ printf("[C] cpolicy FAIL: pipe errno %d\n", errno);
+ return 1;
+ }
+ int forged_recv = err(recv(777, buf, 4, MSG_DONTWAIT));
+ int forged_close = err(close(778));
+ int pipe_recv = err(recv(pipe_fds[0], buf, 4, MSG_DONTWAIT));
+ int pipe_name = err(getsockname(pipe_fds[0], (void *)&any, &(socklen_t){sizeof any}));
+ printf("[C] cpolicy bind-any %d bind-out %d listen-unbound %d udp-out %d raw %d "
+ "forged %d %d pipe %d %d\n",
+ bind_any, bind_out, listen_unbound, udp_out, raw, forged_recv, forged_close,
+ pipe_recv, pipe_name);
+ int confined = bind_any == EACCES && bind_out == EACCES && listen_unbound == EACCES &&
+ udp_out == ENETUNREACH && raw == EPERM;
+ int forged = forged_recv == EBADF && forged_close == EBADF && pipe_recv == ENOTSOCK &&
+ pipe_name == ENOTSOCK;
+ printf("[C] cpolicy %s: confined %d, forged numbers refused %d\n",
+ confined && forged ? "PASS" : "FAIL", confined, forged);
+ fflush(stdout);
+ return confined && forged ? 0 : 1;
+}
diff --git a/userland/linux_guests/c/cpolicy_1.h b/userland/linux_guests/c/cpolicy_1.h
new file mode 100644
index 000000000..6127f8e45
--- /dev/null
+++ b/userland/linux_guests/c/cpolicy_1.h
@@ -0,0 +1,14 @@
+/* cpolicy, part 1 of 1: included once, by cpolicy.c. */
+
+static struct sockaddr_in at(unsigned ip, int port) {
+ struct sockaddr_in sa;
+ memset(&sa, 0, sizeof sa);
+ sa.sin_family = AF_INET;
+ sa.sin_port = htons(port);
+ sa.sin_addr.s_addr = htonl(ip);
+ return sa;
+}
+
+static int err(int rc) {
+ return rc < 0 ? errno : 0;
+}
diff --git a/userland/linux_guests/c/csock.c b/userland/linux_guests/c/csock.c
new file mode 100644
index 000000000..481813ffe
--- /dev/null
+++ b/userland/linux_guests/c/csock.c
@@ -0,0 +1,64 @@
+/*
+ * Sockets, as Linux has them: a socketpair each way, a listener on 127.0.0.1
+ * with its name and accept4's flags, a non-blocking connect that answers
+ * EINPROGRESS and then SO_ERROR 0, a refused port, shutdown(SHUT_WR) giving
+ * the peer end of file while the other way still flows, epoll readiness on a
+ * listener, end of file, EAGAIN on an empty non-blocking receive, MSG_PEEK
+ * and MSG_DONTWAIT, EPIPE after the peer is gone, an accept and a receive
+ * that wait for another thread, the options Go and a C server set, a
+ * socket a forked child shares, a connect a full listener holds, the
+ * options a loopback connection cannot tell apart, and listeners that share
+ * a port with SO_REUSEPORT. Each part prints as it passes and every part
+ * runs, so one run names each part that fails.
+ */
+
+#define _GNU_SOURCE
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+
+#include "csock_1.h"
+#include "csock_2.h"
+#include "csock_3.h"
+#include "csock_4.h"
+#include "csock_5.h"
+#include "csock_6.h"
+#include "csock_7.h"
+#include "csock_8.h"
+#include "csock_9.h"
+#include "csock_10.h"
+#include "csock_11.h"
+
+int main(void) {
+ signal(SIGPIPE, SIG_IGN);
+ int (*const part[])(void) = {
+ pair_stream, pair_dgram, listen_accept, nb_connect, refused, half_close,
+ epoll_listener, eof, empty_recv, peek, epipe, blocking_accept,
+ blocking_recv, options, fork_share, backlog, quiet_options, reuseport,
+ };
+ const int count = sizeof part / sizeof part[0];
+ long t0 = now_ms();
+ int failed = 0;
+ for (int i = 0; i < count; i++) {
+ failed += part[i]();
+ }
+ if (failed) {
+ printf("[C] csock FAIL: %d parts failed, %d passed\n", failed, parts);
+ fflush(stdout);
+ return 1;
+ }
+ printf("[C] csock PASS: %d parts in %ld ms\n", parts, now_ms() - t0);
+ fflush(stdout);
+ return 0;
+}
diff --git a/userland/linux_guests/c/csock_1.h b/userland/linux_guests/c/csock_1.h
new file mode 100644
index 000000000..807f0aea0
--- /dev/null
+++ b/userland/linux_guests/c/csock_1.h
@@ -0,0 +1,67 @@
+/* csock, part 1 of 11: included once, by csock.c. */
+
+static int parts;
+
+static long now_ms(void) {
+ struct timespec ts;
+ clock_gettime(CLOCK_MONOTONIC, &ts);
+ return ts.tv_sec * 1000 + ts.tv_nsec / 1000000;
+}
+
+static void nap_ms(long ms) {
+ struct timespec ts = {ms / 1000, (ms % 1000) * 1000000};
+ nanosleep(&ts, 0);
+}
+
+static int fail(const char *what, long a, long b) {
+ printf("[C] csock FAIL: %s (%ld, %ld)\n", what, a, b);
+ fflush(stdout);
+ return 1;
+}
+
+static void ok(const char *part, const char *detail, long n) {
+ parts++;
+ printf("[C] csock %s ok: %s %ld\n", part, detail, n);
+ fflush(stdout);
+}
+
+static struct sockaddr_in loopback(int port) {
+ struct sockaddr_in sa;
+ memset(&sa, 0, sizeof sa);
+ sa.sin_family = AF_INET;
+ sa.sin_port = htons(port);
+ sa.sin_addr.s_addr = htonl(INADDR_LOOPBACK);
+ return sa;
+}
+
+/* A listener on 127.0.0.1 at a port the kernel picks; its port through *port. */
+static int listener(int *port, int flags) {
+ int s = socket(AF_INET, SOCK_STREAM | flags, 0);
+ if (s < 0) {
+ return -errno;
+ }
+ int one = 1;
+ setsockopt(s, SOL_SOCKET, SO_REUSEADDR, &one, sizeof one);
+ struct sockaddr_in sa = loopback(0);
+ socklen_t len = sizeof sa;
+ if (bind(s, (void *)&sa, sizeof sa) || getsockname(s, (void *)&sa, &len) || listen(s, 8)) {
+ int e = errno;
+ close(s);
+ return -e;
+ }
+ *port = ntohs(sa.sin_port);
+ return s;
+}
+
+static int dial(int port) {
+ int c = socket(AF_INET, SOCK_STREAM, 0);
+ struct sockaddr_in sa = loopback(port);
+ if (c < 0 || connect(c, (void *)&sa, sizeof sa)) {
+ int e = errno;
+ if (c >= 0) {
+ close(c);
+ }
+ return -e;
+ }
+ return c;
+}
diff --git a/userland/linux_guests/c/csock_10.h b/userland/linux_guests/c/csock_10.h
new file mode 100644
index 000000000..b459408e0
--- /dev/null
+++ b/userland/linux_guests/c/csock_10.h
@@ -0,0 +1,33 @@
+/* csock, part 10 of 11: included once, by csock.c. */
+
+/*
+ * The options Linux keeps that a loopback connection cannot tell apart,
+ * read back as Linux reads them, and the ones a socket's kind refuses.
+ */
+static int quiet_options(void) {
+ int t = socket(AF_INET, SOCK_STREAM, 0), u = socket(AF_INET, SOCK_DGRAM, 0);
+ int x = socket(AF_UNIX, SOCK_STREAM, 0);
+ int tos = 0x13, ttl = 32, zero = 0, ut = 5000, prio = 6;
+ setsockopt(t, IPPROTO_IP, IP_TOS, &tos, sizeof tos);
+ setsockopt(t, IPPROTO_IP, IP_TTL, &ttl, sizeof ttl);
+ int bad_ttl = setsockopt(t, IPPROTO_IP, IP_TTL, &zero, sizeof zero) ? errno : 0;
+ setsockopt(t, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut);
+ setsockopt(t, SOL_SOCKET, SO_PRIORITY, &prio, sizeof prio);
+ int got_tos = get_int(t, IPPROTO_IP, IP_TOS), got_ttl = get_int(t, IPPROTO_IP, IP_TTL);
+ int got_ut = get_int(t, IPPROTO_TCP, TCP_USER_TIMEOUT);
+ int got_qa = get_int(t, IPPROTO_TCP, TCP_QUICKACK), got_prio = get_int(t, SOL_SOCKET, SO_PRIORITY);
+ int udp_tcp = setsockopt(u, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut) ? errno : 0;
+ int unix_tcp = setsockopt(x, IPPROTO_TCP, TCP_NODELAY, &prio, sizeof prio) ? errno : 0;
+ close(t);
+ close(u);
+ close(x);
+ /* A TCP socket drops the two ECN bits of the type of service. */
+ if (got_tos != 0x10 || got_ttl != 32 || bad_ttl != EINVAL || got_ut != 5000 || got_qa != 1 ||
+ got_prio != 6 || udp_tcp != ENOPROTOOPT || unix_tcp != EOPNOTSUPP) {
+ printf("[C] csock quiet_options got %d %d %d %d %d %d %d %d\n", got_tos, got_ttl, bad_ttl,
+ got_ut, got_qa, got_prio, udp_tcp, unix_tcp);
+ return fail("quiet_options: kept and read back as Linux does", got_tos, got_ttl);
+ }
+ ok("quiet_options", "6 kept and read back; refusals", unix_tcp);
+ return 0;
+}
diff --git a/userland/linux_guests/c/csock_11.h b/userland/linux_guests/c/csock_11.h
new file mode 100644
index 000000000..4bf6fb89c
--- /dev/null
+++ b/userland/linux_guests/c/csock_11.h
@@ -0,0 +1,52 @@
+/* csock, part 11 of 11: included once, by csock.c. */
+
+/*
+ * Listeners that set SO_REUSEPORT share a port and between them take every
+ * connection; a socket without it cannot bind there, and when one listener
+ * closes the other takes what comes next.
+ */
+static int reuseport(void) {
+ struct sockaddr_in sa = loopback(0);
+ socklen_t len = sizeof sa;
+ int one = 1, l[2], taken[2] = {0, 0};
+ for (int i = 0; i < 2; i++) {
+ l[i] = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0);
+ setsockopt(l[i], SOL_SOCKET, SO_REUSEPORT, &one, sizeof one);
+ if (bind(l[i], (void *)&sa, sizeof sa) || listen(l[i], 16)) {
+ return fail("reuseport: two listeners on one port", i, errno);
+ }
+ getsockname(l[i], (void *)&sa, &len);
+ }
+ int plain = socket(AF_INET, SOCK_STREAM, 0);
+ int refused = bind(plain, (void *)&sa, sizeof sa) ? errno : 0;
+ close(plain);
+ int c[16];
+ for (int i = 0; i < 16; i++) {
+ c[i] = dial(ntohs(sa.sin_port));
+ }
+ for (int i = 0; i < 2; i++) {
+ int a;
+ while ((a = accept(l[i], 0, 0)) >= 0) {
+ taken[i]++;
+ close(a);
+ }
+ }
+ close(l[0]);
+ int late = dial(ntohs(sa.sin_port));
+ struct pollfd p = {l[1], POLLIN, 0};
+ int ready = poll(&p, 1, 3000);
+ int last = accept(l[1], 0, 0);
+ for (int i = 0; i < 16; i++) {
+ close(c[i]);
+ }
+ close(late);
+ close(last);
+ close(l[1]);
+ if (refused != EADDRINUSE || taken[0] + taken[1] != 16 || !taken[0] || !taken[1] ||
+ ready != 1 || last < 0) {
+ printf("[C] csock reuseport got %d %d+%d %d %d\n", refused, taken[0], taken[1], ready, last);
+ return fail("reuseport: shared, spread, and the survivor takes the rest", taken[0], taken[1]);
+ }
+ ok("reuseport", "16 connections spread over 2 listeners; without it errno", refused);
+ return 0;
+}
diff --git a/userland/linux_guests/c/csock_2.h b/userland/linux_guests/c/csock_2.h
new file mode 100644
index 000000000..97feafea4
--- /dev/null
+++ b/userland/linux_guests/c/csock_2.h
@@ -0,0 +1,57 @@
+/* csock, part 2 of 11: included once, by csock.c. */
+
+/* A connected pair over 127.0.0.1: *a is the client, *b the accepted end. */
+static int tcp_pair(int *a, int *b) {
+ int port, l = listener(&port, 0);
+ if (l < 0) {
+ return l;
+ }
+ *a = dial(port);
+ *b = *a < 0 ? -1 : accept(l, 0, 0);
+ close(l);
+ return *a < 0 ? *a : *b < 0 ? -errno : 0;
+}
+
+/* The parts of csock, one Linux behaviour each. Included once, by csock.c. */
+
+static int pair_stream(void) {
+ int sv[2];
+ char buf[8] = {0};
+ if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv)) {
+ return fail("pair_stream: socketpair", -1, errno);
+ }
+ if (write(sv[0], "ping", 4) != 4 || read(sv[1], buf, 8) != 4 || memcmp(buf, "ping", 4)) {
+ return fail("pair_stream: a to b", 0, errno);
+ }
+ if (write(sv[1], "pong!", 5) != 5 || read(sv[0], buf, 8) != 5 || memcmp(buf, "pong!", 5)) {
+ return fail("pair_stream: b to a", 0, errno);
+ }
+ close(sv[0]);
+ long got = read(sv[1], buf, 8);
+ close(sv[1]);
+ if (got != 0) {
+ return fail("pair_stream: end of file after close", got, errno);
+ }
+ ok("pair_stream", "bytes each way, then eof", 9);
+ return 0;
+}
+
+static int pair_dgram(void) {
+ int sv[2];
+ char buf[16];
+ if (socketpair(AF_UNIX, SOCK_DGRAM, 0, sv)) {
+ return fail("pair_dgram: socketpair", -1, errno);
+ }
+ if (write(sv[0], "one", 3) != 3 || write(sv[0], "second", 6) != 6) {
+ return fail("pair_dgram: write", -1, errno);
+ }
+ long a = read(sv[1], buf, sizeof buf);
+ long b = read(sv[1], buf, sizeof buf);
+ close(sv[0]);
+ close(sv[1]);
+ if (a != 3 || b != 6) {
+ return fail("pair_dgram: boundaries kept", a, b);
+ }
+ ok("pair_dgram", "two datagrams, sizes 3 and", b);
+ return 0;
+}
diff --git a/userland/linux_guests/c/csock_3.h b/userland/linux_guests/c/csock_3.h
new file mode 100644
index 000000000..0d2beb006
--- /dev/null
+++ b/userland/linux_guests/c/csock_3.h
@@ -0,0 +1,47 @@
+/* csock, part 3 of 11: included once, by csock.c. */
+
+static int listen_accept(void) {
+ int port, l = listener(&port, SOCK_NONBLOCK);
+ if (l < 0) {
+ return fail("listen_accept: listener", l, 0);
+ }
+ if (port == 0) {
+ return fail("listen_accept: getsockname gave port 0", 0, 0);
+ }
+ if (accept4(l, 0, 0, 0) != -1 || errno != EAGAIN) {
+ return fail("listen_accept: empty non-blocking accept is EAGAIN", errno, EAGAIN);
+ }
+ int c = dial(port);
+ if (c < 0) {
+ return fail("listen_accept: connect", c, 0);
+ }
+ struct sockaddr_in peer, mine;
+ socklen_t plen = sizeof peer, mlen = sizeof mine;
+ int s = accept4(l, (void *)&peer, &plen, SOCK_NONBLOCK | SOCK_CLOEXEC);
+ if (s < 0) {
+ return fail("listen_accept: accept4", -1, errno);
+ }
+ if (!(fcntl(s, F_GETFL) & O_NONBLOCK) || !(fcntl(s, F_GETFD) & FD_CLOEXEC)) {
+ return fail("listen_accept: accept4 flags", fcntl(s, F_GETFL), fcntl(s, F_GETFD));
+ }
+ getsockname(c, (void *)&mine, &mlen);
+ if (peer.sin_port != mine.sin_port || peer.sin_addr.s_addr != htonl(INADDR_LOOPBACK)) {
+ return fail("listen_accept: accept's address is the client's", ntohs(peer.sin_port),
+ ntohs(mine.sin_port));
+ }
+ struct sockaddr_in back;
+ socklen_t blen = sizeof back;
+ getpeername(c, (void *)&back, &blen);
+ if (ntohs(back.sin_port) != port || blen != sizeof back) {
+ return fail("listen_accept: getpeername", ntohs(back.sin_port), port);
+ }
+ char buf[8];
+ if (write(c, "hi", 2) != 2 || read(s, buf, 8) != 2) {
+ return fail("listen_accept: bytes", 0, errno);
+ }
+ close(c);
+ close(s);
+ close(l);
+ ok("listen_accept", "accept4 NONBLOCK|CLOEXEC on port", port > 0);
+ return 0;
+}
diff --git a/userland/linux_guests/c/csock_4.h b/userland/linux_guests/c/csock_4.h
new file mode 100644
index 000000000..c1b1ec64b
--- /dev/null
+++ b/userland/linux_guests/c/csock_4.h
@@ -0,0 +1,52 @@
+/* csock, part 4 of 11: included once, by csock.c. */
+
+/* csock: connecting, and closing one way or both. */
+
+static int nb_connect(void) {
+ int port, l = listener(&port, 0);
+ int c = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0);
+ struct sockaddr_in sa = loopback(port);
+ int rc = connect(c, (void *)&sa, sizeof sa);
+ int e = errno;
+ if (rc != -1 || e != EINPROGRESS) {
+ return fail("nb_connect: EINPROGRESS", rc, e);
+ }
+ struct pollfd p = {c, POLLOUT, 0};
+ if (poll(&p, 1, 5000) != 1 || !(p.revents & POLLOUT)) {
+ return fail("nb_connect: POLLOUT", p.revents, 0);
+ }
+ int err = -1;
+ socklen_t len = sizeof err;
+ if (getsockopt(c, SOL_SOCKET, SO_ERROR, &err, &len) || err != 0) {
+ return fail("nb_connect: SO_ERROR", err, errno);
+ }
+ close(c);
+ close(l);
+ ok("nb_connect", "EINPROGRESS, POLLOUT, SO_ERROR", err);
+ return 0;
+}
+
+/* A port with no listener: a listener is opened and closed to find one. */
+static int refused(void) {
+ int port, l = listener(&port, 0);
+ close(l);
+ int c = dial(port);
+ if (c != -ECONNREFUSED) {
+ return fail("refused: blocking connect", c, -ECONNREFUSED);
+ }
+ int n = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0);
+ struct sockaddr_in sa = loopback(port);
+ int rc = connect(n, (void *)&sa, sizeof sa);
+ int e = errno;
+ struct pollfd p = {n, POLLOUT, 0};
+ poll(&p, 1, 5000);
+ int err = 0;
+ socklen_t len = sizeof err;
+ getsockopt(n, SOL_SOCKET, SO_ERROR, &err, &len);
+ close(n);
+ if (rc != -1 || e != EINPROGRESS || err != ECONNREFUSED || !(p.revents & POLLERR)) {
+ return fail("refused: non-blocking connect then SO_ERROR", e, err);
+ }
+ ok("refused", "ECONNREFUSED both ways, errno", ECONNREFUSED);
+ return 0;
+}
diff --git a/userland/linux_guests/c/csock_5.h b/userland/linux_guests/c/csock_5.h
new file mode 100644
index 000000000..0fd77ed48
--- /dev/null
+++ b/userland/linux_guests/c/csock_5.h
@@ -0,0 +1,69 @@
+/* csock, part 5 of 11: included once, by csock.c. */
+
+static int half_close(void) {
+ int a, b;
+ char buf[8];
+ if (tcp_pair(&a, &b)) {
+ return fail("half_close: pair", 0, errno);
+ }
+ if (shutdown(a, SHUT_WR)) {
+ return fail("half_close: shutdown", -1, errno);
+ }
+ long got = read(b, buf, 8);
+ if (got != 0) {
+ return fail("half_close: peer reads end of file", got, errno);
+ }
+ if (write(b, "back", 4) != 4 || read(a, buf, 8) != 4) {
+ return fail("half_close: the other way still flows", 0, errno);
+ }
+ if (write(a, "x", 1) != -1 || errno != EPIPE) {
+ return fail("half_close: write after SHUT_WR is EPIPE", errno, EPIPE);
+ }
+ if (fcntl(a, F_GETFD) < 0) {
+ return fail("half_close: shutdown closed the descriptor", errno, 0);
+ }
+ close(a);
+ close(b);
+ ok("half_close", "eof one way, 4 bytes back", 4);
+ return 0;
+}
+
+static int epoll_listener(void) {
+ int port, l = listener(&port, SOCK_NONBLOCK);
+ int ep = epoll_create1(0);
+ struct epoll_event ev = {.events = EPOLLIN, .data.fd = l}, out;
+ epoll_ctl(ep, EPOLL_CTL_ADD, l, &ev);
+ int before = epoll_wait(ep, &out, 1, 0);
+ int c = dial(port);
+ int after = epoll_wait(ep, &out, 1, 5000);
+ int s = accept(l, 0, 0);
+ int drained = epoll_wait(ep, &out, 1, 0);
+ close(s);
+ close(c);
+ close(ep);
+ close(l);
+ if (before != 0 || after != 1 || !(out.events & EPOLLIN) || s < 0 || drained != 0) {
+ return fail("epoll_listener: EPOLLIN only while pending", before * 10 + after, drained);
+ }
+ ok("epoll_listener", "EPOLLIN once pending, then", drained);
+ return 0;
+}
+
+static int eof(void) {
+ int a, b;
+ char buf[8];
+ if (tcp_pair(&a, &b)) {
+ return fail("eof: pair", 0, errno);
+ }
+ if (write(a, "last", 4) != 4) {
+ return fail("eof: write", -1, errno);
+ }
+ close(a);
+ long first = read(b, buf, 8), then = read(b, buf, 8);
+ close(b);
+ if (first != 4 || then != 0) {
+ return fail("eof: data then end of file", first, then);
+ }
+ ok("eof", "4 bytes then", then);
+ return 0;
+}
diff --git a/userland/linux_guests/c/csock_6.h b/userland/linux_guests/c/csock_6.h
new file mode 100644
index 000000000..fc011214a
--- /dev/null
+++ b/userland/linux_guests/c/csock_6.h
@@ -0,0 +1,67 @@
+/* csock, part 6 of 11: included once, by csock.c. */
+
+/* csock: receiving without waiting, waiting, options, and fork. */
+
+static int empty_recv(void) {
+ int a, b;
+ char buf[8];
+ if (tcp_pair(&a, &b)) {
+ return fail("empty_recv: pair", 0, errno);
+ }
+ if (recv(b, buf, 8, MSG_DONTWAIT) != -1 || errno != EAGAIN) {
+ return fail("empty_recv: MSG_DONTWAIT", errno, EAGAIN);
+ }
+ fcntl(b, F_SETFL, O_NONBLOCK);
+ if (read(b, buf, 8) != -1 || errno != EAGAIN) {
+ return fail("empty_recv: O_NONBLOCK read", errno, EAGAIN);
+ }
+ close(a);
+ close(b);
+ ok("empty_recv", "EAGAIN, errno", EAGAIN);
+ return 0;
+}
+
+static int peek(void) {
+ int a, b;
+ char buf[8] = {0};
+ if (tcp_pair(&a, &b)) {
+ return fail("peek: pair", 0, errno);
+ }
+ if (write(a, "abc", 3) != 3) {
+ return fail("peek: write", -1, errno);
+ }
+ long p = recv(b, buf, 8, MSG_PEEK);
+ long r = recv(b, buf, 8, 0);
+ close(a);
+ close(b);
+ if (p != 3 || r != 3 || memcmp(buf, "abc", 3)) {
+ return fail("peek: data stays", p, r);
+ }
+ ok("peek", "peeked then read", r);
+ return 0;
+}
+
+static int epipe(void) {
+ int a, b;
+ if (tcp_pair(&a, &b)) {
+ return fail("epipe: pair", 0, errno);
+ }
+ close(b);
+ long first = send(a, "x", 1, MSG_NOSIGNAL);
+ long second = send(a, "x", 1, MSG_NOSIGNAL);
+ int e = errno;
+ close(a);
+ if (first != 1 || second != -1 || e != EPIPE) {
+ return fail("epipe: first send taken, then EPIPE", first, e);
+ }
+ ok("epipe", "second send errno", e);
+ return 0;
+}
+
+static int late_port;
+
+static void *late_dial(void *arg) {
+ (void)arg;
+ nap_ms(100);
+ return (void *)(long)dial(late_port);
+}
diff --git a/userland/linux_guests/c/csock_7.h b/userland/linux_guests/c/csock_7.h
new file mode 100644
index 000000000..872f2ee6b
--- /dev/null
+++ b/userland/linux_guests/c/csock_7.h
@@ -0,0 +1,61 @@
+/* csock, part 7 of 11: included once, by csock.c. */
+
+static int blocking_accept(void) {
+ int l = listener(&late_port, 0);
+ pthread_t t;
+ long t0 = now_ms();
+ pthread_create(&t, 0, late_dial, 0);
+ int s = accept(l, 0, 0);
+ long waited = now_ms() - t0;
+ void *c;
+ pthread_join(t, &c);
+ close((int)(long)c);
+ close(s);
+ close(l);
+ if (s < 0 || waited < 80) {
+ return fail("blocking_accept: waited for the connect", s, waited);
+ }
+ ok("blocking_accept", "waited ms", waited >= 80);
+ return 0;
+}
+
+static int late_fd;
+
+static void *late_write(void *arg) {
+ (void)arg;
+ nap_ms(100);
+ if (write(late_fd, "late", 4) != 4) {
+ printf("[C] csock late_write: errno %d\n", errno);
+ }
+ return 0;
+}
+
+static int blocking_recv(void) {
+ int a, b;
+ char buf[8];
+ if (tcp_pair(&a, &b)) {
+ return fail("blocking_recv: pair", 0, errno);
+ }
+ late_fd = a;
+ pthread_t t;
+ long t0 = now_ms();
+ pthread_create(&t, 0, late_write, 0);
+ long got = recv(b, buf, 8, 0);
+ long waited = now_ms() - t0;
+ pthread_join(t, 0);
+ close(a);
+ close(b);
+ if (got != 4 || waited < 80) {
+ return fail("blocking_recv: waited for the bytes", got, waited);
+ }
+ ok("blocking_recv", "4 bytes after waiting", waited >= 80);
+ return 0;
+}
+
+/* csock: the options a server sets, and a socket a forked child shares. */
+
+static int get_int(int s, int level, int opt) {
+ int v = -1;
+ socklen_t len = sizeof v;
+ return getsockopt(s, level, opt, &v, &len) ? -errno : v;
+}
diff --git a/userland/linux_guests/c/csock_8.h b/userland/linux_guests/c/csock_8.h
new file mode 100644
index 000000000..3a3f7f366
--- /dev/null
+++ b/userland/linux_guests/c/csock_8.h
@@ -0,0 +1,54 @@
+/* csock, part 8 of 11: included once, by csock.c. */
+
+static int options(void) {
+ int port, l = listener(&port, 0);
+ int one = 1;
+ if (get_int(l, SOL_SOCKET, SO_TYPE) != SOCK_STREAM ||
+ get_int(l, SOL_SOCKET, SO_DOMAIN) != AF_INET ||
+ get_int(l, SOL_SOCKET, SO_ACCEPTCONN) != 1 ||
+ get_int(l, SOL_SOCKET, SO_REUSEADDR) != 1) {
+ return fail("options: type, domain, acceptconn, reuseaddr",
+ get_int(l, SOL_SOCKET, SO_TYPE), get_int(l, SOL_SOCKET, SO_ACCEPTCONN));
+ }
+ int c = dial(port);
+ int sets[][2] = {
+ {SOL_SOCKET, SO_KEEPALIVE}, {IPPROTO_TCP, TCP_NODELAY}, {SOL_SOCKET, SO_REUSEPORT},
+ {SOL_SOCKET, SO_BROADCAST},
+ };
+ for (unsigned i = 0; i < sizeof sets / sizeof sets[0]; i++) {
+ if (setsockopt(c, sets[i][0], sets[i][1], &one, sizeof one) ||
+ get_int(c, sets[i][0], sets[i][1]) != 1) {
+ return fail("options: set then read back", sets[i][1], errno);
+ }
+ }
+ int idle = 15;
+ if (setsockopt(c, IPPROTO_TCP, TCP_KEEPIDLE, &idle, sizeof idle) ||
+ get_int(c, IPPROTO_TCP, TCP_KEEPIDLE) != 15) {
+ return fail("options: TCP_KEEPIDLE", get_int(c, IPPROTO_TCP, TCP_KEEPIDLE), errno);
+ }
+ int buf = 65536;
+ if (setsockopt(c, SOL_SOCKET, SO_RCVBUF, &buf, sizeof buf) ||
+ get_int(c, SOL_SOCKET, SO_RCVBUF) != 2 * buf) {
+ return fail("options: SO_RCVBUF doubled", get_int(c, SOL_SOCKET, SO_RCVBUF), 2 * buf);
+ }
+ struct linger lg = {1, 5}, back = {0, 0};
+ socklen_t len = sizeof back;
+ setsockopt(c, SOL_SOCKET, SO_LINGER, &lg, sizeof lg);
+ getsockopt(c, SOL_SOCKET, SO_LINGER, &back, &len);
+ struct timeval tv = {2, 500000}, tb = {0, 0};
+ len = sizeof tb;
+ setsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof tv);
+ getsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tb, &len);
+ int bogus = setsockopt(c, SOL_SOCKET, 9999, &one, sizeof one);
+ int e = errno;
+ close(c);
+ close(l);
+ if (back.l_onoff != 1 || back.l_linger != 5 || tb.tv_sec != 2 || tb.tv_usec != 500000) {
+ return fail("options: SO_LINGER and SO_RCVTIMEO read back", back.l_linger, tb.tv_usec);
+ }
+ if (bogus != -1 || e != ENOPROTOOPT) {
+ return fail("options: an unknown option is ENOPROTOOPT", bogus, e);
+ }
+ ok("options", "11 options, unknown errno", e);
+ return 0;
+}
diff --git a/userland/linux_guests/c/csock_9.h b/userland/linux_guests/c/csock_9.h
new file mode 100644
index 000000000..8286cdd30
--- /dev/null
+++ b/userland/linux_guests/c/csock_9.h
@@ -0,0 +1,72 @@
+/* csock, part 9 of 11: included once, by csock.c. */
+
+/*
+ * The child holds the accepted end after the parent closes its own: the
+ * client still reads the child's bytes, then end of file when the child exits.
+ */
+static int fork_share(void) {
+ int a, b;
+ char buf[8];
+ if (tcp_pair(&a, &b)) {
+ return fail("fork_share: pair", 0, errno);
+ }
+ pid_t kid = fork();
+ if (kid == 0) {
+ close(a);
+ nap_ms(50);
+ _exit(write(b, "kid", 3) == 3 ? 0 : 1);
+ }
+ close(b);
+ long got = read(a, buf, 8);
+ int status = 0;
+ waitpid(kid, &status, 0);
+ long then = read(a, buf, 8);
+ close(a);
+ if (got != 3 || then != 0 || status != 0) {
+ return fail("fork_share: child's bytes, then eof", got, then);
+ }
+ ok("fork_share", "3 bytes from the child, then", then);
+ return 0;
+}
+
+/*
+ * A listener with a backlog of 0 queues one connect; the next non-blocking
+ * one answers EINPROGRESS, is not writable, and a second connect on it is
+ * EALREADY, until an accept makes room and it completes.
+ */
+static int backlog(void) {
+ int s = socket(AF_INET, SOCK_STREAM, 0);
+ struct sockaddr_in sa = loopback(0);
+ socklen_t len = sizeof sa;
+ bind(s, (void *)&sa, sizeof sa);
+ getsockname(s, (void *)&sa, &len);
+ listen(s, 0);
+ int c0 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0);
+ int c1 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0);
+ connect(c0, (void *)&sa, sizeof sa);
+ int r1 = connect(c1, (void *)&sa, sizeof sa);
+ int e1 = errno;
+ struct pollfd p = {c1, POLLOUT, 0};
+ int early = poll(&p, 1, 100);
+ int again = connect(c1, (void *)&sa, sizeof sa) ? errno : 0;
+ int a = accept(s, 0, 0);
+ p.revents = 0;
+ int later = poll(&p, 1, 3000);
+ int err = -1;
+ socklen_t el = sizeof err;
+ getsockopt(c1, SOL_SOCKET, SO_ERROR, &err, &el);
+ struct pollfd q = {s, POLLIN, 0};
+ int b = poll(&q, 1, 3000) == 1 ? accept(s, 0, 0) : -1;
+ close(a);
+ close(b);
+ close(c0);
+ close(c1);
+ close(s);
+ if (r1 != -1 || e1 != EINPROGRESS || early != 0 || again != EALREADY || later != 1 ||
+ err != 0 || b < 0) {
+ printf("[C] csock backlog got %d %d %d %d %d %d %d\n", r1, e1, early, again, later, err, b);
+ return fail("backlog: EINPROGRESS, waits, EALREADY, then connected", e1, again);
+ }
+ ok("backlog", "a full queue's connect completed after accept; EALREADY", again);
+ return 0;
+}
diff --git a/userland/linux_guests/c/cudp.c b/userland/linux_guests/c/cudp.c
new file mode 100644
index 000000000..f379124f5
--- /dev/null
+++ b/userland/linux_guests/c/cudp.c
@@ -0,0 +1,39 @@
+/*
+ * Datagrams on the loopback address, as Linux has them: an echo between an
+ * unbound client and a bound server, each told the other's address; a
+ * connected socket that sends without naming a peer and keeps only its
+ * peer's datagrams; a datagram cut to the buffer with MSG_TRUNC, and
+ * recvmsg's MSG_TRUNC flag; ECONNREFUSED on a connected socket whose peer
+ * port is closed; sendmmsg and recvmmsg; and EDESTADDRREQ with no peer at
+ * all. Each part prints as it passes and every part runs.
+ */
+
+#define _GNU_SOURCE
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+
+#include "cudp_1.h"
+#include "cudp_2.h"
+#include "cudp_3.h"
+
+int main(void) {
+ int (*const part[])(void) = {echo, connected, cut, refused, mmsg, nameless};
+ const int count = sizeof part / sizeof part[0];
+ int failed = 0;
+ for (int i = 0; i < count; i++) {
+ failed += part[i]();
+ }
+ if (failed) {
+ printf("[C] cudp FAIL: %d parts failed, %d passed\n", failed, parts);
+ fflush(stdout);
+ return 1;
+ }
+ printf("[C] cudp PASS: %d parts\n", parts);
+ fflush(stdout);
+ return 0;
+}
diff --git a/userland/linux_guests/c/cudp_1.h b/userland/linux_guests/c/cudp_1.h
new file mode 100644
index 000000000..02ce6c122
--- /dev/null
+++ b/userland/linux_guests/c/cudp_1.h
@@ -0,0 +1,51 @@
+/* cudp, part 1 of 3: included once, by cudp.c. */
+
+static int parts;
+
+static int fail(const char *what, long a, long b) {
+ printf("[C] cudp FAIL: %s (%ld, %ld)\n", what, a, b);
+ fflush(stdout);
+ return 1;
+}
+
+static void ok(const char *part, const char *detail, long n) {
+ parts++;
+ printf("[C] cudp %s ok: %s %ld\n", part, detail, n);
+ fflush(stdout);
+}
+
+/* A datagram socket bound to 127.0.0.1 at a port the kernel picks. */
+static int bound(struct sockaddr_in *at) {
+ int s = socket(AF_INET, SOCK_DGRAM, 0);
+ memset(at, 0, sizeof *at);
+ at->sin_family = AF_INET;
+ at->sin_addr.s_addr = htonl(INADDR_LOOPBACK);
+ socklen_t len = sizeof *at;
+ if (s < 0 || bind(s, (void *)at, sizeof *at) || getsockname(s, (void *)at, &len)) {
+ return -1;
+ }
+ return s;
+}
+
+static int echo(void) {
+ struct sockaddr_in srv, from, back;
+ int s = bound(&srv), c = socket(AF_INET, SOCK_DGRAM, 0);
+ char buf[32];
+ socklen_t flen = sizeof from, blen = sizeof back;
+ if (s < 0 || sendto(c, "ping", 4, 0, (void *)&srv, sizeof srv) != 4) {
+ return fail("echo: send", s, errno);
+ }
+ long got = recvfrom(s, buf, sizeof buf, 0, (void *)&from, &flen);
+ if (got != 4 || from.sin_port == 0 || flen != sizeof from) {
+ return fail("echo: server heard the client and its address", got, ntohs(from.sin_port));
+ }
+ sendto(s, "pong!", 5, 0, (void *)&from, flen);
+ got = recvfrom(c, buf, sizeof buf, 0, (void *)&back, &blen);
+ close(s);
+ close(c);
+ if (got != 5 || back.sin_port != srv.sin_port || memcmp(buf, "pong!", 5)) {
+ return fail("echo: client heard the server", got, ntohs(back.sin_port));
+ }
+ ok("echo", "4 bytes there, back from the server's port", ntohs(back.sin_port) > 0);
+ return 0;
+}
diff --git a/userland/linux_guests/c/cudp_2.h b/userland/linux_guests/c/cudp_2.h
new file mode 100644
index 000000000..96d477852
--- /dev/null
+++ b/userland/linux_guests/c/cudp_2.h
@@ -0,0 +1,68 @@
+/* cudp, part 2 of 3: included once, by cudp.c. */
+
+static int connected(void) {
+ struct sockaddr_in a, b, x;
+ int sa = bound(&a), sb = bound(&b), sx = bound(&x);
+ char buf[8];
+ connect(sb, (void *)&a, sizeof a);
+ if (send(sb, "to-a", 4, 0) != 4 || recv(sa, buf, 8, 0) != 4) {
+ return fail("connected: send without an address", 0, errno);
+ }
+ /* sb keeps only a's datagrams: x's is dropped, a's arrives. */
+ sendto(sx, "noise", 5, 0, (void *)&b, sizeof b);
+ sendto(sa, "real", 4, 0, (void *)&b, sizeof b);
+ long got = recv(sb, buf, 8, 0);
+ long more = recv(sb, buf, 8, MSG_DONTWAIT);
+ int e = errno;
+ close(sa);
+ close(sb);
+ close(sx);
+ if (got != 4 || more != -1 || e != EAGAIN) {
+ return fail("connected: only the peer's datagrams kept", got, more);
+ }
+ ok("connected", "a stranger's datagram dropped, peer's bytes", got);
+ return 0;
+}
+
+/* cudp: truncation, a refused port, several messages at once, and no peer. */
+
+static int cut(void) {
+ struct sockaddr_in a;
+ int s = bound(&a), c = socket(AF_INET, SOCK_DGRAM, 0);
+ char big[100], buf[4];
+ memset(big, 'z', sizeof big);
+ sendto(c, big, sizeof big, 0, (void *)&a, sizeof a);
+ long whole = recv(s, buf, sizeof buf, MSG_TRUNC);
+ long rest = recv(s, buf, sizeof buf, MSG_DONTWAIT);
+ sendto(c, big, sizeof big, 0, (void *)&a, sizeof a);
+ struct iovec iov = {buf, sizeof buf};
+ struct msghdr m = {0};
+ m.msg_iov = &iov;
+ m.msg_iovlen = 1;
+ long cut = recvmsg(s, &m, 0);
+ close(s);
+ close(c);
+ if (whole != 100 || rest != -1 || cut != 4 || !(m.msg_flags & MSG_TRUNC)) {
+ return fail("trunc: cut to 4, whole length 100, flag set", whole, cut);
+ }
+ ok("trunc", "MSG_TRUNC gave the whole length", whole);
+ return 0;
+}
+
+/* A port with no socket: one is bound and closed to find it. */
+static int refused(void) {
+ struct sockaddr_in gone;
+ close(bound(&gone));
+ int c = socket(AF_INET, SOCK_DGRAM, 0);
+ char buf[4];
+ connect(c, (void *)&gone, sizeof gone);
+ long sent = send(c, "x", 1, 0);
+ long got = recv(c, buf, sizeof buf, MSG_DONTWAIT);
+ int e = errno;
+ close(c);
+ if (sent != 1 || got != -1 || e != ECONNREFUSED) {
+ return fail("refused: send taken, then ECONNREFUSED", sent, e);
+ }
+ ok("refused", "connected to a closed port, errno", e);
+ return 0;
+}
diff --git a/userland/linux_guests/c/cudp_3.h b/userland/linux_guests/c/cudp_3.h
new file mode 100644
index 000000000..686e9ea28
--- /dev/null
+++ b/userland/linux_guests/c/cudp_3.h
@@ -0,0 +1,41 @@
+/* cudp, part 3 of 3: included once, by cudp.c. */
+
+static int mmsg(void) {
+ struct sockaddr_in a;
+ int s = bound(&a), c = socket(AF_INET, SOCK_DGRAM, 0);
+ connect(c, (void *)&a, sizeof a);
+ char out[3][4] = {"one", "two", "six"}, in[3][8];
+ struct iovec oi[3], ii[3];
+ struct mmsghdr om[3], im[3];
+ memset(om, 0, sizeof om);
+ memset(im, 0, sizeof im);
+ for (int i = 0; i < 3; i++) {
+ oi[i] = (struct iovec){out[i], 3};
+ ii[i] = (struct iovec){in[i], 8};
+ om[i].msg_hdr.msg_iov = &oi[i];
+ om[i].msg_hdr.msg_iovlen = 1;
+ im[i].msg_hdr.msg_iov = &ii[i];
+ im[i].msg_hdr.msg_iovlen = 1;
+ }
+ int sent = sendmmsg(c, om, 3, 0);
+ int got = recvmmsg(s, im, 3, MSG_WAITFORONE, 0);
+ close(s);
+ close(c);
+ if (sent != 3 || got != 3 || im[2].msg_len != 3 || memcmp(in[2], "six", 3)) {
+ return fail("mmsg: three out, three in", sent, got);
+ }
+ ok("mmsg", "sendmmsg and recvmmsg moved", got);
+ return 0;
+}
+
+static int nameless(void) {
+ int c = socket(AF_INET, SOCK_DGRAM, 0);
+ long sent = send(c, "x", 1, 0);
+ int e = errno;
+ close(c);
+ if (sent != -1 || e != EDESTADDRREQ) {
+ return fail("nameless: no peer is EDESTADDRREQ", sent, e);
+ }
+ ok("nameless", "errno", e);
+ return 0;
+}
diff --git a/userland/linux_guests/c/cunix.c b/userland/linux_guests/c/cunix.c
new file mode 100644
index 000000000..c333e06ab
--- /dev/null
+++ b/userland/linux_guests/c/cunix.c
@@ -0,0 +1,44 @@
+/*
+ * Unix sockets with names, as Linux has them: a listener on a path, its
+ * name and its client's, a connect that completes at once; ENOENT for a path
+ * with nothing there and ECONNREFUSED for one with no listener; a path that
+ * stays after its socket closes until it is unlinked; abstract names and the
+ * sender a datagram reports; a connected datagram socket that refuses
+ * strangers; a name bind chooses; and a connection across fork. Each part
+ * prints as it passes and every part runs.
+ */
+
+#define _GNU_SOURCE
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#define PATH "/tmp/cunix.sock"
+#define GONE "/tmp/cunix.none"
+
+#include "cunix_1.h"
+#include "cunix_2.h"
+#include "cunix_3.h"
+#include "cunix_4.h"
+
+int main(void) {
+ int (*const part[])(void) = {path_stream, left_behind, abstract_dgram,
+ connected_dgram, autobind, across_fork};
+ const int count = sizeof part / sizeof part[0];
+ int failed = 0;
+ for (int i = 0; i < count; i++) {
+ failed += part[i]();
+ }
+ if (failed) {
+ printf("[C] cunix FAIL: %d parts failed, %d passed\n", failed, parts);
+ fflush(stdout);
+ return 1;
+ }
+ printf("[C] cunix PASS: %d parts\n", parts);
+ fflush(stdout);
+ return 0;
+}
diff --git a/userland/linux_guests/c/cunix_1.h b/userland/linux_guests/c/cunix_1.h
new file mode 100644
index 000000000..bf03cf563
--- /dev/null
+++ b/userland/linux_guests/c/cunix_1.h
@@ -0,0 +1,27 @@
+/* cunix, part 1 of 4: included once, by cunix.c. */
+
+static int parts;
+
+static int fail(const char *what, long a, long b) {
+ printf("[C] cunix FAIL: %s (%ld, %ld)\n", what, a, b);
+ fflush(stdout);
+ return 1;
+}
+
+static void ok(const char *part, const char *detail, long n) {
+ parts++;
+ printf("[C] cunix %s ok: %s %ld\n", part, detail, n);
+ fflush(stdout);
+}
+
+/* A sockaddr_un for a path, or for an abstract name when `abs` is set. */
+static socklen_t name(struct sockaddr_un *a, const char *p, int abs) {
+ memset(a, 0, sizeof *a);
+ a->sun_family = AF_UNIX;
+ if (abs) {
+ memcpy(a->sun_path + 1, p, strlen(p));
+ return offsetof(struct sockaddr_un, sun_path) + 1 + strlen(p);
+ }
+ strcpy(a->sun_path, p);
+ return offsetof(struct sockaddr_un, sun_path) + strlen(p) + 1;
+}
diff --git a/userland/linux_guests/c/cunix_2.h b/userland/linux_guests/c/cunix_2.h
new file mode 100644
index 000000000..4f6d31a9f
--- /dev/null
+++ b/userland/linux_guests/c/cunix_2.h
@@ -0,0 +1,50 @@
+/* cunix, part 2 of 4: included once, by cunix.c. */
+
+static int path_stream(void) {
+ struct sockaddr_un a, b;
+ socklen_t l = name(&a, PATH, 0), bl = sizeof b;
+ unlink(PATH);
+ int s = socket(AF_UNIX, SOCK_STREAM, 0);
+ if (listen(s, 1) != -1 || errno != EINVAL) {
+ return fail("path_stream: listen before bind is EINVAL", errno, EINVAL);
+ }
+ if (bind(s, (void *)&a, l) || listen(s, 4)) {
+ return fail("path_stream: bind and listen", -1, errno);
+ }
+ int twice = socket(AF_UNIX, SOCK_STREAM, 0);
+ if (bind(twice, (void *)&a, l) != -1 || errno != EADDRINUSE) {
+ return fail("path_stream: a second bind is EADDRINUSE", errno, EADDRINUSE);
+ }
+ close(twice);
+ int c = socket(AF_UNIX, SOCK_STREAM | SOCK_NONBLOCK, 0);
+ if (connect(c, (void *)&a, l)) {
+ return fail("path_stream: a non-blocking connect completes at once", -1, errno);
+ }
+ int t = accept(s, (void *)&b, &bl);
+ if (t < 0 || bl != 2) {
+ return fail("path_stream: accept names an unnamed client with 2 bytes", t, bl);
+ }
+ bl = sizeof b;
+ getsockname(s, (void *)&b, &bl);
+ if (bl != l || strcmp(b.sun_path, PATH)) {
+ return fail("path_stream: getsockname", bl, l);
+ }
+ bl = sizeof b;
+ getpeername(c, (void *)&b, &bl);
+ if (bl != l || strcmp(b.sun_path, PATH)) {
+ return fail("path_stream: the client's peer is the path", bl, l);
+ }
+ char buf[8];
+ if (write(c, "unix", 4) != 4 || read(t, buf, 8) != 4) {
+ return fail("path_stream: bytes", 0, errno);
+ }
+ close(c);
+ long eof = read(t, buf, 8);
+ close(t);
+ close(s);
+ if (eof != 0) {
+ return fail("path_stream: end of file", eof, 0);
+ }
+ ok("path_stream", "bound, named, connected, 4 bytes, eof; name length", l);
+ return 0;
+}
diff --git a/userland/linux_guests/c/cunix_3.h b/userland/linux_guests/c/cunix_3.h
new file mode 100644
index 000000000..4ea494bc5
--- /dev/null
+++ b/userland/linux_guests/c/cunix_3.h
@@ -0,0 +1,66 @@
+/* cunix, part 3 of 4: included once, by cunix.c. */
+
+/* cunix: what a path leaves behind, abstract datagrams, and fork. */
+
+static int left_behind(void) {
+ struct sockaddr_un a, g;
+ socklen_t l = name(&a, PATH, 0), gl = name(&g, GONE, 0);
+ unlink(GONE);
+ int c = socket(AF_UNIX, SOCK_STREAM, 0);
+ int missing = connect(c, (void *)&g, gl) ? errno : 0;
+ int s = socket(AF_UNIX, SOCK_STREAM, 0);
+ bind(s, (void *)&a, l);
+ int unheard = connect(c, (void *)&a, l) ? errno : 0;
+ close(s);
+ int s2 = socket(AF_UNIX, SOCK_STREAM, 0);
+ int rebind = bind(s2, (void *)&a, l) ? errno : 0;
+ int closed = connect(c, (void *)&a, l) ? errno : 0;
+ unlink(PATH);
+ int after = bind(s2, (void *)&a, l) ? errno : 0;
+ close(s2);
+ close(c);
+ unlink(PATH);
+ if (missing != ENOENT || unheard != ECONNREFUSED || rebind != EADDRINUSE ||
+ closed != ECONNREFUSED || after != 0) {
+ printf("[C] cunix left_behind got %d %d %d %d %d\n", missing, unheard, rebind, closed,
+ after);
+ return fail("left_behind: ENOENT, ECONNREFUSED, EADDRINUSE, ECONNREFUSED, 0", 0, 0);
+ }
+ ok("left_behind", "the path stays until unlink; rebind after it", after);
+ return 0;
+}
+
+static int abstract_dgram(void) {
+ struct sockaddr_un x, y, b, g;
+ socklen_t xl = name(&x, "cunix-x", 1), yl = name(&y, "cunix-y", 1), gl = name(&g, GONE, 0);
+ int rx = socket(AF_UNIX, SOCK_DGRAM, 0), tx = socket(AF_UNIX, SOCK_DGRAM, 0);
+ char buf[8];
+ socklen_t bl = sizeof b;
+ if (bind(rx, (void *)&x, xl) || getsockname(rx, (void *)&b, &bl) || bl != xl) {
+ return fail("abstract_dgram: bind and name", bl, xl);
+ }
+ sendto(tx, "a", 1, 0, (void *)&x, xl);
+ bl = sizeof b;
+ long got = recvfrom(rx, buf, 8, 0, (void *)&b, &bl);
+ if (got != 1 || bl != 0) {
+ return fail("abstract_dgram: an unnamed sender reports length 0", got, bl);
+ }
+ bind(tx, (void *)&y, yl);
+ sendto(tx, "b", 1, 0, (void *)&x, xl);
+ bl = sizeof b;
+ recvfrom(rx, buf, 8, 0, (void *)&b, &bl);
+ if (bl != yl || memcmp(b.sun_path, y.sun_path, yl - 2)) {
+ return fail("abstract_dgram: a named sender reports its name", bl, yl);
+ }
+ int missing = sendto(tx, "c", 1, 0, (void *)&g, gl) < 0 ? errno : 0;
+ int lone = socket(AF_UNIX, SOCK_DGRAM, 0);
+ int nopeer = send(lone, "d", 1, 0) < 0 ? errno : 0;
+ close(lone);
+ close(rx);
+ close(tx);
+ if (missing != ENOENT || nopeer != ENOTCONN) {
+ return fail("abstract_dgram: ENOENT, then ENOTCONN", missing, nopeer);
+ }
+ ok("abstract_dgram", "names reported, no peer errno", nopeer);
+ return 0;
+}
diff --git a/userland/linux_guests/c/cunix_4.h b/userland/linux_guests/c/cunix_4.h
new file mode 100644
index 000000000..c68289a96
--- /dev/null
+++ b/userland/linux_guests/c/cunix_4.h
@@ -0,0 +1,70 @@
+/* cunix, part 4 of 4: included once, by cunix.c. */
+
+/* cunix: a connected datagram socket, a chosen name, and fork. */
+
+static int connected_dgram(void) {
+ struct sockaddr_un r, a;
+ socklen_t rl = name(&r, "cunix-r", 1), al = name(&a, "cunix-a", 1);
+ int rs = socket(AF_UNIX, SOCK_DGRAM, 0), as = socket(AF_UNIX, SOCK_DGRAM, 0);
+ int stranger = socket(AF_UNIX, SOCK_DGRAM, 0);
+ bind(rs, (void *)&r, rl);
+ bind(as, (void *)&a, al);
+ /* r talks only to a: a stranger's datagram to r is refused. */
+ connect(rs, (void *)&a, al);
+ int refused = sendto(stranger, "s", 1, 0, (void *)&r, rl) < 0 ? errno : 0;
+ long sent = send(rs, "to-a", 4, 0);
+ char buf[8];
+ long got = recv(as, buf, 8, 0);
+ close(rs);
+ close(as);
+ close(stranger);
+ if (refused != EPERM || sent != 4 || got != 4) {
+ return fail("connected_dgram: EPERM for a stranger, 4 bytes to the peer", refused, got);
+ }
+ ok("connected_dgram", "a stranger refused, errno", refused);
+ return 0;
+}
+
+static int autobind(void) {
+ struct sockaddr_un a, b;
+ int s = socket(AF_UNIX, SOCK_DGRAM, 0);
+ memset(&a, 0, sizeof a);
+ a.sun_family = AF_UNIX;
+ socklen_t bl = sizeof b;
+ int rc = bind(s, (void *)&a, sizeof(sa_family_t));
+ getsockname(s, (void *)&b, &bl);
+ close(s);
+ /* Linux chooses a NUL and five hex digits. */
+ if (rc || bl != 8 || b.sun_path[0] != 0) {
+ return fail("autobind: a NUL and five hex digits", rc, bl);
+ }
+ ok("autobind", "a chosen abstract name, length", bl);
+ return 0;
+}
+
+static int across_fork(void) {
+ struct sockaddr_un a;
+ socklen_t l = name(&a, PATH, 0);
+ unlink(PATH);
+ int s = socket(AF_UNIX, SOCK_STREAM, 0);
+ bind(s, (void *)&a, l);
+ listen(s, 1);
+ pid_t kid = fork();
+ if (kid == 0) {
+ int c = socket(AF_UNIX, SOCK_STREAM, 0);
+ _exit(connect(c, (void *)&a, l) || write(c, "kid", 3) != 3);
+ }
+ int t = accept(s, 0, 0);
+ char buf[8];
+ long got = read(t, buf, 8);
+ int status = -1;
+ waitpid(kid, &status, 0);
+ close(t);
+ close(s);
+ unlink(PATH);
+ if (got != 3 || status != 0) {
+ return fail("across_fork: the child's connection and bytes", got, status);
+ }
+ ok("across_fork", "bytes from the child", got);
+ return 0;
+}
diff --git a/userland/linux_guests/go/http/go.mod b/userland/linux_guests/go/http/go.mod
new file mode 100644
index 000000000..0fe31ea2f
--- /dev/null
+++ b/userland/linux_guests/go/http/go.mod
@@ -0,0 +1,3 @@
+module nonos/guest/http
+
+go 1.24
diff --git a/userland/linux_guests/go/http/main.go b/userland/linux_guests/go/http/main.go
new file mode 100644
index 000000000..ea13e86b2
--- /dev/null
+++ b/userland/linux_guests/go/http/main.go
@@ -0,0 +1,56 @@
+/*
+ * net/http as Linux runs it, inside one guest: a server on 127.0.0.1:0 and a
+ * client of it. Twenty GETs, each answered 200 with the path it asked for,
+ * over one kept-alive connection, which the server's own count of new
+ * connections shows.
+ */
+package main
+
+import (
+ "fmt"
+ "io"
+ "net"
+ "net/http"
+ "os"
+ "sync/atomic"
+)
+
+func main() {
+ ln, err := net.Listen("tcp", "127.0.0.1:0")
+ if err != nil {
+ fmt.Println("[GO] gohttp FAIL: listen:", err)
+ os.Exit(1)
+ }
+ var conns atomic.Int32
+ srv := &http.Server{
+ Handler: http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ io.WriteString(w, "hello "+r.URL.Path)
+ }),
+ ConnState: func(_ net.Conn, s http.ConnState) {
+ if s == http.StateNew {
+ conns.Add(1)
+ }
+ },
+ }
+ go srv.Serve(ln)
+ client := &http.Client{}
+ const gets = 20
+ for i := 0; i < gets; i++ {
+ r, err := client.Get(fmt.Sprintf("http://%s/%d", ln.Addr(), i))
+ if err != nil {
+ fmt.Println("[GO] gohttp FAIL: get", i, err)
+ os.Exit(1)
+ }
+ body, err := io.ReadAll(r.Body)
+ r.Body.Close()
+ if err != nil || r.StatusCode != 200 || string(body) != fmt.Sprintf("hello /%d", i) {
+ fmt.Println("[GO] gohttp FAIL: reply", i, r.StatusCode, string(body), err)
+ os.Exit(1)
+ }
+ }
+ if n := conns.Load(); n != 1 {
+ fmt.Println("[GO] gohttp FAIL: keep-alive: connections", n)
+ os.Exit(1)
+ }
+ fmt.Printf("[GO] gohttp PASS: %d GETs answered 200 over %d connection\n", gets, conns.Load())
+}