From 71680cab7a7ceab41f49da5510ac4f15ca5a00a4 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 20:03:58 +0000 Subject: [PATCH 01/14] linux: socket call numbers and errnos, transcribed from Linux bind, listen, getsockname, getpeername, socketpair, setsockopt, getsockopt, recvmmsg and sendmmsg had no numbers here, and the errnos a socket call answers with (ENOPROTOOPT, EADDRINUSE, EISCONN, EDOM and the rest) had no names, so a call that needed them could not be written. They are now in abi/nr_sock.rs and abi/errno_sock.rs, taken from arch/x86/entry/syscalls/syscall_64.tbl and include/uapi/asm-generic; nr and errno re-export them with the calls that first use them. --- .../capsule_linux/src/linux/abi/errno_sock.rs | 33 +++++++++++++++++++ userland/capsule_linux/src/linux/abi/mod.rs | 2 ++ .../capsule_linux/src/linux/abi/nr_sock.rs | 28 ++++++++++++++++ 3 files changed, 63 insertions(+) create mode 100644 userland/capsule_linux/src/linux/abi/errno_sock.rs create mode 100644 userland/capsule_linux/src/linux/abi/nr_sock.rs diff --git a/userland/capsule_linux/src/linux/abi/errno_sock.rs b/userland/capsule_linux/src/linux/abi/errno_sock.rs new file mode 100644 index 000000000..d06e632ef --- /dev/null +++ b/userland/capsule_linux/src/linux/abi/errno_sock.rs @@ -0,0 +1,33 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Linux errno values the socket calls answer with, from +//! include/uapi/asm-generic/errno-base.h and errno.h. + +pub const EDOM: i64 = 33; +pub const EDESTADDRREQ: i64 = 89; +pub const EMSGSIZE: i64 = 90; +pub const EPROTOTYPE: i64 = 91; +pub const ENOPROTOOPT: i64 = 92; +pub const EPROTONOSUPPORT: i64 = 93; +pub const ESOCKTNOSUPPORT: i64 = 94; +pub const EOPNOTSUPP: i64 = 95; +pub const EADDRINUSE: i64 = 98; +pub const EADDRNOTAVAIL: i64 = 99; +pub const ENETUNREACH: i64 = 101; +pub const EISCONN: i64 = 106; +pub const ECONNABORTED: i64 = 103; +pub const EALREADY: i64 = 114; diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs index 68e04090a..fab5d81af 100644 --- a/userland/capsule_linux/src/linux/abi/mod.rs +++ b/userland/capsule_linux/src/linux/abi/mod.rs @@ -19,8 +19,10 @@ #![allow(dead_code)] pub mod errno; +pub mod errno_sock; pub mod name; pub mod nr; pub mod nr_path; pub mod nr_high; pub mod nr_sched; +pub mod nr_sock; diff --git a/userland/capsule_linux/src/linux/abi/nr_sock.rs b/userland/capsule_linux/src/linux/abi/nr_sock.rs new file mode 100644 index 000000000..3a8a13fba --- /dev/null +++ b/userland/capsule_linux/src/linux/abi/nr_sock.rs @@ -0,0 +1,28 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Linux x86_64 syscall numbers for sockets, from +//! arch/x86/entry/syscalls/syscall_64.tbl. Same contract as `nr`. + +pub const BIND: u64 = 49; +pub const LISTEN: u64 = 50; +pub const GETSOCKNAME: u64 = 51; +pub const GETPEERNAME: u64 = 52; +pub const SOCKETPAIR: u64 = 53; +pub const SETSOCKOPT: u64 = 54; +pub const GETSOCKOPT: u64 = 55; +pub const RECVMMSG: u64 = 299; +pub const SENDMMSG: u64 = 307; From 6f21ae44e8c1650a7ec0796dd52f453bd5a3828b Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 20:03:58 +0000 Subject: [PATCH 02/14] linux: a run takes a 64 MiB heap A run started in the libc default heap of 16 MiB and read the program it runs whole from the store into it. A 6 MB Go program (net/http) outgrew it while it was read: the personality died with "memory allocation of 8388608 bytes failed" before the guest started. A run now asks for 64 MiB, the largest program the personality accepts (source::MAX_IMAGE), and falls back to the default when the kernel will not give it. An install keeps its 320 MiB. --- userland/capsule_linux/src/linux/heap.rs | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/userland/capsule_linux/src/linux/heap.rs b/userland/capsule_linux/src/linux/heap.rs index b669aa944..e61776739 100644 --- a/userland/capsule_linux/src/linux/heap.rs +++ b/userland/capsule_linux/src/linux/heap.rs @@ -21,9 +21,14 @@ use nonos_libc::{heap_init, heap_init_sized, mk_args}; /// An install holds a distribution's index while it resolves a closure. /// Kali's main is 21 MB fetched and 85 MB inflated, parsed into records -/// beside it; Alpine's is a few. A run takes the default. +/// beside it; Alpine's is a few. const INSTALL_HEAP: usize = 320 << 20; +/// A run holds the program it loads, read whole from the store, beside the +/// family's own state. A 6 MB Go program outgrew the 16 MiB default while it +/// was read; this is the most a program may be (`source::MAX_IMAGE`). +const RUN_HEAP: usize = 64 << 20; + pub fn init() { let mut buf = [0u8; 256]; let n = mk_args(buf.as_mut_ptr(), buf.len()); @@ -35,6 +40,8 @@ pub fn init() { // it is read, with that reason, instead of here without one. let line = b"[LINUX] no room for a large index, installing in the default heap\n"; let _ = nonos_libc::mk_debug(line.as_ptr(), line.len()); + } else if heap_init_sized(RUN_HEAP).is_ok() { + return; } let _ = heap_init(); } From 8b55aaecaceb5a551b0ca66e70fb8ff50239a377 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 20:04:16 +0000 Subject: [PATCH 03/14] linux guests: the store can carry only the guests a boot needs Every guest went into the store the vfs loads, and the vfs loads at most 16 MiB. The set on this branch already used 16161792 bytes, so a new guest, or a Go guest of any size, made the store step fail with "nonos-store-pack: ... vfs loads at most 16777216". LINUX_GUEST_STORE_ONLY, when set, names the guests the store carries; every guest is still built, signed and proven. Unset, the store carries them all, as before. --- userland/linux_guests/Guests.mk | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index c42c8e873..90fed03f3 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -34,6 +34,12 @@ $(NONOS_BAKED_TRUST_DIR)/keys/guest_%_publisher_mldsa65.pub: | $(CAPSULE_SIGN_BI # name, service port, reply port[, prebuilt ELF[, guest path]]. The enrolled # copy is named guest_, so its certificate and trailer cannot collide # with a capsule's. The guest path defaults to /bin/. +# Every guest is built, signed and proven, but the store the vfs loads holds +# at most 16 MiB, which the whole set outgrows. LINUX_GUEST_STORE_ONLY, when +# set, names the guests the store carries, for example +# make LINUX_GUEST_STORE_ONLY="busybox csock" LINUX_GUEST_BOOT_ARGS=... \ +# target/qemu-virtio-blk.img.store.stamp +# and unset it carries them all, as before. define LINUX_GUEST CAPSULE_SLUG := linux-guest-$(1) CAPSULE_HANDLE := linux.guest.$(1) @@ -54,10 +60,12 @@ nonos-mk-check-linux-guest-$(1)-keys: \ $(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_ed25519.pub \ $(NONOS_BAKED_TRUST_DIR)/keys/guest_$(1)_publisher_mldsa65.pub LINUX_GUEST_STORE_DEPS += $$(linux-guest-$(1)_ARTIFACTS) $$(linux-guest-$(1)_ATTESTATION) +ifneq ($(if $(LINUX_GUEST_STORE_ONLY),$(filter $(1),$(LINUX_GUEST_STORE_ONLY)),all),) LINUX_GUEST_STORE_ENTRIES += --entry /linux$(or $(5),/bin/$(1))=$$(linux-guest-$(1)_BIN) \ --entry /linux$(or $(5),/bin/$(1)).nonos_id_cert.bin=$$(linux-guest-$(1)_CERT) \ --entry /linux$(or $(5),/bin/$(1)).manifest.bin=$$(linux-guest-$(1)_MANIFEST) \ --entry /linux$(or $(5),/bin/$(1)).zk_trailer.bin=$$(linux-guest-$(1)_ATTESTATION) +endif endef $(eval $(call LINUX_GUEST,suite,4950,4951)) From 41a8bd1cf04f4fa3ad9418656726034f23d4ebd8 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 20:05:51 +0000 Subject: [PATCH 04/14] linux: a guest's sockets are the family's own, on 127.0.0.0/8 A guest could not use a socket as Linux has one. Every AF_INET stream asked net.sockets for a mixnet socket, which the Linux capsule may not call (it holds no Network capability), so socket answered EIO; a datagram socket was always the in-capsule resolver; bind, listen, accept, accept4, getsockname, getpeername, socketpair, sendmmsg and recvmmsg were unserved; sendmsg and recvmsg served only the display socket; and shutdown closed the descriptor, so SHUT_WR ended both directions and a dup'd descriptor lost its socket. Now every socket a guest opens is an entry in one table the personality keeps for the family (net/sock). On 127.0.0.0/8 nothing leaves the capsule: - bind and listen, only on loopback: any other address is EACCES with a [LINUX] refused line naming it, and raw sockets are EPERM; - connect to a listener links the two ends in the caller's own call; a non-blocking one answers EINPROGRESS then POLLOUT; a port with no listener is ECONNREFUSED, or EINPROGRESS then POLLERR with the error kept for SO_ERROR, as Linux's loopback answers; - accept and accept4 hand out the oldest connection, with SOCK_NONBLOCK and SOCK_CLOEXEC, and a listener with one pending reads POLLIN; - bytes wait in the peer's queue: end of file after the peer closes or shuts its side, ECONNRESET when it closed with bytes unread, and after the peer is gone the first send is taken and the next is EPIPE; - shutdown shuts one direction or both of the socket and leaves the descriptor open; socketpair gives two connected Unix sockets; - datagrams go to whichever socket holds the port, cut to the buffer with MSG_TRUNC; a connected socket keeps only its peer's, and learns of a closed port as ECONNREFUSED; sendmsg, recvmsg, sendmmsg and recvmmsg carry iovecs and addresses; - a datagram to port 53 that no family socket holds still becomes the resolver; one to anywhere else outside is ENETUNREACH, said by name. A stream to anywhere outside the family still goes over the mixnet through net.sockets, as before. A socket belongs to the processes that hold a descriptor naming it: a forked child is added (fork_state), close lets go when the process has no other descriptor on it (call/io.rs now hands close the guest and the descriptor), and a process that ends lets go of everything it held (the Holder field at the end of Guest, dropped with the process). abi/nr.rs and abi/errno.rs re-export the socket numbers and errnos added in abi/nr_sock.rs and abi/errno_sock.rs. --- userland/capsule_linux/src/linux/abi/errno.rs | 2 + userland/capsule_linux/src/linux/abi/nr.rs | 1 + userland/capsule_linux/src/linux/call/io.rs | 4 +- .../src/linux/guest/fork_state.rs | 1 + .../capsule_linux/src/linux/guest/handle.rs | 4 + .../src/linux/guest/handle_new.rs | 1 + .../capsule_linux/src/linux/net/accept.rs | 63 +++++++++++++++ userland/capsule_linux/src/linux/net/bind.rs | 62 ++++++++++++++ userland/capsule_linux/src/linux/net/close.rs | 45 +++++++++++ .../capsule_linux/src/linux/net/connect.rs | 59 +++++--------- .../src/linux/net/connect_dgram.rs | 51 ++++++++++++ .../capsule_linux/src/linux/net/connect_lo.rs | 57 +++++++++++++ .../src/linux/net/connect_out.rs | 78 ++++++++++++++++++ userland/capsule_linux/src/linux/net/dgram.rs | 74 ++++++++++++----- .../capsule_linux/src/linux/net/dns/mod.rs | 2 - userland/capsule_linux/src/linux/net/fd.rs | 60 ++++++++++++++ .../src/linux/net/{dns/open.rs => flags.rs} | 17 ++-- userland/capsule_linux/src/linux/net/iov.rs | 75 +++++++++++++++++ .../capsule_linux/src/linux/net/listen.rs | 52 ++++++++++++ userland/capsule_linux/src/linux/net/mmsg.rs | 77 ++++++++++++++++++ userland/capsule_linux/src/linux/net/mod.rs | 42 +++++++++- userland/capsule_linux/src/linux/net/msg.rs | 79 ++++++++++++++++++ .../capsule_linux/src/linux/net/msg_hdr.rs | 46 +++++++++++ userland/capsule_linux/src/linux/net/name.rs | 53 ++++++++++++ userland/capsule_linux/src/linux/net/pair.rs | 70 ++++++++++++++++ .../capsule_linux/src/linux/net/peer_addr.rs | 55 +++++++++++++ .../capsule_linux/src/linux/net/policy.rs | 57 +++++++++++++ .../src/linux/net/poll_socket.rs | 17 +++- .../capsule_linux/src/linux/net/resolver.rs | 64 +++++++++++++++ .../capsule_linux/src/linux/net/shutdown.rs | 73 +++++++++++++++++ .../capsule_linux/src/linux/net/sock/bind.rs | 42 ++++++++++ .../capsule_linux/src/linux/net/sock/cell.rs | 37 +++++++++ .../capsule_linux/src/linux/net/sock/free.rs | 57 +++++++++++++ .../capsule_linux/src/linux/net/sock/gram.rs | 68 ++++++++++++++++ .../src/linux/net/sock/gram_in.rs | 57 +++++++++++++ .../src/linux/net/sock/holders.rs | 62 ++++++++++++++ .../capsule_linux/src/linux/net/sock/link.rs | 80 +++++++++++++++++++ .../capsule_linux/src/linux/net/sock/mod.rs | 46 +++++++++++ .../capsule_linux/src/linux/net/sock/new.rs | 50 ++++++++++++ .../src/linux/net/{addr.rs => sock/opts.rs} | 33 ++++---- .../capsule_linux/src/linux/net/sock/port.rs | 67 ++++++++++++++++ .../capsule_linux/src/linux/net/sock/ready.rs | 67 ++++++++++++++++ .../capsule_linux/src/linux/net/sock/recv.rs | 49 ++++++++++++ .../capsule_linux/src/linux/net/sock/send.rs | 55 +++++++++++++ .../capsule_linux/src/linux/net/sock/table.rs | 62 ++++++++++++++ .../capsule_linux/src/linux/net/sock/types.rs | 73 +++++++++++++++++ .../capsule_linux/src/linux/net/sockaddr.rs | 51 ++++++++++++ .../src/linux/net/sockaddr_out.rs | 58 ++++++++++++++ .../capsule_linux/src/linux/net/socket.rs | 64 +++++++-------- .../capsule_linux/src/linux/net/stream.rs | 39 ++++----- .../capsule_linux/src/linux/net/xfer_in.rs | 79 ++++++++++++++++++ .../capsule_linux/src/linux/net/xfer_out.rs | 69 ++++++++++++++++ .../src/linux/serve/table_net.rs | 27 +++++-- 53 files changed, 2470 insertions(+), 163 deletions(-) create mode 100644 userland/capsule_linux/src/linux/net/accept.rs create mode 100644 userland/capsule_linux/src/linux/net/bind.rs create mode 100644 userland/capsule_linux/src/linux/net/close.rs create mode 100644 userland/capsule_linux/src/linux/net/connect_dgram.rs create mode 100644 userland/capsule_linux/src/linux/net/connect_lo.rs create mode 100644 userland/capsule_linux/src/linux/net/connect_out.rs create mode 100644 userland/capsule_linux/src/linux/net/fd.rs rename userland/capsule_linux/src/linux/net/{dns/open.rs => flags.rs} (62%) create mode 100644 userland/capsule_linux/src/linux/net/iov.rs create mode 100644 userland/capsule_linux/src/linux/net/listen.rs create mode 100644 userland/capsule_linux/src/linux/net/mmsg.rs create mode 100644 userland/capsule_linux/src/linux/net/msg.rs create mode 100644 userland/capsule_linux/src/linux/net/msg_hdr.rs create mode 100644 userland/capsule_linux/src/linux/net/name.rs create mode 100644 userland/capsule_linux/src/linux/net/pair.rs create mode 100644 userland/capsule_linux/src/linux/net/peer_addr.rs create mode 100644 userland/capsule_linux/src/linux/net/policy.rs create mode 100644 userland/capsule_linux/src/linux/net/resolver.rs create mode 100644 userland/capsule_linux/src/linux/net/shutdown.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/bind.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/cell.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/free.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/gram.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/gram_in.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/holders.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/link.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/mod.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/new.rs rename userland/capsule_linux/src/linux/net/{addr.rs => sock/opts.rs} (51%) create mode 100644 userland/capsule_linux/src/linux/net/sock/port.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/ready.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/recv.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/send.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/table.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/types.rs create mode 100644 userland/capsule_linux/src/linux/net/sockaddr.rs create mode 100644 userland/capsule_linux/src/linux/net/sockaddr_out.rs create mode 100644 userland/capsule_linux/src/linux/net/xfer_in.rs create mode 100644 userland/capsule_linux/src/linux/net/xfer_out.rs diff --git a/userland/capsule_linux/src/linux/abi/errno.rs b/userland/capsule_linux/src/linux/abi/errno.rs index 448c49fa7..188d8ca83 100644 --- a/userland/capsule_linux/src/linux/abi/errno.rs +++ b/userland/capsule_linux/src/linux/abi/errno.rs @@ -16,6 +16,8 @@ //! Linux errno values, and the convention for returning them. +pub use super::errno_sock::*; + pub const EPERM: i64 = 1; pub const ENOENT: i64 = 2; pub const EINTR: i64 = 4; diff --git a/userland/capsule_linux/src/linux/abi/nr.rs b/userland/capsule_linux/src/linux/abi/nr.rs index 1e800243b..0bfdfac31 100644 --- a/userland/capsule_linux/src/linux/abi/nr.rs +++ b/userland/capsule_linux/src/linux/abi/nr.rs @@ -19,6 +19,7 @@ pub use super::nr_high::*; pub use super::nr_sched::*; +pub use super::nr_sock::*; pub const READ: u64 = 0; pub const WRITE: u64 = 1; diff --git a/userland/capsule_linux/src/linux/call/io.rs b/userland/capsule_linux/src/linux/call/io.rs index f5145519b..07d8a2725 100644 --- a/userland/capsule_linux/src/linux/call/io.rs +++ b/userland/capsule_linux/src/linux/call/io.rs @@ -59,8 +59,6 @@ pub fn read(guest: &mut Guest, fd: u64, buf: u64, len: u64) -> u64 { } pub fn close(guest: &mut Guest, fd: u64) -> u64 { - if let Some(h) = guest.socket_handle(fd) { - net::close(h); - } + net::close(guest, fd); file::close(guest, fd) } diff --git a/userland/capsule_linux/src/linux/guest/fork_state.rs b/userland/capsule_linux/src/linux/guest/fork_state.rs index 54b26502a..39074ba25 100644 --- a/userland/capsule_linux/src/linux/guest/fork_state.rs +++ b/userland/capsule_linux/src/linux/guest/fork_state.rs @@ -46,6 +46,7 @@ impl Guest { g.sid = self.sid; g.umask = self.umask; g.links = self.links.clone(); + self.sockets.fork(child, &self.fds); g } } diff --git a/userland/capsule_linux/src/linux/guest/handle.rs b/userland/capsule_linux/src/linux/guest/handle.rs index 1f12ca784..4cfc902bd 100644 --- a/userland/capsule_linux/src/linux/guest/handle.rs +++ b/userland/capsule_linux/src/linux/guest/handle.rs @@ -86,4 +86,8 @@ pub struct Guest { pub blocked: Vec, /// The image's symbolic links, read once and shared by the family. pub links: alloc::rc::Rc, + /// The pid this process holds family sockets under. Here, rather than + /// in the socket table, so that dropping a process that ends lets go of + /// what it held, as Linux closes an exiting process's descriptors. + pub sockets: crate::linux::net::sock::Holder, } diff --git a/userland/capsule_linux/src/linux/guest/handle_new.rs b/userland/capsule_linux/src/linux/guest/handle_new.rs index 0d8b798db..103c11717 100644 --- a/userland/capsule_linux/src/linux/guest/handle_new.rs +++ b/userland/capsule_linux/src/linux/guest/handle_new.rs @@ -61,6 +61,7 @@ impl Guest { sleepers: Vec::new(), blocked: Vec::new(), links: Default::default(), + sockets: crate::linux::net::sock::Holder::new(pid), } } } diff --git a/userland/capsule_linux/src/linux/net/accept.rs b/userland/capsule_linux/src/linux/net/accept.rs new file mode 100644 index 000000000..d6d9979c5 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/accept.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `accept` and `accept4`: the oldest connection a listener has queued, as +//! a new descriptor held by the caller alone. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::close::discard; +use super::fd::{install, sock_of, SOCK_CLOEXEC, SOCK_NONBLOCK}; +use super::sock::{self, Domain, Proto}; + +pub fn accept4(guest: &mut Guest, fd: u64, at: u64, lenp: u64, flags: u64) -> u64 { + if flags & !(SOCK_NONBLOCK | SOCK_CLOEXEC) != 0 { + return errno::fail(errno::EINVAL); + } + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let pid = guest.pid; + let taken = sock::with(|t| { + let s = t.get_mut(id).ok_or(errno::EBADF)?; + if s.proto == Proto::Dgram { + return Err(errno::EOPNOTSUPP); + } + if !s.listening { + return Err(errno::EINVAL); + } + let child = s.pending.pop_front().ok_or(errno::EAGAIN)?; + let c = t.get_mut(child).ok_or(errno::ECONNABORTED)?; + c.holders.push(pid); + Ok((child, c.remote.unwrap_or_default())) + }); + let (child, from) = match taken { + Ok(v) => v, + Err(e) => return errno::fail(e), + }; + let n = install(guest, child, flags); + let Some(slot) = errno::slot(n) else { + return n; + }; + let wrote = super::sockaddr_out::write(guest, at, lenp, Domain::Inet, from); + if errno::slot(wrote).is_none() { + discard(guest, slot as u64); + return wrote; + } + n +} diff --git a/userland/capsule_linux/src/linux/net/bind.rs b/userland/capsule_linux/src/linux/net/bind.rs new file mode 100644 index 000000000..9b6dbb399 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/bind.rs @@ -0,0 +1,62 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `bind` and `listen`, on 127.0.0.0/8 only (`policy`). + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::policy::not_loopback; +use super::sock::{self, Domain}; +use super::sockaddr::{self, is_loopback, AF_INET}; + +pub fn bind(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let (family, mut want) = match sockaddr::read(guest, at, len) { + Ok(v) => v, + Err(e) => return e, + }; + if family != AF_INET { + return errno::fail(errno::EAFNOSUPPORT); + } + if !is_loopback(want.ip) { + return not_loopback("bind", want); + } + sock::with(|t| { + let Some(s) = t.get(id) else { + return errno::fail(errno::EBADF); + }; + if s.domain != Domain::Inet || s.local.is_some() || s.svc.is_some() { + return errno::fail(errno::EINVAL); + } + if want.port == 0 { + match t.ephemeral(s.proto, want.ip) { + Some(port) => want.port = port, + None => return errno::fail(errno::EADDRINUSE), + } + } else if t.in_use(id, want) { + return errno::fail(errno::EADDRINUSE); + } + if let Some(s) = t.get_mut(id) { + s.local = Some(want); + } + errno::ok(0) + }) +} diff --git a/userland/capsule_linux/src/linux/net/close.rs b/userland/capsule_linux/src/linux/net/close.rs new file mode 100644 index 000000000..0dbf62ee9 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/close.rs @@ -0,0 +1,45 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Closing a socket descriptor. The socket goes with the last descriptor +//! naming it in the last process holding it. + +use crate::linux::guest::{Guest, Kind}; + +use super::fd::sock_of; +use super::sock; + +/// Called by close before it clears `fd`: this process lets go of the +/// socket unless another of its descriptors still names it. +pub fn close(guest: &mut Guest, fd: u64) { + let Ok(id) = sock_of(guest, fd) else { + return; + }; + let named_again = guest + .fds + .iter() + .enumerate() + .any(|(i, f)| i as u64 != fd && f.kind == Kind::Socket && f.handle == id); + if !named_again { + sock::with(|t| t.release(id, guest.pid)); + } +} + +/// Close a descriptor this module opened and cannot hand out after all. +pub(super) fn discard(guest: &mut Guest, fd: u64) { + close(guest, fd); + let _ = crate::linux::file::close(guest, fd); +} diff --git a/userland/capsule_linux/src/linux/net/connect.rs b/userland/capsule_linux/src/linux/net/connect.rs index 51f727956..77a26e4f6 100644 --- a/userland/capsule_linux/src/linux/net/connect.rs +++ b/userland/capsule_linux/src/linux/net/connect.rs @@ -14,54 +14,39 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `connect`, for a socket the guest opened here. - -use alloc::vec::Vec; +//! `connect`. On 127.0.0.0/8 the family links the two ends in the caller's +//! own call, as Linux's loopback does, and a non-blocking socket answers +//! EINPROGRESS all the same; anywhere else a stream goes over the mixnet. use crate::linux::abi::errno; use crate::linux::guest::{Guest, Kind}; -use super::addr::inet; -use super::call::call; -use super::dns::host_for; -use super::ops::{OP_CONNECT, OP_CONNECT_HOST}; +use super::fd::{nonblock, sock_of}; +use super::sock::{self, Domain, Proto}; +use super::sockaddr::{self, is_loopback, AF_INET}; pub fn connect(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { - /* - * A program may connect its nameserver socket before writing to - * it. There is nothing to reach: this capsule is the nameserver. - */ + // This capsule is the nameserver, so its socket has nothing to reach. if guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Resolver) { return errno::ok(0); } - let Some(handle) = guest.socket_handle(fd) else { - return errno::fail(errno::ENOTSOCK); + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, }; - let Some((port, ip)) = inet(guest, at, len) else { - return errno::fail(errno::EAFNOSUPPORT); + let (family, to) = match sockaddr::read(guest, at, len) { + Ok(v) => v, + Err(e) => return e, }; - // An address this capsule invented for a name goes back to being the name. - if let Some(host) = host_for(guest, ip) { - return by_host(handle, &host, port); - } - let mut body = Vec::with_capacity(10); - body.extend_from_slice(&handle.to_le_bytes()); - body.extend_from_slice(&ip); - body.extend_from_slice(&port.to_le_bytes()); - match call(OP_CONNECT, &body, 0) { - Some((0, _)) => errno::ok(0), - Some(_) => errno::fail(errno::ECONNREFUSED), - None => errno::fail(errno::EIO), - } -} - -fn by_host(handle: u32, host: &[u8], port: u16) -> u64 { - let Some(body) = super::host_body::host_body(handle, port, host) else { - return errno::fail(errno::EINVAL); + let Some((proto, domain)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.domain))) else { + return errno::fail(errno::EBADF); }; - match call(OP_CONNECT_HOST, &body, 0) { - Some((0, _)) => errno::ok(0), - Some(_) => errno::fail(errno::ECONNREFUSED), - None => errno::fail(errno::EIO), + match (proto, domain) { + // A socketpair end is connected from the start. + (_, Domain::Unix) => errno::fail(errno::EISCONN), + (Proto::Dgram, _) => super::connect_dgram::connect(guest, fd, id, family, to), + _ if family != AF_INET => errno::fail(errno::EAFNOSUPPORT), + _ if is_loopback(to.ip) => super::connect_lo::loopback(id, to, nonblock(guest, fd)), + _ => super::connect_out::connect(guest, id, to), } } diff --git a/userland/capsule_linux/src/linux/net/connect_dgram.rs b/userland/capsule_linux/src/linux/net/connect_dgram.rs new file mode 100644 index 000000000..ee74eb7f9 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/connect_dgram.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `connect` on a datagram socket: it only names where sends go and whose +//! datagrams are kept. AF_UNSPEC forgets it again. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::policy::refuse_out; +use super::sock::{self, Addr}; +use super::sockaddr::{is_loopback, AF_INET, AF_UNSPEC}; + +pub fn connect(guest: &mut Guest, fd: u64, id: u32, family: u16, to: Addr) -> u64 { + if family == AF_UNSPEC { + sock::with(|t| t.get_mut(id).map(|s| s.remote = None)); + return errno::ok(0); + } + if family != AF_INET { + return errno::fail(errno::EAFNOSUPPORT); + } + if super::resolver::is_nameserver(to) { + super::resolver::become_resolver(guest, fd); + return errno::ok(0); + } + if !is_loopback(to.ip) { + return refuse_out("connect", to); + } + sock::with(|t| { + t.autobind(id)?; + if let Some(s) = t.get_mut(id) { + s.remote = Some(to); + s.error = 0; + } + Ok(()) + }) + .map_or_else(errno::fail, |()| errno::ok(0)) +} diff --git a/userland/capsule_linux/src/linux/net/connect_lo.rs b/userland/capsule_linux/src/linux/net/connect_lo.rs new file mode 100644 index 000000000..45d937e5f --- /dev/null +++ b/userland/capsule_linux/src/linux/net/connect_lo.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A stream connect to 127.0.0.0/8, which the family completes in the +//! caller's own call, as Linux's loopback does. + +use crate::linux::abi::errno; + +use super::sock::{self, Addr, Link}; + +pub fn loopback(id: u32, to: Addr, nonblock: bool) -> u64 { + sock::with(|t| { + let Some(s) = t.get_mut(id) else { + return errno::fail(errno::EBADF); + }; + if s.connected || s.listening || s.svc.is_some() { + return errno::fail(errno::EISCONN); + } + s.error = 0; + let bound_here = s.local.is_none(); + if let Err(e) = t.autobind(id) { + return errno::fail(e); + } + let linked = t.link(id, to); + // A connect that fails gives back the port it bound. + if !matches!(linked, Link::Done) && bound_here { + if let Some(s) = t.get_mut(id) { + s.local = None; + } + } + match linked { + Link::Done if nonblock => errno::fail(errno::EINPROGRESS), + Link::Done => errno::ok(0), + Link::Refused if nonblock => { + if let Some(s) = t.get_mut(id) { + s.error = errno::ECONNREFUSED; + } + errno::fail(errno::EINPROGRESS) + } + Link::Refused => errno::fail(errno::ECONNREFUSED), + Link::Full => errno::fail(errno::EAGAIN), + } + }) +} diff --git a/userland/capsule_linux/src/linux/net/connect_out.rs b/userland/capsule_linux/src/linux/net/connect_out.rs new file mode 100644 index 000000000..36b0b2b5e --- /dev/null +++ b/userland/capsule_linux/src/linux/net/connect_out.rs @@ -0,0 +1,78 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A stream to an address outside the family. It goes over the mixnet, +//! never the open network, and the guest holds no capability that could +//! name a socket: there is no second route to disable and no firewall rule +//! to remove. net.sockets holds the stream; the family's entry names it. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::call::call; +use super::dns::host_for; +use super::ops::{DOMAIN, KIND_MIXNET, OP_CONNECT, OP_CONNECT_HOST, OP_SOCKET}; +use super::sock::{self, Addr}; + +pub fn connect(guest: &Guest, id: u32, to: Addr) -> u64 { + if sock::with(|t| t.get(id).is_some_and(|s| s.connected || s.listening || s.svc.is_some())) { + return errno::fail(errno::EISCONN); + } + let mut body = Vec::with_capacity(4); + body.extend_from_slice(&DOMAIN.to_le_bytes()); + body.extend_from_slice(&KIND_MIXNET.to_le_bytes()); + let handle = match call(OP_SOCKET, &body, 8) { + Some((0, out)) if out.len() >= 4 => u32::from_le_bytes([out[0], out[1], out[2], out[3]]), + Some(_) => return errno::fail(errno::ENOMEM), + None => return errno::fail(errno::EIO), + }; + // An address this capsule invented for a name goes back to being the name. + let status = match host_for(guest, to.ip) { + Some(host) => match super::host_body::host_body(handle, to.port, &host) { + Some(body) => call(OP_CONNECT_HOST, &body, 0), + None => { + super::stream::close(handle); + return errno::fail(errno::EINVAL); + } + }, + None => { + let mut body = Vec::with_capacity(10); + body.extend_from_slice(&handle.to_le_bytes()); + body.extend_from_slice(&to.ip); + body.extend_from_slice(&to.port.to_le_bytes()); + call(OP_CONNECT, &body, 0) + } + }; + let answer = match status { + Some((0, _)) => errno::ok(0), + Some(_) => errno::fail(errno::ECONNREFUSED), + None => errno::fail(errno::EIO), + }; + if answer != 0 { + super::stream::close(handle); + return answer; + } + sock::with(|t| { + if let Some(s) = t.get_mut(id) { + s.svc = Some(handle); + s.remote = Some(to); + s.connected = true; + } + }); + answer +} diff --git a/userland/capsule_linux/src/linux/net/dgram.rs b/userland/capsule_linux/src/linux/net/dgram.rs index dd3b31bd5..7eeafc8b1 100644 --- a/userland/capsule_linux/src/linux/net/dgram.rs +++ b/userland/capsule_linux/src/linux/net/dgram.rs @@ -14,36 +14,68 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `sendto` and `recvfrom`, which differ from write and read only in carrying -//! an address. +//! `sendto` and `recvfrom`, which differ from write and read only in +//! carrying an address. + +use alloc::vec; -use crate::linux::call; use crate::linux::guest::{Guest, Kind}; -use super::addr::inet; -use super::dgram_addr::{encode, fill}; -use super::dns; +use super::fd::sock_of; +use super::sock::{self, Domain, Proto}; +use super::sockaddr::is_loopback; -pub fn sendto(guest: &mut Guest, fd: u64, buf: u64, len: u64, at: u64, alen: u64) -> u64 { +pub fn sendto( + guest: &mut Guest, + fd: u64, + buf: u64, + len: u64, + flags: u64, + at: u64, + alen: u64, +) -> u64 { + let to = match super::peer_addr::address(guest, at, alen) { + Ok(to) => to, + Err(e) => return e, + }; if !is_resolver(guest, fd) { - return call::write(guest, fd, buf, len); + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let inet_dgram = sock::with(|t| { + t.get(id).is_some_and(|s| s.proto == Proto::Dgram && s.domain == Domain::Inet) + }); + match to.filter(|_| inet_dgram) { + Some(to) if super::resolver::is_nameserver(to) => { + super::resolver::become_resolver(guest, fd) + } + Some(to) if !is_loopback(to.ip) => return super::policy::refuse_out("sendto", to), + _ => return super::xfer_out::send(guest, id, &vec![(buf, len)], 0, flags, to), + } } - /* - * A program with no `resolv.conf` asks the loopback address, and one with - * a configured nameserver asks that. - */ - let peer = inet(guest, at, alen).unwrap_or((53, [127, 0, 0, 1])); - dns::query(guest, fd, buf, len, encode(peer)) + super::resolver::query(guest, fd, buf, len, to) } -pub fn recvfrom(guest: &mut Guest, fd: u64, buf: u64, len: u64, at: u64, alen: u64) -> u64 { - if !is_resolver(guest, fd) { - return call::read(guest, fd, buf, len); +pub fn recvfrom( + guest: &mut Guest, + fd: u64, + buf: u64, + len: u64, + flags: u64, + at: u64, + alen: u64, +) -> u64 { + if is_resolver(guest, fd) { + return super::resolver::answer(guest, fd, buf, len, at, alen); } - let (got, from) = dns::answer_out(guest, fd, buf, len); - match from { - Some(peer) if at != 0 => fill(guest, at, alen, peer, got), - _ => got, + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + match super::xfer_in::recv(guest, id, &vec![(buf, len)], 0, flags) { + Ok(got) => super::peer_addr::finish(guest, got, flags, at, alen), + Err(e) => e, } } diff --git a/userland/capsule_linux/src/linux/net/dns/mod.rs b/userland/capsule_linux/src/linux/net/dns/mod.rs index 23b9adad3..c530ba42e 100644 --- a/userland/capsule_linux/src/linux/net/dns/mod.rs +++ b/userland/capsule_linux/src/linux/net/dns/mod.rs @@ -19,10 +19,8 @@ mod automap; mod decide; mod name; -mod open; mod reply; mod serve; pub use automap::host_for; -pub use open::open; pub use serve::{answer_out, query}; diff --git a/userland/capsule_linux/src/linux/net/fd.rs b/userland/capsule_linux/src/linux/net/fd.rs new file mode 100644 index 000000000..53bf72c5a --- /dev/null +++ b/userland/capsule_linux/src/linux/net/fd.rs @@ -0,0 +1,60 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! From a descriptor to the family socket it names, and back. + +use crate::linux::abi::errno; +use crate::linux::guest::{Fd, Guest, Kind}; + +use super::sock; + +/// SOCK_NONBLOCK and SOCK_CLOEXEC, which socket, socketpair and accept4 +/// take with the type or as flags: O_NONBLOCK and O_CLOEXEC's values. +pub const SOCK_NONBLOCK: u64 = 0o4000; +pub const SOCK_CLOEXEC: u64 = 0o2_000_000; + +/// The socket `fd` names, or the errno Linux gives for a descriptor that is +/// not one: EBADF when nothing is open there, ENOTSOCK when a file is. +pub fn sock_of(guest: &Guest, fd: u64) -> Result { + match guest.fds.get(fd as usize) { + Some(f) if f.kind == Kind::Socket => Ok(f.handle), + Some(f) if f.is_open() => Err(errno::fail(errno::ENOTSOCK)), + _ => Err(errno::fail(errno::EBADF)), + } +} + +/// A descriptor for socket `id`, with the flags asked for. A socket with no +/// descriptor to name it is let go at once, since nobody could close it. +pub fn install(guest: &mut Guest, id: u32, flags: u64) -> u64 { + match crate::linux::file::install(guest, Fd::socket(id)) { + Some(n) => { + if let Some(f) = guest.fds.get_mut(n as usize) { + f.nonblock = flags & SOCK_NONBLOCK != 0; + f.cloexec = flags & SOCK_CLOEXEC != 0; + } + errno::ok(n) + } + None => { + sock::with(|t| t.release(id, guest.pid)); + errno::fail(errno::EMFILE) + } + } +} + +/// True when `fd` is non-blocking. +pub fn nonblock(guest: &Guest, fd: u64) -> bool { + guest.fds.get(fd as usize).is_some_and(|f| f.nonblock) +} diff --git a/userland/capsule_linux/src/linux/net/dns/open.rs b/userland/capsule_linux/src/linux/net/flags.rs similarity index 62% rename from userland/capsule_linux/src/linux/net/dns/open.rs rename to userland/capsule_linux/src/linux/net/flags.rs index fb3bd77be..5084c12ed 100644 --- a/userland/capsule_linux/src/linux/net/dns/open.rs +++ b/userland/capsule_linux/src/linux/net/flags.rs @@ -14,16 +14,9 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! The descriptor a program opens to reach its nameserver. +//! The flags the send and receive calls take, from include/linux/socket.h. -use crate::linux::abi::errno; -use crate::linux::guest::{Fd, Guest}; - -/// It looks like a datagram socket and holds no handle: there is -/// nothing on the other side of it, which is the point. -pub fn open(guest: &mut Guest) -> u64 { - match crate::linux::file::install(guest, Fd::resolver()) { - Some(n) => errno::ok(n), - None => errno::fail(errno::EMFILE), - } -} +pub const MSG_OOB: u64 = 0x1; +pub const MSG_PEEK: u64 = 0x2; +pub const MSG_TRUNC: u64 = 0x20; +pub const MSG_DONTWAIT: u64 = 0x40; diff --git a/userland/capsule_linux/src/linux/net/iov.rs b/userland/capsule_linux/src/linux/net/iov.rs new file mode 100644 index 000000000..922dff54b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/iov.rs @@ -0,0 +1,75 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The byte vectors the message calls name: one buffer, or an iovec array. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +/// Linux's UIO_MAXIOV. +const IOV_MAX: u64 = 1024; +const IOVEC: usize = 16; + +pub type Iov = Vec<(u64, u64)>; + +/// The `count` entries of the iovec array at `at`. +pub fn read(guest: &Guest, at: u64, count: u64) -> Result { + if count > IOV_MAX { + return Err(errno::fail(errno::EINVAL)); + } + let raw = guest.read(at, count as usize * IOVEC).ok_or(errno::fail(errno::EFAULT))?; + let word = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + Ok((0..count as usize).map(|i| (word(i * IOVEC), word(i * IOVEC + 8))).collect()) +} + +pub fn total(iov: &Iov) -> usize { + iov.iter().map(|&(_, len)| len as usize).sum() +} + +/// The message's bytes from `skip` on, at most `cap` of them. +pub fn gather(guest: &Guest, iov: &Iov, skip: usize, cap: usize) -> Result, u64> { + let mut out = Vec::new(); + let mut pos = 0usize; + for &(base, len) in iov { + let len = len as usize; + let (from, to) = (skip.max(pos), (pos + len).min(skip + cap)); + if from < to { + let part = guest.read(base + (from - pos) as u64, to - from); + out.extend_from_slice(&part.ok_or(errno::fail(errno::EFAULT))?); + } + pos += len; + } + Ok(out) +} + +/// Put `bytes` into the message's buffers starting `skip` bytes in. +pub fn scatter(guest: &Guest, iov: &Iov, skip: usize, bytes: &[u8]) -> Result<(), u64> { + let mut pos = 0usize; + for &(base, len) in iov { + let len = len as usize; + let (from, to) = (skip.max(pos), (pos + len).min(skip + bytes.len())); + if from < to { + let part = &bytes[from - skip..to - skip]; + if guest.write(base + (from - pos) as u64, part) < part.len() as i64 { + return Err(errno::fail(errno::EFAULT)); + } + } + pos += len; + } + Ok(()) +} diff --git a/userland/capsule_linux/src/linux/net/listen.rs b/userland/capsule_linux/src/linux/net/listen.rs new file mode 100644 index 000000000..adb92fa4b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/listen.rs @@ -0,0 +1,52 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `listen`, on a socket bound to 127.0.0.1 (`policy`). + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::policy::not_loopback; +use super::sock::{self, Addr, Domain, Proto}; + +pub fn listen(guest: &mut Guest, fd: u64, backlog: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + sock::with(|t| { + let Some(s) = t.get(id) else { + return errno::fail(errno::EBADF); + }; + match (s.proto, s.domain, s.local) { + (Proto::Dgram, ..) | (_, Domain::Unix, _) => return errno::fail(errno::EOPNOTSUPP), + _ if s.connected || s.svc.is_some() => return errno::fail(errno::EINVAL), + // Linux would bind 0.0.0.0 here, which is not the family's own. + (_, _, None) => return not_loopback("listen", Addr::default()), + (_, _, Some(at)) if !s.listening && t.listener(at).is_some() => { + return errno::fail(errno::EADDRINUSE) + } + _ => {} + } + if let Some(s) = t.get_mut(id) { + s.listening = true; + // An int, and somaxconn's 4096 is the most Linux keeps. + s.backlog = (backlog as i32).clamp(0, 4096) as usize; + } + errno::ok(0) + }) +} diff --git a/userland/capsule_linux/src/linux/net/mmsg.rs b/userland/capsule_linux/src/linux/net/mmsg.rs new file mode 100644 index 000000000..3fe863d9d --- /dev/null +++ b/userland/capsule_linux/src/linux/net/mmsg.rs @@ -0,0 +1,77 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sendmmsg` and `recvmmsg`: several messages in one call. A blocking +//! recvmmsg waits until all `vlen` have come, unless MSG_WAITFORONE; the +//! wait keeps its count between tries (`waits_sock`), so each try starts at +//! message `skip` and answers how many it moved. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::flags::MSG_DONTWAIT; +use super::policy::refuse; + +/// struct mmsghdr on x86_64: a msghdr, then msg_len. +const MMSGHDR: u64 = 64; +const LEN_AT: u64 = 56; +/// Linux's UIO_MAXIOV caps vlen. +pub const MOST: u64 = 1024; +pub const MSG_WAITFORONE: u64 = 0x10000; + +/// `a` is sendmmsg's arguments: fd, vector, vlen, flags. +pub fn sendmmsg(guest: &mut Guest, a: [u64; 6], skip: usize) -> u64 { + each(guest, a, skip, |g, at, first| { + let f = if first { a[3] } else { a[3] | MSG_DONTWAIT }; + super::msg::sendmsg(g, a[0], at, f, 0) + }) +} + +/// `a` is recvmmsg's arguments: fd, vector, vlen, flags, timeout. +pub fn recvmmsg(guest: &mut Guest, a: [u64; 6], skip: usize) -> u64 { + if a[4] != 0 { + return refuse("recvmmsg timeout: SO_RCVTIMEO bounds the wait instead", errno::EINVAL); + } + let flags = a[3] & !MSG_WAITFORONE; + each(guest, a, skip, |g, at, first| { + let f = if first { flags } else { flags | MSG_DONTWAIT }; + super::msg::recvmsg(g, a[0], at, f, 0) + }) +} + +/// Run `one` on each message from `skip` until one fails: the count moved, +/// or the first failure's errno when none was. +fn each( + guest: &mut Guest, + a: [u64; 6], + skip: usize, + mut one: impl FnMut(&mut Guest, u64, bool) -> u64, +) -> u64 { + let vlen = a[2].min(MOST); + let mut moved = 0u64; + for i in skip as u64..vlen { + let at = a[1] + i * MMSGHDR; + let got = one(guest, at, moved == 0); + let Some(n) = errno::slot(got) else { + return if moved == 0 { got } else { errno::ok(moved) }; + }; + if guest.write(at + LEN_AT, &(n as u32).to_le_bytes()) < 4 { + return if moved == 0 { errno::fail(errno::EFAULT) } else { errno::ok(moved) }; + } + moved += 1; + } + errno::ok(moved) +} diff --git a/userland/capsule_linux/src/linux/net/mod.rs b/userland/capsule_linux/src/linux/net/mod.rs index c0155af09..ebd081702 100644 --- a/userland/capsule_linux/src/linux/net/mod.rs +++ b/userland/capsule_linux/src/linux/net/mod.rs @@ -14,30 +14,64 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Sockets, over the net.sockets service. +//! Sockets: the family's own, kept in `sock`, and a stream outside the +//! family, which net.sockets carries over the mixnet. -mod addr; +mod accept; +mod bind; mod call; +mod close; mod connect; +mod connect_lo; +mod connect_dgram; +mod connect_out; mod dgram; mod dgram_addr; -mod host_body; pub mod dns; +mod fd; +mod flags; +mod host_body; +mod iov; +mod listen; +mod mmsg; +mod msg; +mod msg_hdr; +mod name; mod ops; +mod pair; +mod peer_addr; +mod policy; mod poll; mod poll_set; mod poll_socket; pub mod raw; pub mod raw_io; +mod resolver; pub mod route; mod select; +mod shutdown; +pub mod sock; +mod sockaddr; +mod sockaddr_out; mod socket; mod stream; +mod xfer_in; +mod xfer_out; +pub use accept::accept4; +pub use bind::bind; +pub use listen::listen; +pub use close::close; pub use connect::connect; pub use dgram::{recvfrom, sendto}; +pub use mmsg::{recvmmsg, sendmmsg}; +pub use msg::{recvmsg, sendmsg}; +pub use name::{getpeername, getsockname}; +pub use pair::socketpair; pub use poll::{ready, POLLERR, POLLHUP}; pub use poll_set::poll; pub use select::{clear as select_clear, select}; +pub use shutdown::shutdown; pub use socket::socket; -pub use stream::{close, recv, send}; +pub use xfer_in::read as recv; +pub use xfer_out::write as send; diff --git a/userland/capsule_linux/src/linux/net/msg.rs b/userland/capsule_linux/src/linux/net/msg.rs new file mode 100644 index 000000000..a9037c255 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/msg.rs @@ -0,0 +1,79 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sendmsg` and `recvmsg` on a family socket: an iovec, an address, and +//! control data, which only a Unix socket carries on Linux and none here. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::flags::MSG_TRUNC; +use super::msg_hdr::{hdr, CONTROLLEN_AT, FLAGS_AT, NAMELEN_AT}; +use super::policy::refuse; + +/// `skip` bytes of the message went in an earlier try of this same call. +pub fn sendmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let h = match hdr(guest, msg) { + Ok(h) => h, + Err(e) => return e, + }; + if h.controllen != 0 { + return refuse( + "sendmsg control data on a family socket: nothing here reads it", + errno::EINVAL, + ); + } + let to = match super::peer_addr::address(guest, h.name, h.namelen) { + Ok(to) => to, + Err(e) => return e, + }; + if let Some(to) = to.filter(|a| !super::sockaddr::is_loopback(a.ip)) { + return super::policy::refuse_out("sendmsg", to); + } + super::xfer_out::send(guest, id, &h.iov, skip, flags, to) +} + +pub fn recvmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let h = match hdr(guest, msg) { + Ok(h) => h, + Err(e) => return e, + }; + let got = match super::xfer_in::recv(guest, id, &h.iov, skip, flags) { + Ok(got) => got, + Err(e) => return e, + }; + let cut = if got.whole > got.n { MSG_TRUNC as u32 } else { 0 }; + let (name, lenp) = if h.name != 0 { (h.name, msg + NAMELEN_AT) } else { (0, 0) }; + let value = super::peer_addr::finish(guest, got, flags, name, lenp); + if errno::slot(value).is_none() { + return value; + } + if guest.write(msg + CONTROLLEN_AT, &0u64.to_le_bytes()) < 8 + || guest.write(msg + FLAGS_AT, &cut.to_le_bytes()) < 4 + { + return errno::fail(errno::EFAULT); + } + value +} diff --git a/userland/capsule_linux/src/linux/net/msg_hdr.rs b/userland/capsule_linux/src/linux/net/msg_hdr.rs new file mode 100644 index 000000000..e3f60a9c3 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/msg_hdr.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! struct msghdr, as sendmsg and recvmsg find it in guest memory. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::iov; + +/// struct msghdr on x86_64. +const MSGHDR: usize = 56; +pub const NAMELEN_AT: u64 = 8; +pub const CONTROLLEN_AT: u64 = 40; +pub const FLAGS_AT: u64 = 48; + +pub struct Hdr { + pub name: u64, + pub namelen: u64, + pub iov: iov::Iov, + pub controllen: u64, +} + +pub fn hdr(guest: &Guest, msg: u64) -> Result { + let raw = guest.read(msg, MSGHDR).ok_or(errno::fail(errno::EFAULT))?; + let word = |i: usize| u64::from_le_bytes(raw[i..i + 8].try_into().unwrap_or([0; 8])); + Ok(Hdr { + name: word(0), + namelen: u64::from(u32::from_le_bytes([raw[8], raw[9], raw[10], raw[11]])), + iov: iov::read(guest, word(16), word(24))?, + controllen: word(40), + }) +} diff --git a/userland/capsule_linux/src/linux/net/name.rs b/userland/capsule_linux/src/linux/net/name.rs new file mode 100644 index 000000000..c153acf1d --- /dev/null +++ b/userland/capsule_linux/src/linux/net/name.rs @@ -0,0 +1,53 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `getsockname` and `getpeername`. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::sock::{self, Addr, Domain}; + +pub fn getsockname(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let Some((domain, local)) = sock::with(|t| t.get(id).map(|s| (s.domain, s.local))) else { + return errno::fail(errno::EBADF); + }; + // A socket not yet bound is 0.0.0.0, port 0. + super::sockaddr_out::write(guest, at, lenp, domain, local.unwrap_or_default()) +} + +pub fn getpeername(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + // A reset connection is closed, and has no peer; one whose peer only + // shut down, or left cleanly, still does. + let peer: Option<(Domain, Addr)> = sock::with(|t| { + let s = t.get(id)?; + let live = s.connected && !s.broken && s.error == 0; + live.then(|| (s.domain, s.remote.unwrap_or_default())) + }); + match peer { + Some((domain, addr)) => super::sockaddr_out::write(guest, at, lenp, domain, addr), + None => errno::fail(errno::ENOTCONN), + } +} diff --git a/userland/capsule_linux/src/linux/net/pair.rs b/userland/capsule_linux/src/linux/net/pair.rs new file mode 100644 index 000000000..4fb4dc3ec --- /dev/null +++ b/userland/capsule_linux/src/linux/net/pair.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `socketpair`: two connected Unix sockets, both ends in the family. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::{install, SOCK_CLOEXEC, SOCK_NONBLOCK}; +use super::sock::{self, Domain, Proto}; +use super::sockaddr::{AF_INET, AF_UNIX}; + +const SOCK_STREAM: u64 = 1; +const SOCK_DGRAM: u64 = 2; +const TYPE_MASK: u64 = 0xF; + +pub fn socketpair(guest: &mut Guest, family: u64, kind: u64, protocol: u64, out: u64) -> u64 { + let flags = kind & (SOCK_NONBLOCK | SOCK_CLOEXEC); + if kind & !(TYPE_MASK | flags) != 0 { + return errno::fail(errno::EINVAL); + } + match family { + f if f == u64::from(AF_UNIX) => {} + // Linux has no connected pair for the internet families. + f if f == u64::from(AF_INET) => return errno::fail(errno::EOPNOTSUPP), + _ => return errno::fail(errno::EAFNOSUPPORT), + } + let proto = match kind & TYPE_MASK { + SOCK_STREAM => Proto::Stream, + SOCK_DGRAM => Proto::Dgram, + _ => return errno::fail(errno::ESOCKTNOSUPPORT), + }; + if protocol != 0 { + return errno::fail(errno::EPROTONOSUPPORT); + } + let (a, b) = sock::with(|t| t.pair(Domain::Unix, proto, guest.pid)); + let fa = install(guest, a, flags); + let Some(na) = errno::slot(fa) else { + sock::with(|t| t.release(b, guest.pid)); + return fa; + }; + let fb = install(guest, b, flags); + let Some(nb) = errno::slot(fb) else { + super::close::discard(guest, na as u64); + return fb; + }; + let mut pair = [0u8; 8]; + pair[..4].copy_from_slice(&(na as u32).to_le_bytes()); + pair[4..].copy_from_slice(&(nb as u32).to_le_bytes()); + // Linux copies the pair out before it installs either descriptor. + if guest.write(out, &pair) < 8 { + super::close::discard(guest, na as u64); + super::close::discard(guest, nb as u64); + return errno::fail(errno::EFAULT); + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/net/peer_addr.rs b/userland/capsule_linux/src/linux/net/peer_addr.rs new file mode 100644 index 000000000..d43aa4e33 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/peer_addr.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The addresses the send and receive calls carry: the one a send names, +//! and the sender a receive writes back. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::flags::MSG_TRUNC; +use super::sock::Addr; +use super::sockaddr::{self, AF_INET}; +use super::xfer_in::In; + +/// The count a receive answers, with the sender written out. A stream has +/// no sender, and Linux says so with a length of zero. +pub fn finish(guest: &mut Guest, got: In, flags: u64, at: u64, alen: u64) -> u64 { + if at != 0 { + let wrote = match got.from { + Some((domain, from)) => super::sockaddr_out::write(guest, at, alen, domain, from), + None if alen != 0 && guest.write(alen, &0u32.to_le_bytes()) < 4 => { + errno::fail(errno::EFAULT) + } + None => errno::ok(0), + }; + if errno::slot(wrote).is_none() { + return wrote; + } + } + errno::ok(if flags & MSG_TRUNC != 0 { got.whole } else { got.n } as u64) +} + +/// The address a send names, if any: IPv4 only. +pub fn address(guest: &Guest, at: u64, alen: u64) -> Result, u64> { + if at == 0 { + return Ok(None); + } + match sockaddr::read(guest, at, alen)? { + (AF_INET, a) => Ok(Some(a)), + _ => Err(errno::fail(errno::EAFNOSUPPORT)), + } +} diff --git a/userland/capsule_linux/src/linux/net/policy.rs b/userland/capsule_linux/src/linux/net/policy.rs new file mode 100644 index 000000000..46a1227e7 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/policy.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What this capsule lets a guest's sockets reach. A guest binds and +//! listens only on 127.0.0.0/8, which is the family's own: nothing outside +//! the capsule can connect to it. Anything else is EACCES, said by name. +//! Raw and packet sockets reach the link layer, below anything that could +//! confine them, and are EPERM. + +use alloc::format; + +use crate::linux::abi::errno; +use crate::linux::start::say; + +use super::sock::Addr; + +/// EACCES for a bind or listen outside loopback, with the address. +pub fn not_loopback(call: &str, at: Addr) -> u64 { + let [a, b, c, d] = at.ip; + let line = format!( + "[LINUX] refused {call} {a}.{b}.{c}.{d}:{}: a guest listens only on 127.0.0.0/8\n", + at.port + ); + say(line.as_bytes()); + errno::fail(errno::EACCES) +} + +/// ENETUNREACH for a datagram to anywhere outside the family: the mixnet +/// carries streams, and a guest's datagrams have no other way out. +pub fn refuse_out(call: &str, to: Addr) -> u64 { + let [a, b, c, d] = to.ip; + let line = format!( + "[LINUX] refused {call} {a}.{b}.{c}.{d}:{}: a guest's datagrams stay in the family\n", + to.port + ); + say(line.as_bytes()); + errno::fail(errno::ENETUNREACH) +} + +/// `errno` for a call this capsule declines, with why. +pub fn refuse(what: &str, errno: i64) -> u64 { + say(format!("[LINUX] refused {what}\n").as_bytes()); + errno::fail(errno) +} diff --git a/userland/capsule_linux/src/linux/net/poll_socket.rs b/userland/capsule_linux/src/linux/net/poll_socket.rs index 7d5fe6ef8..00a60e68f 100644 --- a/userland/capsule_linux/src/linux/net/poll_socket.rs +++ b/userland/capsule_linux/src/linux/net/poll_socket.rs @@ -14,15 +14,28 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Asking net.sockets whether one handle is ready. +//! A socket's readiness in poll's bits: the family's own sockets answer from +//! the table, a stream outside the family from net.sockets. use super::call::call; use super::ops::{OP_POLL, POLL_READABLE, POLL_WRITABLE}; +use super::sock; const POLLIN: u16 = 0x001; const POLLOUT: u16 = 0x004; +const POLLNVAL: u16 = 0x020; -pub(super) fn socket_bits(handle: u32) -> u16 { +pub(super) fn socket_bits(id: u32) -> u16 { + if let Some(bits) = sock::bits(id) { + return bits; + } + match sock::with(|t| t.get(id).and_then(|s| s.svc)) { + Some(handle) => service_bits(handle), + None => POLLNVAL, + } +} + +fn service_bits(handle: u32) -> u16 { let Some((0, out)) = call(OP_POLL, &handle.to_le_bytes(), 1) else { return 0; }; diff --git a/userland/capsule_linux/src/linux/net/resolver.rs b/userland/capsule_linux/src/linux/net/resolver.rs new file mode 100644 index 000000000..9ca2e7bb3 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/resolver.rs @@ -0,0 +1,64 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A datagram socket that talks to a nameserver. This capsule answers name +//! queries itself (`dns`), so a query never leaves it: a socket that sends +//! to port 53 outside the family, or to a loopback port 53 no family socket +//! holds, becomes the resolver descriptor it always was before a guest +//! could bind a datagram socket of its own. + +use crate::linux::guest::{Fd, Guest}; + +use super::dgram_addr::{encode, fill}; +use super::dns; + +use super::sock::{self, Addr, Proto}; +use super::sockaddr::is_loopback; + +const NAMESERVER_PORT: u16 = 53; + +pub fn is_nameserver(to: Addr) -> bool { + to.port == NAMESERVER_PORT + && (!is_loopback(to.ip) || sock::with(|t| t.bound(Proto::Dgram, to).is_none())) +} + +/// Let go of the socket `fd` names and make `fd` the resolver, keeping its +/// descriptor flags. +pub fn become_resolver(guest: &mut Guest, fd: u64) { + super::close::close(guest, fd); + if let Some(f) = guest.fds.get_mut(fd as usize) { + let (cloexec, nonblock) = (f.cloexec, f.nonblock); + *f = Fd::resolver(); + f.cloexec = cloexec; + f.nonblock = nonblock; + } +} + +/// A query written to the resolver. A program with no `resolv.conf` asks +/// the loopback address, and one with a configured nameserver asks that. +pub fn query(guest: &mut Guest, fd: u64, buf: u64, len: u64, to: Option) -> u64 { + let peer = to.map_or((NAMESERVER_PORT, [127, 0, 0, 1]), |a| (a.port, a.ip)); + dns::query(guest, fd, buf, len, encode(peer)) +} + +/// An answer read from the resolver, with the nameserver it came from. +pub fn answer(guest: &mut Guest, fd: u64, buf: u64, len: u64, at: u64, alen: u64) -> u64 { + let (got, from) = dns::answer_out(guest, fd, buf, len); + match from { + Some(peer) if at != 0 => fill(guest, at, alen, peer, got), + _ => got, + } +} diff --git a/userland/capsule_linux/src/linux/net/shutdown.rs b/userland/capsule_linux/src/linux/net/shutdown.rs new file mode 100644 index 000000000..64540fd0c --- /dev/null +++ b/userland/capsule_linux/src/linux/net/shutdown.rs @@ -0,0 +1,73 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `shutdown`: one direction or both, on the socket rather than the +//! descriptor, which stays open. A dup'd or inherited descriptor sees it too. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::policy::refuse; +use super::sock::{self, Proto}; + +const SHUT_RD: u64 = 0; +const SHUT_WR: u64 = 1; +const SHUT_RDWR: u64 = 2; + +pub fn shutdown(guest: &Guest, fd: u64, how: u64) -> u64 { + if how > SHUT_RDWR { + return errno::fail(errno::EINVAL); + } + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let (rd, wr) = (how != SHUT_WR, how != SHUT_RD); + sock::with(|t| { + let Some(s) = t.get_mut(id) else { + return errno::fail(errno::EBADF); + }; + if s.svc.is_some() { + return refuse( + "shutdown of a stream outside the family: net.sockets has no half-close", + errno::EOPNOTSUPP, + ); + } + if s.listening { + // Shutting a listener's reading side stops it listening. + if rd { + s.listening = false; + for queued in core::mem::take(&mut s.pending) { + t.free(queued, true); + } + } + return errno::ok(0); + } + let connected = s.connected || (s.proto == Proto::Dgram && s.remote.is_some()); + if !connected { + return errno::fail(errno::ENOTCONN); + } + s.rd_shut |= rd; + s.wr_shut |= wr; + let peer = s.peer; + // The peer reads end of file once it has what was already sent. + if let Some(p) = peer.filter(|_| wr).and_then(|p| t.get_mut(p)) { + p.eof = true; + } + errno::ok(0) + }) +} diff --git a/userland/capsule_linux/src/linux/net/sock/bind.rs b/userland/capsule_linux/src/linux/net/sock/bind.rs new file mode 100644 index 000000000..5b74f73f6 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/bind.rs @@ -0,0 +1,42 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A local address for a socket that sends or connects before it binds. + +use crate::linux::abi::errno::EADDRNOTAVAIL; + +use super::table::Socks; +use super::types::{Addr, Domain}; + +/// The loopback route's source address. +const LOOPBACK: [u8; 4] = [127, 0, 0, 1]; + +impl Socks { + /// Bind `id` to 127.0.0.1 and a free ephemeral port, as Linux's + /// autobind does, unless it is bound already or is a socketpair end, + /// which has no address. + pub fn autobind(&mut self, id: u32) -> Result<(), i64> { + let unbound = |s: &&super::types::Sock| s.local.is_none() && s.domain == Domain::Inet; + let Some(proto) = self.get(id).filter(unbound).map(|s| s.proto) else { + return Ok(()); + }; + let port = self.ephemeral(proto, LOOPBACK).ok_or(EADDRNOTAVAIL)?; + if let Some(s) = self.get_mut(id) { + s.local = Some(Addr { ip: LOOPBACK, port }); + } + Ok(()) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/cell.rs b/userland/capsule_linux/src/linux/net/sock/cell.rs new file mode 100644 index 000000000..f98edb5fc --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/cell.rs @@ -0,0 +1,37 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where the table lives. This personality hosts one family and serves it +//! from one thread, so the table is a single process-wide value: net.sockets +//! keeps its own the same way. + +use core::cell::RefCell; + +use super::table::Socks; + +struct One(RefCell); + +// SAFETY: the serve loop is the only thread in this capsule that reaches the +// table; guest threads run in their own processes and only trap into it. +unsafe impl Sync for One {} + +static TABLE: One = One(RefCell::new(Socks::new())); + +/// Run `f` on the table. A call inside `f` that reaches the table again would +/// panic on the borrow, so no function here calls out while holding it. +pub fn with(f: impl FnOnce(&mut Socks) -> R) -> R { + f(&mut TABLE.0.borrow_mut()) +} diff --git a/userland/capsule_linux/src/linux/net/sock/free.rs b/userland/capsule_linux/src/linux/net/sock/free.rs new file mode 100644 index 000000000..36f3b2cbb --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/free.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Letting a socket go, and what its peer sees when it does. + +use crate::linux::abi::errno::ECONNRESET; + +use super::table::Socks; + +impl Socks { + /// `pid` no longer holds `id`. The socket goes when nobody does. + pub fn release(&mut self, id: u32, pid: u32) { + let Some(s) = self.get_mut(id) else { + return; + }; + let held = s.holders.len(); + s.holders.retain(|&p| p != pid); + if held != 0 && s.holders.is_empty() { + self.free(id, false); + } + } + + /// Close `id`. Its peer reads end of file, or ECONNRESET if this end + /// left bytes unread or `reset` is set, which is when Linux sends a reset + /// instead of a FIN. Connections still queued on a listener are reset. + pub fn free(&mut self, id: u32, reset: bool) { + let Some(gone) = self.list.get_mut(id as usize).and_then(Option::take) else { + return; + }; + if let Some(h) = gone.svc { + super::super::stream::close(h); + } + if let Some(p) = gone.peer.and_then(|p| self.get_mut(p)) { + p.peer = None; + p.eof = true; + if reset || !gone.rx.is_empty() { + p.error = ECONNRESET; + } + } + for queued in gone.pending { + self.free(queued, true); + } + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/gram.rs b/userland/capsule_linux/src/linux/net/sock/gram.rs new file mode 100644 index 000000000..f1f81f36f --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/gram.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A datagram into the family: to whichever socket holds the port it is +//! sent to, or a socketpair end's peer. + +use core::mem; + +use crate::linux::abi::errno::{EBADF, ECONNREFUSED, EDESTADDRREQ, EMSGSIZE, ENOTCONN}; + +use super::table::Socks; +use super::types::{Addr, Domain, Proto}; + +/// The largest UDP payload IPv4 carries. +const MAX_GRAM: usize = 65507; + +impl Socks { + /// One datagram from `id` to `to`, or to the address it connected to. A + /// datagram nobody is bound to receive is dropped, as on the wire; a + /// connected sender learns of it as ECONNREFUSED on its next call, which + /// is what the ICMP reply does on Linux. + pub fn send_gram(&mut self, id: u32, to: Option, bytes: &[u8]) -> Result { + let s = self.get_mut(id).ok_or(EBADF)?; + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if bytes.len() > MAX_GRAM { + return Err(EMSGSIZE); + } + let (from, connected) = (s.local.unwrap_or_default(), s.remote.is_some()); + let target = match s.domain { + Domain::Unix => s.peer.ok_or(ECONNREFUSED)?, + Domain::Inet => { + let dest = + to.or(s.remote).ok_or(if connected { ENOTCONN } else { EDESTADDRREQ })?; + match self.bound(Proto::Dgram, dest) { + Some(r) => r, + None => { + if let Some(s) = self.get_mut(id).filter(|_| connected) { + s.error = ECONNREFUSED; + } + return Ok(bytes.len()); + } + } + } + }; + let r = self.get_mut(target).ok_or(ECONNREFUSED)?; + let queued: usize = r.grams.iter().map(|(_, g)| g.len()).sum(); + let wanted = r.remote.is_none_or(|x| x == from); + if wanted && queued + bytes.len() <= r.opts.rcvbuf as usize { + r.grams.push_back((from, bytes.to_vec())); + } + Ok(bytes.len()) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/gram_in.rs b/userland/capsule_linux/src/linux/net/sock/gram_in.rs new file mode 100644 index 000000000..329c9263e --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/gram_in.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A datagram out of the family's queue for one socket. + +use alloc::vec::Vec; +use core::mem; + +use crate::linux::abi::errno::{EAGAIN, EBADF}; + +use super::table::Socks; +use super::types::Addr; + +impl Socks { + /// The next datagram, cut to `want`, with its whole length and sender. + pub fn take_gram(&mut self, id: u32, want: usize, peek: bool) -> Result { + let s = self.get_mut(id).ok_or(EBADF)?; + if let Some((from, gram)) = s.grams.front() { + let got = Got { + bytes: gram[..want.min(gram.len())].to_vec(), + whole: gram.len(), + from: *from, + }; + if !peek { + s.grams.pop_front(); + } + return Ok(got); + } + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if s.rd_shut { + return Ok(Got { bytes: Vec::new(), whole: 0, from: Addr::default() }); + } + Err(EAGAIN) + } +} + +pub struct Got { + pub bytes: Vec, + /// The datagram's length before it was cut to fit. + pub whole: usize, + pub from: Addr, +} diff --git a/userland/capsule_linux/src/linux/net/sock/holders.rs b/userland/capsule_linux/src/linux/net/sock/holders.rs new file mode 100644 index 000000000..572e28b71 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/holders.rs @@ -0,0 +1,62 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which processes hold a socket. A descriptor copied by fork is held by the +//! child as well; the socket is let go when the last holder closes it or +//! ends. A process that ends without closing lets go of everything it held +//! when the family drops it, which is what `Holder` does. + +use alloc::vec::Vec; + +use super::cell::with; +use crate::linux::guest::{Fd, Kind}; + +/// Carried by each hosted process, as the pid its holdings are kept under. +pub struct Holder { + pid: u32, +} + +impl Holder { + pub fn new(pid: u32) -> Holder { + Holder { pid } + } + + /// A forked child holds every socket its parent's descriptors name. + pub fn fork(&self, child: u32, fds: &[Fd]) { + with(|t| { + for f in fds.iter().filter(|f| f.kind == Kind::Socket) { + if let Some(s) = t.get_mut(f.handle) { + if !s.holders.contains(&child) { + s.holders.push(child); + } + } + } + }); + } +} + +impl Drop for Holder { + fn drop(&mut self) { + let pid = self.pid; + with(|t| { + let held: Vec = + t.iter().filter(|(_, s)| s.holders.contains(&pid)).map(|(i, _)| i).collect(); + for id in held { + t.release(id, pid); + } + }); + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/link.rs b/userland/capsule_linux/src/linux/net/sock/link.rs new file mode 100644 index 000000000..ffdc8300b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/link.rs @@ -0,0 +1,80 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Joining two stream ends: a loopback connect, which Linux completes in the +//! caller's own call, and socketpair. + +use super::table::Socks; +use super::types::{Addr, Domain, Proto}; + +pub enum Link { + /// Connected; the listener has one more connection to accept. + Done, + /// Nothing listens there: Linux's loopback answers with a reset. + Refused, + /// The listener's queue is full. + Full, +} + +impl Socks { + /// Connect stream `id`, already bound, to the listener at `to`. + pub fn link(&mut self, id: u32, to: Addr) -> Link { + let Some(l) = self.listener(to) else { + return Link::Refused; + }; + let Some((backlog, queued, opts)) = + self.get(l).map(|s| (s.backlog, s.pending.len(), s.opts)) + else { + return Link::Refused; + }; + // Linux queues one more than the backlog it was given. + if queued > backlog { + return Link::Full; + } + let from = self.get(id).and_then(|s| s.local); + let server = self.open(Domain::Inet, Proto::Stream, None); + if let Some(s) = self.get_mut(server) { + s.local = Some(to); + s.remote = from; + s.peer = Some(id); + s.connected = true; + // An accepted socket starts with its listener's options. + s.opts = opts; + } + if let Some(c) = self.get_mut(id) { + c.remote = Some(to); + c.peer = Some(server); + c.connected = true; + } + if let Some(l) = self.get_mut(l) { + l.pending.push_back(server); + } + Link::Done + } + + /// Two connected sockets, both held by `pid`. + pub fn pair(&mut self, domain: Domain, proto: Proto, pid: u32) -> (u32, u32) { + let a = self.open(domain, proto, Some(pid)); + let b = self.open(domain, proto, Some(pid)); + for (me, other) in [(a, b), (b, a)] { + if let Some(s) = self.get_mut(me) { + s.peer = Some(other); + s.connected = true; + } + } + (a, b) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/mod.rs b/userland/capsule_linux/src/linux/net/sock/mod.rs new file mode 100644 index 000000000..c1a43ab31 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/mod.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The family's sockets: one table this personality keeps for every process +//! it hosts, so a connection between two of them, or two threads of one, +//! never leaves the capsule. A descriptor names an entry by index; the entry +//! records which processes hold it, and is let go when none does. +//! +//! A stream to 127.0.0.0/8 is a pair of entries, each holding what the other +//! wrote. A datagram to a bound loopback port is queued on that entry. A +//! stream to anywhere else is an entry backed by a net.sockets handle. + +mod bind; +mod cell; +mod free; +mod gram; +mod gram_in; +mod holders; +mod link; +mod new; +mod opts; +mod port; +mod ready; +mod recv; +mod send; +mod table; +mod types; + +pub use cell::with; +pub use holders::Holder; +pub use link::Link; +pub use ready::bits; +pub use types::{Addr, Domain, Proto}; diff --git a/userland/capsule_linux/src/linux/net/sock/new.rs b/userland/capsule_linux/src/linux/net/sock/new.rs new file mode 100644 index 000000000..293afe55c --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/new.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A socket as it starts: unbound, unconnected, Linux's default options. + +use alloc::collections::VecDeque; +use alloc::vec; +use alloc::vec::Vec; + +use super::opts::Opts; +use super::types::{Domain, Proto, Sock}; + +impl Sock { + pub fn new(domain: Domain, proto: Proto, pid: Option) -> Sock { + Sock { + domain, + proto, + local: None, + remote: None, + listening: false, + backlog: 0, + pending: VecDeque::new(), + peer: None, + connected: false, + rx: VecDeque::new(), + grams: VecDeque::new(), + eof: false, + wr_shut: false, + rd_shut: false, + broken: false, + error: 0, + opts: Opts::new(proto), + svc: None, + holders: pid.map_or_else(Vec::new, |p| vec![p]), + } + } +} diff --git a/userland/capsule_linux/src/linux/net/addr.rs b/userland/capsule_linux/src/linux/net/sock/opts.rs similarity index 51% rename from userland/capsule_linux/src/linux/net/addr.rs rename to userland/capsule_linux/src/linux/net/sock/opts.rs index 330c461cf..9fb73e378 100644 --- a/userland/capsule_linux/src/linux/net/addr.rs +++ b/userland/capsule_linux/src/linux/net/sock/opts.rs @@ -14,25 +14,24 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . +//! The options a socket keeps, with the values Linux starts a socket with +//! (read from a Linux 6.x host: tcp_rmem[1] and rmem_default). -//! `struct sockaddr_in` out of a guest. +use super::types::Proto; -use crate::linux::guest::Guest; - -/// family(2) port(2) addr(4), and the rest of the sixteen bytes unused. -const SOCKADDR_IN: usize = 16; -const AF_INET: u16 = 2; +#[derive(Clone, Copy)] +pub struct Opts { + pub reuseaddr: bool, + /// As getsockopt reports it; also what the socket's queue holds. + pub rcvbuf: u32, +} -/// Port in network order and address in network order, which is the -/// order net.sockets wants as well, so neither is byte swapped here. -pub fn inet(guest: &Guest, at: u64, len: u64) -> Option<(u16, [u8; 4])> { - if len < SOCKADDR_IN as u64 { - return None; - } - let raw = guest.read(at, SOCKADDR_IN)?; - if u16::from_le_bytes([raw[0], raw[1]]) != AF_INET { - return None; +impl Opts { + pub fn new(proto: Proto) -> Opts { + let rcvbuf = match proto { + Proto::Stream => 131_072, + Proto::Dgram => 212_992, + }; + Opts { reuseaddr: false, rcvbuf } } - let port = u16::from_be_bytes([raw[2], raw[3]]); - Some((port, [raw[4], raw[5], raw[6], raw[7]])) } diff --git a/userland/capsule_linux/src/linux/net/sock/port.rs b/userland/capsule_linux/src/linux/net/sock/port.rs new file mode 100644 index 000000000..c6d136671 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/port.rs @@ -0,0 +1,67 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Ports on the loopback address: who holds one, and a free one when a +//! program asks for port 0 or connects before it binds. + +use super::table::Socks; +use super::types::{Addr, Proto}; + +/// Linux's net.ipv4.ip_local_port_range default. +const FIRST: u16 = 32768; +const LAST: u16 = 60999; + +impl Socks { + /// The socket of this kind bound to exactly this address, if any. + pub fn bound(&self, proto: Proto, at: Addr) -> Option { + self.iter().find(|(_, s)| s.proto == proto && s.local == Some(at)).map(|(i, _)| i) + } + + /// The listener a connect to `at` reaches. + pub fn listener(&self, at: Addr) -> Option { + self.iter() + .find(|(_, s)| s.listening && s.proto == Proto::Stream && s.local == Some(at)) + .map(|(i, _)| i) + } + + /// True when binding `id` to `at` takes a port another socket holds. + /// Two sockets share one only when both set SO_REUSEADDR and neither + /// listens, as Linux allows a restarted server to rebind. + pub fn in_use(&self, id: u32, at: Addr) -> bool { + let Some(me) = self.get(id) else { + return true; + }; + self.iter().any(|(i, s)| { + i != id + && s.proto == me.proto + && s.local.is_some_and(|l| l.port == at.port && (l.ip == at.ip)) + && (s.listening || !(s.opts.reuseaddr && me.opts.reuseaddr)) + }) + } + + /// A port in the ephemeral range no socket of this kind holds on `ip`. + pub fn ephemeral(&mut self, proto: Proto, ip: [u8; 4]) -> Option { + let span = (LAST - FIRST + 1) as u32; + for step in 0..span { + let port = FIRST + ((u32::from(self.next_port) + step) % span) as u16; + if self.bound(proto, Addr { ip, port }).is_none() { + self.next_port = (port - FIRST + 1) % span as u16; + return Some(port); + } + } + None + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/ready.rs b/userland/capsule_linux/src/linux/net/sock/ready.rs new file mode 100644 index 000000000..4a77599e1 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/ready.rs @@ -0,0 +1,67 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What a family socket can do now, in poll's bits, as Linux's tcp_poll, +//! udp_poll and unix_poll report it. + +use super::cell::with; +use super::types::{Proto, Sock}; + +const POLLIN: u16 = 0x001; +const POLLOUT: u16 = 0x004; +const POLLERR: u16 = 0x008; +const POLLHUP: u16 = 0x010; +const POLLRDHUP: u16 = 0x2000; + +/// The bits for `id`, or None when net.sockets holds its state. +pub fn bits(id: u32) -> Option { + with(|t| { + let s = t.get(id).filter(|s| s.svc.is_none())?; + // A stream is writable while its peer has room, or once it is gone. + let room = + s.peer.and_then(|p| t.get(p)).is_none_or(|p| p.rx.len() < p.opts.rcvbuf as usize); + Some(of(s, room)) + }) +} + +fn of(s: &Sock, room: bool) -> u16 { + let err = if s.error != 0 { POLLERR } else { 0 }; + if s.proto == Proto::Dgram { + let rd = if s.grams.is_empty() { 0 } else { POLLIN }; + return err | rd | POLLOUT; + } + if s.listening { + return err | if s.pending.is_empty() { 0 } else { POLLIN }; + } + // Never connected, refused, or reset: Linux's TCP_CLOSE. + if !s.connected || s.broken { + let rd = if s.connected { POLLIN | POLLRDHUP } else { 0 }; + return err | rd | POLLOUT | POLLHUP; + } + let shut_rd = s.eof || s.rd_shut; + let mut set = err; + if !s.rx.is_empty() || shut_rd { + set |= POLLIN; + } + if shut_rd { + set |= POLLRDHUP; + } + if shut_rd && s.wr_shut { + set |= POLLHUP; + } + // After SHUT_WR a write fails at once, so Linux reports it writable. + set | if room || s.wr_shut { POLLOUT } else { 0 } +} diff --git a/userland/capsule_linux/src/linux/net/sock/recv.rs b/userland/capsule_linux/src/linux/net/sock/recv.rs new file mode 100644 index 000000000..730901fce --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/recv.rs @@ -0,0 +1,49 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes out of a family stream. + +use alloc::vec::Vec; +use core::mem; + +use crate::linux::abi::errno::{EAGAIN, EBADF, ENOTCONN}; + +use super::table::Socks; + +impl Socks { + /// Up to `want` bytes of a stream; empty is end of file, EAGAIN is none + /// yet. Bytes come before a pending error, as Linux reads its queue first. + pub fn read(&mut self, id: u32, want: usize, peek: bool) -> Result, i64> { + let s = self.get_mut(id).ok_or(EBADF)?; + if !s.rx.is_empty() && want != 0 { + let n = want.min(s.rx.len()); + return Ok(match peek { + true => s.rx.iter().take(n).copied().collect(), + false => s.rx.drain(..n).collect(), + }); + } + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if s.listening || !s.connected { + return Err(ENOTCONN); + } + if want == 0 || s.eof || s.rd_shut { + return Ok(Vec::new()); + } + Err(EAGAIN) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/send.rs b/userland/capsule_linux/src/linux/net/sock/send.rs new file mode 100644 index 000000000..8b4268b86 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/send.rs @@ -0,0 +1,55 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes into a family stream: to its peer's queue. + +use core::mem; + +use crate::linux::abi::errno::{EAGAIN, EBADF, EPIPE}; + +use super::table::Socks; +use super::types::Domain; + +impl Socks { + /// As many of `bytes` as the peer has room for, EAGAIN for none. + pub fn write(&mut self, id: u32, bytes: &[u8]) -> Result { + let s = self.get_mut(id).ok_or(EBADF)?; + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if !s.connected || s.wr_shut || s.broken { + return Err(EPIPE); + } + let Some(p) = s.peer else { + // TCP takes the first write after the peer left, and the reset + // that answers it makes every later one EPIPE. A Unix socket + // knows at once. + if s.domain == Domain::Unix { + return Err(EPIPE); + } + s.broken = true; + return Ok(bytes.len()); + }; + let peer = self.get_mut(p).ok_or(EPIPE)?; + let room = (peer.opts.rcvbuf as usize).saturating_sub(peer.rx.len()); + if room == 0 && !bytes.is_empty() { + return Err(EAGAIN); + } + let n = room.min(bytes.len()); + peer.rx.extend(&bytes[..n]); + Ok(n) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/table.rs b/userland/capsule_linux/src/linux/net/sock/table.rs new file mode 100644 index 000000000..057346762 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/table.rs @@ -0,0 +1,62 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The table itself: entries by index, a freed index reused. + +use alloc::vec::Vec; + +use super::types::{Domain, Proto, Sock}; + +pub struct Socks { + pub(super) list: Vec>, + /// Where the next ephemeral port search starts. + pub(super) next_port: u16, +} + +impl Socks { + pub const fn new() -> Socks { + Socks { list: Vec::new(), next_port: 0 } + } + + /// A fresh socket held by `pid`, or by nobody yet when `pid` is None (the + /// server end of a connection, until accept hands it out). + pub fn open(&mut self, domain: Domain, proto: Proto, pid: Option) -> u32 { + let sock = Sock::new(domain, proto, pid); + match self.list.iter().position(Option::is_none) { + Some(i) => { + self.list[i] = Some(sock); + i as u32 + } + None => { + self.list.push(Some(sock)); + (self.list.len() - 1) as u32 + } + } + } + + pub fn get(&self, id: u32) -> Option<&Sock> { + self.list.get(id as usize).and_then(Option::as_ref) + } + + pub fn get_mut(&mut self, id: u32) -> Option<&mut Sock> { + self.list.get_mut(id as usize).and_then(Option::as_mut) + } + + /// Every live socket, with its index. + pub fn iter(&self) -> impl Iterator { + self.list.iter().enumerate().filter_map(|(i, s)| s.as_ref().map(|s| (i as u32, s))) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/types.rs b/userland/capsule_linux/src/linux/net/sock/types.rs new file mode 100644 index 000000000..9d54c4dce --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/types.rs @@ -0,0 +1,73 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! One socket, as the family sees it. + +use alloc::collections::VecDeque; +use alloc::vec::Vec; + +use super::opts::Opts; + +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Proto { + Stream, + Dgram, +} + +/// AF_INET, or AF_UNIX for the two ends socketpair makes. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Domain { + Inet, + Unix, +} + +/// An IPv4 address and a port, the port in host order. +#[derive(Clone, Copy, PartialEq, Eq, Default)] +pub struct Addr { + pub ip: [u8; 4], + pub port: u16, +} + +pub struct Sock { + pub domain: Domain, + pub proto: Proto, + pub local: Option, + pub remote: Option, + /// Set by listen, with the queue of connections accept has not taken. + pub listening: bool, + pub backlog: usize, + pub pending: VecDeque, + /// The other end of a stream, until it is let go. + pub peer: Option, + pub connected: bool, + /// Bytes the peer wrote that this end has not read. + pub rx: VecDeque, + /// Datagrams waiting to be read, each with where it came from. + pub grams: VecDeque<(Addr, Vec)>, + /// The peer will write nothing more: it shut its side or it is gone. + pub eof: bool, + pub wr_shut: bool, + pub rd_shut: bool, + /// Written to after the peer was gone: every later write is EPIPE. + pub broken: bool, + /// SO_ERROR: reported once, by the next call that looks. + pub error: i64, + pub opts: Opts, + /// The net.sockets handle, for a stream that reaches outside the family. + pub svc: Option, + /// The processes that hold a descriptor naming this socket. + pub holders: Vec, +} diff --git a/userland/capsule_linux/src/linux/net/sockaddr.rs b/userland/capsule_linux/src/linux/net/sockaddr.rs new file mode 100644 index 000000000..6ca4fc158 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sockaddr.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sockaddr_in` and `sockaddr_un` between guest memory and `Addr`. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::sock::Addr; + +pub const AF_UNSPEC: u16 = 0; +pub const AF_UNIX: u16 = 1; +pub const AF_INET: u16 = 2; +pub const SOCKADDR_IN: usize = 16; + +/// The family and address a guest named; EINVAL for one too short to hold +/// its family, EFAULT for one it cannot read. +pub fn read(guest: &Guest, at: u64, len: u64) -> Result<(u16, Addr), u64> { + if len < 2 { + return Err(errno::fail(errno::EINVAL)); + } + let head = guest.read(at, 2).ok_or(errno::fail(errno::EFAULT))?; + let family = u16::from_le_bytes([head[0], head[1]]); + if family != AF_INET { + return Ok((family, Addr::default())); + } + if len < SOCKADDR_IN as u64 { + return Err(errno::fail(errno::EINVAL)); + } + let raw = guest.read(at, SOCKADDR_IN).ok_or(errno::fail(errno::EFAULT))?; + let port = u16::from_be_bytes([raw[2], raw[3]]); + Ok((family, Addr { ip: [raw[4], raw[5], raw[6], raw[7]], port })) +} + +/// True for 127.0.0.0/8, the only addresses the family keeps to itself. +pub fn is_loopback(ip: [u8; 4]) -> bool { + ip[0] == 127 +} diff --git a/userland/capsule_linux/src/linux/net/sockaddr_out.rs b/userland/capsule_linux/src/linux/net/sockaddr_out.rs new file mode 100644 index 000000000..a8b84b09b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sockaddr_out.rs @@ -0,0 +1,58 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A socket's address written into guest memory. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::sock::{Addr, Domain}; +use super::sockaddr::{AF_INET, AF_UNIX, SOCKADDR_IN}; + +/// Write `addr` at `at`, cut to the length the guest offered at `lenp`, and +/// the whole length back at `lenp`, as Linux's move_addr_to_user does. A +/// socketpair end has an unnamed Unix address: the family alone. +pub fn write(guest: &mut Guest, at: u64, lenp: u64, domain: Domain, addr: Addr) -> u64 { + if at == 0 || lenp == 0 { + return errno::ok(0); + } + let Some(raw) = guest.read(lenp, 4) else { + return errno::fail(errno::EFAULT); + }; + let room = u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]) as i32; + if room < 0 { + return errno::fail(errno::EINVAL); + } + let mut sa = [0u8; SOCKADDR_IN]; + let whole = match domain { + Domain::Unix => { + sa[0..2].copy_from_slice(&AF_UNIX.to_le_bytes()); + 2 + } + Domain::Inet => { + sa[0..2].copy_from_slice(&AF_INET.to_le_bytes()); + sa[2..4].copy_from_slice(&addr.port.to_be_bytes()); + sa[4..8].copy_from_slice(&addr.ip); + SOCKADDR_IN + } + }; + let n = whole.min(room as usize); + if guest.write(at, &sa[..n]) < n as i64 || guest.write(lenp, &(whole as u32).to_le_bytes()) < 4 + { + return errno::fail(errno::EFAULT); + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/net/socket.rs b/userland/capsule_linux/src/linux/net/socket.rs index cdcbd5cbe..adcd11ef2 100644 --- a/userland/capsule_linux/src/linux/net/socket.rs +++ b/userland/capsule_linux/src/linux/net/socket.rs @@ -14,49 +14,45 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `socket` and `connect`, over net.sockets. - -use alloc::vec::Vec; - -use super::ops::{DOMAIN, KIND_MIXNET, OP_SOCKET}; +//! `socket` for AF_INET. The socket is the family's until it connects +//! outside 127.0.0.0/8; only then is net.sockets asked for one. use crate::linux::abi::errno; -use crate::linux::guest::{Fd, Guest}; +use crate::linux::guest::Guest; -use super::call::call; +use super::fd::{install, SOCK_CLOEXEC, SOCK_NONBLOCK}; +use super::policy::refuse; +use super::sock::{self, Domain, Proto}; +use super::sockaddr::AF_INET; -const AF_INET: u64 = 2; const SOCK_STREAM: u64 = 1; const SOCK_DGRAM: u64 = 2; -/// Linux ORs these into the type; neither changes what is opened here. -const TYPE_MASK: u64 = 0xFF; +const SOCK_RAW: u64 = 3; +const TYPE_MASK: u64 = 0xF; +const IPPROTO_TCP: u64 = 6; +const IPPROTO_UDP: u64 = 17; -pub fn socket(guest: &mut Guest, family: u64, kind: u64) -> u64 { - if family != AF_INET { +pub fn socket(guest: &mut Guest, family: u64, kind: u64, protocol: u64) -> u64 { + let flags = kind & (SOCK_NONBLOCK | SOCK_CLOEXEC); + if kind & !(TYPE_MASK | flags) != 0 { + return errno::fail(errno::EINVAL); + } + if family != u64::from(AF_INET) { return errno::fail(errno::EAFNOSUPPORT); } - /* - * A guest's stream goes over the mixnet, never the open network, and it - * holds no capability that could name a socket: there is no second route - * to disable and no firewall rule to remove. - */ - let want = match kind & TYPE_MASK { - SOCK_STREAM => KIND_MIXNET, - SOCK_DGRAM => return super::dns::open(guest), - _ => return errno::fail(errno::ENOSYS), + let (proto, own) = match kind & TYPE_MASK { + SOCK_STREAM => (Proto::Stream, IPPROTO_TCP), + SOCK_DGRAM => (Proto::Dgram, IPPROTO_UDP), + SOCK_RAW => { + return refuse("SOCK_RAW: raw sockets reach below any confinement", errno::EPERM) + } + _ => return errno::fail(errno::ESOCKTNOSUPPORT), }; - let mut body = Vec::with_capacity(4); - body.extend_from_slice(&DOMAIN.to_le_bytes()); - body.extend_from_slice(&want.to_le_bytes()); - let Some((status, out)) = call(OP_SOCKET, &body, 8) else { - return errno::fail(errno::EIO); - }; - if status != 0 || out.len() < 4 { - return errno::fail(errno::ENOMEM); - } - let handle = u32::from_le_bytes([out[0], out[1], out[2], out[3]]); - match crate::linux::file::install(guest, Fd::socket(handle)) { - Some(n) => errno::ok(n), - None => errno::fail(errno::EMFILE), + // Anything but the type's own protocol, MPTCP included, is one this + // stack does not have; Go falls back to TCP on this answer. + if protocol != 0 && protocol != own { + return errno::fail(errno::EPROTONOSUPPORT); } + let id = sock::with(|t| t.open(Domain::Inet, proto, Some(guest.pid))); + install(guest, id, flags) } diff --git a/userland/capsule_linux/src/linux/net/stream.rs b/userland/capsule_linux/src/linux/net/stream.rs index 94c6effd5..943dcb972 100644 --- a/userland/capsule_linux/src/linux/net/stream.rs +++ b/userland/capsule_linux/src/linux/net/stream.rs @@ -15,50 +15,41 @@ // along with this program. If not, see . -//! Bytes on and off a connected socket. +//! Bytes on a stream that reaches outside the family, through net.sockets. use alloc::vec::Vec; use crate::linux::abi::errno; -use crate::linux::guest::Guest; use super::call::call; use super::ops::{OP_CLOSE, OP_RECV, OP_SEND}; -/// One transfer. The server's reply buffer is the ceiling, not this. -const MAX_IO: u64 = 32 << 10; +/// The most net.sockets carries in one call. +pub const MAX_IO: usize = 32 << 10; -pub fn send(guest: &Guest, handle: u32, buf: u64, len: u64) -> u64 { - let take = len.min(MAX_IO); - let Some(bytes) = guest.read(buf, take as usize) else { - return errno::fail(errno::EFAULT); - }; +pub fn send_bytes(handle: u32, bytes: &[u8]) -> u64 { let mut body = Vec::with_capacity(4 + bytes.len()); body.extend_from_slice(&handle.to_le_bytes()); - body.extend_from_slice(&bytes); + body.extend_from_slice(bytes); match call(OP_SEND, &body, 0) { - Some((0, _)) => errno::ok(take), + Some((0, _)) => errno::ok(bytes.len() as u64), Some(_) => errno::fail(errno::EPIPE), None => errno::fail(errno::EIO), } } -pub fn recv(guest: &Guest, handle: u32, buf: u64, len: u64) -> u64 { - let want = len.min(MAX_IO) as usize; - let Some((status, bytes)) = call(OP_RECV, &handle.to_le_bytes(), want) else { - return errno::fail(errno::EIO); +/// Up to `want` bytes. net.sockets answers the same for a quiet stream, a +/// closed one and a reset one, so any refusal reads as ECONNRESET. +pub fn recv_bytes(handle: u32, want: usize) -> Result, u64> { + let want = want.min(MAX_IO); + let Some((status, mut bytes)) = call(OP_RECV, &handle.to_le_bytes(), want) else { + return Err(errno::fail(errno::EIO)); }; if status != 0 { - return errno::fail(errno::ECONNRESET); - } - if bytes.is_empty() { - return errno::ok(0); - } - let n = bytes.len().min(want); - if guest.write(buf, &bytes[..n]) < n as i64 { - return errno::fail(errno::EFAULT); + return Err(errno::fail(errno::ECONNRESET)); } - errno::ok(n as u64) + bytes.truncate(want); + Ok(bytes) } pub fn close(handle: u32) { diff --git a/userland/capsule_linux/src/linux/net/xfer_in.rs b/userland/capsule_linux/src/linux/net/xfer_in.rs new file mode 100644 index 000000000..a0bdaa13b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/xfer_in.rs @@ -0,0 +1,79 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes into a guest from a socket. + +use alloc::vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::flags::{MSG_OOB, MSG_PEEK, MSG_TRUNC}; +use super::iov::{self, Iov}; +use super::sock::{self, Addr, Domain, Proto}; + +pub struct In { + /// Bytes put in the guest's buffers. + pub n: usize, + /// A datagram's length before it was cut to fit. + pub whole: usize, + /// Where a datagram came from; a stream, and an unnamed sender, say + /// nothing. + pub from: Option<(Domain, Addr)>, +} + +/// Receive into `iov` from socket `id`, `skip` bytes of it filled already. +pub fn recv(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64) -> Result { + if flags & MSG_OOB != 0 { + return Err(errno::fail(errno::EINVAL)); + } + let Some((proto, domain, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.domain, s.svc))) + else { + return Err(errno::fail(errno::EBADF)); + }; + let want = iov::total(iov).saturating_sub(skip); + let peek = flags & MSG_PEEK != 0; + if let Some(h) = svc { + let bytes = super::stream::recv_bytes(h, want)?; + iov::scatter(guest, iov, skip, &bytes)?; + return Ok(In { n: bytes.len(), whole: bytes.len(), from: None }); + } + match proto { + Proto::Stream => { + let bytes = sock::with(|t| t.read(id, want, peek)).map_err(errno::fail)?; + // A stream's MSG_TRUNC discards what it would have read. + if flags & MSG_TRUNC == 0 { + iov::scatter(guest, iov, skip, &bytes)?; + } + Ok(In { n: bytes.len(), whole: bytes.len(), from: None }) + } + Proto::Dgram => { + let got = sock::with(|t| t.take_gram(id, want, peek)).map_err(errno::fail)?; + iov::scatter(guest, iov, skip, &got.bytes)?; + // A socketpair's peer has no name, and Linux gives none. + let from = (domain == Domain::Inet).then_some((domain, got.from)); + Ok(In { n: got.bytes.len(), whole: got.whole, from }) + } + } +} + +/// `read` on a socket. +pub fn read(guest: &Guest, id: u32, buf: u64, len: u64) -> u64 { + match recv(guest, id, &vec![(buf, len)], 0, 0) { + Ok(got) => errno::ok(got.n as u64), + Err(e) => e, + } +} diff --git a/userland/capsule_linux/src/linux/net/xfer_out.rs b/userland/capsule_linux/src/linux/net/xfer_out.rs new file mode 100644 index 000000000..aa8a59410 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/xfer_out.rs @@ -0,0 +1,69 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes out of a socket: to a stream's peer, a datagram to the port it is +//! sent to, or a stream outside the family through net.sockets. + +use alloc::vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::flags::MSG_OOB; +use super::iov::{self, Iov}; +use super::sock::{self, Addr, Proto}; + +/// What one send takes from the guest at most: the default receive buffer +/// of a stream's peer, so a single call can fill it. +const STREAM_CAP: usize = 128 << 10; +/// One byte past the largest datagram, so a larger one is seen and refused. +const GRAM_CAP: usize = 65508; + +/// Send the message in `iov` from socket `id`, `skip` bytes of it having +/// gone already: the count sent now, or an errno. +pub fn send(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64, to: Option) -> u64 { + if flags & MSG_OOB != 0 { + return errno::fail(errno::EOPNOTSUPP); + } + let Some((proto, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.svc))) else { + return errno::fail(errno::EBADF); + }; + let cap = match (svc, proto) { + (Some(_), _) => super::stream::MAX_IO, + (None, Proto::Stream) => STREAM_CAP, + (None, Proto::Dgram) => GRAM_CAP, + }; + let bytes = match iov::gather(guest, iov, skip, cap) { + Ok(b) => b, + Err(e) => return e, + }; + if let Some(h) = svc { + return super::stream::send_bytes(h, &bytes); + } + let sent = sock::with(|t| match proto { + Proto::Stream => t.write(id, &bytes), + Proto::Dgram => t.autobind(id).and_then(|()| t.send_gram(id, to, &bytes)), + }); + match sent { + Ok(n) => errno::ok(n as u64), + Err(e) => errno::fail(e), + } +} + +/// `write` on a socket: a send with no flags and no address. +pub fn write(guest: &Guest, id: u32, buf: u64, len: u64) -> u64 { + send(guest, id, &vec![(buf, len)], 0, 0, None) +} diff --git a/userland/capsule_linux/src/linux/serve/table_net.rs b/userland/capsule_linux/src/linux/serve/table_net.rs index cf5902775..fc2ca09e1 100644 --- a/userland/capsule_linux/src/linux/serve/table_net.rs +++ b/userland/capsule_linux/src/linux/serve/table_net.rs @@ -17,23 +17,36 @@ //! Calls that name a socket. use crate::linux::abi::nr; -use crate::linux::call; use crate::linux::guest::Guest; use crate::linux::net; use crate::linux::unix::{self, is_unix}; +/// Every socket call. The ones that can wait reach here only when they +/// cannot: a socket's own calls are routed to `waits_sock` first, and the +/// rest are a resolver's, a display socket's, or not a socket at all. pub fn net_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option { let _ = tid; Some(match nr { nr::SOCKET if a[0] == 1 => unix::socket(guest, a[1]), - nr::SOCKET => net::socket(guest, a[0], a[1]), + nr::SOCKET => net::socket(guest, a[0], a[1], a[2]), + nr::SOCKETPAIR => net::socketpair(guest, a[0], a[1], a[2], a[3]), nr::CONNECT if is_unix(guest, a[0]) => unix::connect(guest, a[0], a[1], a[2]), nr::CONNECT => net::connect(guest, a[0], a[1], a[2]), - nr::SENDTO => net::sendto(guest, a[0], a[1], a[2], a[4], a[5]), - nr::RECVFROM => net::recvfrom(guest, a[0], a[1], a[2], a[4], a[5]), - nr::SHUTDOWN => call::close(guest, a[0]), - nr::SENDMSG => unix::sendmsg(guest, a[0], a[1]), - nr::RECVMSG => unix::recvmsg(guest, a[0], a[1]), + nr::BIND => net::bind(guest, a[0], a[1], a[2]), + nr::LISTEN => net::listen(guest, a[0], a[1]), + nr::ACCEPT => net::accept4(guest, a[0], a[1], a[2], 0), + nr::ACCEPT4 => net::accept4(guest, a[0], a[1], a[2], a[3]), + nr::GETSOCKNAME => net::getsockname(guest, a[0], a[1], a[2]), + nr::GETPEERNAME => net::getpeername(guest, a[0], a[1], a[2]), + nr::SENDTO => net::sendto(guest, a[0], a[1], a[2], a[3], a[4], a[5]), + nr::RECVFROM => net::recvfrom(guest, a[0], a[1], a[2], a[3], a[4], a[5]), + nr::SHUTDOWN => net::shutdown(guest, a[0], a[1]), + nr::SENDMSG if is_unix(guest, a[0]) => unix::sendmsg(guest, a[0], a[1]), + nr::RECVMSG if is_unix(guest, a[0]) => unix::recvmsg(guest, a[0], a[1]), + nr::SENDMSG => net::sendmsg(guest, a[0], a[1], a[2], 0), + nr::RECVMSG => net::recvmsg(guest, a[0], a[1], a[2], 0), + nr::SENDMMSG => net::sendmmsg(guest, a, 0), + nr::RECVMMSG => net::recvmmsg(guest, a, 0), _ => return None, }) } From 887ca99aefe7de12ef10bbd09086460aadcc9b17 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 20:06:41 +0000 Subject: [PATCH 05/14] linux: setsockopt and getsockopt keep and read back what Linux keeps Both were unserved, so every Go listener and dialer, which sets SO_REUSEADDR, TCP_NODELAY and the keepalive options, and every program that reads SO_ERROR after a non-blocking connect, got ENOSYS. Now a socket keeps SO_REUSEADDR, SO_REUSEPORT, SO_KEEPALIVE, SO_BROADCAST, SO_RCVBUF and SO_SNDBUF (doubled and bounded as Linux does), SO_LINGER, SO_RCVTIMEO and SO_SNDTIMEO (EDOM for microseconds out of range), TCP_NODELAY and TCP_KEEPIDLE, TCP_KEEPINTVL and TCP_KEEPCNT (EINVAL outside Linux's range), each with Linux's default, and reads them back; SO_TYPE, SO_DOMAIN, SO_PROTOCOL, SO_ACCEPTCONN and SO_ERROR, which reports a refused connect once, are read. A TCP option on a datagram socket and an IPv6 option on an AF_INET one answer as Linux does. Any other option is ENOPROTOOPT with a [LINUX] refused line naming its level and number. SO_REUSEADDR decides whether a bind may share a port, SO_RCVBUF bounds the socket's queue, and SO_LINGER of zero seconds makes close reset the peer, as on Linux. --- userland/capsule_linux/src/linux/net/mod.rs | 7 ++ .../capsule_linux/src/linux/net/opt_get.rs | 57 ++++++++++++++ .../capsule_linux/src/linux/net/opt_ids.rs | 52 +++++++++++++ .../capsule_linux/src/linux/net/opt_set.rs | 74 +++++++++++++++++++ .../capsule_linux/src/linux/net/opt_time.rs | 42 +++++++++++ .../capsule_linux/src/linux/net/opt_value.rs | 61 +++++++++++++++ .../capsule_linux/src/linux/net/sock/free.rs | 7 +- .../capsule_linux/src/linux/net/sock/mod.rs | 2 +- .../capsule_linux/src/linux/net/sock/opts.rs | 41 ++++++++-- .../src/linux/serve/table_net.rs | 2 + 10 files changed, 335 insertions(+), 10 deletions(-) create mode 100644 userland/capsule_linux/src/linux/net/opt_get.rs create mode 100644 userland/capsule_linux/src/linux/net/opt_ids.rs create mode 100644 userland/capsule_linux/src/linux/net/opt_set.rs create mode 100644 userland/capsule_linux/src/linux/net/opt_time.rs create mode 100644 userland/capsule_linux/src/linux/net/opt_value.rs diff --git a/userland/capsule_linux/src/linux/net/mod.rs b/userland/capsule_linux/src/linux/net/mod.rs index ebd081702..32bd53e81 100644 --- a/userland/capsule_linux/src/linux/net/mod.rs +++ b/userland/capsule_linux/src/linux/net/mod.rs @@ -38,6 +38,11 @@ mod msg; mod msg_hdr; mod name; mod ops; +mod opt_get; +mod opt_ids; +mod opt_set; +mod opt_time; +mod opt_value; mod pair; mod peer_addr; mod policy; @@ -67,6 +72,8 @@ pub use dgram::{recvfrom, sendto}; pub use mmsg::{recvmmsg, sendmmsg}; pub use msg::{recvmsg, sendmsg}; pub use name::{getpeername, getsockname}; +pub use opt_get::getsockopt; +pub use opt_set::setsockopt; pub use pair::socketpair; pub use poll::{ready, POLLERR, POLLHUP}; pub use poll_set::poll; diff --git a/userland/capsule_linux/src/linux/net/opt_get.rs b/userland/capsule_linux/src/linux/net/opt_get.rs new file mode 100644 index 000000000..be6fac3a6 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt_get.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `getsockopt`, for the options `opt_set` keeps and the ones a socket +//! reports about itself. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::sock; + +pub fn getsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, lenp: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let Some(raw) = guest.read(lenp, 4) else { + return errno::fail(errno::EFAULT); + }; + let room = i32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]); + if room < 0 { + return errno::fail(errno::EINVAL); + } + let value = sock::with(|t| t.get_mut(id).map(|s| super::opt_value::value(s, level, name))); + let bytes = match value { + Some(Ok(bytes)) => bytes, + Some(Err(e)) => return e, + None => return errno::fail(errno::EBADF), + }; + let n = bytes.len().min(room as usize); + if guest.write(val, &bytes[..n]) < n as i64 || guest.write(lenp, &(n as u32).to_le_bytes()) < 4 + { + return errno::fail(errno::EFAULT); + } + errno::ok(0) +} + +/// ENOPROTOOPT for an option this capsule does not keep, said by name: a +/// program may depend on it, and Linux would have kept it. +pub fn unknown(call: &str, level: u64, name: u64) -> u64 { + let what = alloc::format!("{call} level {level} option {name}: not kept for a guest socket"); + super::policy::refuse(&what, errno::ENOPROTOOPT) +} diff --git a/userland/capsule_linux/src/linux/net/opt_ids.rs b/userland/capsule_linux/src/linux/net/opt_ids.rs new file mode 100644 index 000000000..a2d4d4277 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt_ids.rs @@ -0,0 +1,52 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Option levels and names, from include/uapi/asm-generic/socket.h, +//! include/uapi/linux/in.h and include/uapi/linux/tcp.h. + +pub const SOL_SOCKET: u64 = 1; +pub const IPPROTO_TCP: u64 = 6; +pub const IPPROTO_IPV6: u64 = 41; + +pub const SO_REUSEADDR: u64 = 2; +pub const SO_TYPE: u64 = 3; +pub const SO_ERROR: u64 = 4; +pub const SO_BROADCAST: u64 = 6; +pub const SO_SNDBUF: u64 = 7; +pub const SO_RCVBUF: u64 = 8; +pub const SO_KEEPALIVE: u64 = 9; +pub const SO_LINGER: u64 = 13; +pub const SO_REUSEPORT: u64 = 15; +pub const SO_RCVTIMEO: u64 = 20; +pub const SO_SNDTIMEO: u64 = 21; +pub const SO_ACCEPTCONN: u64 = 30; +pub const SO_PROTOCOL: u64 = 38; +pub const SO_DOMAIN: u64 = 39; + +pub const TCP_NODELAY: u64 = 1; +pub const TCP_KEEPIDLE: u64 = 4; +pub const TCP_KEEPINTVL: u64 = 5; +pub const TCP_KEEPCNT: u64 = 6; + +/// net.core.rmem_max and wmem_max as Linux ships them: what SO_RCVBUF and +/// SO_SNDBUF are held to before they are doubled. +pub const BUF_MAX: u32 = 212_992; +/// The least Linux keeps: SOCK_MIN_RCVBUF and SOCK_MIN_SNDBUF. +pub const RCVBUF_MIN: u32 = 2304; +pub const SNDBUF_MIN: u32 = 4608; +/// MAX_TCP_KEEPIDLE, MAX_TCP_KEEPINTVL and MAX_TCP_KEEPCNT. +pub const KEEP_MAX: u32 = 32767; +pub const KEEPCNT_MAX: u32 = 127; diff --git a/userland/capsule_linux/src/linux/net/opt_set.rs b/userland/capsule_linux/src/linux/net/opt_set.rs new file mode 100644 index 000000000..9704cf988 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt_set.rs @@ -0,0 +1,74 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `setsockopt`. Every option here is kept and reads back as Linux reads +//! it; one that would change nothing on the family's loopback is still kept, +//! since Linux keeps it too. One this capsule cannot honour is refused. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::opt_ids::*; +use super::opt_time::{keep, timeo}; +use super::sock::{self, Proto}; + +pub fn setsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, len: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let Some(raw) = guest.read(val, (len as usize).min(16)) else { + return errno::fail(errno::EFAULT); + }; + if len < 4 { + return errno::fail(errno::EINVAL); + } + let int = u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]); + let word = + |i: usize| raw.get(i..i + 8).map(|b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8]))); + sock::with(|t| { + let Some(s) = t.get_mut(id) else { + return errno::fail(errno::EBADF); + }; + let proto = s.proto; + let o = &mut s.opts; + match (level, name) { + (SOL_SOCKET, SO_REUSEADDR) => o.reuseaddr = int != 0, + (SOL_SOCKET, SO_REUSEPORT) => o.reuseport = int != 0, + (SOL_SOCKET, SO_KEEPALIVE) => o.keepalive = int != 0, + (SOL_SOCKET, SO_BROADCAST) => o.broadcast = int != 0, + (SOL_SOCKET, SO_RCVBUF) => o.rcvbuf = (int.min(BUF_MAX) * 2).max(RCVBUF_MIN), + (SOL_SOCKET, SO_SNDBUF) => o.sndbuf = (int.min(BUF_MAX) * 2).max(SNDBUF_MIN), + (SOL_SOCKET, SO_LINGER) if len >= 8 => { + o.linger = + (u32::from(int != 0), u32::from_le_bytes([raw[4], raw[5], raw[6], raw[7]])) + } + (SOL_SOCKET, SO_LINGER) => return errno::fail(errno::EINVAL), + (SOL_SOCKET, SO_RCVTIMEO) => return timeo(&mut o.rcvtimeo, word(0), word(8)), + (SOL_SOCKET, SO_SNDTIMEO) => return timeo(&mut o.sndtimeo, word(0), word(8)), + (IPPROTO_TCP, _) if proto == Proto::Dgram => return errno::fail(errno::ENOPROTOOPT), + (IPPROTO_TCP, TCP_NODELAY) => o.nodelay = int != 0, + (IPPROTO_TCP, TCP_KEEPIDLE) => return keep(&mut o.keepidle, int, KEEP_MAX), + (IPPROTO_TCP, TCP_KEEPINTVL) => return keep(&mut o.keepintvl, int, KEEP_MAX), + (IPPROTO_TCP, TCP_KEEPCNT) => return keep(&mut o.keepcnt, int, KEEPCNT_MAX), + // An AF_INET socket has no IPv6 options on Linux either. + (IPPROTO_IPV6, _) => return errno::fail(errno::ENOPROTOOPT), + _ => return super::opt_get::unknown("setsockopt", level, name), + } + errno::ok(0) + }) +} diff --git a/userland/capsule_linux/src/linux/net/opt_time.rs b/userland/capsule_linux/src/linux/net/opt_time.rs new file mode 100644 index 000000000..02712da02 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt_time.rs @@ -0,0 +1,42 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The options that hold a number of seconds: the keepalive times, and the +//! receive and send limits. + +use crate::linux::abi::errno; + +/// A keepalive time or count, 1 up to Linux's most, else EINVAL. +pub fn keep(slot: &mut u32, v: u32, most: u32) -> u64 { + if v < 1 || v > most { + return errno::fail(errno::EINVAL); + } + *slot = v; + errno::ok(0) +} + +/// SO_RCVTIMEO or SO_SNDTIMEO from a struct timeval: EDOM for microseconds +/// out of range, and a negative time is no limit, as Linux treats both. +pub fn timeo(slot: &mut (u64, u64), sec: Option, usec: Option) -> u64 { + let (Some(sec), Some(usec)) = (sec, usec) else { + return errno::fail(errno::EINVAL); + }; + if usec as i64 >= 1_000_000 || (usec as i64) < 0 { + return errno::fail(errno::EDOM); + } + *slot = if (sec as i64) < 0 { (0, 0) } else { (sec, usec) }; + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/net/opt_value.rs b/userland/capsule_linux/src/linux/net/opt_value.rs new file mode 100644 index 000000000..9e1e1e82a --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt_value.rs @@ -0,0 +1,61 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The value getsockopt reads for each option a socket has. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; + +use super::opt_ids::*; +use super::sock::{Domain, Proto, Sock}; + +pub fn value(s: &mut Sock, level: u64, name: u64) -> Result, u64> { + let o = s.opts; + let int = |v: u32| Ok(v.to_le_bytes().to_vec()); + let pair = |a: u64, b: u64| Ok([a.to_le_bytes(), b.to_le_bytes()].concat()); + let stream = s.proto == Proto::Stream; + match (level, name) { + (SOL_SOCKET, SO_TYPE) => int(if stream { 1 } else { 2 }), + (SOL_SOCKET, SO_DOMAIN) => int(if s.domain == Domain::Unix { 1 } else { 2 }), + (SOL_SOCKET, SO_PROTOCOL) => match (s.domain, stream) { + (Domain::Unix, _) => int(0), + (_, true) => int(6), + (_, false) => int(17), + }, + (SOL_SOCKET, SO_ACCEPTCONN) => int(u32::from(s.listening)), + (SOL_SOCKET, SO_ERROR) => int(core::mem::take(&mut s.error) as u32), + (SOL_SOCKET, SO_REUSEADDR) => int(u32::from(o.reuseaddr)), + (SOL_SOCKET, SO_REUSEPORT) => int(u32::from(o.reuseport)), + (SOL_SOCKET, SO_KEEPALIVE) => int(u32::from(o.keepalive)), + (SOL_SOCKET, SO_BROADCAST) => int(u32::from(o.broadcast)), + (SOL_SOCKET, SO_RCVBUF) => int(o.rcvbuf), + (SOL_SOCKET, SO_SNDBUF) => int(o.sndbuf), + (SOL_SOCKET, SO_LINGER) => { + Ok([o.linger.0.to_le_bytes(), o.linger.1.to_le_bytes()].concat()) + } + (SOL_SOCKET, SO_RCVTIMEO) => pair(o.rcvtimeo.0, o.rcvtimeo.1), + (SOL_SOCKET, SO_SNDTIMEO) => pair(o.sndtimeo.0, o.sndtimeo.1), + // A datagram socket has no TCP options, and Linux says so this way. + (IPPROTO_TCP, _) if !stream => Err(errno::fail(errno::EOPNOTSUPP)), + (IPPROTO_TCP, TCP_NODELAY) => int(u32::from(o.nodelay)), + (IPPROTO_TCP, TCP_KEEPIDLE) => int(o.keepidle), + (IPPROTO_TCP, TCP_KEEPINTVL) => int(o.keepintvl), + (IPPROTO_TCP, TCP_KEEPCNT) => int(o.keepcnt), + (IPPROTO_IPV6, _) => Err(errno::fail(errno::EOPNOTSUPP)), + _ => Err(super::opt_get::unknown("getsockopt", level, name)), + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/free.rs b/userland/capsule_linux/src/linux/net/sock/free.rs index 36f3b2cbb..3fcbb640b 100644 --- a/userland/capsule_linux/src/linux/net/sock/free.rs +++ b/userland/capsule_linux/src/linux/net/sock/free.rs @@ -34,8 +34,9 @@ impl Socks { } /// Close `id`. Its peer reads end of file, or ECONNRESET if this end - /// left bytes unread or `reset` is set, which is when Linux sends a reset - /// instead of a FIN. Connections still queued on a listener are reset. + /// left bytes unread, set SO_LINGER to zero seconds, or `reset` is set, + /// which is when Linux sends a reset instead of a FIN. Connections still + /// queued on a listener are reset. pub fn free(&mut self, id: u32, reset: bool) { let Some(gone) = self.list.get_mut(id as usize).and_then(Option::take) else { return; @@ -46,7 +47,7 @@ impl Socks { if let Some(p) = gone.peer.and_then(|p| self.get_mut(p)) { p.peer = None; p.eof = true; - if reset || !gone.rx.is_empty() { + if reset || !gone.rx.is_empty() || gone.opts.linger == (1, 0) { p.error = ECONNRESET; } } diff --git a/userland/capsule_linux/src/linux/net/sock/mod.rs b/userland/capsule_linux/src/linux/net/sock/mod.rs index c1a43ab31..fad03b903 100644 --- a/userland/capsule_linux/src/linux/net/sock/mod.rs +++ b/userland/capsule_linux/src/linux/net/sock/mod.rs @@ -43,4 +43,4 @@ pub use cell::with; pub use holders::Holder; pub use link::Link; pub use ready::bits; -pub use types::{Addr, Domain, Proto}; +pub use types::{Addr, Domain, Proto, Sock}; diff --git a/userland/capsule_linux/src/linux/net/sock/opts.rs b/userland/capsule_linux/src/linux/net/sock/opts.rs index 9fb73e378..5c73bea59 100644 --- a/userland/capsule_linux/src/linux/net/sock/opts.rs +++ b/userland/capsule_linux/src/linux/net/sock/opts.rs @@ -15,23 +15,52 @@ // along with this program. If not, see . //! The options a socket keeps, with the values Linux starts a socket with -//! (read from a Linux 6.x host: tcp_rmem[1] and rmem_default). +//! (read from a Linux 6.x host: tcp_rmem[1], tcp_wmem[1], rmem_default and +//! wmem_default, and the TCP keepalive defaults). use super::types::Proto; #[derive(Clone, Copy)] pub struct Opts { pub reuseaddr: bool, - /// As getsockopt reports it; also what the socket's queue holds. + pub reuseport: bool, + pub keepalive: bool, + pub broadcast: bool, + pub nodelay: bool, + /// As getsockopt reports them: Linux doubles what setsockopt was given. pub rcvbuf: u32, + pub sndbuf: u32, + pub keepidle: u32, + pub keepintvl: u32, + pub keepcnt: u32, + /// SO_LINGER's l_onoff and l_linger. + pub linger: (u32, u32), + /// SO_RCVTIMEO and SO_SNDTIMEO as the guest wrote them, seconds and + /// microseconds, so they read back exactly. + pub rcvtimeo: (u64, u64), + pub sndtimeo: (u64, u64), } impl Opts { pub fn new(proto: Proto) -> Opts { - let rcvbuf = match proto { - Proto::Stream => 131_072, - Proto::Dgram => 212_992, + let (rcvbuf, sndbuf) = match proto { + Proto::Stream => (131_072, 16_384), + Proto::Dgram => (212_992, 212_992), }; - Opts { reuseaddr: false, rcvbuf } + Opts { + reuseaddr: false, + reuseport: false, + keepalive: false, + broadcast: false, + nodelay: false, + rcvbuf, + sndbuf, + keepidle: 7200, + keepintvl: 75, + keepcnt: 9, + linger: (0, 0), + rcvtimeo: (0, 0), + sndtimeo: (0, 0), + } } } diff --git a/userland/capsule_linux/src/linux/serve/table_net.rs b/userland/capsule_linux/src/linux/serve/table_net.rs index fc2ca09e1..b8dd1fd1c 100644 --- a/userland/capsule_linux/src/linux/serve/table_net.rs +++ b/userland/capsule_linux/src/linux/serve/table_net.rs @@ -38,6 +38,8 @@ pub fn net_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option nr::ACCEPT4 => net::accept4(guest, a[0], a[1], a[2], a[3]), nr::GETSOCKNAME => net::getsockname(guest, a[0], a[1], a[2]), nr::GETPEERNAME => net::getpeername(guest, a[0], a[1], a[2]), + nr::SETSOCKOPT => net::setsockopt(guest, a[0], a[1], a[2], a[3], a[4]), + nr::GETSOCKOPT => net::getsockopt(guest, a[0], a[1], a[2], a[3], a[4]), nr::SENDTO => net::sendto(guest, a[0], a[1], a[2], a[3], a[4], a[5]), nr::RECVFROM => net::recvfrom(guest, a[0], a[1], a[2], a[3], a[4], a[5]), nr::SHUTDOWN => net::shutdown(guest, a[0], a[1]), From 201d9b877700a6d8ac55e115d8bcb91b2135ea03 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 21:55:30 +0000 Subject: [PATCH 06/14] linux: socket calls wait as on Linux; a family socket needs no tick A blocking accept on an empty listener, a blocking receive with nothing queued, and a blocking send to a full peer answered EAGAIN at once, since only read and write on a pipe or an eventfd could park; and any wait that watched a socket was looked at every 10 ms, because net.sockets does not say when its streams change. Now accept, accept4, connect, the sends and receives and read, write, readv and writev on a socket are routed to waits_sock (one arm in serve/dispatch.rs, one in waits::attempt). Each is tried at once; a blocking one that cannot finish is parked and tried again after every answer, and a non-blocking one or MSG_DONTWAIT answers EAGAIN. A blocking stream send returns once every byte is queued, a receive with MSG_WAITALL once its buffer is full, and recvmmsg once vlen messages have come unless MSG_WAITFORONE; the count so far is kept per thread between tries. SO_RCVTIMEO and SO_SNDTIMEO bound the wait, which then answers EAGAIN, or the count moved, as Linux does. A family socket changes only in an answer, after which every parked call is tried, so the 10 ms tick (family_waits::next_wait_ms) now runs only while a wait watches a stream net.sockets holds. --- .../capsule_linux/src/linux/net/call_kind.rs | 44 ++++++++++ userland/capsule_linux/src/linux/net/fd.rs | 10 +++ userland/capsule_linux/src/linux/net/flags.rs | 1 + userland/capsule_linux/src/linux/net/mod.rs | 7 ++ .../capsule_linux/src/linux/net/opt_time.rs | 10 ++- .../src/linux/net/poll_socket.rs | 6 ++ .../capsule_linux/src/linux/net/sock/mod.rs | 3 + .../capsule_linux/src/linux/net/sock/opts.rs | 6 ++ .../src/linux/net/sock/progress.rs | 35 ++++++++ .../capsule_linux/src/linux/net/sock/table.rs | 4 +- .../capsule_linux/src/linux/net/try_call.rs | 81 +++++++++++++++++++ .../capsule_linux/src/linux/serve/dispatch.rs | 3 + .../src/linux/serve/family_waits.rs | 12 ++- userland/capsule_linux/src/linux/serve/mod.rs | 2 + .../capsule_linux/src/linux/serve/waits.rs | 1 + .../src/linux/serve/waits_sock.rs | 81 +++++++++++++++++++ .../src/linux/serve/waits_sock_kind.rs | 48 +++++++++++ 17 files changed, 349 insertions(+), 5 deletions(-) create mode 100644 userland/capsule_linux/src/linux/net/call_kind.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/progress.rs create mode 100644 userland/capsule_linux/src/linux/net/try_call.rs create mode 100644 userland/capsule_linux/src/linux/serve/waits_sock.rs create mode 100644 userland/capsule_linux/src/linux/serve/waits_sock_kind.rs diff --git a/userland/capsule_linux/src/linux/net/call_kind.rs b/userland/capsule_linux/src/linux/net/call_kind.rs new file mode 100644 index 000000000..eb7214481 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/call_kind.rs @@ -0,0 +1,44 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What kind of wait a socket call makes: its flags, and whether a +//! blocking one waits until it has moved everything. + +use crate::linux::abi::nr; + +use super::flags::{MSG_DONTWAIT, MSG_WAITALL}; +use super::mmsg; + +/// The call's flags, where it has them. +pub fn flags(n: u64, a: [u64; 6]) -> u64 { + match n { + nr::RECVFROM | nr::SENDTO | nr::RECVMMSG | nr::SENDMMSG | nr::ACCEPT4 => a[3], + nr::RECVMSG | nr::SENDMSG => a[2], + _ => 0, + } +} + +/// True when a blocking call waits until it has moved everything it asked +/// for: a send on a stream, a receive with MSG_WAITALL, and recvmmsg +/// without MSG_WAITFORONE. +pub fn wants_all(stream: bool, n: u64, flags: u64) -> bool { + match n { + nr::WRITE | nr::WRITEV | nr::SENDTO | nr::SENDMSG => stream, + nr::RECVFROM | nr::RECVMSG => stream && flags & MSG_WAITALL != 0, + nr::RECVMMSG => flags & (mmsg::MSG_WAITFORONE | MSG_DONTWAIT) == 0, + _ => false, + } +} diff --git a/userland/capsule_linux/src/linux/net/fd.rs b/userland/capsule_linux/src/linux/net/fd.rs index 53bf72c5a..1da8a45e2 100644 --- a/userland/capsule_linux/src/linux/net/fd.rs +++ b/userland/capsule_linux/src/linux/net/fd.rs @@ -54,6 +54,16 @@ pub fn install(guest: &mut Guest, id: u32, flags: u64) -> u64 { } } +/// The socket `fd` names, if it names one. +pub fn sock_id(guest: &Guest, fd: u64) -> Option { + sock_of(guest, fd).ok() +} + +/// True for a stream socket. +pub fn is_stream(id: u32) -> bool { + sock::with(|t| t.get(id).is_some_and(|s| s.proto == sock::Proto::Stream)) +} + /// True when `fd` is non-blocking. pub fn nonblock(guest: &Guest, fd: u64) -> bool { guest.fds.get(fd as usize).is_some_and(|f| f.nonblock) diff --git a/userland/capsule_linux/src/linux/net/flags.rs b/userland/capsule_linux/src/linux/net/flags.rs index 5084c12ed..c077e41e7 100644 --- a/userland/capsule_linux/src/linux/net/flags.rs +++ b/userland/capsule_linux/src/linux/net/flags.rs @@ -20,3 +20,4 @@ pub const MSG_OOB: u64 = 0x1; pub const MSG_PEEK: u64 = 0x2; pub const MSG_TRUNC: u64 = 0x20; pub const MSG_DONTWAIT: u64 = 0x40; +pub const MSG_WAITALL: u64 = 0x100; diff --git a/userland/capsule_linux/src/linux/net/mod.rs b/userland/capsule_linux/src/linux/net/mod.rs index 32bd53e81..cd42fe43c 100644 --- a/userland/capsule_linux/src/linux/net/mod.rs +++ b/userland/capsule_linux/src/linux/net/mod.rs @@ -19,6 +19,7 @@ mod accept; mod bind; +mod call_kind; mod call; mod close; mod connect; @@ -60,25 +61,31 @@ mod sockaddr; mod sockaddr_out; mod socket; mod stream; +mod try_call; mod xfer_in; mod xfer_out; pub use accept::accept4; pub use bind::bind; +pub use call_kind::{flags as call_flags, wants_all}; pub use listen::listen; pub use close::close; pub use connect::connect; pub use dgram::{recvfrom, sendto}; +pub use fd::{is_stream, sock_id}; pub use mmsg::{recvmmsg, sendmmsg}; pub use msg::{recvmsg, sendmsg}; pub use name::{getpeername, getsockname}; pub use opt_get::getsockopt; pub use opt_set::setsockopt; +pub use opt_time::limit_ms; pub use pair::socketpair; pub use poll::{ready, POLLERR, POLLHUP}; pub use poll_set::poll; +pub use poll_socket::outside; pub use select::{clear as select_clear, select}; pub use shutdown::shutdown; pub use socket::socket; +pub use try_call::try_call; pub use xfer_in::read as recv; pub use xfer_out::write as send; diff --git a/userland/capsule_linux/src/linux/net/opt_time.rs b/userland/capsule_linux/src/linux/net/opt_time.rs index 02712da02..33fe6b1a4 100644 --- a/userland/capsule_linux/src/linux/net/opt_time.rs +++ b/userland/capsule_linux/src/linux/net/opt_time.rs @@ -15,10 +15,12 @@ // along with this program. If not, see . //! The options that hold a number of seconds: the keepalive times, and the -//! receive and send limits. +//! receive and send limits a wait keeps to. use crate::linux::abi::errno; +use super::sock::{self, Opts}; + /// A keepalive time or count, 1 up to Linux's most, else EINVAL. pub fn keep(slot: &mut u32, v: u32, most: u32) -> u64 { if v < 1 || v > most { @@ -40,3 +42,9 @@ pub fn timeo(slot: &mut (u64, u64), sec: Option, usec: Option) -> u64 *slot = if (sec as i64) < 0 { (0, 0) } else { (sec, usec) }; errno::ok(0) } + +/// The limit a receive (`read`) or a send waits for, from its option. +pub fn limit_ms(id: u32, read: bool) -> Option { + sock::with(|t| t.get(id).map(|s| if read { s.opts.rcvtimeo } else { s.opts.sndtimeo })) + .and_then(Opts::limit_ms) +} diff --git a/userland/capsule_linux/src/linux/net/poll_socket.rs b/userland/capsule_linux/src/linux/net/poll_socket.rs index 00a60e68f..bec59ec99 100644 --- a/userland/capsule_linux/src/linux/net/poll_socket.rs +++ b/userland/capsule_linux/src/linux/net/poll_socket.rs @@ -35,6 +35,12 @@ pub(super) fn socket_bits(id: u32) -> u16 { } } +/// True when `id` is a stream net.sockets holds: nothing tells the family +/// when it changes, so a wait on it is looked at again on a tick. +pub fn outside(id: u32) -> bool { + sock::with(|t| t.get(id).is_some_and(|s| s.svc.is_some())) +} + fn service_bits(handle: u32) -> u16 { let Some((0, out)) = call(OP_POLL, &handle.to_le_bytes(), 1) else { return 0; diff --git a/userland/capsule_linux/src/linux/net/sock/mod.rs b/userland/capsule_linux/src/linux/net/sock/mod.rs index fad03b903..3508a4fb3 100644 --- a/userland/capsule_linux/src/linux/net/sock/mod.rs +++ b/userland/capsule_linux/src/linux/net/sock/mod.rs @@ -33,6 +33,7 @@ mod link; mod new; mod opts; mod port; +mod progress; mod ready; mod recv; mod send; @@ -42,5 +43,7 @@ mod types; pub use cell::with; pub use holders::Holder; pub use link::Link; +pub use opts::Opts; +pub use progress::{progress, set_progress}; pub use ready::bits; pub use types::{Addr, Domain, Proto, Sock}; diff --git a/userland/capsule_linux/src/linux/net/sock/opts.rs b/userland/capsule_linux/src/linux/net/sock/opts.rs index 5c73bea59..36532a2ab 100644 --- a/userland/capsule_linux/src/linux/net/sock/opts.rs +++ b/userland/capsule_linux/src/linux/net/sock/opts.rs @@ -63,4 +63,10 @@ impl Opts { sndtimeo: (0, 0), } } + + /// Milliseconds a receive or a send may wait, None for no limit. + pub fn limit_ms((sec, usec): (u64, u64)) -> Option { + let ms = sec.saturating_mul(1000).saturating_add(usec.div_ceil(1000)); + (ms != 0).then_some(ms) + } } diff --git a/userland/capsule_linux/src/linux/net/sock/progress.rs b/userland/capsule_linux/src/linux/net/sock/progress.rs new file mode 100644 index 000000000..a0d9cc555 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/progress.rs @@ -0,0 +1,35 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! How far a waiting call has got. A blocking send on Linux returns once +//! every byte is queued, and a receive with MSG_WAITALL once its buffer is +//! full; a call parked partway keeps its count here, by thread, until it is +//! answered. + +use super::cell::with; + +pub fn progress(tid: u32) -> usize { + with(|t| t.progress.iter().find(|p| p.0 == tid).map_or(0, |p| p.1)) +} + +pub fn set_progress(tid: u32, done: usize) { + with(|t| { + t.progress.retain(|p| p.0 != tid); + if done != 0 { + t.progress.push((tid, done)); + } + }); +} diff --git a/userland/capsule_linux/src/linux/net/sock/table.rs b/userland/capsule_linux/src/linux/net/sock/table.rs index 057346762..09287eebf 100644 --- a/userland/capsule_linux/src/linux/net/sock/table.rs +++ b/userland/capsule_linux/src/linux/net/sock/table.rs @@ -24,11 +24,13 @@ pub struct Socks { pub(super) list: Vec>, /// Where the next ephemeral port search starts. pub(super) next_port: u16, + /// Waiting calls partway through, by thread (`progress`). + pub(super) progress: Vec<(u32, usize)>, } impl Socks { pub const fn new() -> Socks { - Socks { list: Vec::new(), next_port: 0 } + Socks { list: Vec::new(), next_port: 0, progress: Vec::new() } } /// A fresh socket held by `pid`, or by nobody yet when `pid` is None (the diff --git a/userland/capsule_linux/src/linux/net/try_call.rs b/userland/capsule_linux/src/linux/net/try_call.rs new file mode 100644 index 000000000..0af36208b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/try_call.rs @@ -0,0 +1,81 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! One try at a socket call that may wait. The serve loop's `waits_sock` +//! parks a blocking one that cannot finish and tries it again; `done` is +//! how much of it an earlier try moved (bytes, or messages for the mmsg +//! calls), and a try goes on from there. + +use alloc::vec; + +use crate::linux::abi::{errno, nr, nr_path as np}; +use crate::linux::guest::Guest; + +use super::{iov, mmsg}; + +/// The answer, and the whole amount the call asks to move. +pub fn try_call(guest: &mut Guest, n: u64, a: [u64; 6], done: usize) -> (u64, usize) { + let fd = a[0]; + let Ok(id) = super::fd::sock_of(guest, fd) else { + return (errno::fail(errno::EBADF), 0); + }; + let one = |buf: u64, len: u64| vec![(buf, len)]; + match n { + nr::READ => (bytes_in(guest, id, &one(a[1], a[2]), done, 0), a[2] as usize), + nr::WRITE => { + (super::xfer_out::send(guest, id, &one(a[1], a[2]), done, 0, None), a[2] as usize) + } + np::READV | nr::WRITEV => match iov::read(guest, a[1], a[2]) { + Ok(v) if n == nr::WRITEV => { + (super::xfer_out::send(guest, id, &v, done, 0, None), iov::total(&v)) + } + Ok(v) => (bytes_in(guest, id, &v, done, 0), iov::total(&v)), + Err(e) => (e, 0), + }, + nr::RECVFROM if done == 0 => { + (super::recvfrom(guest, fd, a[1], a[2], a[3], a[4], a[5]), a[2] as usize) + } + nr::RECVFROM => (bytes_in(guest, id, &one(a[1], a[2]), done, a[3]), a[2] as usize), + nr::SENDTO if done == 0 => { + (super::sendto(guest, fd, a[1], a[2], a[3], a[4], a[5]), a[2] as usize) + } + nr::SENDTO => { + (super::xfer_out::send(guest, id, &one(a[1], a[2]), done, a[3], None), a[2] as usize) + } + nr::SENDMSG => (super::msg::sendmsg(guest, fd, a[1], a[2], done), msg_len(guest, a[1])), + nr::RECVMSG => (super::msg::recvmsg(guest, fd, a[1], a[2], done), msg_len(guest, a[1])), + nr::SENDMMSG => (mmsg::sendmmsg(guest, a, done), a[2].min(mmsg::MOST) as usize), + nr::RECVMMSG => (mmsg::recvmmsg(guest, a, done), a[2].min(mmsg::MOST) as usize), + nr::ACCEPT => (super::accept::accept4(guest, fd, a[1], a[2], 0), 0), + nr::ACCEPT4 => (super::accept::accept4(guest, fd, a[1], a[2], a[3]), 0), + nr::CONNECT => (super::connect(guest, fd, a[1], a[2]), 0), + _ => (errno::fail(errno::ENOSYS), 0), + } +} + +fn bytes_in(guest: &Guest, id: u32, v: &iov::Iov, done: usize, flags: u64) -> u64 { + match super::xfer_in::recv(guest, id, v, done, flags) { + Ok(got) => errno::ok(got.n as u64), + Err(e) => e, + } +} + +fn msg_len(guest: &Guest, msg: u64) -> usize { + let word = |at: u64| { + guest.read(at, 8).map_or(0, |b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8]))) + }; + iov::read(guest, word(msg + 16), word(msg + 24)).map_or(0, |v| iov::total(&v)) +} diff --git a/userland/capsule_linux/src/linux/serve/dispatch.rs b/userland/capsule_linux/src/linux/serve/dispatch.rs index e292a5f27..3e59f6c87 100644 --- a/userland/capsule_linux/src/linux/serve/dispatch.rs +++ b/userland/capsule_linux/src/linux/serve/dispatch.rs @@ -58,6 +58,9 @@ fn route(guest: &mut Guest, frame: &ForeignFrame) -> Answer { np::CLOCK_NANOSLEEP => { crate::linux::call::clock_nanosleep(guest, frame.pid, a[0], a[1], a[2]) } + n if super::waits_sock::takes(guest, n, a[0]) => { + super::waits_sock::io(guest, frame.pid, n, a) + } nr::READ | nr::WRITE if super::waits::may_wait(guest, frame.nr, a[0]) => { super::waits::io(guest, frame.pid, frame.nr, a) } diff --git a/userland/capsule_linux/src/linux/serve/family_waits.rs b/userland/capsule_linux/src/linux/serve/family_waits.rs index 735f19a44..c3f667f79 100644 --- a/userland/capsule_linux/src/linux/serve/family_waits.rs +++ b/userland/capsule_linux/src/linux/serve/family_waits.rs @@ -25,10 +25,13 @@ use super::waits::{attempt, expire}; use super::waits_fds::watched; use crate::linux::call::now_ms; use crate::linux::guest::Kind; +use crate::linux::net::outside; const CLOCK_MONOTONIC: u64 = 1; -/// How often a wait on a socket is looked at again: its readiness changes -/// with no call for the family to answer. A timer is looked at when it fires. +/// How often a wait on a stream net.sockets holds is looked at again: its +/// readiness changes with no call for the family to answer. A family socket +/// changes only in an answer, after which every wait is tried, so it needs +/// none. A timer is looked at when it fires. const TICK_MS: u64 = 10; impl Family { @@ -73,9 +76,12 @@ impl Family { if let Some(d) = wait.deadline { keep(d.saturating_sub(now)); } + if super::waits_sock::ticks(g, wait) { + keep(TICK_MS); + } for fd in watched(g, wait) { match g.fds.get(fd as usize) { - Some(f) if f.kind == Kind::Socket => keep(TICK_MS), + Some(f) if f.kind == Kind::Socket && outside(f.handle) => keep(TICK_MS), Some(f) if f.kind == Kind::Timer => { // One that has already fired was seen by the last look. let due = self.timers.get(f.handle as usize).map_or(0, |t| t.due); diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index 700f5b2dd..324eed45f 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -40,6 +40,8 @@ mod tally; mod unserved; mod waits; mod waits_fds; +mod waits_sock; +mod waits_sock_kind; mod waits_time; pub use answer::Answer; diff --git a/userland/capsule_linux/src/linux/serve/waits.rs b/userland/capsule_linux/src/linux/serve/waits.rs index 6de3c6d7a..0361528ce 100644 --- a/userland/capsule_linux/src/linux/serve/waits.rs +++ b/userland/capsule_linux/src/linux/serve/waits.rs @@ -79,6 +79,7 @@ pub fn attempt(guest: &mut Guest, wait: &Blocked) -> Option { let a = wait.args; let again = errno::fail(errno::EAGAIN); match wait.nr { + n if super::waits_sock::takes(guest, n, a[0]) => super::waits_sock::attempt(guest, wait), nr::READ => Some(call::read(guest, a[0], a[1], a[2])).filter(|&v| v != again), nr::WRITE => Some(call::write(guest, a[0], a[1], a[2])).filter(|&v| v != again), nr::POLL | np::PPOLL => Some(net::poll(guest, a[0], a[1])).filter(|&v| v != 0), diff --git a/userland/capsule_linux/src/linux/serve/waits_sock.rs b/userland/capsule_linux/src/linux/serve/waits_sock.rs new file mode 100644 index 000000000..5991c2bd6 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/waits_sock.rs @@ -0,0 +1,81 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Socket calls that wait: accept, connect, the sends and receives, and a +//! read or write on a socket. Each is tried at once; a blocking one that +//! cannot finish is parked, and the family tries it again after every +//! answer (`family_waits`). A family socket changes only in an answer, so +//! it needs no tick; a stream net.sockets holds is looked at on one. +//! +//! SO_RCVTIMEO and SO_SNDTIMEO bound the wait, which then answers EAGAIN, +//! or the count moved so far, as Linux does. + +use crate::linux::abi::{errno, nr}; +use crate::linux::call::now_ms; +use crate::linux::guest::{Blocked, Guest}; +use crate::linux::net::{self, sock}; + +use super::answer::Answer; +pub use super::waits_sock_kind::{takes, ticks}; + +const CLOCK_MONOTONIC: u64 = 1; +const MSG_DONTWAIT: u64 = 0x40; + +pub fn io(guest: &mut Guest, tid: u32, n: u64, a: [u64; 6]) -> Answer { + let reads = !matches!( + n, + nr::WRITE | nr::WRITEV | nr::SENDTO | nr::SENDMSG | nr::SENDMMSG | nr::CONNECT + ); + let limit = net::sock_id(guest, a[0]).and_then(|id| net::limit_ms(id, reads)); + let deadline = limit.map(|ms| now_ms(CLOCK_MONOTONIC).unwrap_or(0).saturating_add(ms)); + let wait = Blocked { tid, nr: n, args: a, deadline }; + match attempt(guest, &wait) { + Some(v) => Answer::value(v), + None => { + guest.blocked.push(wait); + Answer::Park + } + } +} + +/// The call's answer if it can give one now, None to go on waiting. +pub fn attempt(guest: &mut Guest, wait: &Blocked) -> Option { + let (n, a, tid) = (wait.nr, wait.args, wait.tid); + let done = sock::progress(tid); + let (value, whole) = net::try_call(guest, n, a, done); + let flags = net::call_flags(n, a); + let blocking = + flags & MSG_DONTWAIT == 0 && !guest.fds.get(a[0] as usize).is_some_and(|f| f.nonblock); + let late = wait.deadline.is_some_and(|d| now_ms(CLOCK_MONOTONIC).is_some_and(|now| d <= now)); + let answer = match errno::slot(value) { + Some(moved) => { + let total = done + moved; + let stream = net::sock_id(guest, a[0]).is_some_and(net::is_stream); + if blocking && moved != 0 && total < whole && !late && net::wants_all(stream, n, flags) + { + sock::set_progress(tid, total); + return None; + } + total as u64 + } + None if value == errno::fail(errno::EAGAIN) && blocking && !late => return None, + // What moved before an error or the time limit is what Linux answers. + None if done != 0 => done as u64, + None => value, + }; + sock::set_progress(tid, 0); + Some(answer) +} diff --git a/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs b/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs new file mode 100644 index 000000000..6e1beafa0 --- /dev/null +++ b/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs @@ -0,0 +1,48 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which calls `waits_sock` takes, and which of its waits need a tick. + +use crate::linux::abi::{nr, nr_path as np}; +use crate::linux::guest::{Blocked, Guest, Kind}; +use crate::linux::net; + +/// True for a call on a socket descriptor that may have to wait. +pub fn takes(guest: &Guest, n: u64, fd: u64) -> bool { + matches!( + n, + nr::READ + | nr::WRITE + | np::READV + | nr::WRITEV + | nr::ACCEPT + | nr::ACCEPT4 + | nr::CONNECT + | nr::SENDTO + | nr::RECVFROM + | nr::SENDMSG + | nr::RECVMSG + | nr::SENDMMSG + | nr::RECVMMSG + ) && guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Socket) +} + +/// True when a parked call waits on a stream net.sockets holds, which +/// nothing but a look tells the family has changed. +pub fn ticks(guest: &Guest, wait: &Blocked) -> bool { + takes(guest, wait.nr, wait.args[0]) + && net::sock_id(guest, wait.args[0]).is_some_and(net::outside) +} From 67ea756875e79018067b2e9584df70cc5480cc0d Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 20:07:18 +0000 Subject: [PATCH 07/14] linux guests: csock, cudp, cpolicy, cidle and gohttp Nothing here proved a guest's sockets, so none of what the family's socket table does could be shown or held to Linux. Five guests, ids 5000 to 5009, each built static and run on a Linux host first as the oracle: - csock (C), fifteen parts, each named when it fails: socketpair both ways, a listener's name and accept4's flags, a non-blocking connect, a refused port, half-close, epoll on a listener, end of file, EAGAIN, MSG_PEEK, EPIPE, an accept and a receive that wait for another thread, the options a server sets, and a socket a forked child shares. Linux: "[C] csock PASS: 15 parts". - cudp (C), six parts: an echo, a connected socket that drops a stranger's datagram, MSG_TRUNC, a refused port, sendmmsg and recvmmsg, and no peer. Linux: "[C] cudp PASS: 6 parts". - cpolicy (C): the answers the capsule's policy gives, which an unconfined Linux does not (its line records what Linux allows), and what descriptor numbers the guest never opened get. - cidle (C): accept blocked while nothing happens for the seconds given, to measure what the serve loop does meanwhile. - gohttp (Go): net/http, a server on 127.0.0.1:0 and its client in one guest, twenty GETs over one kept-alive connection. Linux: "[GO] gohttp PASS: 20 GETs answered 200 over 1 connection". --- userland/linux_guests/Guests.mk | 27 ++++++ userland/linux_guests/c/cidle.c | 54 +++++++++++ userland/linux_guests/c/cpolicy.c | 66 +++++++++++++ userland/linux_guests/c/csock.c | 127 +++++++++++++++++++++++++ userland/linux_guests/c/csock_parts.h | 90 ++++++++++++++++++ userland/linux_guests/c/csock_parts2.h | 118 +++++++++++++++++++++++ userland/linux_guests/c/csock_parts3.h | 113 ++++++++++++++++++++++ userland/linux_guests/c/csock_parts4.h | 88 +++++++++++++++++ userland/linux_guests/c/cudp.c | 108 +++++++++++++++++++++ userland/linux_guests/c/cudp_parts.h | 82 ++++++++++++++++ userland/linux_guests/go/http/go.mod | 3 + userland/linux_guests/go/http/main.go | 54 +++++++++++ 12 files changed, 930 insertions(+) create mode 100644 userland/linux_guests/c/cidle.c create mode 100644 userland/linux_guests/c/cpolicy.c create mode 100644 userland/linux_guests/c/csock.c create mode 100644 userland/linux_guests/c/csock_parts.h create mode 100644 userland/linux_guests/c/csock_parts2.h create mode 100644 userland/linux_guests/c/csock_parts3.h create mode 100644 userland/linux_guests/c/csock_parts4.h create mode 100644 userland/linux_guests/c/cudp.c create mode 100644 userland/linux_guests/c/cudp_parts.h create mode 100644 userland/linux_guests/go/http/go.mod create mode 100644 userland/linux_guests/go/http/main.go diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index 90fed03f3..f4d3b90a4 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -107,6 +107,9 @@ $(eval $(call LINUX_GUEST,gopoll,4976,4977,$(GO_OUT)/poll)) # A goroutine spinning with no call, which only a signal to its running # thread can move off the one CPU the guest has. $(eval $(call LINUX_GUEST,gopreempt,4944,4945,$(GO_OUT)/preempt)) +# net/http inside one guest: a server on 127.0.0.1:0 and its client, twenty +# GETs over one kept-alive connection. +$(eval $(call LINUX_GUEST,gohttp,5002,5003,$(GO_OUT)/http)) # A C guest that faults in a worker thread while main joins: it proves the # whole process ends, as on Linux, and that musl threads run. Static, so no @@ -127,6 +130,30 @@ $(LINUX_GUESTS_C)/cwait: $(LINUX_GUESTS_DIR)/c/cwait.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait)) +# Sockets as Linux has them, on the family's own loopback: socketpair, a +# listener with accept4's flags, a non-blocking connect, a refused port, +# half-close, epoll on a listener, end of file, EAGAIN, MSG_PEEK, EPIPE, an +# accept and a receive that wait, the options a server sets, and fork. +$(LINUX_GUESTS_C)/csock: $(LINUX_GUESTS_DIR)/c/csock.c $(wildcard $(LINUX_GUESTS_DIR)/c/csock_parts*.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,csock,5000,5001,$(LINUX_GUESTS_C)/csock)) + +# Datagrams on the family's loopback: an echo, a connected socket, MSG_TRUNC, +# a refused port, sendmmsg and recvmmsg, and no peer at all. +$(LINUX_GUESTS_C)/cudp: $(LINUX_GUESTS_DIR)/c/cudp.c $(LINUX_GUESTS_DIR)/c/cudp_parts.h + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,cudp,5004,5005,$(LINUX_GUESTS_C)/cudp)) + +# What a guest's sockets may reach, and what a descriptor number alone gets. +$(LINUX_GUESTS_C)/cpolicy: $(LINUX_GUESTS_DIR)/c/cpolicy.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,cpolicy,5006,5007,$(LINUX_GUESTS_C)/cpolicy)) + +# A guest blocked in accept with nothing happening, for the loop's wakeups. +$(LINUX_GUESTS_C)/cidle: $(LINUX_GUESTS_DIR)/c/cidle.c + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,cidle,5008,5009,$(LINUX_GUESTS_C)/cidle)) + # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the # normal image, which does not set NONOS_LINUX_GUESTS, still ships them. diff --git a/userland/linux_guests/c/cidle.c b/userland/linux_guests/c/cidle.c new file mode 100644 index 000000000..3110b7772 --- /dev/null +++ b/userland/linux_guests/c/cidle.c @@ -0,0 +1,54 @@ +// An idle guest: main blocks in accept on 127.0.0.1 while nothing happens +// for the number of seconds given (default 10), then a thread connects. It +// is what the serve loop's wakeups are measured against: a family socket +// changes only in an answer, so the loop has nothing to look at meanwhile. +#include +#include +#include +#include +#include +#include +#include +#include + +static struct sockaddr_in addr; +static int idle_s = 10; + +static long now_ms(void) { + struct timespec ts; + clock_gettime(CLOCK_MONOTONIC, &ts); + return ts.tv_sec * 1000 + ts.tv_nsec / 1000000; +} + +static void *late(void *arg) { + (void)arg; + struct timespec ts = {idle_s, 0}; + nanosleep(&ts, 0); + int c = socket(AF_INET, SOCK_STREAM, 0); + connect(c, (void *)&addr, sizeof addr); + return (void *)(long)c; +} + +int main(int argc, char **argv) { + if (argc > 1) { + idle_s = atoi(argv[1]); + } + int l = socket(AF_INET, SOCK_STREAM, 0); + memset(&addr, 0, sizeof addr); + addr.sin_family = AF_INET; + addr.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + socklen_t len = sizeof addr; + bind(l, (void *)&addr, sizeof addr); + getsockname(l, (void *)&addr, &len); + listen(l, 1); + pthread_t t; + long t0 = now_ms(); + pthread_create(&t, 0, late, 0); + int s = accept(l, 0, 0); + long waited = now_ms() - t0; + pthread_join(t, 0); + printf("[C] cidle %s: accept blocked %ld ms for a connect after %d s\n", + s >= 0 && waited >= idle_s * 1000 ? "PASS" : "FAIL", waited, idle_s); + fflush(stdout); + return s >= 0 ? 0 : 1; +} diff --git a/userland/linux_guests/c/cpolicy.c b/userland/linux_guests/c/cpolicy.c new file mode 100644 index 000000000..bf6d32fc0 --- /dev/null +++ b/userland/linux_guests/c/cpolicy.c @@ -0,0 +1,66 @@ +// What a guest's sockets may reach, and what a descriptor number alone gets. +// Each part prints the errno it got; the NONOS answer is the capsule's policy +// and differs from an unconfined Linux by design, so the host's line records +// what Linux itself would allow. Parts: +// bind-any bind 0.0.0.0:0 NONOS EACCES, Linux 0 +// bind-out bind 10.0.2.15:0 NONOS EACCES, Linux EADDRNOTAVAIL +// listen-unbound listen with no bind NONOS EACCES, Linux 0 +// udp-out sendto 192.0.2.1:9 NONOS ENETUNREACH, Linux 1 +// raw socket(SOCK_RAW) NONOS EPERM, Linux EPERM unless root +// forged recv/close on numbers the guest never opened, and on a +// pipe: EBADF, EBADF, ENOTSOCK on both +#include +#include +#include +#include +#include +#include + +static struct sockaddr_in at(unsigned ip, int port) { + struct sockaddr_in sa; + memset(&sa, 0, sizeof sa); + sa.sin_family = AF_INET; + sa.sin_port = htons(port); + sa.sin_addr.s_addr = htonl(ip); + return sa; +} + +static int err(int rc) { + return rc < 0 ? errno : 0; +} + +int main(void) { + int s = socket(AF_INET, SOCK_STREAM, 0); + struct sockaddr_in any = at(0, 0), out = at(0x0a00020f, 0), far = at(0xc0000201, 9); + int bind_any = err(bind(s, (void *)&any, sizeof any)); + close(s); + s = socket(AF_INET, SOCK_STREAM, 0); + int bind_out = err(bind(s, (void *)&out, sizeof out)); + close(s); + s = socket(AF_INET, SOCK_STREAM, 0); + int listen_unbound = err(listen(s, 1)); + close(s); + int u = socket(AF_INET, SOCK_DGRAM, 0); + int udp_out = sendto(u, "x", 1, 0, (void *)&far, sizeof far) == 1 ? 0 : errno; + close(u); + int raw = err(socket(AF_INET, SOCK_RAW, IPPROTO_ICMP)); + char buf[4]; + int pipe_fds[2]; + pipe(pipe_fds); + int forged_recv = err(recv(777, buf, 4, MSG_DONTWAIT)); + int forged_close = err(close(778)); + int pipe_recv = err(recv(pipe_fds[0], buf, 4, MSG_DONTWAIT)); + int pipe_name = err(getsockname(pipe_fds[0], (void *)&any, &(socklen_t){sizeof any})); + printf("[C] cpolicy bind-any %d bind-out %d listen-unbound %d udp-out %d raw %d " + "forged %d %d pipe %d %d\n", + bind_any, bind_out, listen_unbound, udp_out, raw, forged_recv, forged_close, + pipe_recv, pipe_name); + int confined = bind_any == EACCES && bind_out == EACCES && listen_unbound == EACCES && + udp_out == ENETUNREACH && raw == EPERM; + int forged = forged_recv == EBADF && forged_close == EBADF && pipe_recv == ENOTSOCK && + pipe_name == ENOTSOCK; + printf("[C] cpolicy %s: confined %d, forged numbers refused %d\n", + confined && forged ? "PASS" : "FAIL", confined, forged); + fflush(stdout); + return confined && forged ? 0 : 1; +} diff --git a/userland/linux_guests/c/csock.c b/userland/linux_guests/c/csock.c new file mode 100644 index 000000000..98e9297a3 --- /dev/null +++ b/userland/linux_guests/c/csock.c @@ -0,0 +1,127 @@ +// Sockets, as Linux has them: a socketpair each way, a listener on 127.0.0.1 +// with its name and accept4's flags, a non-blocking connect that answers +// EINPROGRESS and then SO_ERROR 0, a refused port, shutdown(SHUT_WR) giving +// the peer end of file while the other way still flows, epoll readiness on a +// listener, end of file, EAGAIN on an empty non-blocking receive, MSG_PEEK +// and MSG_DONTWAIT, EPIPE after the peer is gone, an accept and a receive +// that wait for another thread, the options Go and a C server set, and a +// socket a forked child shares. Each part prints as it passes and every part +// runs, so one run names each part that fails. +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +static int parts; + +static long now_ms(void) { + struct timespec ts; + clock_gettime(CLOCK_MONOTONIC, &ts); + return ts.tv_sec * 1000 + ts.tv_nsec / 1000000; +} + +static void nap_ms(long ms) { + struct timespec ts = {ms / 1000, (ms % 1000) * 1000000}; + nanosleep(&ts, 0); +} + +static int fail(const char *what, long a, long b) { + printf("[C] csock FAIL: %s (%ld, %ld)\n", what, a, b); + fflush(stdout); + return 1; +} + +static void ok(const char *part, const char *detail, long n) { + parts++; + printf("[C] csock %s ok: %s %ld\n", part, detail, n); + fflush(stdout); +} + +static struct sockaddr_in loopback(int port) { + struct sockaddr_in sa; + memset(&sa, 0, sizeof sa); + sa.sin_family = AF_INET; + sa.sin_port = htons(port); + sa.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + return sa; +} + +// A listener on 127.0.0.1 at a port the kernel picks; its port through *port. +static int listener(int *port, int flags) { + int s = socket(AF_INET, SOCK_STREAM | flags, 0); + if (s < 0) { + return -errno; + } + int one = 1; + setsockopt(s, SOL_SOCKET, SO_REUSEADDR, &one, sizeof one); + struct sockaddr_in sa = loopback(0); + socklen_t len = sizeof sa; + if (bind(s, (void *)&sa, sizeof sa) || getsockname(s, (void *)&sa, &len) || listen(s, 8)) { + int e = errno; + close(s); + return -e; + } + *port = ntohs(sa.sin_port); + return s; +} + +static int dial(int port) { + int c = socket(AF_INET, SOCK_STREAM, 0); + struct sockaddr_in sa = loopback(port); + if (c < 0 || connect(c, (void *)&sa, sizeof sa)) { + int e = errno; + if (c >= 0) { + close(c); + } + return -e; + } + return c; +} + +// A connected pair over 127.0.0.1: *a is the client, *b the accepted end. +static int tcp_pair(int *a, int *b) { + int port, l = listener(&port, 0); + if (l < 0) { + return l; + } + *a = dial(port); + *b = *a < 0 ? -1 : accept(l, 0, 0); + close(l); + return *a < 0 ? *a : *b < 0 ? -errno : 0; +} + +#include "csock_parts.h" + +int main(void) { + signal(SIGPIPE, SIG_IGN); + int (*const part[])(void) = { + pair_stream, pair_dgram, listen_accept, nb_connect, refused, half_close, + epoll_listener, eof, empty_recv, peek, epipe, blocking_accept, + blocking_recv, options, fork_share, + }; + const int count = sizeof part / sizeof part[0]; + long t0 = now_ms(); + int failed = 0; + for (int i = 0; i < count; i++) { + failed += part[i](); + } + if (failed) { + printf("[C] csock FAIL: %d parts failed, %d passed\n", failed, parts); + fflush(stdout); + return 1; + } + printf("[C] csock PASS: %d parts in %ld ms\n", parts, now_ms() - t0); + fflush(stdout); + return 0; +} diff --git a/userland/linux_guests/c/csock_parts.h b/userland/linux_guests/c/csock_parts.h new file mode 100644 index 000000000..b624ec049 --- /dev/null +++ b/userland/linux_guests/c/csock_parts.h @@ -0,0 +1,90 @@ +// The parts of csock, one Linux behaviour each. Included once, by csock.c. + +static int pair_stream(void) { + int sv[2]; + char buf[8] = {0}; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv)) { + return fail("pair_stream: socketpair", -1, errno); + } + if (write(sv[0], "ping", 4) != 4 || read(sv[1], buf, 8) != 4 || memcmp(buf, "ping", 4)) { + return fail("pair_stream: a to b", 0, errno); + } + if (write(sv[1], "pong!", 5) != 5 || read(sv[0], buf, 8) != 5 || memcmp(buf, "pong!", 5)) { + return fail("pair_stream: b to a", 0, errno); + } + close(sv[0]); + long got = read(sv[1], buf, 8); + close(sv[1]); + if (got != 0) { + return fail("pair_stream: end of file after close", got, errno); + } + ok("pair_stream", "bytes each way, then eof", 9); + return 0; +} + +static int pair_dgram(void) { + int sv[2]; + char buf[16]; + if (socketpair(AF_UNIX, SOCK_DGRAM, 0, sv)) { + return fail("pair_dgram: socketpair", -1, errno); + } + write(sv[0], "one", 3); + write(sv[0], "second", 6); + long a = read(sv[1], buf, sizeof buf); + long b = read(sv[1], buf, sizeof buf); + close(sv[0]); + close(sv[1]); + if (a != 3 || b != 6) { + return fail("pair_dgram: boundaries kept", a, b); + } + ok("pair_dgram", "two datagrams, sizes 3 and", b); + return 0; +} + +static int listen_accept(void) { + int port, l = listener(&port, SOCK_NONBLOCK); + if (l < 0) { + return fail("listen_accept: listener", l, 0); + } + if (port == 0) { + return fail("listen_accept: getsockname gave port 0", 0, 0); + } + if (accept4(l, 0, 0, 0) != -1 || errno != EAGAIN) { + return fail("listen_accept: empty non-blocking accept is EAGAIN", errno, EAGAIN); + } + int c = dial(port); + if (c < 0) { + return fail("listen_accept: connect", c, 0); + } + struct sockaddr_in peer, mine; + socklen_t plen = sizeof peer, mlen = sizeof mine; + int s = accept4(l, (void *)&peer, &plen, SOCK_NONBLOCK | SOCK_CLOEXEC); + if (s < 0) { + return fail("listen_accept: accept4", -1, errno); + } + if (!(fcntl(s, F_GETFL) & O_NONBLOCK) || !(fcntl(s, F_GETFD) & FD_CLOEXEC)) { + return fail("listen_accept: accept4 flags", fcntl(s, F_GETFL), fcntl(s, F_GETFD)); + } + getsockname(c, (void *)&mine, &mlen); + if (peer.sin_port != mine.sin_port || peer.sin_addr.s_addr != htonl(INADDR_LOOPBACK)) { + return fail("listen_accept: accept's address is the client's", ntohs(peer.sin_port), + ntohs(mine.sin_port)); + } + struct sockaddr_in back; + socklen_t blen = sizeof back; + getpeername(c, (void *)&back, &blen); + if (ntohs(back.sin_port) != port || blen != sizeof back) { + return fail("listen_accept: getpeername", ntohs(back.sin_port), port); + } + char buf[8]; + if (write(c, "hi", 2) != 2 || read(s, buf, 8) != 2) { + return fail("listen_accept: bytes", 0, errno); + } + close(c); + close(s); + close(l); + ok("listen_accept", "accept4 NONBLOCK|CLOEXEC on port", port > 0); + return 0; +} + +#include "csock_parts2.h" diff --git a/userland/linux_guests/c/csock_parts2.h b/userland/linux_guests/c/csock_parts2.h new file mode 100644 index 000000000..de90a45c0 --- /dev/null +++ b/userland/linux_guests/c/csock_parts2.h @@ -0,0 +1,118 @@ +// csock: connecting, and closing one way or both. + +static int nb_connect(void) { + int port, l = listener(&port, 0); + int c = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + struct sockaddr_in sa = loopback(port); + int rc = connect(c, (void *)&sa, sizeof sa); + int e = errno; + if (rc != -1 || e != EINPROGRESS) { + return fail("nb_connect: EINPROGRESS", rc, e); + } + struct pollfd p = {c, POLLOUT, 0}; + if (poll(&p, 1, 5000) != 1 || !(p.revents & POLLOUT)) { + return fail("nb_connect: POLLOUT", p.revents, 0); + } + int err = -1; + socklen_t len = sizeof err; + if (getsockopt(c, SOL_SOCKET, SO_ERROR, &err, &len) || err != 0) { + return fail("nb_connect: SO_ERROR", err, errno); + } + close(c); + close(l); + ok("nb_connect", "EINPROGRESS, POLLOUT, SO_ERROR", err); + return 0; +} + +// A port with no listener: a listener is opened and closed to find one. +static int refused(void) { + int port, l = listener(&port, 0); + close(l); + int c = dial(port); + if (c != -ECONNREFUSED) { + return fail("refused: blocking connect", c, -ECONNREFUSED); + } + int n = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + struct sockaddr_in sa = loopback(port); + int rc = connect(n, (void *)&sa, sizeof sa); + int e = errno; + struct pollfd p = {n, POLLOUT, 0}; + poll(&p, 1, 5000); + int err = 0; + socklen_t len = sizeof err; + getsockopt(n, SOL_SOCKET, SO_ERROR, &err, &len); + close(n); + if (rc != -1 || e != EINPROGRESS || err != ECONNREFUSED || !(p.revents & POLLERR)) { + return fail("refused: non-blocking connect then SO_ERROR", e, err); + } + ok("refused", "ECONNREFUSED both ways, errno", ECONNREFUSED); + return 0; +} + +static int half_close(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("half_close: pair", 0, errno); + } + if (shutdown(a, SHUT_WR)) { + return fail("half_close: shutdown", -1, errno); + } + long got = read(b, buf, 8); + if (got != 0) { + return fail("half_close: peer reads end of file", got, errno); + } + if (write(b, "back", 4) != 4 || read(a, buf, 8) != 4) { + return fail("half_close: the other way still flows", 0, errno); + } + if (write(a, "x", 1) != -1 || errno != EPIPE) { + return fail("half_close: write after SHUT_WR is EPIPE", errno, EPIPE); + } + if (fcntl(a, F_GETFD) < 0) { + return fail("half_close: shutdown closed the descriptor", errno, 0); + } + close(a); + close(b); + ok("half_close", "eof one way, 4 bytes back", 4); + return 0; +} + +static int epoll_listener(void) { + int port, l = listener(&port, SOCK_NONBLOCK); + int ep = epoll_create1(0); + struct epoll_event ev = {.events = EPOLLIN, .data.fd = l}, out; + epoll_ctl(ep, EPOLL_CTL_ADD, l, &ev); + int before = epoll_wait(ep, &out, 1, 0); + int c = dial(port); + int after = epoll_wait(ep, &out, 1, 5000); + int s = accept(l, 0, 0); + int drained = epoll_wait(ep, &out, 1, 0); + close(s); + close(c); + close(ep); + close(l); + if (before != 0 || after != 1 || !(out.events & EPOLLIN) || s < 0 || drained != 0) { + return fail("epoll_listener: EPOLLIN only while pending", before * 10 + after, drained); + } + ok("epoll_listener", "EPOLLIN once pending, then", drained); + return 0; +} + +static int eof(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("eof: pair", 0, errno); + } + write(a, "last", 4); + close(a); + long first = read(b, buf, 8), then = read(b, buf, 8); + close(b); + if (first != 4 || then != 0) { + return fail("eof: data then end of file", first, then); + } + ok("eof", "4 bytes then", then); + return 0; +} + +#include "csock_parts3.h" diff --git a/userland/linux_guests/c/csock_parts3.h b/userland/linux_guests/c/csock_parts3.h new file mode 100644 index 000000000..bcbc5dbff --- /dev/null +++ b/userland/linux_guests/c/csock_parts3.h @@ -0,0 +1,113 @@ +// csock: receiving without waiting, waiting, options, and fork. + +static int empty_recv(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("empty_recv: pair", 0, errno); + } + if (recv(b, buf, 8, MSG_DONTWAIT) != -1 || errno != EAGAIN) { + return fail("empty_recv: MSG_DONTWAIT", errno, EAGAIN); + } + fcntl(b, F_SETFL, O_NONBLOCK); + if (read(b, buf, 8) != -1 || errno != EAGAIN) { + return fail("empty_recv: O_NONBLOCK read", errno, EAGAIN); + } + close(a); + close(b); + ok("empty_recv", "EAGAIN, errno", EAGAIN); + return 0; +} + +static int peek(void) { + int a, b; + char buf[8] = {0}; + if (tcp_pair(&a, &b)) { + return fail("peek: pair", 0, errno); + } + write(a, "abc", 3); + long p = recv(b, buf, 8, MSG_PEEK); + long r = recv(b, buf, 8, 0); + close(a); + close(b); + if (p != 3 || r != 3 || memcmp(buf, "abc", 3)) { + return fail("peek: data stays", p, r); + } + ok("peek", "peeked then read", r); + return 0; +} + +static int epipe(void) { + int a, b; + if (tcp_pair(&a, &b)) { + return fail("epipe: pair", 0, errno); + } + close(b); + long first = send(a, "x", 1, MSG_NOSIGNAL); + long second = send(a, "x", 1, MSG_NOSIGNAL); + int e = errno; + close(a); + if (first != 1 || second != -1 || e != EPIPE) { + return fail("epipe: first send taken, then EPIPE", first, e); + } + ok("epipe", "second send errno", e); + return 0; +} + +static int late_port; +static void *late_dial(void *arg) { + (void)arg; + nap_ms(100); + return (void *)(long)dial(late_port); +} + +static int blocking_accept(void) { + int l = listener(&late_port, 0); + pthread_t t; + long t0 = now_ms(); + pthread_create(&t, 0, late_dial, 0); + int s = accept(l, 0, 0); + long waited = now_ms() - t0; + void *c; + pthread_join(t, &c); + close((int)(long)c); + close(s); + close(l); + if (s < 0 || waited < 80) { + return fail("blocking_accept: waited for the connect", s, waited); + } + ok("blocking_accept", "waited ms", waited >= 80); + return 0; +} + +static int late_fd; +static void *late_write(void *arg) { + (void)arg; + nap_ms(100); + write(late_fd, "late", 4); + return 0; +} + +static int blocking_recv(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("blocking_recv: pair", 0, errno); + } + late_fd = a; + pthread_t t; + long t0 = now_ms(); + pthread_create(&t, 0, late_write, 0); + long got = recv(b, buf, 8, 0); + long waited = now_ms() - t0; + pthread_join(t, 0); + close(a); + close(b); + if (got != 4 || waited < 80) { + return fail("blocking_recv: waited for the bytes", got, waited); + } + ok("blocking_recv", "4 bytes after waiting", waited >= 80); + return 0; +} + +#include "csock_parts4.h" diff --git a/userland/linux_guests/c/csock_parts4.h b/userland/linux_guests/c/csock_parts4.h new file mode 100644 index 000000000..6d37228e3 --- /dev/null +++ b/userland/linux_guests/c/csock_parts4.h @@ -0,0 +1,88 @@ +// csock: the options a server sets, and a socket a forked child shares. + +static int get_int(int s, int level, int opt) { + int v = -1; + socklen_t len = sizeof v; + return getsockopt(s, level, opt, &v, &len) ? -errno : v; +} + +static int options(void) { + int port, l = listener(&port, 0); + int one = 1; + if (get_int(l, SOL_SOCKET, SO_TYPE) != SOCK_STREAM || + get_int(l, SOL_SOCKET, SO_DOMAIN) != AF_INET || + get_int(l, SOL_SOCKET, SO_ACCEPTCONN) != 1 || + get_int(l, SOL_SOCKET, SO_REUSEADDR) != 1) { + return fail("options: type, domain, acceptconn, reuseaddr", + get_int(l, SOL_SOCKET, SO_TYPE), get_int(l, SOL_SOCKET, SO_ACCEPTCONN)); + } + int c = dial(port); + int sets[][2] = { + {SOL_SOCKET, SO_KEEPALIVE}, {IPPROTO_TCP, TCP_NODELAY}, {SOL_SOCKET, SO_REUSEPORT}, + {SOL_SOCKET, SO_BROADCAST}, + }; + for (unsigned i = 0; i < sizeof sets / sizeof sets[0]; i++) { + if (setsockopt(c, sets[i][0], sets[i][1], &one, sizeof one) || + get_int(c, sets[i][0], sets[i][1]) != 1) { + return fail("options: set then read back", sets[i][1], errno); + } + } + int idle = 15; + if (setsockopt(c, IPPROTO_TCP, TCP_KEEPIDLE, &idle, sizeof idle) || + get_int(c, IPPROTO_TCP, TCP_KEEPIDLE) != 15) { + return fail("options: TCP_KEEPIDLE", get_int(c, IPPROTO_TCP, TCP_KEEPIDLE), errno); + } + int buf = 65536; + if (setsockopt(c, SOL_SOCKET, SO_RCVBUF, &buf, sizeof buf) || + get_int(c, SOL_SOCKET, SO_RCVBUF) != 2 * buf) { + return fail("options: SO_RCVBUF doubled", get_int(c, SOL_SOCKET, SO_RCVBUF), 2 * buf); + } + struct linger lg = {1, 5}, back = {0, 0}; + socklen_t len = sizeof back; + setsockopt(c, SOL_SOCKET, SO_LINGER, &lg, sizeof lg); + getsockopt(c, SOL_SOCKET, SO_LINGER, &back, &len); + struct timeval tv = {2, 500000}, tb = {0, 0}; + len = sizeof tb; + setsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof tv); + getsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tb, &len); + int bogus = setsockopt(c, SOL_SOCKET, 9999, &one, sizeof one); + int e = errno; + close(c); + close(l); + if (back.l_onoff != 1 || back.l_linger != 5 || tb.tv_sec != 2 || tb.tv_usec != 500000) { + return fail("options: SO_LINGER and SO_RCVTIMEO read back", back.l_linger, tb.tv_usec); + } + if (bogus != -1 || e != ENOPROTOOPT) { + return fail("options: an unknown option is ENOPROTOOPT", bogus, e); + } + ok("options", "11 options, unknown errno", e); + return 0; +} + +// The child holds the accepted end after the parent closes its own: the +// client still reads the child's bytes, then end of file when the child exits. +static int fork_share(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("fork_share: pair", 0, errno); + } + pid_t kid = fork(); + if (kid == 0) { + close(a); + nap_ms(50); + write(b, "kid", 3); + _exit(0); + } + close(b); + long got = read(a, buf, 8); + int status = 0; + waitpid(kid, &status, 0); + long then = read(a, buf, 8); + close(a); + if (got != 3 || then != 0 || status != 0) { + return fail("fork_share: child's bytes, then eof", got, then); + } + ok("fork_share", "3 bytes from the child, then", then); + return 0; +} diff --git a/userland/linux_guests/c/cudp.c b/userland/linux_guests/c/cudp.c new file mode 100644 index 000000000..16ecd7892 --- /dev/null +++ b/userland/linux_guests/c/cudp.c @@ -0,0 +1,108 @@ +// Datagrams on the loopback address, as Linux has them: an echo between an +// unbound client and a bound server, each told the other's address; a +// connected socket that sends without naming a peer and keeps only its +// peer's datagrams; a datagram cut to the buffer with MSG_TRUNC, and +// recvmsg's MSG_TRUNC flag; ECONNREFUSED on a connected socket whose peer +// port is closed; sendmmsg and recvmmsg; and EDESTADDRREQ with no peer at +// all. Each part prints as it passes and every part runs. +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include + +static int parts; + +static int fail(const char *what, long a, long b) { + printf("[C] cudp FAIL: %s (%ld, %ld)\n", what, a, b); + fflush(stdout); + return 1; +} + +static void ok(const char *part, const char *detail, long n) { + parts++; + printf("[C] cudp %s ok: %s %ld\n", part, detail, n); + fflush(stdout); +} + +// A datagram socket bound to 127.0.0.1 at a port the kernel picks. +static int bound(struct sockaddr_in *at) { + int s = socket(AF_INET, SOCK_DGRAM, 0); + memset(at, 0, sizeof *at); + at->sin_family = AF_INET; + at->sin_addr.s_addr = htonl(INADDR_LOOPBACK); + socklen_t len = sizeof *at; + if (s < 0 || bind(s, (void *)at, sizeof *at) || getsockname(s, (void *)at, &len)) { + return -1; + } + return s; +} + +static int echo(void) { + struct sockaddr_in srv, from, back; + int s = bound(&srv), c = socket(AF_INET, SOCK_DGRAM, 0); + char buf[32]; + socklen_t flen = sizeof from, blen = sizeof back; + if (s < 0 || sendto(c, "ping", 4, 0, (void *)&srv, sizeof srv) != 4) { + return fail("echo: send", s, errno); + } + long got = recvfrom(s, buf, sizeof buf, 0, (void *)&from, &flen); + if (got != 4 || from.sin_port == 0 || flen != sizeof from) { + return fail("echo: server heard the client and its address", got, ntohs(from.sin_port)); + } + sendto(s, "pong!", 5, 0, (void *)&from, flen); + got = recvfrom(c, buf, sizeof buf, 0, (void *)&back, &blen); + close(s); + close(c); + if (got != 5 || back.sin_port != srv.sin_port || memcmp(buf, "pong!", 5)) { + return fail("echo: client heard the server", got, ntohs(back.sin_port)); + } + ok("echo", "4 bytes there, back from the server's port", ntohs(back.sin_port) > 0); + return 0; +} + +static int connected(void) { + struct sockaddr_in a, b, x; + int sa = bound(&a), sb = bound(&b), sx = bound(&x); + char buf[8]; + connect(sb, (void *)&a, sizeof a); + if (send(sb, "to-a", 4, 0) != 4 || recv(sa, buf, 8, 0) != 4) { + return fail("connected: send without an address", 0, errno); + } + // sb keeps only a's datagrams: x's is dropped, a's arrives. + sendto(sx, "noise", 5, 0, (void *)&b, sizeof b); + sendto(sa, "real", 4, 0, (void *)&b, sizeof b); + long got = recv(sb, buf, 8, 0); + long more = recv(sb, buf, 8, MSG_DONTWAIT); + int e = errno; + close(sa); + close(sb); + close(sx); + if (got != 4 || more != -1 || e != EAGAIN) { + return fail("connected: only the peer's datagrams kept", got, more); + } + ok("connected", "a stranger's datagram dropped, peer's bytes", got); + return 0; +} + +#include "cudp_parts.h" + +int main(void) { + int (*const part[])(void) = {echo, connected, cut, refused, mmsg, nameless}; + const int count = sizeof part / sizeof part[0]; + int failed = 0; + for (int i = 0; i < count; i++) { + failed += part[i](); + } + if (failed) { + printf("[C] cudp FAIL: %d parts failed, %d passed\n", failed, parts); + fflush(stdout); + return 1; + } + printf("[C] cudp PASS: %d parts\n", parts); + fflush(stdout); + return 0; +} diff --git a/userland/linux_guests/c/cudp_parts.h b/userland/linux_guests/c/cudp_parts.h new file mode 100644 index 000000000..ab4b1b22b --- /dev/null +++ b/userland/linux_guests/c/cudp_parts.h @@ -0,0 +1,82 @@ +// cudp: truncation, a refused port, several messages at once, and no peer. + +static int cut(void) { + struct sockaddr_in a; + int s = bound(&a), c = socket(AF_INET, SOCK_DGRAM, 0); + char big[100], buf[4]; + memset(big, 'z', sizeof big); + sendto(c, big, sizeof big, 0, (void *)&a, sizeof a); + long whole = recv(s, buf, sizeof buf, MSG_TRUNC); + long rest = recv(s, buf, sizeof buf, MSG_DONTWAIT); + sendto(c, big, sizeof big, 0, (void *)&a, sizeof a); + struct iovec iov = {buf, sizeof buf}; + struct msghdr m = {0}; + m.msg_iov = &iov; + m.msg_iovlen = 1; + long cut = recvmsg(s, &m, 0); + close(s); + close(c); + if (whole != 100 || rest != -1 || cut != 4 || !(m.msg_flags & MSG_TRUNC)) { + return fail("trunc: cut to 4, whole length 100, flag set", whole, cut); + } + ok("trunc", "MSG_TRUNC gave the whole length", whole); + return 0; +} + +// A port with no socket: one is bound and closed to find it. +static int refused(void) { + struct sockaddr_in gone; + close(bound(&gone)); + int c = socket(AF_INET, SOCK_DGRAM, 0); + char buf[4]; + connect(c, (void *)&gone, sizeof gone); + long sent = send(c, "x", 1, 0); + long got = recv(c, buf, sizeof buf, MSG_DONTWAIT); + int e = errno; + close(c); + if (sent != 1 || got != -1 || e != ECONNREFUSED) { + return fail("refused: send taken, then ECONNREFUSED", sent, e); + } + ok("refused", "connected to a closed port, errno", e); + return 0; +} + +static int mmsg(void) { + struct sockaddr_in a; + int s = bound(&a), c = socket(AF_INET, SOCK_DGRAM, 0); + connect(c, (void *)&a, sizeof a); + char out[3][4] = {"one", "two", "six"}, in[3][8]; + struct iovec oi[3], ii[3]; + struct mmsghdr om[3], im[3]; + memset(om, 0, sizeof om); + memset(im, 0, sizeof im); + for (int i = 0; i < 3; i++) { + oi[i] = (struct iovec){out[i], 3}; + ii[i] = (struct iovec){in[i], 8}; + om[i].msg_hdr.msg_iov = &oi[i]; + om[i].msg_hdr.msg_iovlen = 1; + im[i].msg_hdr.msg_iov = &ii[i]; + im[i].msg_hdr.msg_iovlen = 1; + } + int sent = sendmmsg(c, om, 3, 0); + int got = recvmmsg(s, im, 3, MSG_WAITFORONE, 0); + close(s); + close(c); + if (sent != 3 || got != 3 || im[2].msg_len != 3 || memcmp(in[2], "six", 3)) { + return fail("mmsg: three out, three in", sent, got); + } + ok("mmsg", "sendmmsg and recvmmsg moved", got); + return 0; +} + +static int nameless(void) { + int c = socket(AF_INET, SOCK_DGRAM, 0); + long sent = send(c, "x", 1, 0); + int e = errno; + close(c); + if (sent != -1 || e != EDESTADDRREQ) { + return fail("nameless: no peer is EDESTADDRREQ", sent, e); + } + ok("nameless", "errno", e); + return 0; +} diff --git a/userland/linux_guests/go/http/go.mod b/userland/linux_guests/go/http/go.mod new file mode 100644 index 000000000..0fe31ea2f --- /dev/null +++ b/userland/linux_guests/go/http/go.mod @@ -0,0 +1,3 @@ +module nonos/guest/http + +go 1.24 diff --git a/userland/linux_guests/go/http/main.go b/userland/linux_guests/go/http/main.go new file mode 100644 index 000000000..139431c3b --- /dev/null +++ b/userland/linux_guests/go/http/main.go @@ -0,0 +1,54 @@ +// net/http as Linux runs it, inside one guest: a server on 127.0.0.1:0 and a +// client of it. Twenty GETs, each answered 200 with the path it asked for, +// over one kept-alive connection, which the server's own count of new +// connections shows. +package main + +import ( + "fmt" + "io" + "net" + "net/http" + "os" + "sync/atomic" +) + +func main() { + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + fmt.Println("[GO] gohttp FAIL: listen:", err) + os.Exit(1) + } + var conns atomic.Int32 + srv := &http.Server{ + Handler: http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + io.WriteString(w, "hello "+r.URL.Path) + }), + ConnState: func(_ net.Conn, s http.ConnState) { + if s == http.StateNew { + conns.Add(1) + } + }, + } + go srv.Serve(ln) + client := &http.Client{} + const gets = 20 + for i := 0; i < gets; i++ { + r, err := client.Get(fmt.Sprintf("http://%s/%d", ln.Addr(), i)) + if err != nil { + fmt.Println("[GO] gohttp FAIL: get", i, err) + os.Exit(1) + } + body, err := io.ReadAll(r.Body) + r.Body.Close() + if err != nil || r.StatusCode != 200 || string(body) != fmt.Sprintf("hello /%d", i) { + fmt.Println("[GO] gohttp FAIL: reply", i, r.StatusCode, string(body), err) + os.Exit(1) + } + } + if n := conns.Load(); n != 1 { + fmt.Println("[GO] gohttp FAIL: keep-alive: connections", n) + os.Exit(1) + } + fmt.Printf("[GO] gohttp PASS: %d GETs answered 200 over %d connection\n", gets, conns.Load()) +} From 5922240b3e6452d71db9aba4041f88da71bf5c13 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 21:18:31 +0000 Subject: [PATCH 08/14] linux: Unix sockets with names are the family's own A guest's AF_UNIX socket could only reach the display: socket made a display descriptor, a datagram one was ENOSYS, bind, listen and accept had nothing to do on it, and a connect to any other path was ECONNREFUSED. A Go net.Listen("unix", ...), a local daemon, or two processes meeting at a path could not run. Now an AF_UNIX socket is an entry in the family's table like any other: - bind to a path makes the path a file, as Linux does, EADDRINUSE if anything is there; the file stays after the socket closes, so a rebind is EADDRINUSE and a connect ECONNREFUSED until the guest unlinks it; an abstract name has no file and goes with its socket; bind with the family alone chooses a NUL and five hex digits; - listen needs a name (EINVAL), and a connect to a listener completes in the caller's call, blocking or not; a path with nothing there is ENOENT, one with no listener ECONNREFUSED, a socket of the other type EPROTOTYPE; - getsockname, getpeername, accept and recvfrom report sockaddr_un at Linux's lengths: a path with its NUL, an abstract name without, an unnamed peer as the family alone, an unnamed sender as nothing; - a datagram goes to the socket bound to the name it is sent to; a connected datagram socket sends to its peer, answers ECONNREFUSED once the peer is gone and ENOTCONN if it never had one, and refuses a stranger's datagram with EPERM; - SOCK_RAW is a datagram socket, as on Linux; SOCK_SEQPACKET is refused by name. A connect to the display's path still turns the descriptor into the display connection this capsule serves (unix/), so a Wayland client is unchanged. A socket that goes now clears every socket that pointed at it, not only its own peer: a connected datagram socket points one way, and it kept an index that a later socket could reuse. --- .../capsule_linux/src/linux/net/accept.rs | 10 ++- userland/capsule_linux/src/linux/net/bind.rs | 3 + .../capsule_linux/src/linux/net/connect.rs | 3 +- userland/capsule_linux/src/linux/net/dest.rs | 43 +++++++++ userland/capsule_linux/src/linux/net/dgram.rs | 15 +++- .../capsule_linux/src/linux/net/listen.rs | 5 +- userland/capsule_linux/src/linux/net/mod.rs | 5 ++ userland/capsule_linux/src/linux/net/msg.rs | 6 +- userland/capsule_linux/src/linux/net/name.rs | 22 +++-- .../capsule_linux/src/linux/net/peer_addr.rs | 20 +++-- .../capsule_linux/src/linux/net/sock/free.rs | 7 +- .../capsule_linux/src/linux/net/sock/gram.rs | 68 ++++++++------ .../src/linux/net/sock/gram_dest.rs | 44 ++++++++++ .../src/linux/net/sock/gram_in.rs | 8 +- .../capsule_linux/src/linux/net/sock/link.rs | 49 +++++------ .../capsule_linux/src/linux/net/sock/mod.rs | 5 ++ .../capsule_linux/src/linux/net/sock/name.rs | 45 ++++++++++ .../capsule_linux/src/linux/net/sock/new.rs | 2 + .../capsule_linux/src/linux/net/sock/pair.rs | 35 ++++++++ .../capsule_linux/src/linux/net/sock/types.rs | 6 +- .../src/linux/net/sockaddr_out.rs | 55 +++++++----- .../src/linux/net/sockaddr_un.rs | 49 +++++++++++ .../capsule_linux/src/linux/net/socket.rs | 37 +++++--- .../capsule_linux/src/linux/net/unix_bind.rs | 79 +++++++++++++++++ .../capsule_linux/src/linux/net/unix_calls.rs | 88 +++++++++++++++++++ .../capsule_linux/src/linux/net/unix_name.rs | 63 +++++++++++++ .../capsule_linux/src/linux/net/xfer_in.rs | 12 +-- .../capsule_linux/src/linux/net/xfer_out.rs | 14 ++- .../src/linux/serve/table_net.rs | 1 - userland/capsule_linux/src/linux/unix/mod.rs | 6 +- userland/capsule_linux/src/linux/unix/sock.rs | 22 +---- 31 files changed, 681 insertions(+), 146 deletions(-) create mode 100644 userland/capsule_linux/src/linux/net/dest.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/gram_dest.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/name.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/pair.rs create mode 100644 userland/capsule_linux/src/linux/net/sockaddr_un.rs create mode 100644 userland/capsule_linux/src/linux/net/unix_bind.rs create mode 100644 userland/capsule_linux/src/linux/net/unix_calls.rs create mode 100644 userland/capsule_linux/src/linux/net/unix_name.rs diff --git a/userland/capsule_linux/src/linux/net/accept.rs b/userland/capsule_linux/src/linux/net/accept.rs index d6d9979c5..23c62b6ad 100644 --- a/userland/capsule_linux/src/linux/net/accept.rs +++ b/userland/capsule_linux/src/linux/net/accept.rs @@ -22,7 +22,7 @@ use crate::linux::guest::Guest; use super::close::discard; use super::fd::{install, sock_of, SOCK_CLOEXEC, SOCK_NONBLOCK}; -use super::sock::{self, Domain, Proto}; +use super::sock::{self, Domain, Peer, Proto}; pub fn accept4(guest: &mut Guest, fd: u64, at: u64, lenp: u64, flags: u64) -> u64 { if flags & !(SOCK_NONBLOCK | SOCK_CLOEXEC) != 0 { @@ -44,7 +44,11 @@ pub fn accept4(guest: &mut Guest, fd: u64, at: u64, lenp: u64, flags: u64) -> u6 let child = s.pending.pop_front().ok_or(errno::EAGAIN)?; let c = t.get_mut(child).ok_or(errno::ECONNABORTED)?; c.holders.push(pid); - Ok((child, c.remote.unwrap_or_default())) + let from = match c.domain { + Domain::Inet => Peer::Inet(c.remote.unwrap_or_default()), + Domain::Unix => Peer::Unix(c.upeer.clone()), + }; + Ok((child, from)) }); let (child, from) = match taken { Ok(v) => v, @@ -54,7 +58,7 @@ pub fn accept4(guest: &mut Guest, fd: u64, at: u64, lenp: u64, flags: u64) -> u6 let Some(slot) = errno::slot(n) else { return n; }; - let wrote = super::sockaddr_out::write(guest, at, lenp, Domain::Inet, from); + let wrote = super::sockaddr_out::write(guest, at, lenp, &from); if errno::slot(wrote).is_none() { discard(guest, slot as u64); return wrote; diff --git a/userland/capsule_linux/src/linux/net/bind.rs b/userland/capsule_linux/src/linux/net/bind.rs index 9b6dbb399..9a2f1a72c 100644 --- a/userland/capsule_linux/src/linux/net/bind.rs +++ b/userland/capsule_linux/src/linux/net/bind.rs @@ -29,6 +29,9 @@ pub fn bind(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { Ok(id) => id, Err(e) => return e, }; + if sock::with(|t| t.get(id).is_some_and(|s| s.domain == Domain::Unix)) { + return super::unix_bind::bind(guest, id, at, len); + } let (family, mut want) = match sockaddr::read(guest, at, len) { Ok(v) => v, Err(e) => return e, diff --git a/userland/capsule_linux/src/linux/net/connect.rs b/userland/capsule_linux/src/linux/net/connect.rs index 77a26e4f6..4b1040508 100644 --- a/userland/capsule_linux/src/linux/net/connect.rs +++ b/userland/capsule_linux/src/linux/net/connect.rs @@ -42,8 +42,7 @@ pub fn connect(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { return errno::fail(errno::EBADF); }; match (proto, domain) { - // A socketpair end is connected from the start. - (_, Domain::Unix) => errno::fail(errno::EISCONN), + (_, Domain::Unix) => super::unix_calls::connect(guest, fd, id, proto, at, len), (Proto::Dgram, _) => super::connect_dgram::connect(guest, fd, id, family, to), _ if family != AF_INET => errno::fail(errno::EAFNOSUPPORT), _ if is_loopback(to.ip) => super::connect_lo::loopback(id, to, nonblock(guest, fd)), diff --git a/userland/capsule_linux/src/linux/net/dest.rs b/userland/capsule_linux/src/linux/net/dest.rs new file mode 100644 index 000000000..458b23b6d --- /dev/null +++ b/userland/capsule_linux/src/linux/net/dest.rs @@ -0,0 +1,43 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a send goes: nowhere but the peer for a stream, and for a datagram +//! the address or the Unix name it names, found now. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::peer_addr::To; +use super::sock::{Dest, Domain, Proto}; + +pub fn dest(guest: &Guest, proto: Proto, domain: Domain, to: Option) -> Result { + Ok(match (proto, to) { + (Proto::Stream, _) | (_, None) => Dest::Default, + (_, Some(To::Inet(a))) => Dest::Inet(a), + (_, Some(To::Unix(_))) if domain == Domain::Inet => { + return Err(errno::fail(errno::EAFNOSUPPORT)) + } + (_, Some(To::Unix(ua))) => { + let found = super::unix_name::resolve(guest, &ua) + .ok_or(errno::EINVAL) + .and_then(|name| super::unix_name::find(&name, Proto::Dgram)); + match found { + Ok(t) => Dest::Sock(t), + Err(e) => return Err(errno::fail(e)), + } + } + }) +} diff --git a/userland/capsule_linux/src/linux/net/dgram.rs b/userland/capsule_linux/src/linux/net/dgram.rs index 7eeafc8b1..5999d41d9 100644 --- a/userland/capsule_linux/src/linux/net/dgram.rs +++ b/userland/capsule_linux/src/linux/net/dgram.rs @@ -22,6 +22,7 @@ use alloc::vec; use crate::linux::guest::{Guest, Kind}; use super::fd::sock_of; +use super::peer_addr::To; use super::sock::{self, Domain, Proto}; use super::sockaddr::is_loopback; @@ -46,14 +47,20 @@ pub fn sendto( let inet_dgram = sock::with(|t| { t.get(id).is_some_and(|s| s.proto == Proto::Dgram && s.domain == Domain::Inet) }); - match to.filter(|_| inet_dgram) { - Some(to) if super::resolver::is_nameserver(to) => { + match to { + Some(To::Inet(a)) if inet_dgram && super::resolver::is_nameserver(a) => { super::resolver::become_resolver(guest, fd) } - Some(to) if !is_loopback(to.ip) => return super::policy::refuse_out("sendto", to), - _ => return super::xfer_out::send(guest, id, &vec![(buf, len)], 0, flags, to), + Some(To::Inet(a)) if inet_dgram && !is_loopback(a.ip) => { + return super::policy::refuse_out("sendto", a) + } + to => return super::xfer_out::send(guest, id, &vec![(buf, len)], 0, flags, to), } } + let to = match to { + Some(To::Inet(a)) => Some(a), + _ => None, + }; super::resolver::query(guest, fd, buf, len, to) } diff --git a/userland/capsule_linux/src/linux/net/listen.rs b/userland/capsule_linux/src/linux/net/listen.rs index adb92fa4b..9d6fc71bb 100644 --- a/userland/capsule_linux/src/linux/net/listen.rs +++ b/userland/capsule_linux/src/linux/net/listen.rs @@ -33,8 +33,11 @@ pub fn listen(guest: &mut Guest, fd: u64, backlog: u64) -> u64 { return errno::fail(errno::EBADF); }; match (s.proto, s.domain, s.local) { - (Proto::Dgram, ..) | (_, Domain::Unix, _) => return errno::fail(errno::EOPNOTSUPP), + (Proto::Dgram, ..) => return errno::fail(errno::EOPNOTSUPP), _ if s.connected || s.svc.is_some() => return errno::fail(errno::EINVAL), + // A Unix socket must be bound first: Linux does not name it here. + (_, Domain::Unix, _) if s.uname.is_none() => return errno::fail(errno::EINVAL), + (_, Domain::Unix, _) => {} // Linux would bind 0.0.0.0 here, which is not the family's own. (_, _, None) => return not_loopback("listen", Addr::default()), (_, _, Some(at)) if !s.listening && t.listener(at).is_some() => { diff --git a/userland/capsule_linux/src/linux/net/mod.rs b/userland/capsule_linux/src/linux/net/mod.rs index cd42fe43c..0a199bf09 100644 --- a/userland/capsule_linux/src/linux/net/mod.rs +++ b/userland/capsule_linux/src/linux/net/mod.rs @@ -26,6 +26,7 @@ mod connect; mod connect_lo; mod connect_dgram; mod connect_out; +mod dest; mod dgram; mod dgram_addr; pub mod dns; @@ -59,9 +60,13 @@ mod shutdown; pub mod sock; mod sockaddr; mod sockaddr_out; +mod sockaddr_un; mod socket; mod stream; mod try_call; +mod unix_bind; +mod unix_calls; +mod unix_name; mod xfer_in; mod xfer_out; diff --git a/userland/capsule_linux/src/linux/net/msg.rs b/userland/capsule_linux/src/linux/net/msg.rs index a9037c255..bc49211c0 100644 --- a/userland/capsule_linux/src/linux/net/msg.rs +++ b/userland/capsule_linux/src/linux/net/msg.rs @@ -45,8 +45,10 @@ pub fn sendmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> Ok(to) => to, Err(e) => return e, }; - if let Some(to) = to.filter(|a| !super::sockaddr::is_loopback(a.ip)) { - return super::policy::refuse_out("sendmsg", to); + if let Some(super::peer_addr::To::Inet(a)) = &to { + if !super::sockaddr::is_loopback(a.ip) { + return super::policy::refuse_out("sendmsg", *a); + } } super::xfer_out::send(guest, id, &h.iov, skip, flags, to) } diff --git a/userland/capsule_linux/src/linux/net/name.rs b/userland/capsule_linux/src/linux/net/name.rs index c153acf1d..75580a91a 100644 --- a/userland/capsule_linux/src/linux/net/name.rs +++ b/userland/capsule_linux/src/linux/net/name.rs @@ -20,18 +20,23 @@ use crate::linux::abi::errno; use crate::linux::guest::Guest; use super::fd::sock_of; -use super::sock::{self, Addr, Domain}; +use super::sock::{self, Domain, Peer}; pub fn getsockname(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 { let id = match sock_of(guest, fd) { Ok(id) => id, Err(e) => return e, }; - let Some((domain, local)) = sock::with(|t| t.get(id).map(|s| (s.domain, s.local))) else { + // A socket not yet bound is 0.0.0.0 port 0, or an unnamed Unix socket. + let Some(me) = sock::with(|t| { + t.get(id).map(|s| match s.domain { + Domain::Inet => Peer::Inet(s.local.unwrap_or_default()), + Domain::Unix => Peer::Unix(s.uname.clone()), + }) + }) else { return errno::fail(errno::EBADF); }; - // A socket not yet bound is 0.0.0.0, port 0. - super::sockaddr_out::write(guest, at, lenp, domain, local.unwrap_or_default()) + super::sockaddr_out::write(guest, at, lenp, &me) } pub fn getpeername(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 { @@ -41,13 +46,16 @@ pub fn getpeername(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 { }; // A reset connection is closed, and has no peer; one whose peer only // shut down, or left cleanly, still does. - let peer: Option<(Domain, Addr)> = sock::with(|t| { + let peer: Option = sock::with(|t| { let s = t.get(id)?; let live = s.connected && !s.broken && s.error == 0; - live.then(|| (s.domain, s.remote.unwrap_or_default())) + live.then(|| match s.domain { + Domain::Inet => Peer::Inet(s.remote.unwrap_or_default()), + Domain::Unix => Peer::Unix(s.upeer.clone()), + }) }); match peer { - Some((domain, addr)) => super::sockaddr_out::write(guest, at, lenp, domain, addr), + Some(p) => super::sockaddr_out::write(guest, at, lenp, &p), None => errno::fail(errno::ENOTCONN), } } diff --git a/userland/capsule_linux/src/linux/net/peer_addr.rs b/userland/capsule_linux/src/linux/net/peer_addr.rs index d43aa4e33..29bb3984f 100644 --- a/userland/capsule_linux/src/linux/net/peer_addr.rs +++ b/userland/capsule_linux/src/linux/net/peer_addr.rs @@ -22,15 +22,16 @@ use crate::linux::guest::Guest; use super::flags::MSG_TRUNC; use super::sock::Addr; -use super::sockaddr::{self, AF_INET}; +use super::sockaddr::{self, AF_INET, AF_UNIX}; +use super::sockaddr_un::UAddr; use super::xfer_in::In; /// The count a receive answers, with the sender written out. A stream has /// no sender, and Linux says so with a length of zero. pub fn finish(guest: &mut Guest, got: In, flags: u64, at: u64, alen: u64) -> u64 { if at != 0 { - let wrote = match got.from { - Some((domain, from)) => super::sockaddr_out::write(guest, at, alen, domain, from), + let wrote = match &got.from { + Some(from) => super::sockaddr_out::write(guest, at, alen, from), None if alen != 0 && guest.write(alen, &0u32.to_le_bytes()) < 4 => { errno::fail(errno::EFAULT) } @@ -43,13 +44,20 @@ pub fn finish(guest: &mut Guest, got: In, flags: u64, at: u64, alen: u64) -> u64 errno::ok(if flags & MSG_TRUNC != 0 { got.whole } else { got.n } as u64) } -/// The address a send names, if any: IPv4 only. -pub fn address(guest: &Guest, at: u64, alen: u64) -> Result, u64> { +/// Where a send names: an IPv4 address or a Unix name. +pub enum To { + Inet(Addr), + Unix(UAddr), +} + +/// The address a send names, if any. +pub fn address(guest: &Guest, at: u64, alen: u64) -> Result, u64> { if at == 0 { return Ok(None); } match sockaddr::read(guest, at, alen)? { - (AF_INET, a) => Ok(Some(a)), + (AF_INET, a) => Ok(Some(To::Inet(a))), + (AF_UNIX, _) => Ok(Some(To::Unix(super::sockaddr_un::read(guest, at, alen)?))), _ => Err(errno::fail(errno::EAFNOSUPPORT)), } } diff --git a/userland/capsule_linux/src/linux/net/sock/free.rs b/userland/capsule_linux/src/linux/net/sock/free.rs index 3fcbb640b..06265cd99 100644 --- a/userland/capsule_linux/src/linux/net/sock/free.rs +++ b/userland/capsule_linux/src/linux/net/sock/free.rs @@ -44,10 +44,13 @@ impl Socks { if let Some(h) = gone.svc { super::super::stream::close(h); } - if let Some(p) = gone.peer.and_then(|p| self.get_mut(p)) { + let reset = reset || !gone.rx.is_empty() || gone.opts.linger == (1, 0); + // A stream's peer points back; a connected Unix datagram socket + // points at this one alone. Either is told, and forgets the index. + for p in self.list.iter_mut().flatten().filter(|p| p.peer == Some(id)) { p.peer = None; p.eof = true; - if reset || !gone.rx.is_empty() || gone.opts.linger == (1, 0) { + if reset && p.proto == super::types::Proto::Stream { p.error = ECONNRESET; } } diff --git a/userland/capsule_linux/src/linux/net/sock/gram.rs b/userland/capsule_linux/src/linux/net/sock/gram.rs index f1f81f36f..74fe5f239 100644 --- a/userland/capsule_linux/src/linux/net/sock/gram.rs +++ b/userland/capsule_linux/src/linux/net/sock/gram.rs @@ -14,25 +14,28 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! A datagram into the family: to whichever socket holds the port it is -//! sent to, or a socketpair end's peer. +//! A datagram into the family: to whichever socket holds the port or the +//! Unix name it is sent to, or to a connected socket's peer. use core::mem; -use crate::linux::abi::errno::{EBADF, ECONNREFUSED, EDESTADDRREQ, EMSGSIZE, ENOTCONN}; +use crate::linux::abi::errno::{ + EBADF, ECONNREFUSED, EDESTADDRREQ, EINVAL, EMSGSIZE, ENOTCONN, EPERM, +}; +use super::gram_dest::Dest; +use super::name::Peer; use super::table::Socks; -use super::types::{Addr, Domain, Proto}; +use super::types::Domain; /// The largest UDP payload IPv4 carries. const MAX_GRAM: usize = 65507; impl Socks { - /// One datagram from `id` to `to`, or to the address it connected to. A - /// datagram nobody is bound to receive is dropped, as on the wire; a - /// connected sender learns of it as ECONNREFUSED on its next call, which - /// is what the ICMP reply does on Linux. - pub fn send_gram(&mut self, id: u32, to: Option, bytes: &[u8]) -> Result { + /// One datagram from `id`. One nobody holds the port of is dropped, as + /// on the wire; a connected sender learns of it as ECONNREFUSED on its + /// next call, which is what the ICMP reply does on Linux. + pub fn send_gram(&mut self, id: u32, dest: Dest, bytes: &[u8]) -> Result { let s = self.get_mut(id).ok_or(EBADF)?; if s.error != 0 { return Err(mem::take(&mut s.error)); @@ -40,26 +43,41 @@ impl Socks { if bytes.len() > MAX_GRAM { return Err(EMSGSIZE); } - let (from, connected) = (s.local.unwrap_or_default(), s.remote.is_some()); - let target = match s.domain { - Domain::Unix => s.peer.ok_or(ECONNREFUSED)?, - Domain::Inet => { - let dest = - to.or(s.remote).ok_or(if connected { ENOTCONN } else { EDESTADDRREQ })?; - match self.bound(Proto::Dgram, dest) { + let from = match s.domain { + Domain::Inet => Peer::Inet(s.local.unwrap_or_default()), + Domain::Unix => Peer::Unix(s.uname.clone()), + }; + let (remote, peer, connected, domain) = (s.remote, s.peer, s.connected, s.domain); + let target = match (dest, domain) { + (Dest::Sock(t), Domain::Unix) => t, + (Dest::Default, Domain::Unix) => match peer { + Some(p) => p, + None if connected => return Err(ECONNREFUSED), + None => return Err(ENOTCONN), + }, + (Dest::Inet(to), Domain::Inet) => match self.inet_target(id, to, remote.is_some()) { + Some(r) => r, + None => return Ok(bytes.len()), + }, + (Dest::Default, Domain::Inet) => match remote { + Some(to) => match self.inet_target(id, to, true) { Some(r) => r, - None => { - if let Some(s) = self.get_mut(id).filter(|_| connected) { - s.error = ECONNREFUSED; - } - return Ok(bytes.len()); - } - } - } + None => return Ok(bytes.len()), + }, + None => return Err(EDESTADDRREQ), + }, + _ => return Err(EINVAL), }; let r = self.get_mut(target).ok_or(ECONNREFUSED)?; + // A connected Unix socket takes only from its peer, and says so. + if r.domain == Domain::Unix && r.peer.is_some_and(|p| p != id) && r.connected { + return Err(EPERM); + } let queued: usize = r.grams.iter().map(|(_, g)| g.len()).sum(); - let wanted = r.remote.is_none_or(|x| x == from); + let wanted = match (&from, r.remote) { + (Peer::Inet(a), Some(x)) => *a == x, + _ => true, + }; if wanted && queued + bytes.len() <= r.opts.rcvbuf as usize { r.grams.push_back((from, bytes.to_vec())); } diff --git a/userland/capsule_linux/src/linux/net/sock/gram_dest.rs b/userland/capsule_linux/src/linux/net/sock/gram_dest.rs new file mode 100644 index 000000000..812a081fa --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/gram_dest.rs @@ -0,0 +1,44 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Where a datagram goes, and which socket holds an IPv4 port. + +use crate::linux::abi::errno::ECONNREFUSED; + +use super::table::Socks; +use super::types::{Addr, Proto}; + +/// Where a datagram goes: where the socket connected, an IPv4 address, or +/// a Unix socket already found by its name (`unix_name::find`). +pub enum Dest { + Default, + Inet(Addr), + Sock(u32), +} + +impl Socks { + /// The socket bound to `to`; None drops the datagram, and a connected + /// sender is told on its next call. + pub(super) fn inet_target(&mut self, id: u32, to: Addr, connected: bool) -> Option { + let found = self.bound(Proto::Dgram, to); + if found.is_none() { + if let Some(s) = self.get_mut(id).filter(|_| connected) { + s.error = ECONNREFUSED; + } + } + found + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/gram_in.rs b/userland/capsule_linux/src/linux/net/sock/gram_in.rs index 329c9263e..da4408f62 100644 --- a/userland/capsule_linux/src/linux/net/sock/gram_in.rs +++ b/userland/capsule_linux/src/linux/net/sock/gram_in.rs @@ -21,8 +21,8 @@ use core::mem; use crate::linux::abi::errno::{EAGAIN, EBADF}; +use super::name::Peer; use super::table::Socks; -use super::types::Addr; impl Socks { /// The next datagram, cut to `want`, with its whole length and sender. @@ -32,7 +32,7 @@ impl Socks { let got = Got { bytes: gram[..want.min(gram.len())].to_vec(), whole: gram.len(), - from: *from, + from: from.clone(), }; if !peek { s.grams.pop_front(); @@ -43,7 +43,7 @@ impl Socks { return Err(mem::take(&mut s.error)); } if s.rd_shut { - return Ok(Got { bytes: Vec::new(), whole: 0, from: Addr::default() }); + return Ok(Got { bytes: Vec::new(), whole: 0, from: Peer::Unix(None) }); } Err(EAGAIN) } @@ -53,5 +53,5 @@ pub struct Got { pub bytes: Vec, /// The datagram's length before it was cut to fit. pub whole: usize, - pub from: Addr, + pub from: Peer, } diff --git a/userland/capsule_linux/src/linux/net/sock/link.rs b/userland/capsule_linux/src/linux/net/sock/link.rs index ffdc8300b..c24b04a27 100644 --- a/userland/capsule_linux/src/linux/net/sock/link.rs +++ b/userland/capsule_linux/src/linux/net/sock/link.rs @@ -14,11 +14,11 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Joining two stream ends: a loopback connect, which Linux completes in the -//! caller's own call, and socketpair. +//! Joining two stream ends: a connect on loopback or to a Unix name, which +//! Linux completes in the caller's own call, and socketpair. use super::table::Socks; -use super::types::{Addr, Domain, Proto}; +use super::types::{Addr, Proto}; pub enum Link { /// Connected; the listener has one more connection to accept. @@ -32,30 +32,38 @@ pub enum Link { impl Socks { /// Connect stream `id`, already bound, to the listener at `to`. pub fn link(&mut self, id: u32, to: Addr) -> Link { - let Some(l) = self.listener(to) else { - return Link::Refused; - }; - let Some((backlog, queued, opts)) = - self.get(l).map(|s| (s.backlog, s.pending.len(), s.opts)) - else { + match self.listener(to) { + Some(l) => self.join(id, l), + None => Link::Refused, + } + } + + /// Connect stream `id` to listener `l`: a new server end, with the + /// listener's name and options, waits in its queue for accept. + pub fn join(&mut self, id: u32, l: u32) -> Link { + let Some(ls) = self.get(l) else { return Link::Refused; }; // Linux queues one more than the backlog it was given. - if queued > backlog { + if ls.pending.len() > ls.backlog { return Link::Full; } - let from = self.get(id).and_then(|s| s.local); - let server = self.open(Domain::Inet, Proto::Stream, None); + let (domain, local, uname, opts) = (ls.domain, ls.local, ls.uname.clone(), ls.opts); + let (from, from_name) = self.get(id).map_or((None, None), |c| (c.local, c.uname.clone())); + let server = self.open(domain, Proto::Stream, None); if let Some(s) = self.get_mut(server) { - s.local = Some(to); + s.local = local; s.remote = from; + s.uname = uname.clone(); + s.upeer = from_name; s.peer = Some(id); s.connected = true; // An accepted socket starts with its listener's options. s.opts = opts; } if let Some(c) = self.get_mut(id) { - c.remote = Some(to); + c.remote = local; + c.upeer = uname; c.peer = Some(server); c.connected = true; } @@ -64,17 +72,4 @@ impl Socks { } Link::Done } - - /// Two connected sockets, both held by `pid`. - pub fn pair(&mut self, domain: Domain, proto: Proto, pid: u32) -> (u32, u32) { - let a = self.open(domain, proto, Some(pid)); - let b = self.open(domain, proto, Some(pid)); - for (me, other) in [(a, b), (b, a)] { - if let Some(s) = self.get_mut(me) { - s.peer = Some(other); - s.connected = true; - } - } - (a, b) - } } diff --git a/userland/capsule_linux/src/linux/net/sock/mod.rs b/userland/capsule_linux/src/linux/net/sock/mod.rs index 3508a4fb3..e17344506 100644 --- a/userland/capsule_linux/src/linux/net/sock/mod.rs +++ b/userland/capsule_linux/src/linux/net/sock/mod.rs @@ -27,11 +27,14 @@ mod bind; mod cell; mod free; mod gram; +mod gram_dest; mod gram_in; mod holders; mod link; +mod name; mod new; mod opts; +mod pair; mod port; mod progress; mod ready; @@ -41,8 +44,10 @@ mod table; mod types; pub use cell::with; +pub use gram_dest::Dest; pub use holders::Holder; pub use link::Link; +pub use name::{Peer, UName}; pub use opts::Opts; pub use progress::{progress, set_progress}; pub use ready::bits; diff --git a/userland/capsule_linux/src/linux/net/sock/name.rs b/userland/capsule_linux/src/linux/net/sock/name.rs new file mode 100644 index 000000000..908cfa6fb --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/name.rs @@ -0,0 +1,45 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A Unix socket's name, and who a message came from. + +use alloc::vec::Vec; + +use super::types::Addr; + +/// A bound Unix name. `key` is what two sockets must share to meet: the +/// resolved path, or a NUL then the abstract name. `shown` is the sun_path +/// the guest gave, which getsockname and accept report back. +#[derive(Clone, PartialEq, Eq)] +pub struct UName { + pub key: Vec, + pub shown: Vec, +} + +impl UName { + /// True for a name in the abstract namespace, which no file backs. + pub fn is_abstract(&self) -> bool { + self.key.first() == Some(&0) + } +} + +/// The far end of a message or a connection, as a sockaddr reports it: an +/// IPv4 address, or a Unix name, None for an unnamed socket. +#[derive(Clone)] +pub enum Peer { + Inet(Addr), + Unix(Option), +} diff --git a/userland/capsule_linux/src/linux/net/sock/new.rs b/userland/capsule_linux/src/linux/net/sock/new.rs index 293afe55c..708d7f96f 100644 --- a/userland/capsule_linux/src/linux/net/sock/new.rs +++ b/userland/capsule_linux/src/linux/net/sock/new.rs @@ -45,6 +45,8 @@ impl Sock { opts: Opts::new(proto), svc: None, holders: pid.map_or_else(Vec::new, |p| vec![p]), + uname: None, + upeer: None, } } } diff --git a/userland/capsule_linux/src/linux/net/sock/pair.rs b/userland/capsule_linux/src/linux/net/sock/pair.rs new file mode 100644 index 000000000..04e294ed2 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/pair.rs @@ -0,0 +1,35 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! socketpair: two sockets connected from the start. + +use super::table::Socks; +use super::types::{Domain, Proto}; + +impl Socks { + /// Two connected sockets, both held by `pid`. + pub fn pair(&mut self, domain: Domain, proto: Proto, pid: u32) -> (u32, u32) { + let a = self.open(domain, proto, Some(pid)); + let b = self.open(domain, proto, Some(pid)); + for (me, other) in [(a, b), (b, a)] { + if let Some(s) = self.get_mut(me) { + s.peer = Some(other); + s.connected = true; + } + } + (a, b) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/types.rs b/userland/capsule_linux/src/linux/net/sock/types.rs index 9d54c4dce..1a4a8294b 100644 --- a/userland/capsule_linux/src/linux/net/sock/types.rs +++ b/userland/capsule_linux/src/linux/net/sock/types.rs @@ -19,6 +19,7 @@ use alloc::collections::VecDeque; use alloc::vec::Vec; +use super::name::{Peer, UName}; use super::opts::Opts; #[derive(Clone, Copy, PartialEq, Eq)] @@ -56,7 +57,7 @@ pub struct Sock { /// Bytes the peer wrote that this end has not read. pub rx: VecDeque, /// Datagrams waiting to be read, each with where it came from. - pub grams: VecDeque<(Addr, Vec)>, + pub grams: VecDeque<(Peer, Vec)>, /// The peer will write nothing more: it shut its side or it is gone. pub eof: bool, pub wr_shut: bool, @@ -70,4 +71,7 @@ pub struct Sock { pub svc: Option, /// The processes that hold a descriptor naming this socket. pub holders: Vec, + /// A Unix socket's own name, and its connected peer's. + pub uname: Option, + pub upeer: Option, } diff --git a/userland/capsule_linux/src/linux/net/sockaddr_out.rs b/userland/capsule_linux/src/linux/net/sockaddr_out.rs index a8b84b09b..5a20bd547 100644 --- a/userland/capsule_linux/src/linux/net/sockaddr_out.rs +++ b/userland/capsule_linux/src/linux/net/sockaddr_out.rs @@ -16,16 +16,18 @@ //! A socket's address written into guest memory. +use alloc::vec::Vec; + use crate::linux::abi::errno; use crate::linux::guest::Guest; -use super::sock::{Addr, Domain}; -use super::sockaddr::{AF_INET, AF_UNIX, SOCKADDR_IN}; +use super::sock::Peer; +use super::sockaddr::{AF_INET, AF_UNIX}; -/// Write `addr` at `at`, cut to the length the guest offered at `lenp`, and -/// the whole length back at `lenp`, as Linux's move_addr_to_user does. A -/// socketpair end has an unnamed Unix address: the family alone. -pub fn write(guest: &mut Guest, at: u64, lenp: u64, domain: Domain, addr: Addr) -> u64 { +/// Write `peer` at `at`, cut to the length the guest offered at `lenp`, and +/// the whole length back at `lenp`, as Linux's move_addr_to_user does. An +/// unnamed Unix socket is the family alone; a path carries its NUL. +pub fn write(guest: &mut Guest, at: u64, lenp: u64, peer: &Peer) -> u64 { if at == 0 || lenp == 0 { return errno::ok(0); } @@ -36,23 +38,34 @@ pub fn write(guest: &mut Guest, at: u64, lenp: u64, domain: Domain, addr: Addr) if room < 0 { return errno::fail(errno::EINVAL); } - let mut sa = [0u8; SOCKADDR_IN]; - let whole = match domain { - Domain::Unix => { - sa[0..2].copy_from_slice(&AF_UNIX.to_le_bytes()); - 2 - } - Domain::Inet => { - sa[0..2].copy_from_slice(&AF_INET.to_le_bytes()); - sa[2..4].copy_from_slice(&addr.port.to_be_bytes()); - sa[4..8].copy_from_slice(&addr.ip); - SOCKADDR_IN - } - }; - let n = whole.min(room as usize); - if guest.write(at, &sa[..n]) < n as i64 || guest.write(lenp, &(whole as u32).to_le_bytes()) < 4 + let sa = encode(peer); + let n = sa.len().min(room as usize); + if guest.write(at, &sa[..n]) < n as i64 + || guest.write(lenp, &(sa.len() as u32).to_le_bytes()) < 4 { return errno::fail(errno::EFAULT); } errno::ok(0) } + +fn encode(peer: &Peer) -> Vec { + let mut sa = Vec::with_capacity(16); + match peer { + Peer::Inet(addr) => { + sa.extend_from_slice(&AF_INET.to_le_bytes()); + sa.extend_from_slice(&addr.port.to_be_bytes()); + sa.extend_from_slice(&addr.ip); + sa.resize(16, 0); + } + Peer::Unix(name) => { + sa.extend_from_slice(&AF_UNIX.to_le_bytes()); + if let Some(n) = name { + sa.extend_from_slice(&n.shown); + if !n.is_abstract() { + sa.push(0); + } + } + } + } + sa +} diff --git a/userland/capsule_linux/src/linux/net/sockaddr_un.rs b/userland/capsule_linux/src/linux/net/sockaddr_un.rs new file mode 100644 index 000000000..e16af06c3 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sockaddr_un.rs @@ -0,0 +1,49 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `sockaddr_un` as a guest names it: a path, an abstract name, or only the +//! family, which asks bind for a name of Linux's choosing. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +/// sun_family and the 108 bytes of sun_path. +const SOCKADDR_UN: u64 = 110; + +pub enum UAddr { + Auto, + Path(Vec), + Abstract(Vec), +} + +pub fn read(guest: &Guest, at: u64, len: u64) -> Result { + if !(2..=SOCKADDR_UN).contains(&len) { + return Err(errno::fail(errno::EINVAL)); + } + let raw = guest.read(at, len as usize).ok_or(errno::fail(errno::EFAULT))?; + let path = &raw[2..]; + Ok(match path.first() { + None => UAddr::Auto, + Some(0) => UAddr::Abstract(path[1..].to_vec()), + // A path ends at its first NUL, however long the guest said it was. + Some(_) => { + let end = path.iter().position(|&b| b == 0).unwrap_or(path.len()); + UAddr::Path(path[..end].to_vec()) + } + }) +} diff --git a/userland/capsule_linux/src/linux/net/socket.rs b/userland/capsule_linux/src/linux/net/socket.rs index adcd11ef2..dcca853ed 100644 --- a/userland/capsule_linux/src/linux/net/socket.rs +++ b/userland/capsule_linux/src/linux/net/socket.rs @@ -14,8 +14,9 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `socket` for AF_INET. The socket is the family's until it connects -//! outside 127.0.0.0/8; only then is net.sockets asked for one. +//! `socket` for AF_INET and AF_UNIX. The socket is the family's until it +//! connects outside 127.0.0.0/8 (then net.sockets holds the stream) or to +//! the display's path (then it is the display connection). use crate::linux::abi::errno; use crate::linux::guest::Guest; @@ -23,29 +24,43 @@ use crate::linux::guest::Guest; use super::fd::{install, SOCK_CLOEXEC, SOCK_NONBLOCK}; use super::policy::refuse; use super::sock::{self, Domain, Proto}; -use super::sockaddr::AF_INET; +use super::sockaddr::{AF_INET, AF_UNIX}; const SOCK_STREAM: u64 = 1; const SOCK_DGRAM: u64 = 2; const SOCK_RAW: u64 = 3; +const SOCK_SEQPACKET: u64 = 5; const TYPE_MASK: u64 = 0xF; const IPPROTO_TCP: u64 = 6; const IPPROTO_UDP: u64 = 17; +/// The one protocol a Unix socket takes besides 0. +const PF_UNIX: u64 = 1; pub fn socket(guest: &mut Guest, family: u64, kind: u64, protocol: u64) -> u64 { let flags = kind & (SOCK_NONBLOCK | SOCK_CLOEXEC); if kind & !(TYPE_MASK | flags) != 0 { return errno::fail(errno::EINVAL); } - if family != u64::from(AF_INET) { - return errno::fail(errno::EAFNOSUPPORT); - } - let (proto, own) = match kind & TYPE_MASK { - SOCK_STREAM => (Proto::Stream, IPPROTO_TCP), - SOCK_DGRAM => (Proto::Dgram, IPPROTO_UDP), - SOCK_RAW => { + let domain = match family { + f if f == u64::from(AF_INET) => Domain::Inet, + f if f == u64::from(AF_UNIX) => Domain::Unix, + _ => return errno::fail(errno::EAFNOSUPPORT), + }; + let (proto, own) = match (kind & TYPE_MASK, domain) { + (SOCK_STREAM, Domain::Inet) => (Proto::Stream, IPPROTO_TCP), + (SOCK_DGRAM, Domain::Inet) => (Proto::Dgram, IPPROTO_UDP), + (SOCK_RAW, Domain::Inet) => { return refuse("SOCK_RAW: raw sockets reach below any confinement", errno::EPERM) } + (SOCK_STREAM, Domain::Unix) => (Proto::Stream, PF_UNIX), + // Linux gives a raw Unix socket datagram semantics. + (SOCK_DGRAM | SOCK_RAW, Domain::Unix) => (Proto::Dgram, PF_UNIX), + (SOCK_SEQPACKET, Domain::Unix) => { + return refuse( + "SOCK_SEQPACKET: a Unix stream here does not keep message boundaries", + errno::ESOCKTNOSUPPORT, + ) + } _ => return errno::fail(errno::ESOCKTNOSUPPORT), }; // Anything but the type's own protocol, MPTCP included, is one this @@ -53,6 +68,6 @@ pub fn socket(guest: &mut Guest, family: u64, kind: u64, protocol: u64) -> u64 { if protocol != 0 && protocol != own { return errno::fail(errno::EPROTONOSUPPORT); } - let id = sock::with(|t| t.open(Domain::Inet, proto, Some(guest.pid))); + let id = sock::with(|t| t.open(domain, proto, Some(guest.pid))); install(guest, id, flags) } diff --git a/userland/capsule_linux/src/linux/net/unix_bind.rs b/userland/capsule_linux/src/linux/net/unix_bind.rs new file mode 100644 index 000000000..73f059e35 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/unix_bind.rs @@ -0,0 +1,79 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `bind` on a Unix socket: a path, which becomes a file as on Linux, an +//! abstract name, or family alone, which asks for a name to be chosen. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::file; +use crate::linux::guest::Guest; + +use super::sock::{self, UName}; +use super::sockaddr_un::UAddr; +use super::unix_calls::unix_addr; +use super::unix_name::resolve; + +pub fn bind(guest: &Guest, id: u32, at: u64, len: u64) -> u64 { + let ua = match unix_addr(guest, at, len) { + Ok(ua) => ua, + Err(e) => return e, + }; + if sock::with(|t| t.get(id).is_none_or(|s| s.uname.is_some())) { + return errno::fail(errno::EINVAL); + } + bind_name(guest, id, &ua) +} + +/// Bind Unix socket `id` to `ua`. A path must not exist yet, and becomes an +/// empty file; family alone chooses an abstract name of five hex digits. +fn bind_name(guest: &Guest, id: u32, ua: &UAddr) -> u64 { + let name = match resolve(guest, ua) { + Some(n) => n, + None => match fresh() { + Some(n) => n, + None => return errno::fail(errno::EADDRINUSE), + }, + }; + let taken = sock::with(|t| t.iter().any(|(_, s)| s.uname.as_ref() == Some(&name))); + if taken || (!name.is_abstract() && file::look(&name.key).is_some()) { + return errno::fail(errno::EADDRINUSE); + } + if !name.is_abstract() { + let key = file::key(&name.key); + if key.writable().is_err() { + return errno::fail(errno::EROFS); + } + // The store refuses a file whose directory is missing. + if file::store_write(&key, &[]).is_err() { + return errno::fail(errno::ENOENT); + } + } + sock::with(|t| t.get_mut(id).map(|s| s.uname = Some(name))); + errno::ok(0) +} + +/// Linux's autobind: a NUL and five hex digits, the first unused. +fn fresh() -> Option { + (0u32..0x10_0000).find_map(|n| { + let mut key: Vec = alloc::vec![0]; + key.extend_from_slice(alloc::format!("{n:05x}").as_bytes()); + let name = UName { key: key.clone(), shown: key }; + let used = sock::with(|t| t.iter().any(|(_, s)| s.uname.as_ref() == Some(&name))); + (!used).then_some(name) + }) +} diff --git a/userland/capsule_linux/src/linux/net/unix_calls.rs b/userland/capsule_linux/src/linux/net/unix_calls.rs new file mode 100644 index 000000000..5c9889708 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/unix_calls.rs @@ -0,0 +1,88 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `connect` on a Unix socket. A connect to the display's path +//! turns the descriptor into the display connection this capsule serves +//! itself (`unix`); every other name is the family's own. + +use crate::linux::abi::errno; +use crate::linux::guest::{Fd, Guest}; +use crate::linux::unix; + +use super::sock::{self, Link, Proto}; +use super::sockaddr::{self, AF_UNIX, AF_UNSPEC}; +use super::sockaddr_un::{self, UAddr}; +use super::unix_name::{find, resolve}; + +pub fn connect(guest: &mut Guest, fd: u64, id: u32, proto: Proto, at: u64, len: u64) -> u64 { + if proto == Proto::Dgram && matches!(sockaddr::read(guest, at, len), Ok((AF_UNSPEC, _))) { + sock::with(|t| t.get_mut(id).map(|s| (s.peer, s.upeer, s.connected) = (None, None, false))); + return errno::ok(0); + } + let ua = match unix_addr(guest, at, len) { + Ok(ua) => ua, + Err(e) => return e, + }; + if let (Proto::Stream, UAddr::Path(p)) = (proto, &ua) { + if unix::is_display(p) { + return display(guest, fd, at, len); + } + } + let Some(name) = resolve(guest, &ua) else { + return errno::fail(errno::EINVAL); + }; + let target = match find(&name, proto) { + Ok(t) => t, + Err(e) => return errno::fail(e), + }; + sock::with(|t| { + let s = t.get_mut(id).ok_or(errno::EBADF)?; + match proto { + Proto::Stream if s.connected => Err(errno::EISCONN), + Proto::Stream if s.listening => Err(errno::EINVAL), + // A Unix connect completes in the caller's call, blocking or not. + Proto::Stream => match t.join(id, target) { + Link::Done => Ok(()), + Link::Full => Err(errno::EAGAIN), + Link::Refused => Err(errno::ECONNREFUSED), + }, + Proto::Dgram => { + (s.peer, s.upeer, s.connected) = (Some(target), Some(name), true); + Ok(()) + } + } + }) + .map_or_else(errno::fail, |()| errno::ok(0)) +} + +pub fn unix_addr(guest: &Guest, at: u64, len: u64) -> Result { + match sockaddr::read(guest, at, len)? { + (AF_UNIX, _) => sockaddr_un::read(guest, at, len), + _ => Err(errno::fail(errno::EINVAL)), + } +} + +/// Let go of the family socket and make `fd` the display connection. +fn display(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { + super::close::close(guest, fd); + if let Some(f) = guest.fds.get_mut(fd as usize) { + let (cloexec, nonblock) = (f.cloexec, f.nonblock); + *f = Fd::unix(); + f.cloexec = cloexec; + f.nonblock = nonblock; + } + unix::connect(guest, fd, at, len) +} diff --git a/userland/capsule_linux/src/linux/net/unix_name.rs b/userland/capsule_linux/src/linux/net/unix_name.rs new file mode 100644 index 000000000..dfb85edd2 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/unix_name.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Unix names. A path is a file, as on Linux: bind makes it, it stays after +//! the socket closes until the guest unlinks it, and a connect to a path +//! with no listener is ECONNREFUSED if the file is there and ENOENT if not. +//! An abstract name has no file and goes with its socket. Neither is seen +//! outside the family: only the family's sockets are bound to them. + +use crate::linux::abi::errno; +use crate::linux::file; +use crate::linux::guest::Guest; + +use super::sock::{self, Domain, Proto, UName}; +use super::sockaddr_un::UAddr; + +/// The name `ua` means for this guest; None asks for one to be chosen. +pub fn resolve(guest: &Guest, ua: &UAddr) -> Option { + match ua { + UAddr::Auto => None, + UAddr::Path(p) => Some(UName { key: file::visible(&guest.cwd, p), shown: p.clone() }), + UAddr::Abstract(n) => { + let mut key = alloc::vec![0u8]; + key.extend_from_slice(n); + Some(UName { key: key.clone(), shown: key }) + } + } +} + +/// The family socket of kind `proto` bound to `name`, the one a connect or +/// a send reaches: a stream's must listen. Otherwise Linux's answer. +pub fn find(name: &UName, proto: Proto) -> Result { + // An accepted connection carries its listener's name, as on Linux; the + // listener is the one a connect reaches. + let found = sock::with(|t| { + let named = + || t.iter().filter(|(_, s)| s.domain == Domain::Unix && s.uname.as_ref() == Some(name)); + named() + .find(|(_, s)| s.listening) + .or_else(|| named().next()) + .map(|(i, s)| (i, s.proto, s.listening)) + }); + match found { + Some((_, p, _)) if p != proto => Err(errno::EPROTOTYPE), + Some((i, Proto::Dgram, _)) | Some((i, _, true)) => Ok(i), + Some(_) => Err(errno::ECONNREFUSED), + None if name.is_abstract() || file::look(&name.key).is_some() => Err(errno::ECONNREFUSED), + None => Err(errno::ENOENT), + } +} diff --git a/userland/capsule_linux/src/linux/net/xfer_in.rs b/userland/capsule_linux/src/linux/net/xfer_in.rs index a0bdaa13b..2cb501676 100644 --- a/userland/capsule_linux/src/linux/net/xfer_in.rs +++ b/userland/capsule_linux/src/linux/net/xfer_in.rs @@ -23,7 +23,7 @@ use crate::linux::guest::Guest; use super::flags::{MSG_OOB, MSG_PEEK, MSG_TRUNC}; use super::iov::{self, Iov}; -use super::sock::{self, Addr, Domain, Proto}; +use super::sock::{self, Peer, Proto}; pub struct In { /// Bytes put in the guest's buffers. @@ -32,7 +32,7 @@ pub struct In { pub whole: usize, /// Where a datagram came from; a stream, and an unnamed sender, say /// nothing. - pub from: Option<(Domain, Addr)>, + pub from: Option, } /// Receive into `iov` from socket `id`, `skip` bytes of it filled already. @@ -40,8 +40,7 @@ pub fn recv(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64) -> Resul if flags & MSG_OOB != 0 { return Err(errno::fail(errno::EINVAL)); } - let Some((proto, domain, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.domain, s.svc))) - else { + let Some((proto, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.svc))) else { return Err(errno::fail(errno::EBADF)); }; let want = iov::total(iov).saturating_sub(skip); @@ -64,7 +63,10 @@ pub fn recv(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64) -> Resul let got = sock::with(|t| t.take_gram(id, want, peek)).map_err(errno::fail)?; iov::scatter(guest, iov, skip, &got.bytes)?; // A socketpair's peer has no name, and Linux gives none. - let from = (domain == Domain::Inet).then_some((domain, got.from)); + let from = match got.from { + Peer::Unix(None) => None, + named => Some(named), + }; Ok(In { n: got.bytes.len(), whole: got.whole, from }) } } diff --git a/userland/capsule_linux/src/linux/net/xfer_out.rs b/userland/capsule_linux/src/linux/net/xfer_out.rs index aa8a59410..97fe43fc8 100644 --- a/userland/capsule_linux/src/linux/net/xfer_out.rs +++ b/userland/capsule_linux/src/linux/net/xfer_out.rs @@ -24,7 +24,8 @@ use crate::linux::guest::Guest; use super::flags::MSG_OOB; use super::iov::{self, Iov}; -use super::sock::{self, Addr, Proto}; +use super::peer_addr::To; +use super::sock::{self, Proto}; /// What one send takes from the guest at most: the default receive buffer /// of a stream's peer, so a single call can fill it. @@ -34,11 +35,12 @@ const GRAM_CAP: usize = 65508; /// Send the message in `iov` from socket `id`, `skip` bytes of it having /// gone already: the count sent now, or an errno. -pub fn send(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64, to: Option) -> u64 { +pub fn send(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64, to: Option) -> u64 { if flags & MSG_OOB != 0 { return errno::fail(errno::EOPNOTSUPP); } - let Some((proto, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.svc))) else { + let Some((proto, domain, svc)) = sock::with(|t| t.get(id).map(|s| (s.proto, s.domain, s.svc))) + else { return errno::fail(errno::EBADF); }; let cap = match (svc, proto) { @@ -53,9 +55,13 @@ pub fn send(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64, to: Opti if let Some(h) = svc { return super::stream::send_bytes(h, &bytes); } + let dest = match super::dest::dest(guest, proto, domain, to) { + Ok(d) => d, + Err(e) => return e, + }; let sent = sock::with(|t| match proto { Proto::Stream => t.write(id, &bytes), - Proto::Dgram => t.autobind(id).and_then(|()| t.send_gram(id, to, &bytes)), + Proto::Dgram => t.autobind(id).and_then(|()| t.send_gram(id, dest, &bytes)), }); match sent { Ok(n) => errno::ok(n as u64), diff --git a/userland/capsule_linux/src/linux/serve/table_net.rs b/userland/capsule_linux/src/linux/serve/table_net.rs index b8dd1fd1c..119cd7a7f 100644 --- a/userland/capsule_linux/src/linux/serve/table_net.rs +++ b/userland/capsule_linux/src/linux/serve/table_net.rs @@ -27,7 +27,6 @@ use crate::linux::unix::{self, is_unix}; pub fn net_ops(guest: &mut Guest, tid: u32, nr: u64, a: [u64; 6]) -> Option { let _ = tid; Some(match nr { - nr::SOCKET if a[0] == 1 => unix::socket(guest, a[1]), nr::SOCKET => net::socket(guest, a[0], a[1], a[2]), nr::SOCKETPAIR => net::socketpair(guest, a[0], a[1], a[2], a[3]), nr::CONNECT if is_unix(guest, a[0]) => unix::connect(guest, a[0], a[1], a[2]), diff --git a/userland/capsule_linux/src/linux/unix/mod.rs b/userland/capsule_linux/src/linux/unix/mod.rs index b03ee7949..44e0e2aa1 100644 --- a/userland/capsule_linux/src/linux/unix/mod.rs +++ b/userland/capsule_linux/src/linux/unix/mod.rs @@ -15,7 +15,8 @@ // along with this program. If not, see . -//! Unix domain sockets, both ends inside this capsule. +//! The display connection: a Unix socket whose far end is this capsule. +//! Every other Unix socket is the family's own (`net::sock`). mod conn; mod give; @@ -32,5 +33,6 @@ pub use conn::Conn; pub use sock::is_unix; pub use recvmsg::recvmsg; pub use sendmsg::sendmsg; -pub use sock::{connect, socket}; +pub use path::is_display; +pub use sock::connect; pub use sock_io::{recv, send}; diff --git a/userland/capsule_linux/src/linux/unix/sock.rs b/userland/capsule_linux/src/linux/unix/sock.rs index 2e290cd5e..d6d16a1fb 100644 --- a/userland/capsule_linux/src/linux/unix/sock.rs +++ b/userland/capsule_linux/src/linux/unix/sock.rs @@ -14,35 +14,21 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - -//! The four calls a client makes on a display socket. +//! Connecting to the display, and telling a display socket apart. use crate::linux::abi::errno; -use crate::linux::guest::{Fd, Guest, Kind}; +use crate::linux::guest::{Guest, Kind}; use super::path::{is_display, sun_path}; -const SOCK_STREAM: u64 = 1; -const TYPE_MASK: u64 = 0xFF; - -pub fn socket(guest: &mut Guest, kind: u64) -> u64 { - if kind & TYPE_MASK != SOCK_STREAM { - return errno::fail(errno::ENOSYS); - } - match crate::linux::file::install(guest, Fd::unix()) { - Some(n) => errno::ok(n), - None => errno::fail(errno::EMFILE), - } -} - pub fn connect(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { let Some(path) = sun_path(guest, at, len) else { return errno::fail(errno::EINVAL); }; if !is_display(&path) { /* - * Nothing else listens in here, and a client that reaches a socket - * which silently accepts would block forever on a reply. + * Only the display is served here; every other name is a family + * socket's (net::unix_calls), which never reaches this call. */ return errno::fail(errno::ECONNREFUSED); } From 2f72985ceccf80427f6d631d182e5a838da1542c Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 21:18:31 +0000 Subject: [PATCH 09/14] linux guests: cunix, Unix sockets with names Nothing proved a named Unix socket. cunix (C, ids 5010 and 5011), six parts, each named when it fails: a listener on a path with its name, its client's and a connect that completes at once; ENOENT, ECONNREFUSED and EADDRINUSE for what a path leaves behind, and a rebind after unlink; abstract datagrams and the sender each reports; a connected datagram socket that refuses a stranger with EPERM; autobind; and a connection across fork. Linux: "[C] cunix PASS: 6 parts". --- userland/linux_guests/Guests.mk | 6 ++ userland/linux_guests/c/cunix.c | 96 ++++++++++++++++++++++++++ userland/linux_guests/c/cunix_parts.h | 66 ++++++++++++++++++ userland/linux_guests/c/cunix_parts2.h | 86 +++++++++++++++++++++++ 4 files changed, 254 insertions(+) create mode 100644 userland/linux_guests/c/cunix.c create mode 100644 userland/linux_guests/c/cunix_parts.h create mode 100644 userland/linux_guests/c/cunix_parts2.h diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index f4d3b90a4..ea197ab4a 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -154,6 +154,12 @@ $(LINUX_GUESTS_C)/cidle: $(LINUX_GUESTS_DIR)/c/cidle.c @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cidle,5008,5009,$(LINUX_GUESTS_C)/cidle)) +# Unix sockets with names: a path, what it leaves behind, abstract names, +# a connected datagram socket, autobind, and a connection across fork. +$(LINUX_GUESTS_C)/cunix: $(LINUX_GUESTS_DIR)/c/cunix.c $(wildcard $(LINUX_GUESTS_DIR)/c/cunix_parts*.h) + @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< +$(eval $(call LINUX_GUEST,cunix,5010,5011,$(LINUX_GUESTS_C)/cunix)) + # The Linux-guest test store is about guests, not the desktop's media and demo # capsules. Drop both so the signed guest set fits the vfs load budget; the # normal image, which does not set NONOS_LINUX_GUESTS, still ships them. diff --git a/userland/linux_guests/c/cunix.c b/userland/linux_guests/c/cunix.c new file mode 100644 index 000000000..dc6309e99 --- /dev/null +++ b/userland/linux_guests/c/cunix.c @@ -0,0 +1,96 @@ +// Unix sockets with names, as Linux has them: a listener on a path, its +// name and its client's, a connect that completes at once; ENOENT for a path +// with nothing there and ECONNREFUSED for one with no listener; a path that +// stays after its socket closes until it is unlinked; abstract names and the +// sender a datagram reports; a connected datagram socket that refuses +// strangers; a name bind chooses; and a connection across fork. Each part +// prints as it passes and every part runs. +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include + +#define PATH "/tmp/cunix.sock" +#define GONE "/tmp/cunix.none" + +static int parts; + +static int fail(const char *what, long a, long b) { + printf("[C] cunix FAIL: %s (%ld, %ld)\n", what, a, b); + fflush(stdout); + return 1; +} + +static void ok(const char *part, const char *detail, long n) { + parts++; + printf("[C] cunix %s ok: %s %ld\n", part, detail, n); + fflush(stdout); +} + +// A sockaddr_un for a path, or for an abstract name when `abs` is set. +static socklen_t name(struct sockaddr_un *a, const char *p, int abs) { + memset(a, 0, sizeof *a); + a->sun_family = AF_UNIX; + if (abs) { + memcpy(a->sun_path + 1, p, strlen(p)); + return offsetof(struct sockaddr_un, sun_path) + 1 + strlen(p); + } + strcpy(a->sun_path, p); + return offsetof(struct sockaddr_un, sun_path) + strlen(p) + 1; +} + +static int path_stream(void) { + struct sockaddr_un a, b; + socklen_t l = name(&a, PATH, 0), bl = sizeof b; + unlink(PATH); + int s = socket(AF_UNIX, SOCK_STREAM, 0); + if (listen(s, 1) != -1 || errno != EINVAL) { + return fail("path_stream: listen before bind is EINVAL", errno, EINVAL); + } + if (bind(s, (void *)&a, l) || listen(s, 4)) { + return fail("path_stream: bind and listen", -1, errno); + } + int twice = socket(AF_UNIX, SOCK_STREAM, 0); + if (bind(twice, (void *)&a, l) != -1 || errno != EADDRINUSE) { + return fail("path_stream: a second bind is EADDRINUSE", errno, EADDRINUSE); + } + close(twice); + int c = socket(AF_UNIX, SOCK_STREAM | SOCK_NONBLOCK, 0); + if (connect(c, (void *)&a, l)) { + return fail("path_stream: a non-blocking connect completes at once", -1, errno); + } + int t = accept(s, (void *)&b, &bl); + if (t < 0 || bl != 2) { + return fail("path_stream: accept names an unnamed client with 2 bytes", t, bl); + } + bl = sizeof b; + getsockname(s, (void *)&b, &bl); + if (bl != l || strcmp(b.sun_path, PATH)) { + return fail("path_stream: getsockname", bl, l); + } + bl = sizeof b; + getpeername(c, (void *)&b, &bl); + if (bl != l || strcmp(b.sun_path, PATH)) { + return fail("path_stream: the client's peer is the path", bl, l); + } + char buf[8]; + if (write(c, "unix", 4) != 4 || read(t, buf, 8) != 4) { + return fail("path_stream: bytes", 0, errno); + } + close(c); + long eof = read(t, buf, 8); + close(t); + close(s); + if (eof != 0) { + return fail("path_stream: end of file", eof, 0); + } + ok("path_stream", "bound, named, connected, 4 bytes, eof; name length", l); + return 0; +} + +#include "cunix_parts.h" diff --git a/userland/linux_guests/c/cunix_parts.h b/userland/linux_guests/c/cunix_parts.h new file mode 100644 index 000000000..a49728c64 --- /dev/null +++ b/userland/linux_guests/c/cunix_parts.h @@ -0,0 +1,66 @@ +// cunix: what a path leaves behind, abstract datagrams, and fork. + +static int left_behind(void) { + struct sockaddr_un a, g; + socklen_t l = name(&a, PATH, 0), gl = name(&g, GONE, 0); + unlink(GONE); + int c = socket(AF_UNIX, SOCK_STREAM, 0); + int missing = connect(c, (void *)&g, gl) ? errno : 0; + int s = socket(AF_UNIX, SOCK_STREAM, 0); + bind(s, (void *)&a, l); + int unheard = connect(c, (void *)&a, l) ? errno : 0; + close(s); + int s2 = socket(AF_UNIX, SOCK_STREAM, 0); + int rebind = bind(s2, (void *)&a, l) ? errno : 0; + int closed = connect(c, (void *)&a, l) ? errno : 0; + unlink(PATH); + int after = bind(s2, (void *)&a, l) ? errno : 0; + close(s2); + close(c); + unlink(PATH); + if (missing != ENOENT || unheard != ECONNREFUSED || rebind != EADDRINUSE || + closed != ECONNREFUSED || after != 0) { + printf("[C] cunix left_behind got %d %d %d %d %d\n", missing, unheard, rebind, closed, + after); + return fail("left_behind: ENOENT, ECONNREFUSED, EADDRINUSE, ECONNREFUSED, 0", 0, 0); + } + ok("left_behind", "the path stays until unlink; rebind after it", after); + return 0; +} + +static int abstract_dgram(void) { + struct sockaddr_un x, y, b, g; + socklen_t xl = name(&x, "cunix-x", 1), yl = name(&y, "cunix-y", 1), gl = name(&g, GONE, 0); + int rx = socket(AF_UNIX, SOCK_DGRAM, 0), tx = socket(AF_UNIX, SOCK_DGRAM, 0); + char buf[8]; + socklen_t bl = sizeof b; + if (bind(rx, (void *)&x, xl) || getsockname(rx, (void *)&b, &bl) || bl != xl) { + return fail("abstract_dgram: bind and name", bl, xl); + } + sendto(tx, "a", 1, 0, (void *)&x, xl); + bl = sizeof b; + long got = recvfrom(rx, buf, 8, 0, (void *)&b, &bl); + if (got != 1 || bl != 0) { + return fail("abstract_dgram: an unnamed sender reports length 0", got, bl); + } + bind(tx, (void *)&y, yl); + sendto(tx, "b", 1, 0, (void *)&x, xl); + bl = sizeof b; + recvfrom(rx, buf, 8, 0, (void *)&b, &bl); + if (bl != yl || memcmp(b.sun_path, y.sun_path, yl - 2)) { + return fail("abstract_dgram: a named sender reports its name", bl, yl); + } + int missing = sendto(tx, "c", 1, 0, (void *)&g, gl) < 0 ? errno : 0; + int lone = socket(AF_UNIX, SOCK_DGRAM, 0); + int nopeer = send(lone, "d", 1, 0) < 0 ? errno : 0; + close(lone); + close(rx); + close(tx); + if (missing != ENOENT || nopeer != ENOTCONN) { + return fail("abstract_dgram: ENOENT, then ENOTCONN", missing, nopeer); + } + ok("abstract_dgram", "names reported, no peer errno", nopeer); + return 0; +} + +#include "cunix_parts2.h" diff --git a/userland/linux_guests/c/cunix_parts2.h b/userland/linux_guests/c/cunix_parts2.h new file mode 100644 index 000000000..7f7a79ed3 --- /dev/null +++ b/userland/linux_guests/c/cunix_parts2.h @@ -0,0 +1,86 @@ +// cunix: a connected datagram socket, a chosen name, and fork. + +static int connected_dgram(void) { + struct sockaddr_un r, a; + socklen_t rl = name(&r, "cunix-r", 1), al = name(&a, "cunix-a", 1); + int rs = socket(AF_UNIX, SOCK_DGRAM, 0), as = socket(AF_UNIX, SOCK_DGRAM, 0); + int stranger = socket(AF_UNIX, SOCK_DGRAM, 0); + bind(rs, (void *)&r, rl); + bind(as, (void *)&a, al); + // r talks only to a: a stranger's datagram to r is refused. + connect(rs, (void *)&a, al); + int refused = sendto(stranger, "s", 1, 0, (void *)&r, rl) < 0 ? errno : 0; + long sent = send(rs, "to-a", 4, 0); + char buf[8]; + long got = recv(as, buf, 8, 0); + close(rs); + close(as); + close(stranger); + if (refused != EPERM || sent != 4 || got != 4) { + return fail("connected_dgram: EPERM for a stranger, 4 bytes to the peer", refused, got); + } + ok("connected_dgram", "a stranger refused, errno", refused); + return 0; +} + +static int autobind(void) { + struct sockaddr_un a, b; + int s = socket(AF_UNIX, SOCK_DGRAM, 0); + memset(&a, 0, sizeof a); + a.sun_family = AF_UNIX; + socklen_t bl = sizeof b; + int rc = bind(s, (void *)&a, sizeof(sa_family_t)); + getsockname(s, (void *)&b, &bl); + close(s); + // Linux chooses a NUL and five hex digits. + if (rc || bl != 8 || b.sun_path[0] != 0) { + return fail("autobind: a NUL and five hex digits", rc, bl); + } + ok("autobind", "a chosen abstract name, length", bl); + return 0; +} + +static int across_fork(void) { + struct sockaddr_un a; + socklen_t l = name(&a, PATH, 0); + unlink(PATH); + int s = socket(AF_UNIX, SOCK_STREAM, 0); + bind(s, (void *)&a, l); + listen(s, 1); + pid_t kid = fork(); + if (kid == 0) { + int c = socket(AF_UNIX, SOCK_STREAM, 0); + _exit(connect(c, (void *)&a, l) || write(c, "kid", 3) != 3); + } + int t = accept(s, 0, 0); + char buf[8]; + long got = read(t, buf, 8); + int status = -1; + waitpid(kid, &status, 0); + close(t); + close(s); + unlink(PATH); + if (got != 3 || status != 0) { + return fail("across_fork: the child's connection and bytes", got, status); + } + ok("across_fork", "bytes from the child", got); + return 0; +} + +int main(void) { + int (*const part[])(void) = {path_stream, left_behind, abstract_dgram, + connected_dgram, autobind, across_fork}; + const int count = sizeof part / sizeof part[0]; + int failed = 0; + for (int i = 0; i < count; i++) { + failed += part[i](); + } + if (failed) { + printf("[C] cunix FAIL: %d parts failed, %d passed\n", failed, parts); + fflush(stdout); + return 1; + } + printf("[C] cunix PASS: %d parts\n", parts); + fflush(stdout); + return 0; +} From 7004fc885532a38040b0401862f14a9e10eb03fe Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 21:20:25 +0000 Subject: [PATCH 10/14] linux: a full listener holds a non-blocking connect until accept A non-blocking connect to a listener whose queue was full answered EAGAIN. Linux answers EINPROGRESS: the connect stays in SYN_SENT, not writable, a second connect on it is EALREADY, and it completes once an accept makes room. Now the listener keeps such connects in order (net/sock/syn.rs) and completes them at the accept that makes room; Linux completes them at its next SYN retransmit, so only the timing differs. Closing the listener, or shutting its reading side, refuses them with ECONNREFUSED. A blocking connect to a full listener still waits, as before. csock gains a sixteenth part, backlog, which shows it. Linux: "[C] csock PASS: 16 parts". --- .../capsule_linux/src/linux/net/accept.rs | 1 + .../capsule_linux/src/linux/net/connect_lo.rs | 9 +++ .../capsule_linux/src/linux/net/shutdown.rs | 7 +- .../capsule_linux/src/linux/net/sock/free.rs | 4 ++ .../capsule_linux/src/linux/net/sock/mod.rs | 1 + .../capsule_linux/src/linux/net/sock/new.rs | 2 + .../capsule_linux/src/linux/net/sock/ready.rs | 4 ++ .../capsule_linux/src/linux/net/sock/syn.rs | 66 +++++++++++++++++++ .../capsule_linux/src/linux/net/sock/types.rs | 4 ++ userland/linux_guests/c/csock.c | 9 +-- userland/linux_guests/c/csock_parts4.h | 39 +++++++++++ 11 files changed, 140 insertions(+), 6 deletions(-) create mode 100644 userland/capsule_linux/src/linux/net/sock/syn.rs diff --git a/userland/capsule_linux/src/linux/net/accept.rs b/userland/capsule_linux/src/linux/net/accept.rs index 23c62b6ad..bbf6f926c 100644 --- a/userland/capsule_linux/src/linux/net/accept.rs +++ b/userland/capsule_linux/src/linux/net/accept.rs @@ -42,6 +42,7 @@ pub fn accept4(guest: &mut Guest, fd: u64, at: u64, lenp: u64, flags: u64) -> u6 return Err(errno::EINVAL); } let child = s.pending.pop_front().ok_or(errno::EAGAIN)?; + t.make_room(id); let c = t.get_mut(child).ok_or(errno::ECONNABORTED)?; c.holders.push(pid); let from = match c.domain { diff --git a/userland/capsule_linux/src/linux/net/connect_lo.rs b/userland/capsule_linux/src/linux/net/connect_lo.rs index 45d937e5f..5be4f5ba4 100644 --- a/userland/capsule_linux/src/linux/net/connect_lo.rs +++ b/userland/capsule_linux/src/linux/net/connect_lo.rs @@ -26,6 +26,9 @@ pub fn loopback(id: u32, to: Addr, nonblock: bool) -> u64 { let Some(s) = t.get_mut(id) else { return errno::fail(errno::EBADF); }; + if s.connecting { + return errno::fail(errno::EALREADY); + } if s.connected || s.listening || s.svc.is_some() { return errno::fail(errno::EISCONN); } @@ -35,6 +38,12 @@ pub fn loopback(id: u32, to: Addr, nonblock: bool) -> u64 { return errno::fail(e); } let linked = t.link(id, to); + // A full listener keeps a non-blocking connect until accept makes + // room, as Linux's SYN_SENT does. + if let (Link::Full, true, Some(l)) = (&linked, nonblock, t.listener(to)) { + t.wait_room(id, l); + return errno::fail(errno::EINPROGRESS); + } // A connect that fails gives back the port it bound. if !matches!(linked, Link::Done) && bound_here { if let Some(s) = t.get_mut(id) { diff --git a/userland/capsule_linux/src/linux/net/shutdown.rs b/userland/capsule_linux/src/linux/net/shutdown.rs index 64540fd0c..5a8d3d8e2 100644 --- a/userland/capsule_linux/src/linux/net/shutdown.rs +++ b/userland/capsule_linux/src/linux/net/shutdown.rs @@ -51,9 +51,12 @@ pub fn shutdown(guest: &Guest, fd: u64, how: u64) -> u64 { // Shutting a listener's reading side stops it listening. if rd { s.listening = false; - for queued in core::mem::take(&mut s.pending) { - t.free(queued, true); + let (queued, waiting) = + (core::mem::take(&mut s.pending), core::mem::take(&mut s.syn)); + for q in queued { + t.free(q, true); } + t.refuse_waiting(waiting.into_iter()); } return errno::ok(0); } diff --git a/userland/capsule_linux/src/linux/net/sock/free.rs b/userland/capsule_linux/src/linux/net/sock/free.rs index 06265cd99..81d9ab53c 100644 --- a/userland/capsule_linux/src/linux/net/sock/free.rs +++ b/userland/capsule_linux/src/linux/net/sock/free.rs @@ -57,5 +57,9 @@ impl Socks { for queued in gone.pending { self.free(queued, true); } + self.refuse_waiting(gone.syn.into_iter()); + for l in self.list.iter_mut().flatten() { + l.syn.retain(|&c| c != id); + } } } diff --git a/userland/capsule_linux/src/linux/net/sock/mod.rs b/userland/capsule_linux/src/linux/net/sock/mod.rs index e17344506..d1aac09d9 100644 --- a/userland/capsule_linux/src/linux/net/sock/mod.rs +++ b/userland/capsule_linux/src/linux/net/sock/mod.rs @@ -40,6 +40,7 @@ mod progress; mod ready; mod recv; mod send; +mod syn; mod table; mod types; diff --git a/userland/capsule_linux/src/linux/net/sock/new.rs b/userland/capsule_linux/src/linux/net/sock/new.rs index 708d7f96f..81749c62d 100644 --- a/userland/capsule_linux/src/linux/net/sock/new.rs +++ b/userland/capsule_linux/src/linux/net/sock/new.rs @@ -33,6 +33,8 @@ impl Sock { listening: false, backlog: 0, pending: VecDeque::new(), + syn: VecDeque::new(), + connecting: false, peer: None, connected: false, rx: VecDeque::new(), diff --git a/userland/capsule_linux/src/linux/net/sock/ready.rs b/userland/capsule_linux/src/linux/net/sock/ready.rs index 4a77599e1..51537e408 100644 --- a/userland/capsule_linux/src/linux/net/sock/ready.rs +++ b/userland/capsule_linux/src/linux/net/sock/ready.rs @@ -46,6 +46,10 @@ fn of(s: &Sock, room: bool) -> u16 { if s.listening { return err | if s.pending.is_empty() { 0 } else { POLLIN }; } + // A connect waiting for room: SYN_SENT, neither readable nor writable. + if s.connecting { + return err; + } // Never connected, refused, or reset: Linux's TCP_CLOSE. if !s.connected || s.broken { let rd = if s.connected { POLLIN | POLLRDHUP } else { 0 }; diff --git a/userland/capsule_linux/src/linux/net/sock/syn.rs b/userland/capsule_linux/src/linux/net/sock/syn.rs new file mode 100644 index 000000000..623a75a29 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/syn.rs @@ -0,0 +1,66 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Connects a full listener turned away. On Linux a non-blocking connect to +//! a listener whose queue is full answers EINPROGRESS and completes when an +//! accept makes room (its SYN is sent again); here it completes at that +//! accept. Until then the socket is not writable, and a second connect is +//! EALREADY. + +use crate::linux::abi::errno::ECONNREFUSED; + +use super::table::Socks; + +impl Socks { + /// Queue uid=0(root) gid=0(root) groups=0(root)'s connect on listener `l`. + pub fn wait_room(&mut self, id: u32, l: u32) { + if let Some(c) = self.get_mut(id) { + c.connecting = true; + } + if let Some(ls) = self.get_mut(l) { + ls.syn.push_back(id); + } + } + + /// Complete the connects waiting on `l` while its queue has room. + pub fn make_room(&mut self, l: u32) { + loop { + let Some(ls) = self.get_mut(l) else { + return; + }; + if ls.pending.len() > ls.backlog { + return; + } + let Some(c) = ls.syn.pop_front() else { + return; + }; + if let Some(cs) = self.get_mut(c) { + cs.connecting = false; + self.join(c, l); + } + } + } + + /// Listener `l` is gone: the connects waiting on it are refused. + pub fn refuse_waiting(&mut self, syn: impl Iterator) { + for c in syn { + if let Some(cs) = self.get_mut(c) { + cs.connecting = false; + cs.error = ECONNREFUSED; + } + } + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/types.rs b/userland/capsule_linux/src/linux/net/sock/types.rs index 1a4a8294b..7ad528821 100644 --- a/userland/capsule_linux/src/linux/net/sock/types.rs +++ b/userland/capsule_linux/src/linux/net/sock/types.rs @@ -51,6 +51,10 @@ pub struct Sock { pub listening: bool, pub backlog: usize, pub pending: VecDeque, + /// Connects the full queue turned away, oldest first (`syn`). + pub syn: VecDeque, + /// This end's connect waits in a listener's `syn` queue. + pub connecting: bool, /// The other end of a stream, until it is let go. pub peer: Option, pub connected: bool, diff --git a/userland/linux_guests/c/csock.c b/userland/linux_guests/c/csock.c index 98e9297a3..c60506cea 100644 --- a/userland/linux_guests/c/csock.c +++ b/userland/linux_guests/c/csock.c @@ -4,9 +4,10 @@ // the peer end of file while the other way still flows, epoll readiness on a // listener, end of file, EAGAIN on an empty non-blocking receive, MSG_PEEK // and MSG_DONTWAIT, EPIPE after the peer is gone, an accept and a receive -// that wait for another thread, the options Go and a C server set, and a -// socket a forked child shares. Each part prints as it passes and every part -// runs, so one run names each part that fails. +// that wait for another thread, the options Go and a C server set, a +// socket a forked child shares, and a connect a full listener holds. Each +// part prints as it passes and every part runs, so one run names each part +// that fails. #define _GNU_SOURCE #include #include @@ -108,7 +109,7 @@ int main(void) { int (*const part[])(void) = { pair_stream, pair_dgram, listen_accept, nb_connect, refused, half_close, epoll_listener, eof, empty_recv, peek, epipe, blocking_accept, - blocking_recv, options, fork_share, + blocking_recv, options, fork_share, backlog, }; const int count = sizeof part / sizeof part[0]; long t0 = now_ms(); diff --git a/userland/linux_guests/c/csock_parts4.h b/userland/linux_guests/c/csock_parts4.h index 6d37228e3..0644025e3 100644 --- a/userland/linux_guests/c/csock_parts4.h +++ b/userland/linux_guests/c/csock_parts4.h @@ -86,3 +86,42 @@ static int fork_share(void) { ok("fork_share", "3 bytes from the child, then", then); return 0; } + +// A listener with a backlog of 0 queues one connect; the next non-blocking +// one answers EINPROGRESS, is not writable, and a second connect on it is +// EALREADY, until an accept makes room and it completes. +static int backlog(void) { + int s = socket(AF_INET, SOCK_STREAM, 0); + struct sockaddr_in sa = loopback(0); + socklen_t len = sizeof sa; + bind(s, (void *)&sa, sizeof sa); + getsockname(s, (void *)&sa, &len); + listen(s, 0); + int c0 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + int c1 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + connect(c0, (void *)&sa, sizeof sa); + int r1 = connect(c1, (void *)&sa, sizeof sa); + int e1 = errno; + struct pollfd p = {c1, POLLOUT, 0}; + int early = poll(&p, 1, 100); + int again = connect(c1, (void *)&sa, sizeof sa) ? errno : 0; + int a = accept(s, 0, 0); + p.revents = 0; + int later = poll(&p, 1, 3000); + int err = -1; + socklen_t el = sizeof err; + getsockopt(c1, SOL_SOCKET, SO_ERROR, &err, &el); + int b = accept(s, 0, 0); + close(a); + close(b); + close(c0); + close(c1); + close(s); + if (r1 != -1 || e1 != EINPROGRESS || early != 0 || again != EALREADY || later != 1 || + err != 0 || b < 0) { + printf("[C] csock backlog got %d %d %d %d %d %d %d\n", r1, e1, early, again, later, err, b); + return fail("backlog: EINPROGRESS, waits, EALREADY, then connected", e1, again); + } + ok("backlog", "a full queue's connect completed after accept; EALREADY", again); + return 0; +} From be5f745bdb2e32129c9c8fbcf03edd23a17fcaaa Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 21:22:17 +0000 Subject: [PATCH 11/14] linux: the options a loopback connection cannot tell apart are kept IP_TOS, IP_TTL, SO_PRIORITY, TCP_USER_TIMEOUT, TCP_QUICKACK and TCP_FASTOPEN were ENOPROTOOPT with a refused line, though Linux keeps each of them and nothing on a loopback connection changes with them; a TCP option on a Unix socket was kept, where Linux answers EOPNOTSUPP. Now each is kept with Linux's starting value and checked as Linux checks it (a TCP socket drops the two ECN bits of IP_TOS; IP_TTL takes 1 to 255, and -1 for the default of 64; a negative user timeout or Fast Open queue is EINVAL), and read back. A Unix socket answers EOPNOTSUPP for any level but SOL_SOCKET, and a datagram socket keeps answering ENOPROTOOPT for a TCP option. csock gains a seventeenth part, quiet_options. Linux: "[C] csock PASS: 17 parts". --- userland/capsule_linux/src/linux/net/mod.rs | 1 + .../capsule_linux/src/linux/net/opt_ids.rs | 8 +++ .../capsule_linux/src/linux/net/opt_more.rs | 68 +++++++++++++++++++ .../capsule_linux/src/linux/net/opt_set.rs | 13 +++- .../capsule_linux/src/linux/net/opt_value.rs | 7 ++ .../capsule_linux/src/linux/net/sock/mod.rs | 2 + .../capsule_linux/src/linux/net/sock/opts.rs | 2 + .../src/linux/net/sock/opts_more.rs | 33 +++++++++ userland/linux_guests/c/csock.c | 2 +- userland/linux_guests/c/csock_parts4.h | 30 ++++++++ 10 files changed, 163 insertions(+), 3 deletions(-) create mode 100644 userland/capsule_linux/src/linux/net/opt_more.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/opts_more.rs diff --git a/userland/capsule_linux/src/linux/net/mod.rs b/userland/capsule_linux/src/linux/net/mod.rs index 0a199bf09..73996cbbb 100644 --- a/userland/capsule_linux/src/linux/net/mod.rs +++ b/userland/capsule_linux/src/linux/net/mod.rs @@ -42,6 +42,7 @@ mod name; mod ops; mod opt_get; mod opt_ids; +mod opt_more; mod opt_set; mod opt_time; mod opt_value; diff --git a/userland/capsule_linux/src/linux/net/opt_ids.rs b/userland/capsule_linux/src/linux/net/opt_ids.rs index a2d4d4277..bbeb753ee 100644 --- a/userland/capsule_linux/src/linux/net/opt_ids.rs +++ b/userland/capsule_linux/src/linux/net/opt_ids.rs @@ -18,6 +18,7 @@ //! include/uapi/linux/in.h and include/uapi/linux/tcp.h. pub const SOL_SOCKET: u64 = 1; +pub const IPPROTO_IP: u64 = 0; pub const IPPROTO_TCP: u64 = 6; pub const IPPROTO_IPV6: u64 = 41; @@ -36,10 +37,17 @@ pub const SO_ACCEPTCONN: u64 = 30; pub const SO_PROTOCOL: u64 = 38; pub const SO_DOMAIN: u64 = 39; +pub const IP_TOS: u64 = 1; +pub const IP_TTL: u64 = 2; +pub const SO_PRIORITY: u64 = 12; + pub const TCP_NODELAY: u64 = 1; pub const TCP_KEEPIDLE: u64 = 4; pub const TCP_KEEPINTVL: u64 = 5; pub const TCP_KEEPCNT: u64 = 6; +pub const TCP_QUICKACK: u64 = 12; +pub const TCP_USER_TIMEOUT: u64 = 18; +pub const TCP_FASTOPEN: u64 = 23; /// net.core.rmem_max and wmem_max as Linux ships them: what SO_RCVBUF and /// SO_SNDBUF are held to before they are doubled. diff --git a/userland/capsule_linux/src/linux/net/opt_more.rs b/userland/capsule_linux/src/linux/net/opt_more.rs new file mode 100644 index 000000000..241cc3a61 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt_more.rs @@ -0,0 +1,68 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Options Linux keeps that change nothing a loopback connection can see: +//! the IP type of service and time to live, the priority a queueing +//! discipline would use, and TCP's user timeout, quick-ack and Fast Open +//! queue. Each is kept, checked as Linux checks it, and read back. + +use crate::linux::abi::errno; + +use super::opt_ids::{ + IPPROTO_IP, IPPROTO_TCP, IP_TOS, IP_TTL, SOL_SOCKET, SO_PRIORITY, TCP_FASTOPEN, TCP_QUICKACK, + TCP_USER_TIMEOUT, +}; +use super::sock::{More, Proto}; + +/// Linux's net.ipv4.ip_default_ttl. +const DEFAULT_TTL: u32 = 64; +/// The ECN bits of the type of service, which a TCP socket does not keep. +const ECN_MASK: u32 = 3; + +pub fn known(level: u64, name: u64) -> bool { + matches!( + (level, name), + (IPPROTO_IP, IP_TOS | IP_TTL) + | (SOL_SOCKET, SO_PRIORITY) + | (IPPROTO_TCP, TCP_USER_TIMEOUT | TCP_QUICKACK | TCP_FASTOPEN) + ) +} + +pub fn set(m: &mut More, proto: Proto, level: u64, name: u64, v: u32) -> u64 { + match (level, name) { + (IPPROTO_IP, IP_TOS) if proto == Proto::Stream => m.tos = v & 0xff & !ECN_MASK, + (IPPROTO_IP, IP_TOS) => m.tos = v & 0xff, + (IPPROTO_IP, IP_TTL) if v as i32 == -1 => m.ttl = DEFAULT_TTL, + (IPPROTO_IP, IP_TTL) if (1..=255).contains(&v) => m.ttl = v, + (SOL_SOCKET, SO_PRIORITY) => m.priority = v, + (IPPROTO_TCP, TCP_USER_TIMEOUT) if (v as i32) >= 0 => m.user_timeout = v, + (IPPROTO_TCP, TCP_QUICKACK) => m.quickack = v != 0, + (IPPROTO_TCP, TCP_FASTOPEN) if (v as i32) >= 0 => m.fastopen = v, + _ => return errno::fail(errno::EINVAL), + } + errno::ok(0) +} + +pub fn get(m: &More, level: u64, name: u64) -> u32 { + match (level, name) { + (IPPROTO_IP, IP_TOS) => m.tos, + (IPPROTO_IP, IP_TTL) => m.ttl, + (SOL_SOCKET, SO_PRIORITY) => m.priority, + (IPPROTO_TCP, TCP_USER_TIMEOUT) => m.user_timeout, + (IPPROTO_TCP, TCP_QUICKACK) => u32::from(m.quickack), + _ => m.fastopen, + } +} diff --git a/userland/capsule_linux/src/linux/net/opt_set.rs b/userland/capsule_linux/src/linux/net/opt_set.rs index 9704cf988..dff9835f4 100644 --- a/userland/capsule_linux/src/linux/net/opt_set.rs +++ b/userland/capsule_linux/src/linux/net/opt_set.rs @@ -24,7 +24,7 @@ use crate::linux::guest::Guest; use super::fd::sock_of; use super::opt_ids::*; use super::opt_time::{keep, timeo}; -use super::sock::{self, Proto}; +use super::sock::{self, Domain, Proto}; pub fn setsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, len: u64) -> u64 { let id = match sock_of(guest, fd) { @@ -44,9 +44,18 @@ pub fn setsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, len: let Some(s) = t.get_mut(id) else { return errno::fail(errno::EBADF); }; - let proto = s.proto; + let (proto, domain) = (s.proto, s.domain); let o = &mut s.opts; match (level, name) { + // A Unix socket has only socket-level options on Linux. + (l, _) if domain == Domain::Unix && l != SOL_SOCKET => { + return errno::fail(errno::EOPNOTSUPP) + } + (l, n) + if super::opt_more::known(l, n) && !(l == IPPROTO_TCP && proto == Proto::Dgram) => + { + return super::opt_more::set(&mut o.more, proto, l, n, int) + } (SOL_SOCKET, SO_REUSEADDR) => o.reuseaddr = int != 0, (SOL_SOCKET, SO_REUSEPORT) => o.reuseport = int != 0, (SOL_SOCKET, SO_KEEPALIVE) => o.keepalive = int != 0, diff --git a/userland/capsule_linux/src/linux/net/opt_value.rs b/userland/capsule_linux/src/linux/net/opt_value.rs index 9e1e1e82a..7c9eee16b 100644 --- a/userland/capsule_linux/src/linux/net/opt_value.rs +++ b/userland/capsule_linux/src/linux/net/opt_value.rs @@ -49,6 +49,13 @@ pub fn value(s: &mut Sock, level: u64, name: u64) -> Result, u64> { } (SOL_SOCKET, SO_RCVTIMEO) => pair(o.rcvtimeo.0, o.rcvtimeo.1), (SOL_SOCKET, SO_SNDTIMEO) => pair(o.sndtimeo.0, o.sndtimeo.1), + (l, _) if s.domain == Domain::Unix && l != SOL_SOCKET => { + Err(errno::fail(errno::EOPNOTSUPP)) + } + (IPPROTO_TCP, _) if !stream && super::opt_more::known(level, name) => { + Err(errno::fail(errno::EOPNOTSUPP)) + } + (l, n) if super::opt_more::known(l, n) => int(super::opt_more::get(&o.more, l, n)), // A datagram socket has no TCP options, and Linux says so this way. (IPPROTO_TCP, _) if !stream => Err(errno::fail(errno::EOPNOTSUPP)), (IPPROTO_TCP, TCP_NODELAY) => int(u32::from(o.nodelay)), diff --git a/userland/capsule_linux/src/linux/net/sock/mod.rs b/userland/capsule_linux/src/linux/net/sock/mod.rs index d1aac09d9..d4f0bb729 100644 --- a/userland/capsule_linux/src/linux/net/sock/mod.rs +++ b/userland/capsule_linux/src/linux/net/sock/mod.rs @@ -34,6 +34,7 @@ mod link; mod name; mod new; mod opts; +mod opts_more; mod pair; mod port; mod progress; @@ -49,6 +50,7 @@ pub use gram_dest::Dest; pub use holders::Holder; pub use link::Link; pub use name::{Peer, UName}; +pub use opts_more::More; pub use opts::Opts; pub use progress::{progress, set_progress}; pub use ready::bits; diff --git a/userland/capsule_linux/src/linux/net/sock/opts.rs b/userland/capsule_linux/src/linux/net/sock/opts.rs index 36532a2ab..4b6d268d8 100644 --- a/userland/capsule_linux/src/linux/net/sock/opts.rs +++ b/userland/capsule_linux/src/linux/net/sock/opts.rs @@ -39,6 +39,7 @@ pub struct Opts { /// microseconds, so they read back exactly. pub rcvtimeo: (u64, u64), pub sndtimeo: (u64, u64), + pub more: super::opts_more::More, } impl Opts { @@ -61,6 +62,7 @@ impl Opts { linger: (0, 0), rcvtimeo: (0, 0), sndtimeo: (0, 0), + more: Default::default(), } } diff --git a/userland/capsule_linux/src/linux/net/sock/opts_more.rs b/userland/capsule_linux/src/linux/net/sock/opts_more.rs new file mode 100644 index 000000000..1e10aa6e0 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/opts_more.rs @@ -0,0 +1,33 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The options `opt_more` keeps, with Linux's starting values. + +#[derive(Clone, Copy)] +pub struct More { + pub tos: u32, + pub ttl: u32, + pub priority: u32, + pub user_timeout: u32, + pub quickack: bool, + pub fastopen: u32, +} + +impl Default for More { + fn default() -> More { + More { tos: 0, ttl: 64, priority: 0, user_timeout: 0, quickack: true, fastopen: 0 } + } +} diff --git a/userland/linux_guests/c/csock.c b/userland/linux_guests/c/csock.c index c60506cea..2dc78193b 100644 --- a/userland/linux_guests/c/csock.c +++ b/userland/linux_guests/c/csock.c @@ -109,7 +109,7 @@ int main(void) { int (*const part[])(void) = { pair_stream, pair_dgram, listen_accept, nb_connect, refused, half_close, epoll_listener, eof, empty_recv, peek, epipe, blocking_accept, - blocking_recv, options, fork_share, backlog, + blocking_recv, options, fork_share, backlog, quiet_options, }; const int count = sizeof part / sizeof part[0]; long t0 = now_ms(); diff --git a/userland/linux_guests/c/csock_parts4.h b/userland/linux_guests/c/csock_parts4.h index 0644025e3..97968d831 100644 --- a/userland/linux_guests/c/csock_parts4.h +++ b/userland/linux_guests/c/csock_parts4.h @@ -125,3 +125,33 @@ static int backlog(void) { ok("backlog", "a full queue's connect completed after accept; EALREADY", again); return 0; } + +// The options Linux keeps that a loopback connection cannot tell apart, +// read back as Linux reads them, and the ones a socket's kind refuses. +static int quiet_options(void) { + int t = socket(AF_INET, SOCK_STREAM, 0), u = socket(AF_INET, SOCK_DGRAM, 0); + int x = socket(AF_UNIX, SOCK_STREAM, 0); + int tos = 0x13, ttl = 32, zero = 0, ut = 5000, prio = 6; + setsockopt(t, IPPROTO_IP, IP_TOS, &tos, sizeof tos); + setsockopt(t, IPPROTO_IP, IP_TTL, &ttl, sizeof ttl); + int bad_ttl = setsockopt(t, IPPROTO_IP, IP_TTL, &zero, sizeof zero) ? errno : 0; + setsockopt(t, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut); + setsockopt(t, SOL_SOCKET, SO_PRIORITY, &prio, sizeof prio); + int got_tos = get_int(t, IPPROTO_IP, IP_TOS), got_ttl = get_int(t, IPPROTO_IP, IP_TTL); + int got_ut = get_int(t, IPPROTO_TCP, TCP_USER_TIMEOUT); + int got_qa = get_int(t, IPPROTO_TCP, TCP_QUICKACK), got_prio = get_int(t, SOL_SOCKET, SO_PRIORITY); + int udp_tcp = setsockopt(u, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut) ? errno : 0; + int unix_tcp = setsockopt(x, IPPROTO_TCP, TCP_NODELAY, &prio, sizeof prio) ? errno : 0; + close(t); + close(u); + close(x); + // A TCP socket drops the two ECN bits of the type of service. + if (got_tos != 0x10 || got_ttl != 32 || bad_ttl != EINVAL || got_ut != 5000 || got_qa != 1 || + got_prio != 6 || udp_tcp != ENOPROTOOPT || unix_tcp != EOPNOTSUPP) { + printf("[C] csock quiet_options got %d %d %d %d %d %d %d %d\n", got_tos, got_ttl, bad_ttl, + got_ut, got_qa, got_prio, udp_tcp, unix_tcp); + return fail("quiet_options: kept and read back as Linux does", got_tos, got_ttl); + } + ok("quiet_options", "6 kept and read back; refusals", unix_tcp); + return 0; +} From 7a9428bd4b99f9292e3de34f8d662ff0d40d3fb5 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 21:23:29 +0000 Subject: [PATCH 12/14] linux: listeners that set SO_REUSEPORT share a port A second listener on an address answered EADDRINUSE with SO_REUSEPORT set on both, which Linux allows: a server that runs one listener per worker could not start its second. Now sockets that each set SO_REUSEPORT may bind and listen on one address, and a connect goes to one of them by its own port, as Linux spreads connections by a hash of each; one without the option still answers EADDRINUSE, and when one listener closes the others take what comes next. csock gains an eighteenth part, reuseport. Linux: "[C] csock PASS: 18 parts". --- .../capsule_linux/src/linux/net/connect_lo.rs | 3 +- .../capsule_linux/src/linux/net/listen.rs | 10 +++- .../capsule_linux/src/linux/net/sock/link.rs | 3 +- .../capsule_linux/src/linux/net/sock/port.rs | 20 +++++--- userland/linux_guests/c/csock.c | 9 ++-- userland/linux_guests/c/csock_parts4.h | 49 +++++++++++++++++++ 6 files changed, 80 insertions(+), 14 deletions(-) diff --git a/userland/capsule_linux/src/linux/net/connect_lo.rs b/userland/capsule_linux/src/linux/net/connect_lo.rs index 5be4f5ba4..df85325f5 100644 --- a/userland/capsule_linux/src/linux/net/connect_lo.rs +++ b/userland/capsule_linux/src/linux/net/connect_lo.rs @@ -40,7 +40,8 @@ pub fn loopback(id: u32, to: Addr, nonblock: bool) -> u64 { let linked = t.link(id, to); // A full listener keeps a non-blocking connect until accept makes // room, as Linux's SYN_SENT does. - if let (Link::Full, true, Some(l)) = (&linked, nonblock, t.listener(to)) { + let from = t.get(id).and_then(|s| s.local).map_or(0, |a| a.port); + if let (Link::Full, true, Some(l)) = (&linked, nonblock, t.listener(to, from)) { t.wait_room(id, l); return errno::fail(errno::EINPROGRESS); } diff --git a/userland/capsule_linux/src/linux/net/listen.rs b/userland/capsule_linux/src/linux/net/listen.rs index 9d6fc71bb..e6313b4b8 100644 --- a/userland/capsule_linux/src/linux/net/listen.rs +++ b/userland/capsule_linux/src/linux/net/listen.rs @@ -40,7 +40,15 @@ pub fn listen(guest: &mut Guest, fd: u64, backlog: u64) -> u64 { (_, Domain::Unix, _) => {} // Linux would bind 0.0.0.0 here, which is not the family's own. (_, _, None) => return not_loopback("listen", Addr::default()), - (_, _, Some(at)) if !s.listening && t.listener(at).is_some() => { + // Listeners share an address only when each set SO_REUSEPORT. + (_, _, Some(at)) + if !s.listening + && t.iter().any(|(_, o)| { + o.listening + && o.local == Some(at) + && !(o.opts.reuseport && s.opts.reuseport) + }) => + { return errno::fail(errno::EADDRINUSE) } _ => {} diff --git a/userland/capsule_linux/src/linux/net/sock/link.rs b/userland/capsule_linux/src/linux/net/sock/link.rs index c24b04a27..2fafa8b06 100644 --- a/userland/capsule_linux/src/linux/net/sock/link.rs +++ b/userland/capsule_linux/src/linux/net/sock/link.rs @@ -32,7 +32,8 @@ pub enum Link { impl Socks { /// Connect stream `id`, already bound, to the listener at `to`. pub fn link(&mut self, id: u32, to: Addr) -> Link { - match self.listener(to) { + let from = self.get(id).and_then(|s| s.local).map_or(0, |a| a.port); + match self.listener(to, from) { Some(l) => self.join(id, l), None => Link::Refused, } diff --git a/userland/capsule_linux/src/linux/net/sock/port.rs b/userland/capsule_linux/src/linux/net/sock/port.rs index c6d136671..e006c3dcb 100644 --- a/userland/capsule_linux/src/linux/net/sock/port.rs +++ b/userland/capsule_linux/src/linux/net/sock/port.rs @@ -30,16 +30,21 @@ impl Socks { self.iter().find(|(_, s)| s.proto == proto && s.local == Some(at)).map(|(i, _)| i) } - /// The listener a connect to `at` reaches. - pub fn listener(&self, at: Addr) -> Option { - self.iter() - .find(|(_, s)| s.listening && s.proto == Proto::Stream && s.local == Some(at)) + /// The listener a connect to `at` from port `from` reaches. Listeners + /// that share a port with SO_REUSEPORT take connections by the + /// connecting port, as Linux spreads them by a hash of the connection. + pub fn listener(&self, at: Addr, from: u16) -> Option { + let group: alloc::vec::Vec = self + .iter() + .filter(|(_, s)| s.listening && s.proto == Proto::Stream && s.local == Some(at)) .map(|(i, _)| i) + .collect(); + group.get(usize::from(from) % group.len().max(1)).copied() } /// True when binding `id` to `at` takes a port another socket holds. - /// Two sockets share one only when both set SO_REUSEADDR and neither - /// listens, as Linux allows a restarted server to rebind. + /// Two sockets share one when both set SO_REUSEPORT, or when both set + /// SO_REUSEADDR and the other does not listen, as Linux allows. pub fn in_use(&self, id: u32, at: Addr) -> bool { let Some(me) = self.get(id) else { return true; @@ -47,7 +52,8 @@ impl Socks { self.iter().any(|(i, s)| { i != id && s.proto == me.proto - && s.local.is_some_and(|l| l.port == at.port && (l.ip == at.ip)) + && s.local == Some(at) + && !(s.opts.reuseport && me.opts.reuseport) && (s.listening || !(s.opts.reuseaddr && me.opts.reuseaddr)) }) } diff --git a/userland/linux_guests/c/csock.c b/userland/linux_guests/c/csock.c index 2dc78193b..b392bb1fe 100644 --- a/userland/linux_guests/c/csock.c +++ b/userland/linux_guests/c/csock.c @@ -5,9 +5,10 @@ // listener, end of file, EAGAIN on an empty non-blocking receive, MSG_PEEK // and MSG_DONTWAIT, EPIPE after the peer is gone, an accept and a receive // that wait for another thread, the options Go and a C server set, a -// socket a forked child shares, and a connect a full listener holds. Each -// part prints as it passes and every part runs, so one run names each part -// that fails. +// socket a forked child shares, a connect a full listener holds, the +// options a loopback connection cannot tell apart, and listeners that share +// a port with SO_REUSEPORT. Each part prints as it passes and every part +// runs, so one run names each part that fails. #define _GNU_SOURCE #include #include @@ -109,7 +110,7 @@ int main(void) { int (*const part[])(void) = { pair_stream, pair_dgram, listen_accept, nb_connect, refused, half_close, epoll_listener, eof, empty_recv, peek, epipe, blocking_accept, - blocking_recv, options, fork_share, backlog, quiet_options, + blocking_recv, options, fork_share, backlog, quiet_options, reuseport, }; const int count = sizeof part / sizeof part[0]; long t0 = now_ms(); diff --git a/userland/linux_guests/c/csock_parts4.h b/userland/linux_guests/c/csock_parts4.h index 97968d831..1953a6bf1 100644 --- a/userland/linux_guests/c/csock_parts4.h +++ b/userland/linux_guests/c/csock_parts4.h @@ -155,3 +155,52 @@ static int quiet_options(void) { ok("quiet_options", "6 kept and read back; refusals", unix_tcp); return 0; } + +// Listeners that set SO_REUSEPORT share a port and between them take every +// connection; a socket without it cannot bind there, and when one listener +// closes the other takes what comes next. +static int reuseport(void) { + struct sockaddr_in sa = loopback(0); + socklen_t len = sizeof sa; + int one = 1, l[2], taken[2] = {0, 0}; + for (int i = 0; i < 2; i++) { + l[i] = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + setsockopt(l[i], SOL_SOCKET, SO_REUSEPORT, &one, sizeof one); + if (bind(l[i], (void *)&sa, sizeof sa) || listen(l[i], 16)) { + return fail("reuseport: two listeners on one port", i, errno); + } + getsockname(l[i], (void *)&sa, &len); + } + int plain = socket(AF_INET, SOCK_STREAM, 0); + int refused = bind(plain, (void *)&sa, sizeof sa) ? errno : 0; + close(plain); + int c[16]; + for (int i = 0; i < 16; i++) { + c[i] = dial(ntohs(sa.sin_port)); + } + for (int i = 0; i < 2; i++) { + int a; + while ((a = accept(l[i], 0, 0)) >= 0) { + taken[i]++; + close(a); + } + } + close(l[0]); + int late = dial(ntohs(sa.sin_port)); + struct pollfd p = {l[1], POLLIN, 0}; + int ready = poll(&p, 1, 3000); + int last = accept(l[1], 0, 0); + for (int i = 0; i < 16; i++) { + close(c[i]); + } + close(late); + close(last); + close(l[1]); + if (refused != EADDRINUSE || taken[0] + taken[1] != 16 || !taken[0] || !taken[1] || + ready != 1 || last < 0) { + printf("[C] csock reuseport got %d %d+%d %d %d\n", refused, taken[0], taken[1], ready, last); + return fail("reuseport: shared, spread, and the survivor takes the rest", taken[0], taken[1]); + } + ok("reuseport", "16 connections spread over 2 listeners; without it errno", refused); + return 0; +} From d629c36ef8bad7003a0262cbc5919debb8fde140 Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 21:56:43 +0000 Subject: [PATCH 13/14] linux: net files hold at most 75 lines, comments in /* */ Every file under net/ and serve/waits_sock*.rs now holds at most 75 lines. The Unix name calls move to net/named/, the options to net/opt/, and the longer table methods split by what they do (deliver, put, take, unlisten, gram_target, kinds). mod.rs files hold module declarations and re-exports only; net/api.rs carries the re-exports the rest of the personality calls. Plain comments are /* */; doc comments stay /// so rustdoc still reads them. The lines this branch adds to guest/handle.rs, serve/dispatch.rs and serve/family_waits.rs are shortened; those three files were over 75 lines before this branch and are not split here. No call answers differently: csock, cudp, cunix, cpolicy and cidle pass on NONOS as they did before the split. --- userland/capsule_linux/src/linux/abi/mod.rs | 2 +- userland/capsule_linux/src/linux/abi/nr.rs | 1 - .../capsule_linux/src/linux/guest/handle.rs | 4 +- userland/capsule_linux/src/linux/net/api.rs | 42 ++++++++++ userland/capsule_linux/src/linux/net/bind.rs | 2 +- .../capsule_linux/src/linux/net/call_kind.rs | 21 ++++- userland/capsule_linux/src/linux/net/cap.rs | 33 ++++++++ .../capsule_linux/src/linux/net/connect.rs | 4 +- .../src/linux/net/connect_dial.rs | 57 +++++++++++++ .../capsule_linux/src/linux/net/connect_lo.rs | 8 +- .../src/linux/net/connect_out.rs | 36 ++------ userland/capsule_linux/src/linux/net/dest.rs | 4 +- userland/capsule_linux/src/linux/net/dgram.rs | 27 +----- .../capsule_linux/src/linux/net/listen.rs | 8 +- userland/capsule_linux/src/linux/net/mmsg.rs | 30 +------ .../capsule_linux/src/linux/net/mmsg_each.rs | 50 +++++++++++ userland/capsule_linux/src/linux/net/mod.rs | 47 +++-------- userland/capsule_linux/src/linux/net/msg.rs | 35 +------- .../capsule_linux/src/linux/net/msg_recv.rs | 52 ++++++++++++ userland/capsule_linux/src/linux/net/name.rs | 8 +- .../net/{sockaddr_un.rs => named/addr.rs} | 11 ++- .../capsule_linux/src/linux/net/named/auto.rs | 32 +++++++ .../linux/net/{unix_bind.rs => named/bind.rs} | 23 ++--- .../net/{unix_calls.rs => named/connect.rs} | 32 ++----- .../src/linux/net/named/display.rs | 32 +++++++ .../capsule_linux/src/linux/net/named/mod.rs | 30 +++++++ .../linux/net/{unix_name.rs => named/name.rs} | 10 ++- .../capsule_linux/src/linux/net/opt/apply.rs | 63 ++++++++++++++ .../src/linux/net/{opt_get.rs => opt/get.rs} | 8 +- .../src/linux/net/{opt_ids.rs => opt/ids.rs} | 0 .../capsule_linux/src/linux/net/opt/mod.rs | 29 +++++++ .../linux/net/{opt_more.rs => opt/more.rs} | 4 +- .../capsule_linux/src/linux/net/opt/set.rs | 42 ++++++++++ .../linux/net/{opt_time.rs => opt/time.rs} | 2 +- .../linux/net/{opt_value.rs => opt/value.rs} | 12 +-- .../capsule_linux/src/linux/net/opt_set.rs | 83 ------------------- userland/capsule_linux/src/linux/net/pair.rs | 4 +- .../capsule_linux/src/linux/net/peer_addr.rs | 4 +- .../capsule_linux/src/linux/net/recvfrom.rs | 46 ++++++++++ .../capsule_linux/src/linux/net/shutdown.rs | 12 +-- .../capsule_linux/src/linux/net/sock/cell.rs | 6 +- .../src/linux/net/sock/deliver.rs | 31 +++++++ .../capsule_linux/src/linux/net/sock/free.rs | 6 +- .../capsule_linux/src/linux/net/sock/gram.rs | 58 +++---------- .../src/linux/net/sock/gram_in.rs | 11 +-- .../src/linux/net/sock/gram_target.rs | 70 ++++++++++++++++ .../capsule_linux/src/linux/net/sock/kinds.rs | 47 +++++++++++ .../capsule_linux/src/linux/net/sock/link.rs | 14 +--- .../capsule_linux/src/linux/net/sock/mod.rs | 10 ++- .../capsule_linux/src/linux/net/sock/name.rs | 8 +- .../capsule_linux/src/linux/net/sock/put.rs | 57 +++++++++++++ .../capsule_linux/src/linux/net/sock/ready.rs | 8 +- .../capsule_linux/src/linux/net/sock/send.rs | 31 +------ .../capsule_linux/src/linux/net/sock/syn.rs | 12 --- .../capsule_linux/src/linux/net/sock/take.rs | 51 ++++++++++++ .../capsule_linux/src/linux/net/sock/types.rs | 26 +----- .../src/linux/net/sock/unlisten.rs | 48 +++++++++++ .../capsule_linux/src/linux/net/socket.rs | 8 +- .../capsule_linux/src/linux/net/stream.rs | 1 - .../capsule_linux/src/linux/net/try_call.rs | 19 +---- .../capsule_linux/src/linux/net/xfer_in.rs | 26 ++---- .../capsule_linux/src/linux/net/xfer_out.rs | 23 ++--- .../capsule_linux/src/linux/serve/dispatch.rs | 5 +- .../src/linux/serve/family_waits.rs | 10 +-- userland/capsule_linux/src/linux/serve/mod.rs | 2 +- .../src/linux/serve/waits_sock.rs | 25 ++---- .../src/linux/serve/waits_sock_kind.rs | 16 +++- userland/capsule_linux/src/linux/unix/mod.rs | 5 +- 68 files changed, 1028 insertions(+), 556 deletions(-) create mode 100644 userland/capsule_linux/src/linux/net/api.rs create mode 100644 userland/capsule_linux/src/linux/net/cap.rs create mode 100644 userland/capsule_linux/src/linux/net/connect_dial.rs create mode 100644 userland/capsule_linux/src/linux/net/mmsg_each.rs create mode 100644 userland/capsule_linux/src/linux/net/msg_recv.rs rename userland/capsule_linux/src/linux/net/{sockaddr_un.rs => named/addr.rs} (79%) create mode 100644 userland/capsule_linux/src/linux/net/named/auto.rs rename userland/capsule_linux/src/linux/net/{unix_bind.rs => named/bind.rs} (77%) rename userland/capsule_linux/src/linux/net/{unix_calls.rs => named/connect.rs} (72%) create mode 100644 userland/capsule_linux/src/linux/net/named/display.rs create mode 100644 userland/capsule_linux/src/linux/net/named/mod.rs rename userland/capsule_linux/src/linux/net/{unix_name.rs => named/name.rs} (92%) create mode 100644 userland/capsule_linux/src/linux/net/opt/apply.rs rename userland/capsule_linux/src/linux/net/{opt_get.rs => opt/get.rs} (89%) rename userland/capsule_linux/src/linux/net/{opt_ids.rs => opt/ids.rs} (100%) create mode 100644 userland/capsule_linux/src/linux/net/opt/mod.rs rename userland/capsule_linux/src/linux/net/{opt_more.rs => opt/more.rs} (97%) create mode 100644 userland/capsule_linux/src/linux/net/opt/set.rs rename userland/capsule_linux/src/linux/net/{opt_time.rs => opt/time.rs} (97%) rename userland/capsule_linux/src/linux/net/{opt_value.rs => opt/value.rs} (88%) delete mode 100644 userland/capsule_linux/src/linux/net/opt_set.rs create mode 100644 userland/capsule_linux/src/linux/net/recvfrom.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/deliver.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/gram_target.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/kinds.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/put.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/take.rs create mode 100644 userland/capsule_linux/src/linux/net/sock/unlisten.rs diff --git a/userland/capsule_linux/src/linux/abi/mod.rs b/userland/capsule_linux/src/linux/abi/mod.rs index fab5d81af..319e970d7 100644 --- a/userland/capsule_linux/src/linux/abi/mod.rs +++ b/userland/capsule_linux/src/linux/abi/mod.rs @@ -22,7 +22,7 @@ pub mod errno; pub mod errno_sock; pub mod name; pub mod nr; -pub mod nr_path; pub mod nr_high; +pub mod nr_path; pub mod nr_sched; pub mod nr_sock; diff --git a/userland/capsule_linux/src/linux/abi/nr.rs b/userland/capsule_linux/src/linux/abi/nr.rs index 0bfdfac31..b489413e7 100644 --- a/userland/capsule_linux/src/linux/abi/nr.rs +++ b/userland/capsule_linux/src/linux/abi/nr.rs @@ -14,7 +14,6 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - //! Linux x86_64 syscall numbers, by family. pub use super::nr_high::*; diff --git a/userland/capsule_linux/src/linux/guest/handle.rs b/userland/capsule_linux/src/linux/guest/handle.rs index 4cfc902bd..8aaedd7c8 100644 --- a/userland/capsule_linux/src/linux/guest/handle.rs +++ b/userland/capsule_linux/src/linux/guest/handle.rs @@ -86,8 +86,6 @@ pub struct Guest { pub blocked: Vec, /// The image's symbolic links, read once and shared by the family. pub links: alloc::rc::Rc, - /// The pid this process holds family sockets under. Here, rather than - /// in the socket table, so that dropping a process that ends lets go of - /// what it held, as Linux closes an exiting process's descriptors. + /// Lets go of this process's family sockets when it is dropped (net::sock). pub sockets: crate::linux::net::sock::Holder, } diff --git a/userland/capsule_linux/src/linux/net/api.rs b/userland/capsule_linux/src/linux/net/api.rs new file mode 100644 index 000000000..455637135 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/api.rs @@ -0,0 +1,42 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What the rest of the personality calls on sockets. + +pub use super::accept::accept4; +pub use super::bind::bind; +pub use super::call_kind::{flags as call_flags, wants_all}; +pub use super::close::close; +pub use super::connect::connect; +pub use super::dgram::sendto; +pub use super::fd::{is_stream, sock_id}; +pub use super::listen::listen; +pub use super::mmsg::{recvmmsg, sendmmsg}; +pub use super::msg::sendmsg; +pub use super::msg_recv::recvmsg; +pub use super::name::{getpeername, getsockname}; +pub use super::opt::{getsockopt, limit_ms, setsockopt}; +pub use super::pair::socketpair; +pub use super::poll::{ready, POLLERR, POLLHUP}; +pub use super::poll_set::poll; +pub use super::poll_socket::outside; +pub use super::recvfrom::recvfrom; +pub use super::select::{clear as select_clear, select}; +pub use super::shutdown::shutdown; +pub use super::socket::socket; +pub use super::try_call::try_call; +pub use super::xfer_in::read as recv; +pub use super::xfer_out::write as send; diff --git a/userland/capsule_linux/src/linux/net/bind.rs b/userland/capsule_linux/src/linux/net/bind.rs index 9a2f1a72c..b8e002b26 100644 --- a/userland/capsule_linux/src/linux/net/bind.rs +++ b/userland/capsule_linux/src/linux/net/bind.rs @@ -30,7 +30,7 @@ pub fn bind(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { Err(e) => return e, }; if sock::with(|t| t.get(id).is_some_and(|s| s.domain == Domain::Unix)) { - return super::unix_bind::bind(guest, id, at, len); + return super::named::bind(guest, id, at, len); } let (family, mut want) = match sockaddr::read(guest, at, len) { Ok(v) => v, diff --git a/userland/capsule_linux/src/linux/net/call_kind.rs b/userland/capsule_linux/src/linux/net/call_kind.rs index eb7214481..6a28e0a56 100644 --- a/userland/capsule_linux/src/linux/net/call_kind.rs +++ b/userland/capsule_linux/src/linux/net/call_kind.rs @@ -17,10 +17,11 @@ //! What kind of wait a socket call makes: its flags, and whether a //! blocking one waits until it has moved everything. -use crate::linux::abi::nr; +use crate::linux::abi::{errno, nr}; +use crate::linux::guest::Guest; use super::flags::{MSG_DONTWAIT, MSG_WAITALL}; -use super::mmsg; +use super::{iov, mmsg}; /// The call's flags, where it has them. pub fn flags(n: u64, a: [u64; 6]) -> u64 { @@ -42,3 +43,19 @@ pub fn wants_all(stream: bool, n: u64, flags: u64) -> bool { _ => false, } } + +/// A receive's answer: the count, or the errno. +pub(super) fn bytes_in(guest: &Guest, id: u32, v: &iov::Iov, done: usize, flags: u64) -> u64 { + match super::xfer_in::recv(guest, id, v, done, flags) { + Ok(got) => errno::ok(got.n as u64), + Err(e) => e, + } +} + +/// The bytes a msghdr's iovecs ask for. +pub(super) fn msg_len(guest: &Guest, msg: u64) -> usize { + let word = |at: u64| { + guest.read(at, 8).map_or(0, |b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8]))) + }; + iov::read(guest, word(msg + 16), word(msg + 24)).map_or(0, |v| iov::total(&v)) +} diff --git a/userland/capsule_linux/src/linux/net/cap.rs b/userland/capsule_linux/src/linux/net/cap.rs new file mode 100644 index 000000000..e51aa5d03 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/cap.rs @@ -0,0 +1,33 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! How many bytes one send takes from the guest. + +/// What one send takes from the guest at most: the default receive buffer +/// of a stream's peer, so a single call can fill it. +const STREAM_CAP: usize = 128 << 10; +/// One byte past the largest datagram, so a larger one is seen and refused. +const GRAM_CAP: usize = 65508; + +/// The most one send gathers: what net.sockets carries in a call, a +/// stream peer's default queue, or one byte past the largest datagram. +pub fn cap(outside: bool, stream: bool) -> usize { + match (outside, stream) { + (true, _) => super::stream::MAX_IO, + (false, true) => STREAM_CAP, + (false, false) => GRAM_CAP, + } +} diff --git a/userland/capsule_linux/src/linux/net/connect.rs b/userland/capsule_linux/src/linux/net/connect.rs index 4b1040508..8d630bc56 100644 --- a/userland/capsule_linux/src/linux/net/connect.rs +++ b/userland/capsule_linux/src/linux/net/connect.rs @@ -26,7 +26,7 @@ use super::sock::{self, Domain, Proto}; use super::sockaddr::{self, is_loopback, AF_INET}; pub fn connect(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { - // This capsule is the nameserver, so its socket has nothing to reach. + /* This capsule is the nameserver, so its socket has nothing to reach. */ if guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Resolver) { return errno::ok(0); } @@ -42,7 +42,7 @@ pub fn connect(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { return errno::fail(errno::EBADF); }; match (proto, domain) { - (_, Domain::Unix) => super::unix_calls::connect(guest, fd, id, proto, at, len), + (_, Domain::Unix) => super::named::connect(guest, fd, id, proto, at, len), (Proto::Dgram, _) => super::connect_dgram::connect(guest, fd, id, family, to), _ if family != AF_INET => errno::fail(errno::EAFNOSUPPORT), _ if is_loopback(to.ip) => super::connect_lo::loopback(id, to, nonblock(guest, fd)), diff --git a/userland/capsule_linux/src/linux/net/connect_dial.rs b/userland/capsule_linux/src/linux/net/connect_dial.rs new file mode 100644 index 000000000..6febedff5 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/connect_dial.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The two calls a stream outside the family makes to net.sockets: a +//! mixnet socket, and a connect to an address or to the name this capsule +//! invented it for. + +use alloc::vec::Vec; + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::call::call; +use super::dns::host_for; +use super::ops::{DOMAIN, KIND_MIXNET, OP_CONNECT, OP_CONNECT_HOST, OP_SOCKET}; +use super::sock::Addr; + +pub fn open() -> Result { + let mut body = Vec::with_capacity(4); + body.extend_from_slice(&DOMAIN.to_le_bytes()); + body.extend_from_slice(&KIND_MIXNET.to_le_bytes()); + match call(OP_SOCKET, &body, 8) { + Some((0, out)) if out.len() >= 4 => { + Ok(u32::from_le_bytes([out[0], out[1], out[2], out[3]])) + } + Some(_) => Err(errno::fail(errno::ENOMEM)), + None => Err(errno::fail(errno::EIO)), + } +} + +/// The service's answer to a connect; None when it did not answer. +pub fn dial(guest: &Guest, handle: u32, to: Addr) -> Result)>, u64> { + /* An address this capsule invented for a name goes back to being the name. */ + if let Some(host) = host_for(guest, to.ip) { + let body = super::host_body::host_body(handle, to.port, &host) + .ok_or(errno::fail(errno::EINVAL))?; + return Ok(call(OP_CONNECT_HOST, &body, 0)); + } + let mut body = Vec::with_capacity(10); + body.extend_from_slice(&handle.to_le_bytes()); + body.extend_from_slice(&to.ip); + body.extend_from_slice(&to.port.to_le_bytes()); + Ok(call(OP_CONNECT, &body, 0)) +} diff --git a/userland/capsule_linux/src/linux/net/connect_lo.rs b/userland/capsule_linux/src/linux/net/connect_lo.rs index df85325f5..53b1a0706 100644 --- a/userland/capsule_linux/src/linux/net/connect_lo.rs +++ b/userland/capsule_linux/src/linux/net/connect_lo.rs @@ -38,14 +38,16 @@ pub fn loopback(id: u32, to: Addr, nonblock: bool) -> u64 { return errno::fail(e); } let linked = t.link(id, to); - // A full listener keeps a non-blocking connect until accept makes - // room, as Linux's SYN_SENT does. + /* + * A full listener keeps a non-blocking connect until accept makes + * room, as Linux's SYN_SENT does. + */ let from = t.get(id).and_then(|s| s.local).map_or(0, |a| a.port); if let (Link::Full, true, Some(l)) = (&linked, nonblock, t.listener(to, from)) { t.wait_room(id, l); return errno::fail(errno::EINPROGRESS); } - // A connect that fails gives back the port it bound. + /* A connect that fails gives back the port it bound. */ if !matches!(linked, Link::Done) && bound_here { if let Some(s) = t.get_mut(id) { s.local = None; diff --git a/userland/capsule_linux/src/linux/net/connect_out.rs b/userland/capsule_linux/src/linux/net/connect_out.rs index 36b0b2b5e..d6a73f635 100644 --- a/userland/capsule_linux/src/linux/net/connect_out.rs +++ b/userland/capsule_linux/src/linux/net/connect_out.rs @@ -19,43 +19,25 @@ //! name a socket: there is no second route to disable and no firewall rule //! to remove. net.sockets holds the stream; the family's entry names it. -use alloc::vec::Vec; - use crate::linux::abi::errno; use crate::linux::guest::Guest; -use super::call::call; -use super::dns::host_for; -use super::ops::{DOMAIN, KIND_MIXNET, OP_CONNECT, OP_CONNECT_HOST, OP_SOCKET}; +use super::connect_dial::{dial, open}; use super::sock::{self, Addr}; pub fn connect(guest: &Guest, id: u32, to: Addr) -> u64 { if sock::with(|t| t.get(id).is_some_and(|s| s.connected || s.listening || s.svc.is_some())) { return errno::fail(errno::EISCONN); } - let mut body = Vec::with_capacity(4); - body.extend_from_slice(&DOMAIN.to_le_bytes()); - body.extend_from_slice(&KIND_MIXNET.to_le_bytes()); - let handle = match call(OP_SOCKET, &body, 8) { - Some((0, out)) if out.len() >= 4 => u32::from_le_bytes([out[0], out[1], out[2], out[3]]), - Some(_) => return errno::fail(errno::ENOMEM), - None => return errno::fail(errno::EIO), + let handle = match open() { + Ok(h) => h, + Err(e) => return e, }; - // An address this capsule invented for a name goes back to being the name. - let status = match host_for(guest, to.ip) { - Some(host) => match super::host_body::host_body(handle, to.port, &host) { - Some(body) => call(OP_CONNECT_HOST, &body, 0), - None => { - super::stream::close(handle); - return errno::fail(errno::EINVAL); - } - }, - None => { - let mut body = Vec::with_capacity(10); - body.extend_from_slice(&handle.to_le_bytes()); - body.extend_from_slice(&to.ip); - body.extend_from_slice(&to.port.to_le_bytes()); - call(OP_CONNECT, &body, 0) + let status = match dial(guest, handle, to) { + Ok(s) => s, + Err(e) => { + super::stream::close(handle); + return e; } }; let answer = match status { diff --git a/userland/capsule_linux/src/linux/net/dest.rs b/userland/capsule_linux/src/linux/net/dest.rs index 458b23b6d..6e5bfd9f0 100644 --- a/userland/capsule_linux/src/linux/net/dest.rs +++ b/userland/capsule_linux/src/linux/net/dest.rs @@ -31,9 +31,9 @@ pub fn dest(guest: &Guest, proto: Proto, domain: Domain, to: Option) -> Resu return Err(errno::fail(errno::EAFNOSUPPORT)) } (_, Some(To::Unix(ua))) => { - let found = super::unix_name::resolve(guest, &ua) + let found = super::named::resolve(guest, &ua) .ok_or(errno::EINVAL) - .and_then(|name| super::unix_name::find(&name, Proto::Dgram)); + .and_then(|name| super::named::find(&name, Proto::Dgram)); match found { Ok(t) => Dest::Sock(t), Err(e) => return Err(errno::fail(e)), diff --git a/userland/capsule_linux/src/linux/net/dgram.rs b/userland/capsule_linux/src/linux/net/dgram.rs index 5999d41d9..ab32f9bc2 100644 --- a/userland/capsule_linux/src/linux/net/dgram.rs +++ b/userland/capsule_linux/src/linux/net/dgram.rs @@ -14,8 +14,7 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `sendto` and `recvfrom`, which differ from write and read only in -//! carrying an address. +//! `sendto`, which differs from write only in carrying an address. use alloc::vec; @@ -64,28 +63,6 @@ pub fn sendto( super::resolver::query(guest, fd, buf, len, to) } -pub fn recvfrom( - guest: &mut Guest, - fd: u64, - buf: u64, - len: u64, - flags: u64, - at: u64, - alen: u64, -) -> u64 { - if is_resolver(guest, fd) { - return super::resolver::answer(guest, fd, buf, len, at, alen); - } - let id = match sock_of(guest, fd) { - Ok(id) => id, - Err(e) => return e, - }; - match super::xfer_in::recv(guest, id, &vec![(buf, len)], 0, flags) { - Ok(got) => super::peer_addr::finish(guest, got, flags, at, alen), - Err(e) => e, - } -} - -fn is_resolver(guest: &Guest, fd: u64) -> bool { +pub(super) fn is_resolver(guest: &Guest, fd: u64) -> bool { guest.fds.get(fd as usize).is_some_and(|f| f.kind == Kind::Resolver) } diff --git a/userland/capsule_linux/src/linux/net/listen.rs b/userland/capsule_linux/src/linux/net/listen.rs index e6313b4b8..a22e3ce57 100644 --- a/userland/capsule_linux/src/linux/net/listen.rs +++ b/userland/capsule_linux/src/linux/net/listen.rs @@ -35,12 +35,12 @@ pub fn listen(guest: &mut Guest, fd: u64, backlog: u64) -> u64 { match (s.proto, s.domain, s.local) { (Proto::Dgram, ..) => return errno::fail(errno::EOPNOTSUPP), _ if s.connected || s.svc.is_some() => return errno::fail(errno::EINVAL), - // A Unix socket must be bound first: Linux does not name it here. + /* A Unix socket must be bound first: Linux does not name it here. */ (_, Domain::Unix, _) if s.uname.is_none() => return errno::fail(errno::EINVAL), (_, Domain::Unix, _) => {} - // Linux would bind 0.0.0.0 here, which is not the family's own. + /* Linux would bind 0.0.0.0 here, which is not the family's own. */ (_, _, None) => return not_loopback("listen", Addr::default()), - // Listeners share an address only when each set SO_REUSEPORT. + /* Listeners share an address only when each set SO_REUSEPORT. */ (_, _, Some(at)) if !s.listening && t.iter().any(|(_, o)| { @@ -55,7 +55,7 @@ pub fn listen(guest: &mut Guest, fd: u64, backlog: u64) -> u64 { } if let Some(s) = t.get_mut(id) { s.listening = true; - // An int, and somaxconn's 4096 is the most Linux keeps. + /* An int, and somaxconn's 4096 is the most Linux keeps. */ s.backlog = (backlog as i32).clamp(0, 4096) as usize; } errno::ok(0) diff --git a/userland/capsule_linux/src/linux/net/mmsg.rs b/userland/capsule_linux/src/linux/net/mmsg.rs index 3fe863d9d..174adaa3c 100644 --- a/userland/capsule_linux/src/linux/net/mmsg.rs +++ b/userland/capsule_linux/src/linux/net/mmsg.rs @@ -23,11 +23,9 @@ use crate::linux::abi::errno; use crate::linux::guest::Guest; use super::flags::MSG_DONTWAIT; +use super::mmsg_each::each; use super::policy::refuse; -/// struct mmsghdr on x86_64: a msghdr, then msg_len. -const MMSGHDR: u64 = 64; -const LEN_AT: u64 = 56; /// Linux's UIO_MAXIOV caps vlen. pub const MOST: u64 = 1024; pub const MSG_WAITFORONE: u64 = 0x10000; @@ -48,30 +46,6 @@ pub fn recvmmsg(guest: &mut Guest, a: [u64; 6], skip: usize) -> u64 { let flags = a[3] & !MSG_WAITFORONE; each(guest, a, skip, |g, at, first| { let f = if first { flags } else { flags | MSG_DONTWAIT }; - super::msg::recvmsg(g, a[0], at, f, 0) + super::msg_recv::recvmsg(g, a[0], at, f, 0) }) } - -/// Run `one` on each message from `skip` until one fails: the count moved, -/// or the first failure's errno when none was. -fn each( - guest: &mut Guest, - a: [u64; 6], - skip: usize, - mut one: impl FnMut(&mut Guest, u64, bool) -> u64, -) -> u64 { - let vlen = a[2].min(MOST); - let mut moved = 0u64; - for i in skip as u64..vlen { - let at = a[1] + i * MMSGHDR; - let got = one(guest, at, moved == 0); - let Some(n) = errno::slot(got) else { - return if moved == 0 { got } else { errno::ok(moved) }; - }; - if guest.write(at + LEN_AT, &(n as u32).to_le_bytes()) < 4 { - return if moved == 0 { errno::fail(errno::EFAULT) } else { errno::ok(moved) }; - } - moved += 1; - } - errno::ok(moved) -} diff --git a/userland/capsule_linux/src/linux/net/mmsg_each.rs b/userland/capsule_linux/src/linux/net/mmsg_each.rs new file mode 100644 index 000000000..a047e1447 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/mmsg_each.rs @@ -0,0 +1,50 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The messages of one sendmmsg or recvmmsg, one after another. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::mmsg::MOST; + +/// struct mmsghdr on x86_64: a msghdr, then msg_len. +const MMSGHDR: u64 = 64; +const LEN_AT: u64 = 56; + +/// Run `one` on each message from `skip` until one fails: the count moved, +/// or the first failure's errno when none was. +pub fn each( + guest: &mut Guest, + a: [u64; 6], + skip: usize, + mut one: impl FnMut(&mut Guest, u64, bool) -> u64, +) -> u64 { + let vlen = a[2].min(MOST); + let mut moved = 0u64; + for i in skip as u64..vlen { + let at = a[1] + i * MMSGHDR; + let got = one(guest, at, moved == 0); + let Some(n) = errno::slot(got) else { + return if moved == 0 { got } else { errno::ok(moved) }; + }; + if guest.write(at + LEN_AT, &(n as u32).to_le_bytes()) < 4 { + return if moved == 0 { errno::fail(errno::EFAULT) } else { errno::ok(moved) }; + } + moved += 1; + } + errno::ok(moved) +} diff --git a/userland/capsule_linux/src/linux/net/mod.rs b/userland/capsule_linux/src/linux/net/mod.rs index 73996cbbb..78c0cc53b 100644 --- a/userland/capsule_linux/src/linux/net/mod.rs +++ b/userland/capsule_linux/src/linux/net/mod.rs @@ -18,13 +18,16 @@ //! family, which net.sockets carries over the mixnet. mod accept; +mod api; mod bind; -mod call_kind; mod call; +mod call_kind; +mod cap; mod close; mod connect; -mod connect_lo; mod connect_dgram; +mod connect_dial; +mod connect_lo; mod connect_out; mod dest; mod dgram; @@ -36,16 +39,14 @@ mod host_body; mod iov; mod listen; mod mmsg; +mod mmsg_each; mod msg; mod msg_hdr; +mod msg_recv; mod name; +mod named; mod ops; -mod opt_get; -mod opt_ids; -mod opt_more; -mod opt_set; -mod opt_time; -mod opt_value; +mod opt; mod pair; mod peer_addr; mod policy; @@ -54,6 +55,7 @@ mod poll_set; mod poll_socket; pub mod raw; pub mod raw_io; +mod recvfrom; mod resolver; pub mod route; mod select; @@ -61,37 +63,10 @@ mod shutdown; pub mod sock; mod sockaddr; mod sockaddr_out; -mod sockaddr_un; mod socket; mod stream; mod try_call; -mod unix_bind; -mod unix_calls; -mod unix_name; mod xfer_in; mod xfer_out; -pub use accept::accept4; -pub use bind::bind; -pub use call_kind::{flags as call_flags, wants_all}; -pub use listen::listen; -pub use close::close; -pub use connect::connect; -pub use dgram::{recvfrom, sendto}; -pub use fd::{is_stream, sock_id}; -pub use mmsg::{recvmmsg, sendmmsg}; -pub use msg::{recvmsg, sendmsg}; -pub use name::{getpeername, getsockname}; -pub use opt_get::getsockopt; -pub use opt_set::setsockopt; -pub use opt_time::limit_ms; -pub use pair::socketpair; -pub use poll::{ready, POLLERR, POLLHUP}; -pub use poll_set::poll; -pub use poll_socket::outside; -pub use select::{clear as select_clear, select}; -pub use shutdown::shutdown; -pub use socket::socket; -pub use try_call::try_call; -pub use xfer_in::read as recv; -pub use xfer_out::write as send; +pub use api::*; diff --git a/userland/capsule_linux/src/linux/net/msg.rs b/userland/capsule_linux/src/linux/net/msg.rs index bc49211c0..091482db4 100644 --- a/userland/capsule_linux/src/linux/net/msg.rs +++ b/userland/capsule_linux/src/linux/net/msg.rs @@ -14,15 +14,13 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! `sendmsg` and `recvmsg` on a family socket: an iovec, an address, and -//! control data, which only a Unix socket carries on Linux and none here. +//! `sendmsg` on a family socket: an iovec and an address. use crate::linux::abi::errno; use crate::linux::guest::Guest; use super::fd::sock_of; -use super::flags::MSG_TRUNC; -use super::msg_hdr::{hdr, CONTROLLEN_AT, FLAGS_AT, NAMELEN_AT}; +use super::msg_hdr::hdr; use super::policy::refuse; /// `skip` bytes of the message went in an earlier try of this same call. @@ -37,7 +35,7 @@ pub fn sendmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> }; if h.controllen != 0 { return refuse( - "sendmsg control data on a family socket: nothing here reads it", + "sendmsg control data on a family socket: SCM_RIGHTS is not carried yet", errno::EINVAL, ); } @@ -52,30 +50,3 @@ pub fn sendmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> } super::xfer_out::send(guest, id, &h.iov, skip, flags, to) } - -pub fn recvmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> u64 { - let id = match sock_of(guest, fd) { - Ok(id) => id, - Err(e) => return e, - }; - let h = match hdr(guest, msg) { - Ok(h) => h, - Err(e) => return e, - }; - let got = match super::xfer_in::recv(guest, id, &h.iov, skip, flags) { - Ok(got) => got, - Err(e) => return e, - }; - let cut = if got.whole > got.n { MSG_TRUNC as u32 } else { 0 }; - let (name, lenp) = if h.name != 0 { (h.name, msg + NAMELEN_AT) } else { (0, 0) }; - let value = super::peer_addr::finish(guest, got, flags, name, lenp); - if errno::slot(value).is_none() { - return value; - } - if guest.write(msg + CONTROLLEN_AT, &0u64.to_le_bytes()) < 8 - || guest.write(msg + FLAGS_AT, &cut.to_le_bytes()) < 4 - { - return errno::fail(errno::EFAULT); - } - value -} diff --git a/userland/capsule_linux/src/linux/net/msg_recv.rs b/userland/capsule_linux/src/linux/net/msg_recv.rs new file mode 100644 index 000000000..5efafbc97 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/msg_recv.rs @@ -0,0 +1,52 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `recvmsg` on a family socket: bytes into an iovec, and the sender's +//! address. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use super::fd::sock_of; +use super::flags::MSG_TRUNC; +use super::msg_hdr::{hdr, CONTROLLEN_AT, FLAGS_AT, NAMELEN_AT}; + +pub fn recvmsg(guest: &mut Guest, fd: u64, msg: u64, flags: u64, skip: usize) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let h = match hdr(guest, msg) { + Ok(h) => h, + Err(e) => return e, + }; + let got = match super::xfer_in::recv(guest, id, &h.iov, skip, flags) { + Ok(got) => got, + Err(e) => return e, + }; + let cut = if got.whole > got.n { MSG_TRUNC as u32 } else { 0 }; + let (name, lenp) = if h.name != 0 { (h.name, msg + NAMELEN_AT) } else { (0, 0) }; + let value = super::peer_addr::finish(guest, got, flags, name, lenp); + if errno::slot(value).is_none() { + return value; + } + if guest.write(msg + CONTROLLEN_AT, &0u64.to_le_bytes()) < 8 + || guest.write(msg + FLAGS_AT, &cut.to_le_bytes()) < 4 + { + return errno::fail(errno::EFAULT); + } + value +} diff --git a/userland/capsule_linux/src/linux/net/name.rs b/userland/capsule_linux/src/linux/net/name.rs index 75580a91a..962cfaec1 100644 --- a/userland/capsule_linux/src/linux/net/name.rs +++ b/userland/capsule_linux/src/linux/net/name.rs @@ -27,7 +27,7 @@ pub fn getsockname(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 { Ok(id) => id, Err(e) => return e, }; - // A socket not yet bound is 0.0.0.0 port 0, or an unnamed Unix socket. + /* A socket not yet bound is 0.0.0.0 port 0, or an unnamed Unix socket. */ let Some(me) = sock::with(|t| { t.get(id).map(|s| match s.domain { Domain::Inet => Peer::Inet(s.local.unwrap_or_default()), @@ -44,8 +44,10 @@ pub fn getpeername(guest: &mut Guest, fd: u64, at: u64, lenp: u64) -> u64 { Ok(id) => id, Err(e) => return e, }; - // A reset connection is closed, and has no peer; one whose peer only - // shut down, or left cleanly, still does. + /* + * A reset connection is closed, and has no peer; one whose peer only + * shut down, or left cleanly, still does. + */ let peer: Option = sock::with(|t| { let s = t.get(id)?; let live = s.connected && !s.broken && s.error == 0; diff --git a/userland/capsule_linux/src/linux/net/sockaddr_un.rs b/userland/capsule_linux/src/linux/net/named/addr.rs similarity index 79% rename from userland/capsule_linux/src/linux/net/sockaddr_un.rs rename to userland/capsule_linux/src/linux/net/named/addr.rs index e16af06c3..28fef072d 100644 --- a/userland/capsule_linux/src/linux/net/sockaddr_un.rs +++ b/userland/capsule_linux/src/linux/net/named/addr.rs @@ -21,6 +21,7 @@ use alloc::vec::Vec; use crate::linux::abi::errno; use crate::linux::guest::Guest; +use crate::linux::net::sockaddr::{self, AF_UNIX}; /// sun_family and the 108 bytes of sun_path. const SOCKADDR_UN: u64 = 110; @@ -40,10 +41,18 @@ pub fn read(guest: &Guest, at: u64, len: u64) -> Result { Ok(match path.first() { None => UAddr::Auto, Some(0) => UAddr::Abstract(path[1..].to_vec()), - // A path ends at its first NUL, however long the guest said it was. + /* A path ends at its first NUL, however long the guest said it was. */ Some(_) => { let end = path.iter().position(|&b| b == 0).unwrap_or(path.len()); UAddr::Path(path[..end].to_vec()) } }) } + +/// The name a bind or connect gives: EINVAL for another family. +pub fn unix_addr(guest: &Guest, at: u64, len: u64) -> Result { + match sockaddr::read(guest, at, len)? { + (AF_UNIX, _) => read(guest, at, len), + _ => Err(errno::fail(errno::EINVAL)), + } +} diff --git a/userland/capsule_linux/src/linux/net/named/auto.rs b/userland/capsule_linux/src/linux/net/named/auto.rs new file mode 100644 index 000000000..6faec8b6c --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/auto.rs @@ -0,0 +1,32 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The name bind chooses when a Unix socket is given its family alone. + +use alloc::vec::Vec; + +use crate::linux::net::sock::{self, UName}; + +/// Linux's autobind: a NUL and five hex digits, the first unused. +pub fn fresh() -> Option { + (0u32..0x10_0000).find_map(|n| { + let mut key: Vec = alloc::vec![0]; + key.extend_from_slice(alloc::format!("{n:05x}").as_bytes()); + let name = UName { key: key.clone(), shown: key }; + let used = sock::with(|t| t.iter().any(|(_, s)| s.uname.as_ref() == Some(&name))); + (!used).then_some(name) + }) +} diff --git a/userland/capsule_linux/src/linux/net/unix_bind.rs b/userland/capsule_linux/src/linux/net/named/bind.rs similarity index 77% rename from userland/capsule_linux/src/linux/net/unix_bind.rs rename to userland/capsule_linux/src/linux/net/named/bind.rs index 73f059e35..75b99e4ee 100644 --- a/userland/capsule_linux/src/linux/net/unix_bind.rs +++ b/userland/capsule_linux/src/linux/net/named/bind.rs @@ -17,16 +17,14 @@ //! `bind` on a Unix socket: a path, which becomes a file as on Linux, an //! abstract name, or family alone, which asks for a name to be chosen. -use alloc::vec::Vec; - use crate::linux::abi::errno; use crate::linux::file; use crate::linux::guest::Guest; -use super::sock::{self, UName}; -use super::sockaddr_un::UAddr; -use super::unix_calls::unix_addr; -use super::unix_name::resolve; +use super::addr::{unix_addr, UAddr}; +use super::auto::fresh; +use super::name::resolve; +use crate::linux::net::sock; pub fn bind(guest: &Guest, id: u32, at: u64, len: u64) -> u64 { let ua = match unix_addr(guest, at, len) { @@ -58,7 +56,7 @@ fn bind_name(guest: &Guest, id: u32, ua: &UAddr) -> u64 { if key.writable().is_err() { return errno::fail(errno::EROFS); } - // The store refuses a file whose directory is missing. + /* The store refuses a file whose directory is missing. */ if file::store_write(&key, &[]).is_err() { return errno::fail(errno::ENOENT); } @@ -66,14 +64,3 @@ fn bind_name(guest: &Guest, id: u32, ua: &UAddr) -> u64 { sock::with(|t| t.get_mut(id).map(|s| s.uname = Some(name))); errno::ok(0) } - -/// Linux's autobind: a NUL and five hex digits, the first unused. -fn fresh() -> Option { - (0u32..0x10_0000).find_map(|n| { - let mut key: Vec = alloc::vec![0]; - key.extend_from_slice(alloc::format!("{n:05x}").as_bytes()); - let name = UName { key: key.clone(), shown: key }; - let used = sock::with(|t| t.iter().any(|(_, s)| s.uname.as_ref() == Some(&name))); - (!used).then_some(name) - }) -} diff --git a/userland/capsule_linux/src/linux/net/unix_calls.rs b/userland/capsule_linux/src/linux/net/named/connect.rs similarity index 72% rename from userland/capsule_linux/src/linux/net/unix_calls.rs rename to userland/capsule_linux/src/linux/net/named/connect.rs index 5c9889708..e030d1abd 100644 --- a/userland/capsule_linux/src/linux/net/unix_calls.rs +++ b/userland/capsule_linux/src/linux/net/named/connect.rs @@ -19,13 +19,14 @@ //! itself (`unix`); every other name is the family's own. use crate::linux::abi::errno; -use crate::linux::guest::{Fd, Guest}; +use crate::linux::guest::Guest; +use crate::linux::net::sock::{self, Link, Proto}; +use crate::linux::net::sockaddr::{self, AF_UNSPEC}; use crate::linux::unix; -use super::sock::{self, Link, Proto}; -use super::sockaddr::{self, AF_UNIX, AF_UNSPEC}; -use super::sockaddr_un::{self, UAddr}; -use super::unix_name::{find, resolve}; +use super::addr::{unix_addr, UAddr}; +use super::display::display; +use super::name::{find, resolve}; pub fn connect(guest: &mut Guest, fd: u64, id: u32, proto: Proto, at: u64, len: u64) -> u64 { if proto == Proto::Dgram && matches!(sockaddr::read(guest, at, len), Ok((AF_UNSPEC, _))) { @@ -53,7 +54,7 @@ pub fn connect(guest: &mut Guest, fd: u64, id: u32, proto: Proto, at: u64, len: match proto { Proto::Stream if s.connected => Err(errno::EISCONN), Proto::Stream if s.listening => Err(errno::EINVAL), - // A Unix connect completes in the caller's call, blocking or not. + /* A Unix connect completes in the caller's call, blocking or not. */ Proto::Stream => match t.join(id, target) { Link::Done => Ok(()), Link::Full => Err(errno::EAGAIN), @@ -67,22 +68,3 @@ pub fn connect(guest: &mut Guest, fd: u64, id: u32, proto: Proto, at: u64, len: }) .map_or_else(errno::fail, |()| errno::ok(0)) } - -pub fn unix_addr(guest: &Guest, at: u64, len: u64) -> Result { - match sockaddr::read(guest, at, len)? { - (AF_UNIX, _) => sockaddr_un::read(guest, at, len), - _ => Err(errno::fail(errno::EINVAL)), - } -} - -/// Let go of the family socket and make `fd` the display connection. -fn display(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { - super::close::close(guest, fd); - if let Some(f) = guest.fds.get_mut(fd as usize) { - let (cloexec, nonblock) = (f.cloexec, f.nonblock); - *f = Fd::unix(); - f.cloexec = cloexec; - f.nonblock = nonblock; - } - unix::connect(guest, fd, at, len) -} diff --git a/userland/capsule_linux/src/linux/net/named/display.rs b/userland/capsule_linux/src/linux/net/named/display.rs new file mode 100644 index 000000000..ef6cc8755 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/display.rs @@ -0,0 +1,32 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! The display's path: the one Unix name this capsule answers itself. + +use crate::linux::guest::{Fd, Guest}; +use crate::linux::unix; + +/// Let go of the family socket and make `fd` the display connection. +pub fn display(guest: &mut Guest, fd: u64, at: u64, len: u64) -> u64 { + crate::linux::net::close::close(guest, fd); + if let Some(f) = guest.fds.get_mut(fd as usize) { + let (cloexec, nonblock) = (f.cloexec, f.nonblock); + *f = Fd::unix(); + f.cloexec = cloexec; + f.nonblock = nonblock; + } + unix::connect(guest, fd, at, len) +} diff --git a/userland/capsule_linux/src/linux/net/named/mod.rs b/userland/capsule_linux/src/linux/net/named/mod.rs new file mode 100644 index 000000000..ec9aac5fa --- /dev/null +++ b/userland/capsule_linux/src/linux/net/named/mod.rs @@ -0,0 +1,30 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Unix sockets with names: the sockaddr_un a guest gives, the names a +//! family socket is bound to, and bind and connect on them. + +mod addr; +mod auto; +mod bind; +mod connect; +mod display; +mod name; + +pub use addr::{read as read_uaddr, UAddr}; +pub use bind::bind; +pub use connect::connect; +pub use name::{find, resolve}; diff --git a/userland/capsule_linux/src/linux/net/unix_name.rs b/userland/capsule_linux/src/linux/net/named/name.rs similarity index 92% rename from userland/capsule_linux/src/linux/net/unix_name.rs rename to userland/capsule_linux/src/linux/net/named/name.rs index dfb85edd2..121bcc83d 100644 --- a/userland/capsule_linux/src/linux/net/unix_name.rs +++ b/userland/capsule_linux/src/linux/net/named/name.rs @@ -24,8 +24,8 @@ use crate::linux::abi::errno; use crate::linux::file; use crate::linux::guest::Guest; -use super::sock::{self, Domain, Proto, UName}; -use super::sockaddr_un::UAddr; +use super::addr::UAddr; +use crate::linux::net::sock::{self, Domain, Proto, UName}; /// The name `ua` means for this guest; None asks for one to be chosen. pub fn resolve(guest: &Guest, ua: &UAddr) -> Option { @@ -43,8 +43,10 @@ pub fn resolve(guest: &Guest, ua: &UAddr) -> Option { /// The family socket of kind `proto` bound to `name`, the one a connect or /// a send reaches: a stream's must listen. Otherwise Linux's answer. pub fn find(name: &UName, proto: Proto) -> Result { - // An accepted connection carries its listener's name, as on Linux; the - // listener is the one a connect reaches. + /* + * An accepted connection carries its listener's name, as on Linux; the + * listener is the one a connect reaches. + */ let found = sock::with(|t| { let named = || t.iter().filter(|(_, s)| s.domain == Domain::Unix && s.uname.as_ref() == Some(name)); diff --git a/userland/capsule_linux/src/linux/net/opt/apply.rs b/userland/capsule_linux/src/linux/net/opt/apply.rs new file mode 100644 index 000000000..dc3610b25 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/apply.rs @@ -0,0 +1,63 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! One option set on one socket: kept as Linux keeps it, or refused. + +use crate::linux::abi::errno; +use crate::linux::net::sock::{Domain, Proto, Sock}; + +use super::ids::*; +use super::time::{keep, timeo}; + +/// `raw` holds up to sixteen bytes of the value; `len` is what the guest +/// said it gave, at least four. +pub fn apply(s: &mut Sock, (level, name): (u64, u64), raw: &[u8], len: u64) -> u64 { + let int = u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]); + let word = + |i: usize| raw.get(i..i + 8).map(|b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8]))); + let (proto, domain) = (s.proto, s.domain); + let o = &mut s.opts; + match (level, name) { + /* A Unix socket has only socket-level options on Linux. */ + (l, _) if domain == Domain::Unix && l != SOL_SOCKET => { + return errno::fail(errno::EOPNOTSUPP) + } + (l, n) if super::more::known(l, n) && !(l == IPPROTO_TCP && proto == Proto::Dgram) => { + return super::more::set(&mut o.more, proto, l, n, int) + } + (SOL_SOCKET, SO_REUSEADDR) => o.reuseaddr = int != 0, + (SOL_SOCKET, SO_REUSEPORT) => o.reuseport = int != 0, + (SOL_SOCKET, SO_KEEPALIVE) => o.keepalive = int != 0, + (SOL_SOCKET, SO_BROADCAST) => o.broadcast = int != 0, + (SOL_SOCKET, SO_RCVBUF) => o.rcvbuf = (int.min(BUF_MAX) * 2).max(RCVBUF_MIN), + (SOL_SOCKET, SO_SNDBUF) => o.sndbuf = (int.min(BUF_MAX) * 2).max(SNDBUF_MIN), + (SOL_SOCKET, SO_LINGER) if len >= 8 => { + o.linger = (u32::from(int != 0), u32::from_le_bytes([raw[4], raw[5], raw[6], raw[7]])) + } + (SOL_SOCKET, SO_LINGER) => return errno::fail(errno::EINVAL), + (SOL_SOCKET, SO_RCVTIMEO) => return timeo(&mut o.rcvtimeo, word(0), word(8)), + (SOL_SOCKET, SO_SNDTIMEO) => return timeo(&mut o.sndtimeo, word(0), word(8)), + (IPPROTO_TCP, _) if proto == Proto::Dgram => return errno::fail(errno::ENOPROTOOPT), + (IPPROTO_TCP, TCP_NODELAY) => o.nodelay = int != 0, + (IPPROTO_TCP, TCP_KEEPIDLE) => return keep(&mut o.keepidle, int, KEEP_MAX), + (IPPROTO_TCP, TCP_KEEPINTVL) => return keep(&mut o.keepintvl, int, KEEP_MAX), + (IPPROTO_TCP, TCP_KEEPCNT) => return keep(&mut o.keepcnt, int, KEEPCNT_MAX), + /* An AF_INET socket has no IPv6 options on Linux either. */ + (IPPROTO_IPV6, _) => return errno::fail(errno::ENOPROTOOPT), + _ => return super::get::unknown("setsockopt", level, name), + } + errno::ok(0) +} diff --git a/userland/capsule_linux/src/linux/net/opt_get.rs b/userland/capsule_linux/src/linux/net/opt/get.rs similarity index 89% rename from userland/capsule_linux/src/linux/net/opt_get.rs rename to userland/capsule_linux/src/linux/net/opt/get.rs index be6fac3a6..67209a09b 100644 --- a/userland/capsule_linux/src/linux/net/opt_get.rs +++ b/userland/capsule_linux/src/linux/net/opt/get.rs @@ -20,8 +20,8 @@ use crate::linux::abi::errno; use crate::linux::guest::Guest; -use super::fd::sock_of; -use super::sock; +use crate::linux::net::fd::sock_of; +use crate::linux::net::sock; pub fn getsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, lenp: u64) -> u64 { let id = match sock_of(guest, fd) { @@ -35,7 +35,7 @@ pub fn getsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, lenp: if room < 0 { return errno::fail(errno::EINVAL); } - let value = sock::with(|t| t.get_mut(id).map(|s| super::opt_value::value(s, level, name))); + let value = sock::with(|t| t.get_mut(id).map(|s| super::value::value(s, level, name))); let bytes = match value { Some(Ok(bytes)) => bytes, Some(Err(e)) => return e, @@ -53,5 +53,5 @@ pub fn getsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, lenp: /// program may depend on it, and Linux would have kept it. pub fn unknown(call: &str, level: u64, name: u64) -> u64 { let what = alloc::format!("{call} level {level} option {name}: not kept for a guest socket"); - super::policy::refuse(&what, errno::ENOPROTOOPT) + crate::linux::net::policy::refuse(&what, errno::ENOPROTOOPT) } diff --git a/userland/capsule_linux/src/linux/net/opt_ids.rs b/userland/capsule_linux/src/linux/net/opt/ids.rs similarity index 100% rename from userland/capsule_linux/src/linux/net/opt_ids.rs rename to userland/capsule_linux/src/linux/net/opt/ids.rs diff --git a/userland/capsule_linux/src/linux/net/opt/mod.rs b/userland/capsule_linux/src/linux/net/opt/mod.rs new file mode 100644 index 000000000..bdeca38af --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/mod.rs @@ -0,0 +1,29 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Socket options: what setsockopt keeps and getsockopt reads back. + +mod apply; +mod get; +mod ids; +mod more; +mod set; +mod time; +mod value; + +pub use get::getsockopt; +pub use set::setsockopt; +pub use time::limit_ms; diff --git a/userland/capsule_linux/src/linux/net/opt_more.rs b/userland/capsule_linux/src/linux/net/opt/more.rs similarity index 97% rename from userland/capsule_linux/src/linux/net/opt_more.rs rename to userland/capsule_linux/src/linux/net/opt/more.rs index 241cc3a61..4e5368c65 100644 --- a/userland/capsule_linux/src/linux/net/opt_more.rs +++ b/userland/capsule_linux/src/linux/net/opt/more.rs @@ -21,11 +21,11 @@ use crate::linux::abi::errno; -use super::opt_ids::{ +use super::ids::{ IPPROTO_IP, IPPROTO_TCP, IP_TOS, IP_TTL, SOL_SOCKET, SO_PRIORITY, TCP_FASTOPEN, TCP_QUICKACK, TCP_USER_TIMEOUT, }; -use super::sock::{More, Proto}; +use crate::linux::net::sock::{More, Proto}; /// Linux's net.ipv4.ip_default_ttl. const DEFAULT_TTL: u32 = 64; diff --git a/userland/capsule_linux/src/linux/net/opt/set.rs b/userland/capsule_linux/src/linux/net/opt/set.rs new file mode 100644 index 000000000..3a432a200 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/opt/set.rs @@ -0,0 +1,42 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `setsockopt`. Every option here is kept and reads back as Linux reads +//! it; one that would change nothing on the family's loopback is still kept, +//! since Linux keeps it too. One this capsule cannot honour is refused. + +use crate::linux::abi::errno; +use crate::linux::guest::Guest; + +use crate::linux::net::fd::sock_of; +use crate::linux::net::sock; + +pub fn setsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, len: u64) -> u64 { + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + let Some(raw) = guest.read(val, (len as usize).min(16)) else { + return errno::fail(errno::EFAULT); + }; + if len < 4 { + return errno::fail(errno::EINVAL); + } + sock::with(|t| match t.get_mut(id) { + Some(s) => super::apply::apply(s, (level, name), &raw, len), + None => errno::fail(errno::EBADF), + }) +} diff --git a/userland/capsule_linux/src/linux/net/opt_time.rs b/userland/capsule_linux/src/linux/net/opt/time.rs similarity index 97% rename from userland/capsule_linux/src/linux/net/opt_time.rs rename to userland/capsule_linux/src/linux/net/opt/time.rs index 33fe6b1a4..f68652bd6 100644 --- a/userland/capsule_linux/src/linux/net/opt_time.rs +++ b/userland/capsule_linux/src/linux/net/opt/time.rs @@ -19,7 +19,7 @@ use crate::linux::abi::errno; -use super::sock::{self, Opts}; +use crate::linux::net::sock::{self, Opts}; /// A keepalive time or count, 1 up to Linux's most, else EINVAL. pub fn keep(slot: &mut u32, v: u32, most: u32) -> u64 { diff --git a/userland/capsule_linux/src/linux/net/opt_value.rs b/userland/capsule_linux/src/linux/net/opt/value.rs similarity index 88% rename from userland/capsule_linux/src/linux/net/opt_value.rs rename to userland/capsule_linux/src/linux/net/opt/value.rs index 7c9eee16b..684b4a795 100644 --- a/userland/capsule_linux/src/linux/net/opt_value.rs +++ b/userland/capsule_linux/src/linux/net/opt/value.rs @@ -20,8 +20,8 @@ use alloc::vec::Vec; use crate::linux::abi::errno; -use super::opt_ids::*; -use super::sock::{Domain, Proto, Sock}; +use super::ids::*; +use crate::linux::net::sock::{Domain, Proto, Sock}; pub fn value(s: &mut Sock, level: u64, name: u64) -> Result, u64> { let o = s.opts; @@ -52,17 +52,17 @@ pub fn value(s: &mut Sock, level: u64, name: u64) -> Result, u64> { (l, _) if s.domain == Domain::Unix && l != SOL_SOCKET => { Err(errno::fail(errno::EOPNOTSUPP)) } - (IPPROTO_TCP, _) if !stream && super::opt_more::known(level, name) => { + (IPPROTO_TCP, _) if !stream && super::more::known(level, name) => { Err(errno::fail(errno::EOPNOTSUPP)) } - (l, n) if super::opt_more::known(l, n) => int(super::opt_more::get(&o.more, l, n)), - // A datagram socket has no TCP options, and Linux says so this way. + (l, n) if super::more::known(l, n) => int(super::more::get(&o.more, l, n)), + /* A datagram socket has no TCP options, and Linux says so this way. */ (IPPROTO_TCP, _) if !stream => Err(errno::fail(errno::EOPNOTSUPP)), (IPPROTO_TCP, TCP_NODELAY) => int(u32::from(o.nodelay)), (IPPROTO_TCP, TCP_KEEPIDLE) => int(o.keepidle), (IPPROTO_TCP, TCP_KEEPINTVL) => int(o.keepintvl), (IPPROTO_TCP, TCP_KEEPCNT) => int(o.keepcnt), (IPPROTO_IPV6, _) => Err(errno::fail(errno::EOPNOTSUPP)), - _ => Err(super::opt_get::unknown("getsockopt", level, name)), + _ => Err(super::get::unknown("getsockopt", level, name)), } } diff --git a/userland/capsule_linux/src/linux/net/opt_set.rs b/userland/capsule_linux/src/linux/net/opt_set.rs deleted file mode 100644 index dff9835f4..000000000 --- a/userland/capsule_linux/src/linux/net/opt_set.rs +++ /dev/null @@ -1,83 +0,0 @@ -// NONOS Operating System -// Copyright (C) 2026 NONOS Contributors -// -// This program is free software: you can redistribute it and/or modify -// it under the terms of the GNU Affero General Public License as published by -// the Free Software Foundation, either version 3 of the License, or -// (at your option) any later version. -// -// This program is distributed in the hope that it will be useful, -// but WITHOUT ANY WARRANTY; without even the implied warranty of -// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -// GNU Affero General Public License for more details. -// -// You should have received a copy of the GNU Affero General Public License -// along with this program. If not, see . - -//! `setsockopt`. Every option here is kept and reads back as Linux reads -//! it; one that would change nothing on the family's loopback is still kept, -//! since Linux keeps it too. One this capsule cannot honour is refused. - -use crate::linux::abi::errno; -use crate::linux::guest::Guest; - -use super::fd::sock_of; -use super::opt_ids::*; -use super::opt_time::{keep, timeo}; -use super::sock::{self, Domain, Proto}; - -pub fn setsockopt(guest: &Guest, fd: u64, level: u64, name: u64, val: u64, len: u64) -> u64 { - let id = match sock_of(guest, fd) { - Ok(id) => id, - Err(e) => return e, - }; - let Some(raw) = guest.read(val, (len as usize).min(16)) else { - return errno::fail(errno::EFAULT); - }; - if len < 4 { - return errno::fail(errno::EINVAL); - } - let int = u32::from_le_bytes([raw[0], raw[1], raw[2], raw[3]]); - let word = - |i: usize| raw.get(i..i + 8).map(|b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8]))); - sock::with(|t| { - let Some(s) = t.get_mut(id) else { - return errno::fail(errno::EBADF); - }; - let (proto, domain) = (s.proto, s.domain); - let o = &mut s.opts; - match (level, name) { - // A Unix socket has only socket-level options on Linux. - (l, _) if domain == Domain::Unix && l != SOL_SOCKET => { - return errno::fail(errno::EOPNOTSUPP) - } - (l, n) - if super::opt_more::known(l, n) && !(l == IPPROTO_TCP && proto == Proto::Dgram) => - { - return super::opt_more::set(&mut o.more, proto, l, n, int) - } - (SOL_SOCKET, SO_REUSEADDR) => o.reuseaddr = int != 0, - (SOL_SOCKET, SO_REUSEPORT) => o.reuseport = int != 0, - (SOL_SOCKET, SO_KEEPALIVE) => o.keepalive = int != 0, - (SOL_SOCKET, SO_BROADCAST) => o.broadcast = int != 0, - (SOL_SOCKET, SO_RCVBUF) => o.rcvbuf = (int.min(BUF_MAX) * 2).max(RCVBUF_MIN), - (SOL_SOCKET, SO_SNDBUF) => o.sndbuf = (int.min(BUF_MAX) * 2).max(SNDBUF_MIN), - (SOL_SOCKET, SO_LINGER) if len >= 8 => { - o.linger = - (u32::from(int != 0), u32::from_le_bytes([raw[4], raw[5], raw[6], raw[7]])) - } - (SOL_SOCKET, SO_LINGER) => return errno::fail(errno::EINVAL), - (SOL_SOCKET, SO_RCVTIMEO) => return timeo(&mut o.rcvtimeo, word(0), word(8)), - (SOL_SOCKET, SO_SNDTIMEO) => return timeo(&mut o.sndtimeo, word(0), word(8)), - (IPPROTO_TCP, _) if proto == Proto::Dgram => return errno::fail(errno::ENOPROTOOPT), - (IPPROTO_TCP, TCP_NODELAY) => o.nodelay = int != 0, - (IPPROTO_TCP, TCP_KEEPIDLE) => return keep(&mut o.keepidle, int, KEEP_MAX), - (IPPROTO_TCP, TCP_KEEPINTVL) => return keep(&mut o.keepintvl, int, KEEP_MAX), - (IPPROTO_TCP, TCP_KEEPCNT) => return keep(&mut o.keepcnt, int, KEEPCNT_MAX), - // An AF_INET socket has no IPv6 options on Linux either. - (IPPROTO_IPV6, _) => return errno::fail(errno::ENOPROTOOPT), - _ => return super::opt_get::unknown("setsockopt", level, name), - } - errno::ok(0) - }) -} diff --git a/userland/capsule_linux/src/linux/net/pair.rs b/userland/capsule_linux/src/linux/net/pair.rs index 4fb4dc3ec..e5b5cf3fd 100644 --- a/userland/capsule_linux/src/linux/net/pair.rs +++ b/userland/capsule_linux/src/linux/net/pair.rs @@ -34,7 +34,7 @@ pub fn socketpair(guest: &mut Guest, family: u64, kind: u64, protocol: u64, out: } match family { f if f == u64::from(AF_UNIX) => {} - // Linux has no connected pair for the internet families. + /* Linux has no connected pair for the internet families. */ f if f == u64::from(AF_INET) => return errno::fail(errno::EOPNOTSUPP), _ => return errno::fail(errno::EAFNOSUPPORT), } @@ -60,7 +60,7 @@ pub fn socketpair(guest: &mut Guest, family: u64, kind: u64, protocol: u64, out: let mut pair = [0u8; 8]; pair[..4].copy_from_slice(&(na as u32).to_le_bytes()); pair[4..].copy_from_slice(&(nb as u32).to_le_bytes()); - // Linux copies the pair out before it installs either descriptor. + /* Linux copies the pair out before it installs either descriptor. */ if guest.write(out, &pair) < 8 { super::close::discard(guest, na as u64); super::close::discard(guest, nb as u64); diff --git a/userland/capsule_linux/src/linux/net/peer_addr.rs b/userland/capsule_linux/src/linux/net/peer_addr.rs index 29bb3984f..8c89528e3 100644 --- a/userland/capsule_linux/src/linux/net/peer_addr.rs +++ b/userland/capsule_linux/src/linux/net/peer_addr.rs @@ -21,9 +21,9 @@ use crate::linux::abi::errno; use crate::linux::guest::Guest; use super::flags::MSG_TRUNC; +use super::named::UAddr; use super::sock::Addr; use super::sockaddr::{self, AF_INET, AF_UNIX}; -use super::sockaddr_un::UAddr; use super::xfer_in::In; /// The count a receive answers, with the sender written out. A stream has @@ -57,7 +57,7 @@ pub fn address(guest: &Guest, at: u64, alen: u64) -> Result, u64> { } match sockaddr::read(guest, at, alen)? { (AF_INET, a) => Ok(Some(To::Inet(a))), - (AF_UNIX, _) => Ok(Some(To::Unix(super::sockaddr_un::read(guest, at, alen)?))), + (AF_UNIX, _) => Ok(Some(To::Unix(super::named::read_uaddr(guest, at, alen)?))), _ => Err(errno::fail(errno::EAFNOSUPPORT)), } } diff --git a/userland/capsule_linux/src/linux/net/recvfrom.rs b/userland/capsule_linux/src/linux/net/recvfrom.rs new file mode 100644 index 000000000..f0179b9b7 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/recvfrom.rs @@ -0,0 +1,46 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! `recvfrom`: a read that says where the bytes came from. + +use alloc::vec; + +use crate::linux::guest::Guest; + +use super::dgram::is_resolver; +use super::fd::sock_of; + +pub fn recvfrom( + guest: &mut Guest, + fd: u64, + buf: u64, + len: u64, + flags: u64, + at: u64, + alen: u64, +) -> u64 { + if is_resolver(guest, fd) { + return super::resolver::answer(guest, fd, buf, len, at, alen); + } + let id = match sock_of(guest, fd) { + Ok(id) => id, + Err(e) => return e, + }; + match super::xfer_in::recv(guest, id, &vec![(buf, len)], 0, flags) { + Ok(got) => super::peer_addr::finish(guest, got, flags, at, alen), + Err(e) => e, + } +} diff --git a/userland/capsule_linux/src/linux/net/shutdown.rs b/userland/capsule_linux/src/linux/net/shutdown.rs index 5a8d3d8e2..ce4cd7cb1 100644 --- a/userland/capsule_linux/src/linux/net/shutdown.rs +++ b/userland/capsule_linux/src/linux/net/shutdown.rs @@ -48,15 +48,9 @@ pub fn shutdown(guest: &Guest, fd: u64, how: u64) -> u64 { ); } if s.listening { - // Shutting a listener's reading side stops it listening. + /* Shutting a listener's reading side stops it listening. */ if rd { - s.listening = false; - let (queued, waiting) = - (core::mem::take(&mut s.pending), core::mem::take(&mut s.syn)); - for q in queued { - t.free(q, true); - } - t.refuse_waiting(waiting.into_iter()); + t.unlisten(id); } return errno::ok(0); } @@ -67,7 +61,7 @@ pub fn shutdown(guest: &Guest, fd: u64, how: u64) -> u64 { s.rd_shut |= rd; s.wr_shut |= wr; let peer = s.peer; - // The peer reads end of file once it has what was already sent. + /* The peer reads end of file once it has what was already sent. */ if let Some(p) = peer.filter(|_| wr).and_then(|p| t.get_mut(p)) { p.eof = true; } diff --git a/userland/capsule_linux/src/linux/net/sock/cell.rs b/userland/capsule_linux/src/linux/net/sock/cell.rs index f98edb5fc..47216b6e8 100644 --- a/userland/capsule_linux/src/linux/net/sock/cell.rs +++ b/userland/capsule_linux/src/linux/net/sock/cell.rs @@ -24,8 +24,10 @@ use super::table::Socks; struct One(RefCell); -// SAFETY: the serve loop is the only thread in this capsule that reaches the -// table; guest threads run in their own processes and only trap into it. +/* + * SAFETY: the serve loop is the only thread in this capsule that reaches the + * table; guest threads run in their own processes and only trap into it. + */ unsafe impl Sync for One {} static TABLE: One = One(RefCell::new(Socks::new())); diff --git a/userland/capsule_linux/src/linux/net/sock/deliver.rs b/userland/capsule_linux/src/linux/net/sock/deliver.rs new file mode 100644 index 000000000..a55a6bb70 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/deliver.rs @@ -0,0 +1,31 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes from one socket into the family: a stream's to its peer, a +//! datagram to where `dest` says, binding the sender first as Linux does. + +use super::gram_dest::Dest; +use super::table::Socks; +use super::types::Proto; + +impl Socks { + pub fn deliver(&mut self, id: u32, dest: Dest, bytes: &[u8]) -> Result { + match self.get(id).map(|s| s.proto) { + Some(Proto::Stream) => self.write(id, bytes), + _ => self.autobind(id).and_then(|()| self.send_gram(id, dest, bytes)), + } + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/free.rs b/userland/capsule_linux/src/linux/net/sock/free.rs index 81d9ab53c..3cbc55df6 100644 --- a/userland/capsule_linux/src/linux/net/sock/free.rs +++ b/userland/capsule_linux/src/linux/net/sock/free.rs @@ -45,8 +45,10 @@ impl Socks { super::super::stream::close(h); } let reset = reset || !gone.rx.is_empty() || gone.opts.linger == (1, 0); - // A stream's peer points back; a connected Unix datagram socket - // points at this one alone. Either is told, and forgets the index. + /* + * A stream's peer points back; a connected Unix datagram socket + * points at this one alone. Either is told, and forgets the index. + */ for p in self.list.iter_mut().flatten().filter(|p| p.peer == Some(id)) { p.peer = None; p.eof = true; diff --git a/userland/capsule_linux/src/linux/net/sock/gram.rs b/userland/capsule_linux/src/linux/net/sock/gram.rs index 74fe5f239..8552f6de8 100644 --- a/userland/capsule_linux/src/linux/net/sock/gram.rs +++ b/userland/capsule_linux/src/linux/net/sock/gram.rs @@ -17,69 +17,33 @@ //! A datagram into the family: to whichever socket holds the port or the //! Unix name it is sent to, or to a connected socket's peer. -use core::mem; - -use crate::linux::abi::errno::{ - EBADF, ECONNREFUSED, EDESTADDRREQ, EINVAL, EMSGSIZE, ENOTCONN, EPERM, -}; +use crate::linux::abi::errno::{ECONNREFUSED, EPERM}; use super::gram_dest::Dest; -use super::name::Peer; +use super::name::{Gram, Peer}; use super::table::Socks; use super::types::Domain; -/// The largest UDP payload IPv4 carries. -const MAX_GRAM: usize = 65507; - impl Socks { /// One datagram from `id`. One nobody holds the port of is dropped, as - /// on the wire; a connected sender learns of it as ECONNREFUSED on its - /// next call, which is what the ICMP reply does on Linux. + /// on the wire. pub fn send_gram(&mut self, id: u32, dest: Dest, bytes: &[u8]) -> Result { - let s = self.get_mut(id).ok_or(EBADF)?; - if s.error != 0 { - return Err(mem::take(&mut s.error)); - } - if bytes.len() > MAX_GRAM { - return Err(EMSGSIZE); - } - let from = match s.domain { - Domain::Inet => Peer::Inet(s.local.unwrap_or_default()), - Domain::Unix => Peer::Unix(s.uname.clone()), - }; - let (remote, peer, connected, domain) = (s.remote, s.peer, s.connected, s.domain); - let target = match (dest, domain) { - (Dest::Sock(t), Domain::Unix) => t, - (Dest::Default, Domain::Unix) => match peer { - Some(p) => p, - None if connected => return Err(ECONNREFUSED), - None => return Err(ENOTCONN), - }, - (Dest::Inet(to), Domain::Inet) => match self.inet_target(id, to, remote.is_some()) { - Some(r) => r, - None => return Ok(bytes.len()), - }, - (Dest::Default, Domain::Inet) => match remote { - Some(to) => match self.inet_target(id, to, true) { - Some(r) => r, - None => return Ok(bytes.len()), - }, - None => return Err(EDESTADDRREQ), - }, - _ => return Err(EINVAL), + let target = self.gram_target(id, dest, bytes.len())?; + let (Some(t), Some(from)) = (target, self.sender(id)) else { + return Ok(bytes.len()); }; - let r = self.get_mut(target).ok_or(ECONNREFUSED)?; - // A connected Unix socket takes only from its peer, and says so. - if r.domain == Domain::Unix && r.peer.is_some_and(|p| p != id) && r.connected { + let r = self.get_mut(t).ok_or(ECONNREFUSED)?; + /* A connected Unix socket takes only from its peer, and says so. */ + if r.domain == Domain::Unix && r.connected && r.peer.is_some_and(|p| p != id) { return Err(EPERM); } - let queued: usize = r.grams.iter().map(|(_, g)| g.len()).sum(); + let queued: usize = r.grams.iter().map(|g| g.bytes.len()).sum(); let wanted = match (&from, r.remote) { (Peer::Inet(a), Some(x)) => *a == x, _ => true, }; if wanted && queued + bytes.len() <= r.opts.rcvbuf as usize { - r.grams.push_back((from, bytes.to_vec())); + r.grams.push_back(Gram { from, bytes: bytes.to_vec() }); } Ok(bytes.len()) } diff --git a/userland/capsule_linux/src/linux/net/sock/gram_in.rs b/userland/capsule_linux/src/linux/net/sock/gram_in.rs index da4408f62..328f2cc78 100644 --- a/userland/capsule_linux/src/linux/net/sock/gram_in.rs +++ b/userland/capsule_linux/src/linux/net/sock/gram_in.rs @@ -28,16 +28,13 @@ impl Socks { /// The next datagram, cut to `want`, with its whole length and sender. pub fn take_gram(&mut self, id: u32, want: usize, peek: bool) -> Result { let s = self.get_mut(id).ok_or(EBADF)?; - if let Some((from, gram)) = s.grams.front() { - let got = Got { - bytes: gram[..want.min(gram.len())].to_vec(), - whole: gram.len(), - from: from.clone(), - }; + if let Some(g) = s.grams.front() { + let (bytes, whole, from) = + (g.bytes[..want.min(g.bytes.len())].to_vec(), g.bytes.len(), g.from.clone()); if !peek { s.grams.pop_front(); } - return Ok(got); + return Ok(Got { bytes, whole, from }); } if s.error != 0 { return Err(mem::take(&mut s.error)); diff --git a/userland/capsule_linux/src/linux/net/sock/gram_target.rs b/userland/capsule_linux/src/linux/net/sock/gram_target.rs new file mode 100644 index 000000000..1fcdc62f5 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/gram_target.rs @@ -0,0 +1,70 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Which socket a datagram goes to, or why it goes nowhere. + +use core::mem; + +use crate::linux::abi::errno::{EBADF, ECONNREFUSED, EDESTADDRREQ, EINVAL, EMSGSIZE, ENOTCONN}; + +use super::gram_dest::Dest; +use super::name::Peer; +use super::table::Socks; +use super::types::Domain; + +/// The largest UDP payload IPv4 carries. +const MAX_GRAM: usize = 65507; + +impl Socks { + /// The socket a datagram from `id` goes to; None drops it unseen. + pub(super) fn gram_target( + &mut self, + id: u32, + dest: Dest, + len: usize, + ) -> Result, i64> { + let s = self.get_mut(id).ok_or(EBADF)?; + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if len > MAX_GRAM { + return Err(EMSGSIZE); + } + let (remote, peer, connected, domain) = (s.remote, s.peer, s.connected, s.domain); + match (dest, domain) { + (Dest::Sock(t), Domain::Unix) => Ok(Some(t)), + (Dest::Default, Domain::Unix) => match peer { + Some(p) => Ok(Some(p)), + None if connected => Err(ECONNREFUSED), + None => Err(ENOTCONN), + }, + (Dest::Inet(to), Domain::Inet) => Ok(self.inet_target(id, to, remote.is_some())), + (Dest::Default, Domain::Inet) => match remote { + Some(to) => Ok(self.inet_target(id, to, true)), + None => Err(EDESTADDRREQ), + }, + _ => Err(EINVAL), + } + } + + /// How a datagram from `id` names its sender. + pub(super) fn sender(&self, id: u32) -> Option { + self.get(id).map(|s| match s.domain { + Domain::Inet => Peer::Inet(s.local.unwrap_or_default()), + Domain::Unix => Peer::Unix(s.uname.clone()), + }) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/kinds.rs b/userland/capsule_linux/src/linux/net/sock/kinds.rs new file mode 100644 index 000000000..e84a90ebd --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/kinds.rs @@ -0,0 +1,47 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! What kind of socket, in which family, at which IPv4 address. + +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Proto { + Stream, + Dgram, +} + +/// AF_INET, or AF_UNIX for the two ends socketpair makes. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum Domain { + Inet, + Unix, +} + +/// An IPv4 address and a port, the port in host order. +#[derive(Clone, Copy, PartialEq, Eq, Default)] +pub struct Addr { + pub ip: [u8; 4], + pub port: u16, +} + +/// How a stream connect to a listener went. +pub enum Link { + /// Connected; the listener has one more connection to accept. + Done, + /// Nothing listens there: Linux's loopback answers with a reset. + Refused, + /// The listener's queue is full. + Full, +} diff --git a/userland/capsule_linux/src/linux/net/sock/link.rs b/userland/capsule_linux/src/linux/net/sock/link.rs index 2fafa8b06..4a4a3d56b 100644 --- a/userland/capsule_linux/src/linux/net/sock/link.rs +++ b/userland/capsule_linux/src/linux/net/sock/link.rs @@ -17,18 +17,10 @@ //! Joining two stream ends: a connect on loopback or to a Unix name, which //! Linux completes in the caller's own call, and socketpair. +use super::kinds::Link; use super::table::Socks; use super::types::{Addr, Proto}; -pub enum Link { - /// Connected; the listener has one more connection to accept. - Done, - /// Nothing listens there: Linux's loopback answers with a reset. - Refused, - /// The listener's queue is full. - Full, -} - impl Socks { /// Connect stream `id`, already bound, to the listener at `to`. pub fn link(&mut self, id: u32, to: Addr) -> Link { @@ -45,7 +37,7 @@ impl Socks { let Some(ls) = self.get(l) else { return Link::Refused; }; - // Linux queues one more than the backlog it was given. + /* Linux queues one more than the backlog it was given. */ if ls.pending.len() > ls.backlog { return Link::Full; } @@ -59,7 +51,7 @@ impl Socks { s.upeer = from_name; s.peer = Some(id); s.connected = true; - // An accepted socket starts with its listener's options. + /* An accepted socket starts with its listener's options. */ s.opts = opts; } if let Some(c) = self.get_mut(id) { diff --git a/userland/capsule_linux/src/linux/net/sock/mod.rs b/userland/capsule_linux/src/linux/net/sock/mod.rs index d4f0bb729..2c0193c71 100644 --- a/userland/capsule_linux/src/linux/net/sock/mod.rs +++ b/userland/capsule_linux/src/linux/net/sock/mod.rs @@ -25,11 +25,14 @@ mod bind; mod cell; +mod deliver; mod free; mod gram; mod gram_dest; mod gram_in; +mod gram_target; mod holders; +mod kinds; mod link; mod name; mod new; @@ -38,20 +41,23 @@ mod opts_more; mod pair; mod port; mod progress; +mod put; mod ready; mod recv; mod send; mod syn; mod table; +mod take; mod types; +mod unlisten; pub use cell::with; pub use gram_dest::Dest; pub use holders::Holder; -pub use link::Link; +pub use kinds::Link; pub use name::{Peer, UName}; -pub use opts_more::More; pub use opts::Opts; +pub use opts_more::More; pub use progress::{progress, set_progress}; pub use ready::bits; pub use types::{Addr, Domain, Proto, Sock}; diff --git a/userland/capsule_linux/src/linux/net/sock/name.rs b/userland/capsule_linux/src/linux/net/sock/name.rs index 908cfa6fb..665eac23d 100644 --- a/userland/capsule_linux/src/linux/net/sock/name.rs +++ b/userland/capsule_linux/src/linux/net/sock/name.rs @@ -14,7 +14,7 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! A Unix socket's name, and who a message came from. +//! A Unix socket's name, who a message came from, and a datagram. use alloc::vec::Vec; @@ -43,3 +43,9 @@ pub enum Peer { Inet(Addr), Unix(Option), } + +/// A datagram waiting to be read, with its sender. +pub struct Gram { + pub from: Peer, + pub bytes: Vec, +} diff --git a/userland/capsule_linux/src/linux/net/sock/put.rs b/userland/capsule_linux/src/linux/net/sock/put.rs new file mode 100644 index 000000000..b7332f101 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/put.rs @@ -0,0 +1,57 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes into a family stream's peer, as many as it has room for. + +use core::mem; + +use crate::linux::abi::errno::{EAGAIN, EBADF, EPIPE}; + +use super::table::Socks; +use super::types::Domain; + +impl Socks { + /// The bytes, and the peer that took them. + pub(super) fn put(&mut self, id: u32, bytes: &[u8]) -> Result<(usize, Option), i64> { + let s = self.get_mut(id).ok_or(EBADF)?; + if s.error != 0 { + return Err(mem::take(&mut s.error)); + } + if !s.connected || s.wr_shut || s.broken { + return Err(EPIPE); + } + let Some(p) = s.peer else { + /* + * TCP takes the first write after the peer left, and the reset + * that answers it makes every later one EPIPE. A Unix socket + * knows at once. + */ + if s.domain == Domain::Unix { + return Err(EPIPE); + } + s.broken = true; + return Ok((bytes.len(), None)); + }; + let peer = self.get_mut(p).ok_or(EPIPE)?; + let room = (peer.opts.rcvbuf as usize).saturating_sub(peer.rx.len()); + if room == 0 && !bytes.is_empty() { + return Err(EAGAIN); + } + let n = room.min(bytes.len()); + peer.rx.extend(&bytes[..n]); + Ok((n, Some(p))) + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/ready.rs b/userland/capsule_linux/src/linux/net/sock/ready.rs index 51537e408..07a994043 100644 --- a/userland/capsule_linux/src/linux/net/sock/ready.rs +++ b/userland/capsule_linux/src/linux/net/sock/ready.rs @@ -30,7 +30,7 @@ const POLLRDHUP: u16 = 0x2000; pub fn bits(id: u32) -> Option { with(|t| { let s = t.get(id).filter(|s| s.svc.is_none())?; - // A stream is writable while its peer has room, or once it is gone. + /* A stream is writable while its peer has room, or once it is gone. */ let room = s.peer.and_then(|p| t.get(p)).is_none_or(|p| p.rx.len() < p.opts.rcvbuf as usize); Some(of(s, room)) @@ -46,11 +46,11 @@ fn of(s: &Sock, room: bool) -> u16 { if s.listening { return err | if s.pending.is_empty() { 0 } else { POLLIN }; } - // A connect waiting for room: SYN_SENT, neither readable nor writable. + /* A connect waiting for room: SYN_SENT, neither readable nor writable. */ if s.connecting { return err; } - // Never connected, refused, or reset: Linux's TCP_CLOSE. + /* Never connected, refused, or reset: Linux's TCP_CLOSE. */ if !s.connected || s.broken { let rd = if s.connected { POLLIN | POLLRDHUP } else { 0 }; return err | rd | POLLOUT | POLLHUP; @@ -66,6 +66,6 @@ fn of(s: &Sock, room: bool) -> u16 { if shut_rd && s.wr_shut { set |= POLLHUP; } - // After SHUT_WR a write fails at once, so Linux reports it writable. + /* After SHUT_WR a write fails at once, so Linux reports it writable. */ set | if room || s.wr_shut { POLLOUT } else { 0 } } diff --git a/userland/capsule_linux/src/linux/net/sock/send.rs b/userland/capsule_linux/src/linux/net/sock/send.rs index 8b4268b86..cdda1d092 100644 --- a/userland/capsule_linux/src/linux/net/sock/send.rs +++ b/userland/capsule_linux/src/linux/net/sock/send.rs @@ -16,40 +16,11 @@ //! Bytes into a family stream: to its peer's queue. -use core::mem; - -use crate::linux::abi::errno::{EAGAIN, EBADF, EPIPE}; - use super::table::Socks; -use super::types::Domain; impl Socks { /// As many of `bytes` as the peer has room for, EAGAIN for none. pub fn write(&mut self, id: u32, bytes: &[u8]) -> Result { - let s = self.get_mut(id).ok_or(EBADF)?; - if s.error != 0 { - return Err(mem::take(&mut s.error)); - } - if !s.connected || s.wr_shut || s.broken { - return Err(EPIPE); - } - let Some(p) = s.peer else { - // TCP takes the first write after the peer left, and the reset - // that answers it makes every later one EPIPE. A Unix socket - // knows at once. - if s.domain == Domain::Unix { - return Err(EPIPE); - } - s.broken = true; - return Ok(bytes.len()); - }; - let peer = self.get_mut(p).ok_or(EPIPE)?; - let room = (peer.opts.rcvbuf as usize).saturating_sub(peer.rx.len()); - if room == 0 && !bytes.is_empty() { - return Err(EAGAIN); - } - let n = room.min(bytes.len()); - peer.rx.extend(&bytes[..n]); - Ok(n) + self.put(id, bytes).map(|(n, _)| n) } } diff --git a/userland/capsule_linux/src/linux/net/sock/syn.rs b/userland/capsule_linux/src/linux/net/sock/syn.rs index 623a75a29..0a29cb8ce 100644 --- a/userland/capsule_linux/src/linux/net/sock/syn.rs +++ b/userland/capsule_linux/src/linux/net/sock/syn.rs @@ -20,8 +20,6 @@ //! accept. Until then the socket is not writable, and a second connect is //! EALREADY. -use crate::linux::abi::errno::ECONNREFUSED; - use super::table::Socks; impl Socks { @@ -53,14 +51,4 @@ impl Socks { } } } - - /// Listener `l` is gone: the connects waiting on it are refused. - pub fn refuse_waiting(&mut self, syn: impl Iterator) { - for c in syn { - if let Some(cs) = self.get_mut(c) { - cs.connecting = false; - cs.error = ECONNREFUSED; - } - } - } } diff --git a/userland/capsule_linux/src/linux/net/sock/take.rs b/userland/capsule_linux/src/linux/net/sock/take.rs new file mode 100644 index 000000000..af6f3fbc9 --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/take.rs @@ -0,0 +1,51 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Bytes out of one family socket: a stream's next bytes, or its next +//! datagram with the name of whoever sent it. + +use alloc::vec::Vec; + +use crate::linux::abi::errno::EBADF; + +use super::name::Peer; +use super::table::Socks; +use super::types::Proto; + +/// The bytes, a datagram's length before it was cut, and its sender. +pub type Taken = (Vec, usize, Option); + +impl Socks { + pub fn take(&mut self, id: u32, want: usize, peek: bool) -> Result { + match self.get(id).map(|s| s.proto) { + None => Err(EBADF), + Some(Proto::Stream) => { + let bytes = self.read(id, want, peek)?; + let n = bytes.len(); + Ok((bytes, n, None)) + } + Some(Proto::Dgram) => { + let got = self.take_gram(id, want, peek)?; + /* A socketpair's peer has no name, and Linux gives none. */ + let from = match got.from { + Peer::Unix(None) => None, + named => Some(named), + }; + Ok((got.bytes, got.whole, from)) + } + } + } +} diff --git a/userland/capsule_linux/src/linux/net/sock/types.rs b/userland/capsule_linux/src/linux/net/sock/types.rs index 7ad528821..513b60d11 100644 --- a/userland/capsule_linux/src/linux/net/sock/types.rs +++ b/userland/capsule_linux/src/linux/net/sock/types.rs @@ -19,28 +19,10 @@ use alloc::collections::VecDeque; use alloc::vec::Vec; -use super::name::{Peer, UName}; +use super::name::{Gram, UName}; use super::opts::Opts; -#[derive(Clone, Copy, PartialEq, Eq)] -pub enum Proto { - Stream, - Dgram, -} - -/// AF_INET, or AF_UNIX for the two ends socketpair makes. -#[derive(Clone, Copy, PartialEq, Eq)] -pub enum Domain { - Inet, - Unix, -} - -/// An IPv4 address and a port, the port in host order. -#[derive(Clone, Copy, PartialEq, Eq, Default)] -pub struct Addr { - pub ip: [u8; 4], - pub port: u16, -} +pub use super::kinds::{Addr, Domain, Proto}; pub struct Sock { pub domain: Domain, @@ -60,8 +42,8 @@ pub struct Sock { pub connected: bool, /// Bytes the peer wrote that this end has not read. pub rx: VecDeque, - /// Datagrams waiting to be read, each with where it came from. - pub grams: VecDeque<(Peer, Vec)>, + /// Datagrams waiting to be read. + pub grams: VecDeque, /// The peer will write nothing more: it shut its side or it is gone. pub eof: bool, pub wr_shut: bool, diff --git a/userland/capsule_linux/src/linux/net/sock/unlisten.rs b/userland/capsule_linux/src/linux/net/sock/unlisten.rs new file mode 100644 index 000000000..4a57a1a3b --- /dev/null +++ b/userland/capsule_linux/src/linux/net/sock/unlisten.rs @@ -0,0 +1,48 @@ +// NONOS Operating System +// Copyright (C) 2026 NONOS Contributors +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! A listener that stops: what it queued is reset, and the connects that +//! wait for its room are refused. + +use crate::linux::abi::errno::ECONNREFUSED; + +use super::table::Socks; + +impl Socks { + /// Listener `l` is gone: the connects waiting on it are refused. + pub fn refuse_waiting(&mut self, syn: impl Iterator) { + for c in syn { + if let Some(cs) = self.get_mut(c) { + cs.connecting = false; + cs.error = ECONNREFUSED; + } + } + } + + /// Listener `l` stops listening: what it queued is reset, and the + /// connects waiting for room are refused. + pub fn unlisten(&mut self, l: u32) { + let Some(s) = self.get_mut(l) else { + return; + }; + s.listening = false; + let (queued, waiting) = (core::mem::take(&mut s.pending), core::mem::take(&mut s.syn)); + for q in queued { + self.free(q, true); + } + self.refuse_waiting(waiting.into_iter()); + } +} diff --git a/userland/capsule_linux/src/linux/net/socket.rs b/userland/capsule_linux/src/linux/net/socket.rs index dcca853ed..85d09936f 100644 --- a/userland/capsule_linux/src/linux/net/socket.rs +++ b/userland/capsule_linux/src/linux/net/socket.rs @@ -53,7 +53,7 @@ pub fn socket(guest: &mut Guest, family: u64, kind: u64, protocol: u64) -> u64 { return refuse("SOCK_RAW: raw sockets reach below any confinement", errno::EPERM) } (SOCK_STREAM, Domain::Unix) => (Proto::Stream, PF_UNIX), - // Linux gives a raw Unix socket datagram semantics. + /* Linux gives a raw Unix socket datagram semantics. */ (SOCK_DGRAM | SOCK_RAW, Domain::Unix) => (Proto::Dgram, PF_UNIX), (SOCK_SEQPACKET, Domain::Unix) => { return refuse( @@ -63,8 +63,10 @@ pub fn socket(guest: &mut Guest, family: u64, kind: u64, protocol: u64) -> u64 { } _ => return errno::fail(errno::ESOCKTNOSUPPORT), }; - // Anything but the type's own protocol, MPTCP included, is one this - // stack does not have; Go falls back to TCP on this answer. + /* + * Anything but the type's own protocol, MPTCP included, is one this + * stack does not have; Go falls back to TCP on this answer. + */ if protocol != 0 && protocol != own { return errno::fail(errno::EPROTONOSUPPORT); } diff --git a/userland/capsule_linux/src/linux/net/stream.rs b/userland/capsule_linux/src/linux/net/stream.rs index 943dcb972..0123a5c07 100644 --- a/userland/capsule_linux/src/linux/net/stream.rs +++ b/userland/capsule_linux/src/linux/net/stream.rs @@ -14,7 +14,6 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - //! Bytes on a stream that reaches outside the family, through net.sockets. use alloc::vec::Vec; diff --git a/userland/capsule_linux/src/linux/net/try_call.rs b/userland/capsule_linux/src/linux/net/try_call.rs index 0af36208b..90203c3e4 100644 --- a/userland/capsule_linux/src/linux/net/try_call.rs +++ b/userland/capsule_linux/src/linux/net/try_call.rs @@ -24,6 +24,7 @@ use alloc::vec; use crate::linux::abi::{errno, nr, nr_path as np}; use crate::linux::guest::Guest; +use super::call_kind::{bytes_in, msg_len}; use super::{iov, mmsg}; /// The answer, and the whole amount the call asks to move. @@ -56,7 +57,9 @@ pub fn try_call(guest: &mut Guest, n: u64, a: [u64; 6], done: usize) -> (u64, us (super::xfer_out::send(guest, id, &one(a[1], a[2]), done, a[3], None), a[2] as usize) } nr::SENDMSG => (super::msg::sendmsg(guest, fd, a[1], a[2], done), msg_len(guest, a[1])), - nr::RECVMSG => (super::msg::recvmsg(guest, fd, a[1], a[2], done), msg_len(guest, a[1])), + nr::RECVMSG => { + (super::msg_recv::recvmsg(guest, fd, a[1], a[2], done), msg_len(guest, a[1])) + } nr::SENDMMSG => (mmsg::sendmmsg(guest, a, done), a[2].min(mmsg::MOST) as usize), nr::RECVMMSG => (mmsg::recvmmsg(guest, a, done), a[2].min(mmsg::MOST) as usize), nr::ACCEPT => (super::accept::accept4(guest, fd, a[1], a[2], 0), 0), @@ -65,17 +68,3 @@ pub fn try_call(guest: &mut Guest, n: u64, a: [u64; 6], done: usize) -> (u64, us _ => (errno::fail(errno::ENOSYS), 0), } } - -fn bytes_in(guest: &Guest, id: u32, v: &iov::Iov, done: usize, flags: u64) -> u64 { - match super::xfer_in::recv(guest, id, v, done, flags) { - Ok(got) => errno::ok(got.n as u64), - Err(e) => e, - } -} - -fn msg_len(guest: &Guest, msg: u64) -> usize { - let word = |at: u64| { - guest.read(at, 8).map_or(0, |b| u64::from_le_bytes(b.try_into().unwrap_or([0; 8]))) - }; - iov::read(guest, word(msg + 16), word(msg + 24)).map_or(0, |v| iov::total(&v)) -} diff --git a/userland/capsule_linux/src/linux/net/xfer_in.rs b/userland/capsule_linux/src/linux/net/xfer_in.rs index 2cb501676..ddfbdb409 100644 --- a/userland/capsule_linux/src/linux/net/xfer_in.rs +++ b/userland/capsule_linux/src/linux/net/xfer_in.rs @@ -14,7 +14,7 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Bytes into a guest from a socket. +//! Bytes into a guest from a family socket or a stream net.sockets holds. use alloc::vec; @@ -50,26 +50,12 @@ pub fn recv(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64) -> Resul iov::scatter(guest, iov, skip, &bytes)?; return Ok(In { n: bytes.len(), whole: bytes.len(), from: None }); } - match proto { - Proto::Stream => { - let bytes = sock::with(|t| t.read(id, want, peek)).map_err(errno::fail)?; - // A stream's MSG_TRUNC discards what it would have read. - if flags & MSG_TRUNC == 0 { - iov::scatter(guest, iov, skip, &bytes)?; - } - Ok(In { n: bytes.len(), whole: bytes.len(), from: None }) - } - Proto::Dgram => { - let got = sock::with(|t| t.take_gram(id, want, peek)).map_err(errno::fail)?; - iov::scatter(guest, iov, skip, &got.bytes)?; - // A socketpair's peer has no name, and Linux gives none. - let from = match got.from { - Peer::Unix(None) => None, - named => Some(named), - }; - Ok(In { n: got.bytes.len(), whole: got.whole, from }) - } + let (bytes, whole, from) = sock::with(|t| t.take(id, want, peek)).map_err(errno::fail)?; + /* A stream's MSG_TRUNC discards what it would have read. */ + if proto != Proto::Stream || flags & MSG_TRUNC == 0 { + iov::scatter(guest, iov, skip, &bytes)?; } + Ok(In { n: bytes.len(), whole, from }) } /// `read` on a socket. diff --git a/userland/capsule_linux/src/linux/net/xfer_out.rs b/userland/capsule_linux/src/linux/net/xfer_out.rs index 97fe43fc8..d2ab7fb63 100644 --- a/userland/capsule_linux/src/linux/net/xfer_out.rs +++ b/userland/capsule_linux/src/linux/net/xfer_out.rs @@ -14,8 +14,9 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Bytes out of a socket: to a stream's peer, a datagram to the port it is -//! sent to, or a stream outside the family through net.sockets. +//! Bytes out of a socket: to a stream's peer, a datagram to the port or the +//! Unix name it is sent to, or a stream outside the family through +//! net.sockets. use alloc::vec; @@ -27,12 +28,6 @@ use super::iov::{self, Iov}; use super::peer_addr::To; use super::sock::{self, Proto}; -/// What one send takes from the guest at most: the default receive buffer -/// of a stream's peer, so a single call can fill it. -const STREAM_CAP: usize = 128 << 10; -/// One byte past the largest datagram, so a larger one is seen and refused. -const GRAM_CAP: usize = 65508; - /// Send the message in `iov` from socket `id`, `skip` bytes of it having /// gone already: the count sent now, or an errno. pub fn send(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64, to: Option) -> u64 { @@ -43,11 +38,7 @@ pub fn send(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64, to: Opti else { return errno::fail(errno::EBADF); }; - let cap = match (svc, proto) { - (Some(_), _) => super::stream::MAX_IO, - (None, Proto::Stream) => STREAM_CAP, - (None, Proto::Dgram) => GRAM_CAP, - }; + let cap = super::cap::cap(svc.is_some(), proto == Proto::Stream); let bytes = match iov::gather(guest, iov, skip, cap) { Ok(b) => b, Err(e) => return e, @@ -59,11 +50,7 @@ pub fn send(guest: &Guest, id: u32, iov: &Iov, skip: usize, flags: u64, to: Opti Ok(d) => d, Err(e) => return e, }; - let sent = sock::with(|t| match proto { - Proto::Stream => t.write(id, &bytes), - Proto::Dgram => t.autobind(id).and_then(|()| t.send_gram(id, dest, &bytes)), - }); - match sent { + match sock::with(|t| t.deliver(id, dest, &bytes)) { Ok(n) => errno::ok(n as u64), Err(e) => errno::fail(e), } diff --git a/userland/capsule_linux/src/linux/serve/dispatch.rs b/userland/capsule_linux/src/linux/serve/dispatch.rs index 3e59f6c87..4523e707f 100644 --- a/userland/capsule_linux/src/linux/serve/dispatch.rs +++ b/userland/capsule_linux/src/linux/serve/dispatch.rs @@ -21,6 +21,7 @@ use nonos_libc::ForeignFrame; use super::answer::Answer; use super::table::plain; +use super::waits_sock; use crate::linux::abi::{nr, nr_path as np}; use crate::linux::call::{clone, exit_thread, futex}; use crate::linux::guest::Guest; @@ -58,9 +59,7 @@ fn route(guest: &mut Guest, frame: &ForeignFrame) -> Answer { np::CLOCK_NANOSLEEP => { crate::linux::call::clock_nanosleep(guest, frame.pid, a[0], a[1], a[2]) } - n if super::waits_sock::takes(guest, n, a[0]) => { - super::waits_sock::io(guest, frame.pid, n, a) - } + n if waits_sock::takes(guest, n, a[0]) => waits_sock::io(guest, frame.pid, n, a), nr::READ | nr::WRITE if super::waits::may_wait(guest, frame.nr, a[0]) => { super::waits::io(guest, frame.pid, frame.nr, a) } diff --git a/userland/capsule_linux/src/linux/serve/family_waits.rs b/userland/capsule_linux/src/linux/serve/family_waits.rs index c3f667f79..6f50f98fb 100644 --- a/userland/capsule_linux/src/linux/serve/family_waits.rs +++ b/userland/capsule_linux/src/linux/serve/family_waits.rs @@ -28,10 +28,8 @@ use crate::linux::guest::Kind; use crate::linux::net::outside; const CLOCK_MONOTONIC: u64 = 1; -/// How often a wait on a stream net.sockets holds is looked at again: its -/// readiness changes with no call for the family to answer. A family socket -/// changes only in an answer, after which every wait is tried, so it needs -/// none. A timer is looked at when it fires. +/// How often a wait on a stream net.sockets holds is looked at again; a family +/// socket changes only in an answer. A timer is looked at when it fires. const TICK_MS: u64 = 10; impl Family { @@ -76,9 +74,7 @@ impl Family { if let Some(d) = wait.deadline { keep(d.saturating_sub(now)); } - if super::waits_sock::ticks(g, wait) { - keep(TICK_MS); - } + super::waits_sock::ticks(g, wait).then(|| keep(TICK_MS)); for fd in watched(g, wait) { match g.fds.get(fd as usize) { Some(f) if f.kind == Kind::Socket && outside(f.handle) => keep(TICK_MS), diff --git a/userland/capsule_linux/src/linux/serve/mod.rs b/userland/capsule_linux/src/linux/serve/mod.rs index 324eed45f..12527be3b 100644 --- a/userland/capsule_linux/src/linux/serve/mod.rs +++ b/userland/capsule_linux/src/linux/serve/mod.rs @@ -28,8 +28,8 @@ mod family_waits; mod loop_impl; mod pid_map; mod pid_ns; -mod refused; mod pid_out; +mod refused; mod table; mod table_file; mod table_link; diff --git a/userland/capsule_linux/src/linux/serve/waits_sock.rs b/userland/capsule_linux/src/linux/serve/waits_sock.rs index 5991c2bd6..9a7fede40 100644 --- a/userland/capsule_linux/src/linux/serve/waits_sock.rs +++ b/userland/capsule_linux/src/linux/serve/waits_sock.rs @@ -14,34 +14,25 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Socket calls that wait: accept, connect, the sends and receives, and a -//! read or write on a socket. Each is tried at once; a blocking one that -//! cannot finish is parked, and the family tries it again after every -//! answer (`family_waits`). A family socket changes only in an answer, so -//! it needs no tick; a stream net.sockets holds is looked at on one. -//! -//! SO_RCVTIMEO and SO_SNDTIMEO bound the wait, which then answers EAGAIN, -//! or the count moved so far, as Linux does. +//! Socket calls that wait. Each is tried at once; a blocking one that cannot +//! finish is parked and tried again after every answer (`family_waits`). +//! SO_RCVTIMEO and SO_SNDTIMEO bound the wait, which then answers EAGAIN, or +//! the count moved so far, as Linux does. -use crate::linux::abi::{errno, nr}; +use crate::linux::abi::errno; use crate::linux::call::now_ms; use crate::linux::guest::{Blocked, Guest}; use crate::linux::net::{self, sock}; use super::answer::Answer; +use super::waits_sock_kind::deadline; pub use super::waits_sock_kind::{takes, ticks}; const CLOCK_MONOTONIC: u64 = 1; const MSG_DONTWAIT: u64 = 0x40; pub fn io(guest: &mut Guest, tid: u32, n: u64, a: [u64; 6]) -> Answer { - let reads = !matches!( - n, - nr::WRITE | nr::WRITEV | nr::SENDTO | nr::SENDMSG | nr::SENDMMSG | nr::CONNECT - ); - let limit = net::sock_id(guest, a[0]).and_then(|id| net::limit_ms(id, reads)); - let deadline = limit.map(|ms| now_ms(CLOCK_MONOTONIC).unwrap_or(0).saturating_add(ms)); - let wait = Blocked { tid, nr: n, args: a, deadline }; + let wait = Blocked { tid, nr: n, args: a, deadline: deadline(guest, n, a[0]) }; match attempt(guest, &wait) { Some(v) => Answer::value(v), None => { @@ -72,7 +63,7 @@ pub fn attempt(guest: &mut Guest, wait: &Blocked) -> Option { total as u64 } None if value == errno::fail(errno::EAGAIN) && blocking && !late => return None, - // What moved before an error or the time limit is what Linux answers. + /* What moved before an error or the time limit is what Linux answers. */ None if done != 0 => done as u64, None => value, }; diff --git a/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs b/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs index 6e1beafa0..31f2bf7d0 100644 --- a/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs +++ b/userland/capsule_linux/src/linux/serve/waits_sock_kind.rs @@ -14,12 +14,16 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! Which calls `waits_sock` takes, and which of its waits need a tick. +//! Which calls `waits_sock` takes, which of its waits need a tick, and +//! when a wait's time runs out. use crate::linux::abi::{nr, nr_path as np}; +use crate::linux::call::now_ms; use crate::linux::guest::{Blocked, Guest, Kind}; use crate::linux::net; +const CLOCK_MONOTONIC: u64 = 1; + /// True for a call on a socket descriptor that may have to wait. pub fn takes(guest: &Guest, n: u64, fd: u64) -> bool { matches!( @@ -46,3 +50,13 @@ pub fn ticks(guest: &Guest, wait: &Blocked) -> bool { takes(guest, wait.nr, wait.args[0]) && net::sock_id(guest, wait.args[0]).is_some_and(net::outside) } + +/// When a call's SO_RCVTIMEO or SO_SNDTIMEO runs out, if it has one. +pub fn deadline(guest: &Guest, n: u64, fd: u64) -> Option { + let reads = !matches!( + n, + nr::WRITE | nr::WRITEV | nr::SENDTO | nr::SENDMSG | nr::SENDMMSG | nr::CONNECT + ); + let limit = net::sock_id(guest, fd).and_then(|id| net::limit_ms(id, reads))?; + Some(now_ms(CLOCK_MONOTONIC).unwrap_or(0).saturating_add(limit)) +} diff --git a/userland/capsule_linux/src/linux/unix/mod.rs b/userland/capsule_linux/src/linux/unix/mod.rs index 44e0e2aa1..b84a27e89 100644 --- a/userland/capsule_linux/src/linux/unix/mod.rs +++ b/userland/capsule_linux/src/linux/unix/mod.rs @@ -14,7 +14,6 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . - //! The display connection: a Unix socket whose far end is this capsule. //! Every other Unix socket is the family's own (`net::sock`). @@ -30,9 +29,9 @@ mod sock; mod sock_io; pub use conn::Conn; -pub use sock::is_unix; +pub use path::is_display; pub use recvmsg::recvmsg; pub use sendmsg::sendmsg; -pub use path::is_display; pub use sock::connect; +pub use sock::is_unix; pub use sock_io::{recv, send}; From de69c36507643c771dad6a23697a83844b7df73d Mon Sep 17 00:00:00 2001 From: eKisNonos Date: Mon, 28 Sep 2026 21:58:10 +0000 Subject: [PATCH 14/14] linux guests: socket guests split into parts of at most 72 lines csock, cudp, cunix, cpolicy and cidle keep their parts and their output; each is now .c with its parts in _N.h, none over 72 lines, and comments in /* */, as are gohttp's. Guests.mk picks the parts up with a wildcard, so a changed part rebuilds its guest. Every write and pipe result is now checked, so a guest built with glibc's -Wall -Wextra builds without a warning; a write that fails is named as its part's failure instead of passing unseen. backlog's second accept now waits at most 3 s for the held connect; before, a listener that never held it blocked csock there for good instead of failing the part by name. On Linux: csock PASS 18 parts, cudp PASS 6, cunix PASS 6, cidle PASS; cpolicy reports "confined 0" there, as root on an unconfined host must. --- userland/linux_guests/Guests.mk | 10 +- userland/linux_guests/c/cidle.c | 29 +-- userland/linux_guests/c/cidle_1.h | 20 ++ userland/linux_guests/c/cpolicy.c | 43 ++-- userland/linux_guests/c/cpolicy_1.h | 14 ++ userland/linux_guests/c/csock.c | 121 +++------- userland/linux_guests/c/csock_1.h | 67 ++++++ userland/linux_guests/c/csock_10.h | 33 +++ userland/linux_guests/c/csock_11.h | 52 +++++ userland/linux_guests/c/csock_2.h | 57 +++++ .../c/{csock_parts.h => csock_3.h} | 45 +--- userland/linux_guests/c/csock_4.h | 52 +++++ .../c/{csock_parts2.h => csock_5.h} | 57 +---- .../c/{csock_parts3.h => csock_6.h} | 60 +---- userland/linux_guests/c/csock_7.h | 61 ++++++ userland/linux_guests/c/csock_8.h | 54 +++++ userland/linux_guests/c/csock_9.h | 72 ++++++ userland/linux_guests/c/csock_parts4.h | 206 ------------------ userland/linux_guests/c/cudp.c | 95 ++------ userland/linux_guests/c/cudp_1.h | 51 +++++ .../linux_guests/c/{cudp_parts.h => cudp_2.h} | 70 +++--- userland/linux_guests/c/cudp_3.h | 41 ++++ userland/linux_guests/c/cunix.c | 108 +++------ userland/linux_guests/c/cunix_1.h | 27 +++ userland/linux_guests/c/cunix_2.h | 50 +++++ .../c/{cunix_parts.h => cunix_3.h} | 6 +- .../c/{cunix_parts2.h => cunix_4.h} | 26 +-- userland/linux_guests/go/http/main.go | 10 +- 28 files changed, 806 insertions(+), 731 deletions(-) create mode 100644 userland/linux_guests/c/cidle_1.h create mode 100644 userland/linux_guests/c/cpolicy_1.h create mode 100644 userland/linux_guests/c/csock_1.h create mode 100644 userland/linux_guests/c/csock_10.h create mode 100644 userland/linux_guests/c/csock_11.h create mode 100644 userland/linux_guests/c/csock_2.h rename userland/linux_guests/c/{csock_parts.h => csock_3.h} (55%) create mode 100644 userland/linux_guests/c/csock_4.h rename userland/linux_guests/c/{csock_parts2.h => csock_5.h} (52%) rename userland/linux_guests/c/{csock_parts3.h => csock_6.h} (53%) create mode 100644 userland/linux_guests/c/csock_7.h create mode 100644 userland/linux_guests/c/csock_8.h create mode 100644 userland/linux_guests/c/csock_9.h delete mode 100644 userland/linux_guests/c/csock_parts4.h create mode 100644 userland/linux_guests/c/cudp_1.h rename userland/linux_guests/c/{cudp_parts.h => cudp_2.h} (50%) create mode 100644 userland/linux_guests/c/cudp_3.h create mode 100644 userland/linux_guests/c/cunix_1.h create mode 100644 userland/linux_guests/c/cunix_2.h rename userland/linux_guests/c/{cunix_parts.h => cunix_3.h} (95%) rename userland/linux_guests/c/{cunix_parts2.h => cunix_4.h} (73%) diff --git a/userland/linux_guests/Guests.mk b/userland/linux_guests/Guests.mk index ea197ab4a..900254bac 100644 --- a/userland/linux_guests/Guests.mk +++ b/userland/linux_guests/Guests.mk @@ -134,29 +134,29 @@ $(eval $(call LINUX_GUEST,cwait,4978,4979,$(LINUX_GUESTS_C)/cwait)) # listener with accept4's flags, a non-blocking connect, a refused port, # half-close, epoll on a listener, end of file, EAGAIN, MSG_PEEK, EPIPE, an # accept and a receive that wait, the options a server sets, and fork. -$(LINUX_GUESTS_C)/csock: $(LINUX_GUESTS_DIR)/c/csock.c $(wildcard $(LINUX_GUESTS_DIR)/c/csock_parts*.h) +$(LINUX_GUESTS_C)/csock: $(LINUX_GUESTS_DIR)/c/csock.c $(wildcard $(LINUX_GUESTS_DIR)/c/csock_*.h) @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,csock,5000,5001,$(LINUX_GUESTS_C)/csock)) # Datagrams on the family's loopback: an echo, a connected socket, MSG_TRUNC, # a refused port, sendmmsg and recvmmsg, and no peer at all. -$(LINUX_GUESTS_C)/cudp: $(LINUX_GUESTS_DIR)/c/cudp.c $(LINUX_GUESTS_DIR)/c/cudp_parts.h +$(LINUX_GUESTS_C)/cudp: $(LINUX_GUESTS_DIR)/c/cudp.c $(wildcard $(LINUX_GUESTS_DIR)/c/cudp_*.h) @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cudp,5004,5005,$(LINUX_GUESTS_C)/cudp)) # What a guest's sockets may reach, and what a descriptor number alone gets. -$(LINUX_GUESTS_C)/cpolicy: $(LINUX_GUESTS_DIR)/c/cpolicy.c +$(LINUX_GUESTS_C)/cpolicy: $(LINUX_GUESTS_DIR)/c/cpolicy.c $(wildcard $(LINUX_GUESTS_DIR)/c/cpolicy_*.h) @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cpolicy,5006,5007,$(LINUX_GUESTS_C)/cpolicy)) # A guest blocked in accept with nothing happening, for the loop's wakeups. -$(LINUX_GUESTS_C)/cidle: $(LINUX_GUESTS_DIR)/c/cidle.c +$(LINUX_GUESTS_C)/cidle: $(LINUX_GUESTS_DIR)/c/cidle.c $(wildcard $(LINUX_GUESTS_DIR)/c/cidle_*.h) @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cidle,5008,5009,$(LINUX_GUESTS_C)/cidle)) # Unix sockets with names: a path, what it leaves behind, abstract names, # a connected datagram socket, autobind, and a connection across fork. -$(LINUX_GUESTS_C)/cunix: $(LINUX_GUESTS_DIR)/c/cunix.c $(wildcard $(LINUX_GUESTS_DIR)/c/cunix_parts*.h) +$(LINUX_GUESTS_C)/cunix: $(LINUX_GUESTS_DIR)/c/cunix.c $(wildcard $(LINUX_GUESTS_DIR)/c/cunix_*.h) @mkdir -p $(@D) && musl-gcc -O2 -static -o $@ $< $(eval $(call LINUX_GUEST,cunix,5010,5011,$(LINUX_GUESTS_C)/cunix)) diff --git a/userland/linux_guests/c/cidle.c b/userland/linux_guests/c/cidle.c index 3110b7772..d94d14006 100644 --- a/userland/linux_guests/c/cidle.c +++ b/userland/linux_guests/c/cidle.c @@ -1,7 +1,10 @@ -// An idle guest: main blocks in accept on 127.0.0.1 while nothing happens -// for the number of seconds given (default 10), then a thread connects. It -// is what the serve loop's wakeups are measured against: a family socket -// changes only in an answer, so the loop has nothing to look at meanwhile. +/* + * An idle guest: main blocks in accept on 127.0.0.1 while nothing happens + * for the number of seconds given (default 10), then a thread connects. It + * is what the serve loop's wakeups are measured against: a family socket + * changes only in an answer, so the loop has nothing to look at meanwhile. + */ + #include #include #include @@ -11,23 +14,7 @@ #include #include -static struct sockaddr_in addr; -static int idle_s = 10; - -static long now_ms(void) { - struct timespec ts; - clock_gettime(CLOCK_MONOTONIC, &ts); - return ts.tv_sec * 1000 + ts.tv_nsec / 1000000; -} - -static void *late(void *arg) { - (void)arg; - struct timespec ts = {idle_s, 0}; - nanosleep(&ts, 0); - int c = socket(AF_INET, SOCK_STREAM, 0); - connect(c, (void *)&addr, sizeof addr); - return (void *)(long)c; -} +#include "cidle_1.h" int main(int argc, char **argv) { if (argc > 1) { diff --git a/userland/linux_guests/c/cidle_1.h b/userland/linux_guests/c/cidle_1.h new file mode 100644 index 000000000..e624b6e19 --- /dev/null +++ b/userland/linux_guests/c/cidle_1.h @@ -0,0 +1,20 @@ +/* cidle, part 1 of 1: included once, by cidle.c. */ + +static struct sockaddr_in addr; + +static int idle_s = 10; + +static long now_ms(void) { + struct timespec ts; + clock_gettime(CLOCK_MONOTONIC, &ts); + return ts.tv_sec * 1000 + ts.tv_nsec / 1000000; +} + +static void *late(void *arg) { + (void)arg; + struct timespec ts = {idle_s, 0}; + nanosleep(&ts, 0); + int c = socket(AF_INET, SOCK_STREAM, 0); + connect(c, (void *)&addr, sizeof addr); + return (void *)(long)c; +} diff --git a/userland/linux_guests/c/cpolicy.c b/userland/linux_guests/c/cpolicy.c index bf6d32fc0..0ff76e759 100644 --- a/userland/linux_guests/c/cpolicy.c +++ b/userland/linux_guests/c/cpolicy.c @@ -1,14 +1,17 @@ -// What a guest's sockets may reach, and what a descriptor number alone gets. -// Each part prints the errno it got; the NONOS answer is the capsule's policy -// and differs from an unconfined Linux by design, so the host's line records -// what Linux itself would allow. Parts: -// bind-any bind 0.0.0.0:0 NONOS EACCES, Linux 0 -// bind-out bind 10.0.2.15:0 NONOS EACCES, Linux EADDRNOTAVAIL -// listen-unbound listen with no bind NONOS EACCES, Linux 0 -// udp-out sendto 192.0.2.1:9 NONOS ENETUNREACH, Linux 1 -// raw socket(SOCK_RAW) NONOS EPERM, Linux EPERM unless root -// forged recv/close on numbers the guest never opened, and on a -// pipe: EBADF, EBADF, ENOTSOCK on both +/* + * What a guest's sockets may reach, and what a descriptor number alone gets. + * Each part prints the errno it got; the NONOS answer is the capsule's policy + * and differs from an unconfined Linux by design, so the host's line records + * what Linux itself would allow. Parts: + * bind-any bind 0.0.0.0:0 NONOS EACCES, Linux 0 + * bind-out bind 10.0.2.15:0 NONOS EACCES, Linux EADDRNOTAVAIL + * listen-unbound listen with no bind NONOS EACCES, Linux 0 + * udp-out sendto 192.0.2.1:9 NONOS ENETUNREACH, Linux 1 + * raw socket(SOCK_RAW) NONOS EPERM, Linux EPERM unless root + * forged recv/close on numbers the guest never opened, and on a + * pipe: EBADF, EBADF, ENOTSOCK on both + */ + #include #include #include @@ -16,18 +19,7 @@ #include #include -static struct sockaddr_in at(unsigned ip, int port) { - struct sockaddr_in sa; - memset(&sa, 0, sizeof sa); - sa.sin_family = AF_INET; - sa.sin_port = htons(port); - sa.sin_addr.s_addr = htonl(ip); - return sa; -} - -static int err(int rc) { - return rc < 0 ? errno : 0; -} +#include "cpolicy_1.h" int main(void) { int s = socket(AF_INET, SOCK_STREAM, 0); @@ -46,7 +38,10 @@ int main(void) { int raw = err(socket(AF_INET, SOCK_RAW, IPPROTO_ICMP)); char buf[4]; int pipe_fds[2]; - pipe(pipe_fds); + if (pipe(pipe_fds)) { + printf("[C] cpolicy FAIL: pipe errno %d\n", errno); + return 1; + } int forged_recv = err(recv(777, buf, 4, MSG_DONTWAIT)); int forged_close = err(close(778)); int pipe_recv = err(recv(pipe_fds[0], buf, 4, MSG_DONTWAIT)); diff --git a/userland/linux_guests/c/cpolicy_1.h b/userland/linux_guests/c/cpolicy_1.h new file mode 100644 index 000000000..6127f8e45 --- /dev/null +++ b/userland/linux_guests/c/cpolicy_1.h @@ -0,0 +1,14 @@ +/* cpolicy, part 1 of 1: included once, by cpolicy.c. */ + +static struct sockaddr_in at(unsigned ip, int port) { + struct sockaddr_in sa; + memset(&sa, 0, sizeof sa); + sa.sin_family = AF_INET; + sa.sin_port = htons(port); + sa.sin_addr.s_addr = htonl(ip); + return sa; +} + +static int err(int rc) { + return rc < 0 ? errno : 0; +} diff --git a/userland/linux_guests/c/csock.c b/userland/linux_guests/c/csock.c index b392bb1fe..481813ffe 100644 --- a/userland/linux_guests/c/csock.c +++ b/userland/linux_guests/c/csock.c @@ -1,14 +1,17 @@ -// Sockets, as Linux has them: a socketpair each way, a listener on 127.0.0.1 -// with its name and accept4's flags, a non-blocking connect that answers -// EINPROGRESS and then SO_ERROR 0, a refused port, shutdown(SHUT_WR) giving -// the peer end of file while the other way still flows, epoll readiness on a -// listener, end of file, EAGAIN on an empty non-blocking receive, MSG_PEEK -// and MSG_DONTWAIT, EPIPE after the peer is gone, an accept and a receive -// that wait for another thread, the options Go and a C server set, a -// socket a forked child shares, a connect a full listener holds, the -// options a loopback connection cannot tell apart, and listeners that share -// a port with SO_REUSEPORT. Each part prints as it passes and every part -// runs, so one run names each part that fails. +/* + * Sockets, as Linux has them: a socketpair each way, a listener on 127.0.0.1 + * with its name and accept4's flags, a non-blocking connect that answers + * EINPROGRESS and then SO_ERROR 0, a refused port, shutdown(SHUT_WR) giving + * the peer end of file while the other way still flows, epoll readiness on a + * listener, end of file, EAGAIN on an empty non-blocking receive, MSG_PEEK + * and MSG_DONTWAIT, EPIPE after the peer is gone, an accept and a receive + * that wait for another thread, the options Go and a C server set, a + * socket a forked child shares, a connect a full listener holds, the + * options a loopback connection cannot tell apart, and listeners that share + * a port with SO_REUSEPORT. Each part prints as it passes and every part + * runs, so one run names each part that fails. + */ + #define _GNU_SOURCE #include #include @@ -25,92 +28,24 @@ #include #include -static int parts; - -static long now_ms(void) { - struct timespec ts; - clock_gettime(CLOCK_MONOTONIC, &ts); - return ts.tv_sec * 1000 + ts.tv_nsec / 1000000; -} - -static void nap_ms(long ms) { - struct timespec ts = {ms / 1000, (ms % 1000) * 1000000}; - nanosleep(&ts, 0); -} - -static int fail(const char *what, long a, long b) { - printf("[C] csock FAIL: %s (%ld, %ld)\n", what, a, b); - fflush(stdout); - return 1; -} - -static void ok(const char *part, const char *detail, long n) { - parts++; - printf("[C] csock %s ok: %s %ld\n", part, detail, n); - fflush(stdout); -} - -static struct sockaddr_in loopback(int port) { - struct sockaddr_in sa; - memset(&sa, 0, sizeof sa); - sa.sin_family = AF_INET; - sa.sin_port = htons(port); - sa.sin_addr.s_addr = htonl(INADDR_LOOPBACK); - return sa; -} - -// A listener on 127.0.0.1 at a port the kernel picks; its port through *port. -static int listener(int *port, int flags) { - int s = socket(AF_INET, SOCK_STREAM | flags, 0); - if (s < 0) { - return -errno; - } - int one = 1; - setsockopt(s, SOL_SOCKET, SO_REUSEADDR, &one, sizeof one); - struct sockaddr_in sa = loopback(0); - socklen_t len = sizeof sa; - if (bind(s, (void *)&sa, sizeof sa) || getsockname(s, (void *)&sa, &len) || listen(s, 8)) { - int e = errno; - close(s); - return -e; - } - *port = ntohs(sa.sin_port); - return s; -} - -static int dial(int port) { - int c = socket(AF_INET, SOCK_STREAM, 0); - struct sockaddr_in sa = loopback(port); - if (c < 0 || connect(c, (void *)&sa, sizeof sa)) { - int e = errno; - if (c >= 0) { - close(c); - } - return -e; - } - return c; -} - -// A connected pair over 127.0.0.1: *a is the client, *b the accepted end. -static int tcp_pair(int *a, int *b) { - int port, l = listener(&port, 0); - if (l < 0) { - return l; - } - *a = dial(port); - *b = *a < 0 ? -1 : accept(l, 0, 0); - close(l); - return *a < 0 ? *a : *b < 0 ? -errno : 0; -} - -#include "csock_parts.h" +#include "csock_1.h" +#include "csock_2.h" +#include "csock_3.h" +#include "csock_4.h" +#include "csock_5.h" +#include "csock_6.h" +#include "csock_7.h" +#include "csock_8.h" +#include "csock_9.h" +#include "csock_10.h" +#include "csock_11.h" int main(void) { signal(SIGPIPE, SIG_IGN); int (*const part[])(void) = { - pair_stream, pair_dgram, listen_accept, nb_connect, refused, half_close, - epoll_listener, eof, empty_recv, peek, epipe, blocking_accept, - blocking_recv, options, fork_share, backlog, quiet_options, reuseport, + pair_stream, pair_dgram, listen_accept, nb_connect, refused, half_close, + epoll_listener, eof, empty_recv, peek, epipe, blocking_accept, + blocking_recv, options, fork_share, backlog, quiet_options, reuseport, }; const int count = sizeof part / sizeof part[0]; long t0 = now_ms(); diff --git a/userland/linux_guests/c/csock_1.h b/userland/linux_guests/c/csock_1.h new file mode 100644 index 000000000..807f0aea0 --- /dev/null +++ b/userland/linux_guests/c/csock_1.h @@ -0,0 +1,67 @@ +/* csock, part 1 of 11: included once, by csock.c. */ + +static int parts; + +static long now_ms(void) { + struct timespec ts; + clock_gettime(CLOCK_MONOTONIC, &ts); + return ts.tv_sec * 1000 + ts.tv_nsec / 1000000; +} + +static void nap_ms(long ms) { + struct timespec ts = {ms / 1000, (ms % 1000) * 1000000}; + nanosleep(&ts, 0); +} + +static int fail(const char *what, long a, long b) { + printf("[C] csock FAIL: %s (%ld, %ld)\n", what, a, b); + fflush(stdout); + return 1; +} + +static void ok(const char *part, const char *detail, long n) { + parts++; + printf("[C] csock %s ok: %s %ld\n", part, detail, n); + fflush(stdout); +} + +static struct sockaddr_in loopback(int port) { + struct sockaddr_in sa; + memset(&sa, 0, sizeof sa); + sa.sin_family = AF_INET; + sa.sin_port = htons(port); + sa.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + return sa; +} + +/* A listener on 127.0.0.1 at a port the kernel picks; its port through *port. */ +static int listener(int *port, int flags) { + int s = socket(AF_INET, SOCK_STREAM | flags, 0); + if (s < 0) { + return -errno; + } + int one = 1; + setsockopt(s, SOL_SOCKET, SO_REUSEADDR, &one, sizeof one); + struct sockaddr_in sa = loopback(0); + socklen_t len = sizeof sa; + if (bind(s, (void *)&sa, sizeof sa) || getsockname(s, (void *)&sa, &len) || listen(s, 8)) { + int e = errno; + close(s); + return -e; + } + *port = ntohs(sa.sin_port); + return s; +} + +static int dial(int port) { + int c = socket(AF_INET, SOCK_STREAM, 0); + struct sockaddr_in sa = loopback(port); + if (c < 0 || connect(c, (void *)&sa, sizeof sa)) { + int e = errno; + if (c >= 0) { + close(c); + } + return -e; + } + return c; +} diff --git a/userland/linux_guests/c/csock_10.h b/userland/linux_guests/c/csock_10.h new file mode 100644 index 000000000..b459408e0 --- /dev/null +++ b/userland/linux_guests/c/csock_10.h @@ -0,0 +1,33 @@ +/* csock, part 10 of 11: included once, by csock.c. */ + +/* + * The options Linux keeps that a loopback connection cannot tell apart, + * read back as Linux reads them, and the ones a socket's kind refuses. + */ +static int quiet_options(void) { + int t = socket(AF_INET, SOCK_STREAM, 0), u = socket(AF_INET, SOCK_DGRAM, 0); + int x = socket(AF_UNIX, SOCK_STREAM, 0); + int tos = 0x13, ttl = 32, zero = 0, ut = 5000, prio = 6; + setsockopt(t, IPPROTO_IP, IP_TOS, &tos, sizeof tos); + setsockopt(t, IPPROTO_IP, IP_TTL, &ttl, sizeof ttl); + int bad_ttl = setsockopt(t, IPPROTO_IP, IP_TTL, &zero, sizeof zero) ? errno : 0; + setsockopt(t, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut); + setsockopt(t, SOL_SOCKET, SO_PRIORITY, &prio, sizeof prio); + int got_tos = get_int(t, IPPROTO_IP, IP_TOS), got_ttl = get_int(t, IPPROTO_IP, IP_TTL); + int got_ut = get_int(t, IPPROTO_TCP, TCP_USER_TIMEOUT); + int got_qa = get_int(t, IPPROTO_TCP, TCP_QUICKACK), got_prio = get_int(t, SOL_SOCKET, SO_PRIORITY); + int udp_tcp = setsockopt(u, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut) ? errno : 0; + int unix_tcp = setsockopt(x, IPPROTO_TCP, TCP_NODELAY, &prio, sizeof prio) ? errno : 0; + close(t); + close(u); + close(x); + /* A TCP socket drops the two ECN bits of the type of service. */ + if (got_tos != 0x10 || got_ttl != 32 || bad_ttl != EINVAL || got_ut != 5000 || got_qa != 1 || + got_prio != 6 || udp_tcp != ENOPROTOOPT || unix_tcp != EOPNOTSUPP) { + printf("[C] csock quiet_options got %d %d %d %d %d %d %d %d\n", got_tos, got_ttl, bad_ttl, + got_ut, got_qa, got_prio, udp_tcp, unix_tcp); + return fail("quiet_options: kept and read back as Linux does", got_tos, got_ttl); + } + ok("quiet_options", "6 kept and read back; refusals", unix_tcp); + return 0; +} diff --git a/userland/linux_guests/c/csock_11.h b/userland/linux_guests/c/csock_11.h new file mode 100644 index 000000000..4bf6fb89c --- /dev/null +++ b/userland/linux_guests/c/csock_11.h @@ -0,0 +1,52 @@ +/* csock, part 11 of 11: included once, by csock.c. */ + +/* + * Listeners that set SO_REUSEPORT share a port and between them take every + * connection; a socket without it cannot bind there, and when one listener + * closes the other takes what comes next. + */ +static int reuseport(void) { + struct sockaddr_in sa = loopback(0); + socklen_t len = sizeof sa; + int one = 1, l[2], taken[2] = {0, 0}; + for (int i = 0; i < 2; i++) { + l[i] = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + setsockopt(l[i], SOL_SOCKET, SO_REUSEPORT, &one, sizeof one); + if (bind(l[i], (void *)&sa, sizeof sa) || listen(l[i], 16)) { + return fail("reuseport: two listeners on one port", i, errno); + } + getsockname(l[i], (void *)&sa, &len); + } + int plain = socket(AF_INET, SOCK_STREAM, 0); + int refused = bind(plain, (void *)&sa, sizeof sa) ? errno : 0; + close(plain); + int c[16]; + for (int i = 0; i < 16; i++) { + c[i] = dial(ntohs(sa.sin_port)); + } + for (int i = 0; i < 2; i++) { + int a; + while ((a = accept(l[i], 0, 0)) >= 0) { + taken[i]++; + close(a); + } + } + close(l[0]); + int late = dial(ntohs(sa.sin_port)); + struct pollfd p = {l[1], POLLIN, 0}; + int ready = poll(&p, 1, 3000); + int last = accept(l[1], 0, 0); + for (int i = 0; i < 16; i++) { + close(c[i]); + } + close(late); + close(last); + close(l[1]); + if (refused != EADDRINUSE || taken[0] + taken[1] != 16 || !taken[0] || !taken[1] || + ready != 1 || last < 0) { + printf("[C] csock reuseport got %d %d+%d %d %d\n", refused, taken[0], taken[1], ready, last); + return fail("reuseport: shared, spread, and the survivor takes the rest", taken[0], taken[1]); + } + ok("reuseport", "16 connections spread over 2 listeners; without it errno", refused); + return 0; +} diff --git a/userland/linux_guests/c/csock_2.h b/userland/linux_guests/c/csock_2.h new file mode 100644 index 000000000..97feafea4 --- /dev/null +++ b/userland/linux_guests/c/csock_2.h @@ -0,0 +1,57 @@ +/* csock, part 2 of 11: included once, by csock.c. */ + +/* A connected pair over 127.0.0.1: *a is the client, *b the accepted end. */ +static int tcp_pair(int *a, int *b) { + int port, l = listener(&port, 0); + if (l < 0) { + return l; + } + *a = dial(port); + *b = *a < 0 ? -1 : accept(l, 0, 0); + close(l); + return *a < 0 ? *a : *b < 0 ? -errno : 0; +} + +/* The parts of csock, one Linux behaviour each. Included once, by csock.c. */ + +static int pair_stream(void) { + int sv[2]; + char buf[8] = {0}; + if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv)) { + return fail("pair_stream: socketpair", -1, errno); + } + if (write(sv[0], "ping", 4) != 4 || read(sv[1], buf, 8) != 4 || memcmp(buf, "ping", 4)) { + return fail("pair_stream: a to b", 0, errno); + } + if (write(sv[1], "pong!", 5) != 5 || read(sv[0], buf, 8) != 5 || memcmp(buf, "pong!", 5)) { + return fail("pair_stream: b to a", 0, errno); + } + close(sv[0]); + long got = read(sv[1], buf, 8); + close(sv[1]); + if (got != 0) { + return fail("pair_stream: end of file after close", got, errno); + } + ok("pair_stream", "bytes each way, then eof", 9); + return 0; +} + +static int pair_dgram(void) { + int sv[2]; + char buf[16]; + if (socketpair(AF_UNIX, SOCK_DGRAM, 0, sv)) { + return fail("pair_dgram: socketpair", -1, errno); + } + if (write(sv[0], "one", 3) != 3 || write(sv[0], "second", 6) != 6) { + return fail("pair_dgram: write", -1, errno); + } + long a = read(sv[1], buf, sizeof buf); + long b = read(sv[1], buf, sizeof buf); + close(sv[0]); + close(sv[1]); + if (a != 3 || b != 6) { + return fail("pair_dgram: boundaries kept", a, b); + } + ok("pair_dgram", "two datagrams, sizes 3 and", b); + return 0; +} diff --git a/userland/linux_guests/c/csock_parts.h b/userland/linux_guests/c/csock_3.h similarity index 55% rename from userland/linux_guests/c/csock_parts.h rename to userland/linux_guests/c/csock_3.h index b624ec049..0d2beb006 100644 --- a/userland/linux_guests/c/csock_parts.h +++ b/userland/linux_guests/c/csock_3.h @@ -1,45 +1,4 @@ -// The parts of csock, one Linux behaviour each. Included once, by csock.c. - -static int pair_stream(void) { - int sv[2]; - char buf[8] = {0}; - if (socketpair(AF_UNIX, SOCK_STREAM, 0, sv)) { - return fail("pair_stream: socketpair", -1, errno); - } - if (write(sv[0], "ping", 4) != 4 || read(sv[1], buf, 8) != 4 || memcmp(buf, "ping", 4)) { - return fail("pair_stream: a to b", 0, errno); - } - if (write(sv[1], "pong!", 5) != 5 || read(sv[0], buf, 8) != 5 || memcmp(buf, "pong!", 5)) { - return fail("pair_stream: b to a", 0, errno); - } - close(sv[0]); - long got = read(sv[1], buf, 8); - close(sv[1]); - if (got != 0) { - return fail("pair_stream: end of file after close", got, errno); - } - ok("pair_stream", "bytes each way, then eof", 9); - return 0; -} - -static int pair_dgram(void) { - int sv[2]; - char buf[16]; - if (socketpair(AF_UNIX, SOCK_DGRAM, 0, sv)) { - return fail("pair_dgram: socketpair", -1, errno); - } - write(sv[0], "one", 3); - write(sv[0], "second", 6); - long a = read(sv[1], buf, sizeof buf); - long b = read(sv[1], buf, sizeof buf); - close(sv[0]); - close(sv[1]); - if (a != 3 || b != 6) { - return fail("pair_dgram: boundaries kept", a, b); - } - ok("pair_dgram", "two datagrams, sizes 3 and", b); - return 0; -} +/* csock, part 3 of 11: included once, by csock.c. */ static int listen_accept(void) { int port, l = listener(&port, SOCK_NONBLOCK); @@ -86,5 +45,3 @@ static int listen_accept(void) { ok("listen_accept", "accept4 NONBLOCK|CLOEXEC on port", port > 0); return 0; } - -#include "csock_parts2.h" diff --git a/userland/linux_guests/c/csock_4.h b/userland/linux_guests/c/csock_4.h new file mode 100644 index 000000000..c1b1ec64b --- /dev/null +++ b/userland/linux_guests/c/csock_4.h @@ -0,0 +1,52 @@ +/* csock, part 4 of 11: included once, by csock.c. */ + +/* csock: connecting, and closing one way or both. */ + +static int nb_connect(void) { + int port, l = listener(&port, 0); + int c = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + struct sockaddr_in sa = loopback(port); + int rc = connect(c, (void *)&sa, sizeof sa); + int e = errno; + if (rc != -1 || e != EINPROGRESS) { + return fail("nb_connect: EINPROGRESS", rc, e); + } + struct pollfd p = {c, POLLOUT, 0}; + if (poll(&p, 1, 5000) != 1 || !(p.revents & POLLOUT)) { + return fail("nb_connect: POLLOUT", p.revents, 0); + } + int err = -1; + socklen_t len = sizeof err; + if (getsockopt(c, SOL_SOCKET, SO_ERROR, &err, &len) || err != 0) { + return fail("nb_connect: SO_ERROR", err, errno); + } + close(c); + close(l); + ok("nb_connect", "EINPROGRESS, POLLOUT, SO_ERROR", err); + return 0; +} + +/* A port with no listener: a listener is opened and closed to find one. */ +static int refused(void) { + int port, l = listener(&port, 0); + close(l); + int c = dial(port); + if (c != -ECONNREFUSED) { + return fail("refused: blocking connect", c, -ECONNREFUSED); + } + int n = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + struct sockaddr_in sa = loopback(port); + int rc = connect(n, (void *)&sa, sizeof sa); + int e = errno; + struct pollfd p = {n, POLLOUT, 0}; + poll(&p, 1, 5000); + int err = 0; + socklen_t len = sizeof err; + getsockopt(n, SOL_SOCKET, SO_ERROR, &err, &len); + close(n); + if (rc != -1 || e != EINPROGRESS || err != ECONNREFUSED || !(p.revents & POLLERR)) { + return fail("refused: non-blocking connect then SO_ERROR", e, err); + } + ok("refused", "ECONNREFUSED both ways, errno", ECONNREFUSED); + return 0; +} diff --git a/userland/linux_guests/c/csock_parts2.h b/userland/linux_guests/c/csock_5.h similarity index 52% rename from userland/linux_guests/c/csock_parts2.h rename to userland/linux_guests/c/csock_5.h index de90a45c0..0fd77ed48 100644 --- a/userland/linux_guests/c/csock_parts2.h +++ b/userland/linux_guests/c/csock_5.h @@ -1,53 +1,4 @@ -// csock: connecting, and closing one way or both. - -static int nb_connect(void) { - int port, l = listener(&port, 0); - int c = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); - struct sockaddr_in sa = loopback(port); - int rc = connect(c, (void *)&sa, sizeof sa); - int e = errno; - if (rc != -1 || e != EINPROGRESS) { - return fail("nb_connect: EINPROGRESS", rc, e); - } - struct pollfd p = {c, POLLOUT, 0}; - if (poll(&p, 1, 5000) != 1 || !(p.revents & POLLOUT)) { - return fail("nb_connect: POLLOUT", p.revents, 0); - } - int err = -1; - socklen_t len = sizeof err; - if (getsockopt(c, SOL_SOCKET, SO_ERROR, &err, &len) || err != 0) { - return fail("nb_connect: SO_ERROR", err, errno); - } - close(c); - close(l); - ok("nb_connect", "EINPROGRESS, POLLOUT, SO_ERROR", err); - return 0; -} - -// A port with no listener: a listener is opened and closed to find one. -static int refused(void) { - int port, l = listener(&port, 0); - close(l); - int c = dial(port); - if (c != -ECONNREFUSED) { - return fail("refused: blocking connect", c, -ECONNREFUSED); - } - int n = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); - struct sockaddr_in sa = loopback(port); - int rc = connect(n, (void *)&sa, sizeof sa); - int e = errno; - struct pollfd p = {n, POLLOUT, 0}; - poll(&p, 1, 5000); - int err = 0; - socklen_t len = sizeof err; - getsockopt(n, SOL_SOCKET, SO_ERROR, &err, &len); - close(n); - if (rc != -1 || e != EINPROGRESS || err != ECONNREFUSED || !(p.revents & POLLERR)) { - return fail("refused: non-blocking connect then SO_ERROR", e, err); - } - ok("refused", "ECONNREFUSED both ways, errno", ECONNREFUSED); - return 0; -} +/* csock, part 5 of 11: included once, by csock.c. */ static int half_close(void) { int a, b; @@ -104,7 +55,9 @@ static int eof(void) { if (tcp_pair(&a, &b)) { return fail("eof: pair", 0, errno); } - write(a, "last", 4); + if (write(a, "last", 4) != 4) { + return fail("eof: write", -1, errno); + } close(a); long first = read(b, buf, 8), then = read(b, buf, 8); close(b); @@ -114,5 +67,3 @@ static int eof(void) { ok("eof", "4 bytes then", then); return 0; } - -#include "csock_parts3.h" diff --git a/userland/linux_guests/c/csock_parts3.h b/userland/linux_guests/c/csock_6.h similarity index 53% rename from userland/linux_guests/c/csock_parts3.h rename to userland/linux_guests/c/csock_6.h index bcbc5dbff..fc011214a 100644 --- a/userland/linux_guests/c/csock_parts3.h +++ b/userland/linux_guests/c/csock_6.h @@ -1,4 +1,6 @@ -// csock: receiving without waiting, waiting, options, and fork. +/* csock, part 6 of 11: included once, by csock.c. */ + +/* csock: receiving without waiting, waiting, options, and fork. */ static int empty_recv(void) { int a, b; @@ -25,7 +27,9 @@ static int peek(void) { if (tcp_pair(&a, &b)) { return fail("peek: pair", 0, errno); } - write(a, "abc", 3); + if (write(a, "abc", 3) != 3) { + return fail("peek: write", -1, errno); + } long p = recv(b, buf, 8, MSG_PEEK); long r = recv(b, buf, 8, 0); close(a); @@ -55,59 +59,9 @@ static int epipe(void) { } static int late_port; + static void *late_dial(void *arg) { (void)arg; nap_ms(100); return (void *)(long)dial(late_port); } - -static int blocking_accept(void) { - int l = listener(&late_port, 0); - pthread_t t; - long t0 = now_ms(); - pthread_create(&t, 0, late_dial, 0); - int s = accept(l, 0, 0); - long waited = now_ms() - t0; - void *c; - pthread_join(t, &c); - close((int)(long)c); - close(s); - close(l); - if (s < 0 || waited < 80) { - return fail("blocking_accept: waited for the connect", s, waited); - } - ok("blocking_accept", "waited ms", waited >= 80); - return 0; -} - -static int late_fd; -static void *late_write(void *arg) { - (void)arg; - nap_ms(100); - write(late_fd, "late", 4); - return 0; -} - -static int blocking_recv(void) { - int a, b; - char buf[8]; - if (tcp_pair(&a, &b)) { - return fail("blocking_recv: pair", 0, errno); - } - late_fd = a; - pthread_t t; - long t0 = now_ms(); - pthread_create(&t, 0, late_write, 0); - long got = recv(b, buf, 8, 0); - long waited = now_ms() - t0; - pthread_join(t, 0); - close(a); - close(b); - if (got != 4 || waited < 80) { - return fail("blocking_recv: waited for the bytes", got, waited); - } - ok("blocking_recv", "4 bytes after waiting", waited >= 80); - return 0; -} - -#include "csock_parts4.h" diff --git a/userland/linux_guests/c/csock_7.h b/userland/linux_guests/c/csock_7.h new file mode 100644 index 000000000..872f2ee6b --- /dev/null +++ b/userland/linux_guests/c/csock_7.h @@ -0,0 +1,61 @@ +/* csock, part 7 of 11: included once, by csock.c. */ + +static int blocking_accept(void) { + int l = listener(&late_port, 0); + pthread_t t; + long t0 = now_ms(); + pthread_create(&t, 0, late_dial, 0); + int s = accept(l, 0, 0); + long waited = now_ms() - t0; + void *c; + pthread_join(t, &c); + close((int)(long)c); + close(s); + close(l); + if (s < 0 || waited < 80) { + return fail("blocking_accept: waited for the connect", s, waited); + } + ok("blocking_accept", "waited ms", waited >= 80); + return 0; +} + +static int late_fd; + +static void *late_write(void *arg) { + (void)arg; + nap_ms(100); + if (write(late_fd, "late", 4) != 4) { + printf("[C] csock late_write: errno %d\n", errno); + } + return 0; +} + +static int blocking_recv(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("blocking_recv: pair", 0, errno); + } + late_fd = a; + pthread_t t; + long t0 = now_ms(); + pthread_create(&t, 0, late_write, 0); + long got = recv(b, buf, 8, 0); + long waited = now_ms() - t0; + pthread_join(t, 0); + close(a); + close(b); + if (got != 4 || waited < 80) { + return fail("blocking_recv: waited for the bytes", got, waited); + } + ok("blocking_recv", "4 bytes after waiting", waited >= 80); + return 0; +} + +/* csock: the options a server sets, and a socket a forked child shares. */ + +static int get_int(int s, int level, int opt) { + int v = -1; + socklen_t len = sizeof v; + return getsockopt(s, level, opt, &v, &len) ? -errno : v; +} diff --git a/userland/linux_guests/c/csock_8.h b/userland/linux_guests/c/csock_8.h new file mode 100644 index 000000000..3a3f7f366 --- /dev/null +++ b/userland/linux_guests/c/csock_8.h @@ -0,0 +1,54 @@ +/* csock, part 8 of 11: included once, by csock.c. */ + +static int options(void) { + int port, l = listener(&port, 0); + int one = 1; + if (get_int(l, SOL_SOCKET, SO_TYPE) != SOCK_STREAM || + get_int(l, SOL_SOCKET, SO_DOMAIN) != AF_INET || + get_int(l, SOL_SOCKET, SO_ACCEPTCONN) != 1 || + get_int(l, SOL_SOCKET, SO_REUSEADDR) != 1) { + return fail("options: type, domain, acceptconn, reuseaddr", + get_int(l, SOL_SOCKET, SO_TYPE), get_int(l, SOL_SOCKET, SO_ACCEPTCONN)); + } + int c = dial(port); + int sets[][2] = { + {SOL_SOCKET, SO_KEEPALIVE}, {IPPROTO_TCP, TCP_NODELAY}, {SOL_SOCKET, SO_REUSEPORT}, + {SOL_SOCKET, SO_BROADCAST}, + }; + for (unsigned i = 0; i < sizeof sets / sizeof sets[0]; i++) { + if (setsockopt(c, sets[i][0], sets[i][1], &one, sizeof one) || + get_int(c, sets[i][0], sets[i][1]) != 1) { + return fail("options: set then read back", sets[i][1], errno); + } + } + int idle = 15; + if (setsockopt(c, IPPROTO_TCP, TCP_KEEPIDLE, &idle, sizeof idle) || + get_int(c, IPPROTO_TCP, TCP_KEEPIDLE) != 15) { + return fail("options: TCP_KEEPIDLE", get_int(c, IPPROTO_TCP, TCP_KEEPIDLE), errno); + } + int buf = 65536; + if (setsockopt(c, SOL_SOCKET, SO_RCVBUF, &buf, sizeof buf) || + get_int(c, SOL_SOCKET, SO_RCVBUF) != 2 * buf) { + return fail("options: SO_RCVBUF doubled", get_int(c, SOL_SOCKET, SO_RCVBUF), 2 * buf); + } + struct linger lg = {1, 5}, back = {0, 0}; + socklen_t len = sizeof back; + setsockopt(c, SOL_SOCKET, SO_LINGER, &lg, sizeof lg); + getsockopt(c, SOL_SOCKET, SO_LINGER, &back, &len); + struct timeval tv = {2, 500000}, tb = {0, 0}; + len = sizeof tb; + setsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof tv); + getsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tb, &len); + int bogus = setsockopt(c, SOL_SOCKET, 9999, &one, sizeof one); + int e = errno; + close(c); + close(l); + if (back.l_onoff != 1 || back.l_linger != 5 || tb.tv_sec != 2 || tb.tv_usec != 500000) { + return fail("options: SO_LINGER and SO_RCVTIMEO read back", back.l_linger, tb.tv_usec); + } + if (bogus != -1 || e != ENOPROTOOPT) { + return fail("options: an unknown option is ENOPROTOOPT", bogus, e); + } + ok("options", "11 options, unknown errno", e); + return 0; +} diff --git a/userland/linux_guests/c/csock_9.h b/userland/linux_guests/c/csock_9.h new file mode 100644 index 000000000..8286cdd30 --- /dev/null +++ b/userland/linux_guests/c/csock_9.h @@ -0,0 +1,72 @@ +/* csock, part 9 of 11: included once, by csock.c. */ + +/* + * The child holds the accepted end after the parent closes its own: the + * client still reads the child's bytes, then end of file when the child exits. + */ +static int fork_share(void) { + int a, b; + char buf[8]; + if (tcp_pair(&a, &b)) { + return fail("fork_share: pair", 0, errno); + } + pid_t kid = fork(); + if (kid == 0) { + close(a); + nap_ms(50); + _exit(write(b, "kid", 3) == 3 ? 0 : 1); + } + close(b); + long got = read(a, buf, 8); + int status = 0; + waitpid(kid, &status, 0); + long then = read(a, buf, 8); + close(a); + if (got != 3 || then != 0 || status != 0) { + return fail("fork_share: child's bytes, then eof", got, then); + } + ok("fork_share", "3 bytes from the child, then", then); + return 0; +} + +/* + * A listener with a backlog of 0 queues one connect; the next non-blocking + * one answers EINPROGRESS, is not writable, and a second connect on it is + * EALREADY, until an accept makes room and it completes. + */ +static int backlog(void) { + int s = socket(AF_INET, SOCK_STREAM, 0); + struct sockaddr_in sa = loopback(0); + socklen_t len = sizeof sa; + bind(s, (void *)&sa, sizeof sa); + getsockname(s, (void *)&sa, &len); + listen(s, 0); + int c0 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + int c1 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); + connect(c0, (void *)&sa, sizeof sa); + int r1 = connect(c1, (void *)&sa, sizeof sa); + int e1 = errno; + struct pollfd p = {c1, POLLOUT, 0}; + int early = poll(&p, 1, 100); + int again = connect(c1, (void *)&sa, sizeof sa) ? errno : 0; + int a = accept(s, 0, 0); + p.revents = 0; + int later = poll(&p, 1, 3000); + int err = -1; + socklen_t el = sizeof err; + getsockopt(c1, SOL_SOCKET, SO_ERROR, &err, &el); + struct pollfd q = {s, POLLIN, 0}; + int b = poll(&q, 1, 3000) == 1 ? accept(s, 0, 0) : -1; + close(a); + close(b); + close(c0); + close(c1); + close(s); + if (r1 != -1 || e1 != EINPROGRESS || early != 0 || again != EALREADY || later != 1 || + err != 0 || b < 0) { + printf("[C] csock backlog got %d %d %d %d %d %d %d\n", r1, e1, early, again, later, err, b); + return fail("backlog: EINPROGRESS, waits, EALREADY, then connected", e1, again); + } + ok("backlog", "a full queue's connect completed after accept; EALREADY", again); + return 0; +} diff --git a/userland/linux_guests/c/csock_parts4.h b/userland/linux_guests/c/csock_parts4.h deleted file mode 100644 index 1953a6bf1..000000000 --- a/userland/linux_guests/c/csock_parts4.h +++ /dev/null @@ -1,206 +0,0 @@ -// csock: the options a server sets, and a socket a forked child shares. - -static int get_int(int s, int level, int opt) { - int v = -1; - socklen_t len = sizeof v; - return getsockopt(s, level, opt, &v, &len) ? -errno : v; -} - -static int options(void) { - int port, l = listener(&port, 0); - int one = 1; - if (get_int(l, SOL_SOCKET, SO_TYPE) != SOCK_STREAM || - get_int(l, SOL_SOCKET, SO_DOMAIN) != AF_INET || - get_int(l, SOL_SOCKET, SO_ACCEPTCONN) != 1 || - get_int(l, SOL_SOCKET, SO_REUSEADDR) != 1) { - return fail("options: type, domain, acceptconn, reuseaddr", - get_int(l, SOL_SOCKET, SO_TYPE), get_int(l, SOL_SOCKET, SO_ACCEPTCONN)); - } - int c = dial(port); - int sets[][2] = { - {SOL_SOCKET, SO_KEEPALIVE}, {IPPROTO_TCP, TCP_NODELAY}, {SOL_SOCKET, SO_REUSEPORT}, - {SOL_SOCKET, SO_BROADCAST}, - }; - for (unsigned i = 0; i < sizeof sets / sizeof sets[0]; i++) { - if (setsockopt(c, sets[i][0], sets[i][1], &one, sizeof one) || - get_int(c, sets[i][0], sets[i][1]) != 1) { - return fail("options: set then read back", sets[i][1], errno); - } - } - int idle = 15; - if (setsockopt(c, IPPROTO_TCP, TCP_KEEPIDLE, &idle, sizeof idle) || - get_int(c, IPPROTO_TCP, TCP_KEEPIDLE) != 15) { - return fail("options: TCP_KEEPIDLE", get_int(c, IPPROTO_TCP, TCP_KEEPIDLE), errno); - } - int buf = 65536; - if (setsockopt(c, SOL_SOCKET, SO_RCVBUF, &buf, sizeof buf) || - get_int(c, SOL_SOCKET, SO_RCVBUF) != 2 * buf) { - return fail("options: SO_RCVBUF doubled", get_int(c, SOL_SOCKET, SO_RCVBUF), 2 * buf); - } - struct linger lg = {1, 5}, back = {0, 0}; - socklen_t len = sizeof back; - setsockopt(c, SOL_SOCKET, SO_LINGER, &lg, sizeof lg); - getsockopt(c, SOL_SOCKET, SO_LINGER, &back, &len); - struct timeval tv = {2, 500000}, tb = {0, 0}; - len = sizeof tb; - setsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof tv); - getsockopt(c, SOL_SOCKET, SO_RCVTIMEO, &tb, &len); - int bogus = setsockopt(c, SOL_SOCKET, 9999, &one, sizeof one); - int e = errno; - close(c); - close(l); - if (back.l_onoff != 1 || back.l_linger != 5 || tb.tv_sec != 2 || tb.tv_usec != 500000) { - return fail("options: SO_LINGER and SO_RCVTIMEO read back", back.l_linger, tb.tv_usec); - } - if (bogus != -1 || e != ENOPROTOOPT) { - return fail("options: an unknown option is ENOPROTOOPT", bogus, e); - } - ok("options", "11 options, unknown errno", e); - return 0; -} - -// The child holds the accepted end after the parent closes its own: the -// client still reads the child's bytes, then end of file when the child exits. -static int fork_share(void) { - int a, b; - char buf[8]; - if (tcp_pair(&a, &b)) { - return fail("fork_share: pair", 0, errno); - } - pid_t kid = fork(); - if (kid == 0) { - close(a); - nap_ms(50); - write(b, "kid", 3); - _exit(0); - } - close(b); - long got = read(a, buf, 8); - int status = 0; - waitpid(kid, &status, 0); - long then = read(a, buf, 8); - close(a); - if (got != 3 || then != 0 || status != 0) { - return fail("fork_share: child's bytes, then eof", got, then); - } - ok("fork_share", "3 bytes from the child, then", then); - return 0; -} - -// A listener with a backlog of 0 queues one connect; the next non-blocking -// one answers EINPROGRESS, is not writable, and a second connect on it is -// EALREADY, until an accept makes room and it completes. -static int backlog(void) { - int s = socket(AF_INET, SOCK_STREAM, 0); - struct sockaddr_in sa = loopback(0); - socklen_t len = sizeof sa; - bind(s, (void *)&sa, sizeof sa); - getsockname(s, (void *)&sa, &len); - listen(s, 0); - int c0 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); - int c1 = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); - connect(c0, (void *)&sa, sizeof sa); - int r1 = connect(c1, (void *)&sa, sizeof sa); - int e1 = errno; - struct pollfd p = {c1, POLLOUT, 0}; - int early = poll(&p, 1, 100); - int again = connect(c1, (void *)&sa, sizeof sa) ? errno : 0; - int a = accept(s, 0, 0); - p.revents = 0; - int later = poll(&p, 1, 3000); - int err = -1; - socklen_t el = sizeof err; - getsockopt(c1, SOL_SOCKET, SO_ERROR, &err, &el); - int b = accept(s, 0, 0); - close(a); - close(b); - close(c0); - close(c1); - close(s); - if (r1 != -1 || e1 != EINPROGRESS || early != 0 || again != EALREADY || later != 1 || - err != 0 || b < 0) { - printf("[C] csock backlog got %d %d %d %d %d %d %d\n", r1, e1, early, again, later, err, b); - return fail("backlog: EINPROGRESS, waits, EALREADY, then connected", e1, again); - } - ok("backlog", "a full queue's connect completed after accept; EALREADY", again); - return 0; -} - -// The options Linux keeps that a loopback connection cannot tell apart, -// read back as Linux reads them, and the ones a socket's kind refuses. -static int quiet_options(void) { - int t = socket(AF_INET, SOCK_STREAM, 0), u = socket(AF_INET, SOCK_DGRAM, 0); - int x = socket(AF_UNIX, SOCK_STREAM, 0); - int tos = 0x13, ttl = 32, zero = 0, ut = 5000, prio = 6; - setsockopt(t, IPPROTO_IP, IP_TOS, &tos, sizeof tos); - setsockopt(t, IPPROTO_IP, IP_TTL, &ttl, sizeof ttl); - int bad_ttl = setsockopt(t, IPPROTO_IP, IP_TTL, &zero, sizeof zero) ? errno : 0; - setsockopt(t, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut); - setsockopt(t, SOL_SOCKET, SO_PRIORITY, &prio, sizeof prio); - int got_tos = get_int(t, IPPROTO_IP, IP_TOS), got_ttl = get_int(t, IPPROTO_IP, IP_TTL); - int got_ut = get_int(t, IPPROTO_TCP, TCP_USER_TIMEOUT); - int got_qa = get_int(t, IPPROTO_TCP, TCP_QUICKACK), got_prio = get_int(t, SOL_SOCKET, SO_PRIORITY); - int udp_tcp = setsockopt(u, IPPROTO_TCP, TCP_USER_TIMEOUT, &ut, sizeof ut) ? errno : 0; - int unix_tcp = setsockopt(x, IPPROTO_TCP, TCP_NODELAY, &prio, sizeof prio) ? errno : 0; - close(t); - close(u); - close(x); - // A TCP socket drops the two ECN bits of the type of service. - if (got_tos != 0x10 || got_ttl != 32 || bad_ttl != EINVAL || got_ut != 5000 || got_qa != 1 || - got_prio != 6 || udp_tcp != ENOPROTOOPT || unix_tcp != EOPNOTSUPP) { - printf("[C] csock quiet_options got %d %d %d %d %d %d %d %d\n", got_tos, got_ttl, bad_ttl, - got_ut, got_qa, got_prio, udp_tcp, unix_tcp); - return fail("quiet_options: kept and read back as Linux does", got_tos, got_ttl); - } - ok("quiet_options", "6 kept and read back; refusals", unix_tcp); - return 0; -} - -// Listeners that set SO_REUSEPORT share a port and between them take every -// connection; a socket without it cannot bind there, and when one listener -// closes the other takes what comes next. -static int reuseport(void) { - struct sockaddr_in sa = loopback(0); - socklen_t len = sizeof sa; - int one = 1, l[2], taken[2] = {0, 0}; - for (int i = 0; i < 2; i++) { - l[i] = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0); - setsockopt(l[i], SOL_SOCKET, SO_REUSEPORT, &one, sizeof one); - if (bind(l[i], (void *)&sa, sizeof sa) || listen(l[i], 16)) { - return fail("reuseport: two listeners on one port", i, errno); - } - getsockname(l[i], (void *)&sa, &len); - } - int plain = socket(AF_INET, SOCK_STREAM, 0); - int refused = bind(plain, (void *)&sa, sizeof sa) ? errno : 0; - close(plain); - int c[16]; - for (int i = 0; i < 16; i++) { - c[i] = dial(ntohs(sa.sin_port)); - } - for (int i = 0; i < 2; i++) { - int a; - while ((a = accept(l[i], 0, 0)) >= 0) { - taken[i]++; - close(a); - } - } - close(l[0]); - int late = dial(ntohs(sa.sin_port)); - struct pollfd p = {l[1], POLLIN, 0}; - int ready = poll(&p, 1, 3000); - int last = accept(l[1], 0, 0); - for (int i = 0; i < 16; i++) { - close(c[i]); - } - close(late); - close(last); - close(l[1]); - if (refused != EADDRINUSE || taken[0] + taken[1] != 16 || !taken[0] || !taken[1] || - ready != 1 || last < 0) { - printf("[C] csock reuseport got %d %d+%d %d %d\n", refused, taken[0], taken[1], ready, last); - return fail("reuseport: shared, spread, and the survivor takes the rest", taken[0], taken[1]); - } - ok("reuseport", "16 connections spread over 2 listeners; without it errno", refused); - return 0; -} diff --git a/userland/linux_guests/c/cudp.c b/userland/linux_guests/c/cudp.c index 16ecd7892..f379124f5 100644 --- a/userland/linux_guests/c/cudp.c +++ b/userland/linux_guests/c/cudp.c @@ -1,10 +1,13 @@ -// Datagrams on the loopback address, as Linux has them: an echo between an -// unbound client and a bound server, each told the other's address; a -// connected socket that sends without naming a peer and keeps only its -// peer's datagrams; a datagram cut to the buffer with MSG_TRUNC, and -// recvmsg's MSG_TRUNC flag; ECONNREFUSED on a connected socket whose peer -// port is closed; sendmmsg and recvmmsg; and EDESTADDRREQ with no peer at -// all. Each part prints as it passes and every part runs. +/* + * Datagrams on the loopback address, as Linux has them: an echo between an + * unbound client and a bound server, each told the other's address; a + * connected socket that sends without naming a peer and keeps only its + * peer's datagrams; a datagram cut to the buffer with MSG_TRUNC, and + * recvmsg's MSG_TRUNC flag; ECONNREFUSED on a connected socket whose peer + * port is closed; sendmmsg and recvmmsg; and EDESTADDRREQ with no peer at + * all. Each part prints as it passes and every part runs. + */ + #define _GNU_SOURCE #include #include @@ -14,81 +17,9 @@ #include #include -static int parts; - -static int fail(const char *what, long a, long b) { - printf("[C] cudp FAIL: %s (%ld, %ld)\n", what, a, b); - fflush(stdout); - return 1; -} - -static void ok(const char *part, const char *detail, long n) { - parts++; - printf("[C] cudp %s ok: %s %ld\n", part, detail, n); - fflush(stdout); -} - -// A datagram socket bound to 127.0.0.1 at a port the kernel picks. -static int bound(struct sockaddr_in *at) { - int s = socket(AF_INET, SOCK_DGRAM, 0); - memset(at, 0, sizeof *at); - at->sin_family = AF_INET; - at->sin_addr.s_addr = htonl(INADDR_LOOPBACK); - socklen_t len = sizeof *at; - if (s < 0 || bind(s, (void *)at, sizeof *at) || getsockname(s, (void *)at, &len)) { - return -1; - } - return s; -} - -static int echo(void) { - struct sockaddr_in srv, from, back; - int s = bound(&srv), c = socket(AF_INET, SOCK_DGRAM, 0); - char buf[32]; - socklen_t flen = sizeof from, blen = sizeof back; - if (s < 0 || sendto(c, "ping", 4, 0, (void *)&srv, sizeof srv) != 4) { - return fail("echo: send", s, errno); - } - long got = recvfrom(s, buf, sizeof buf, 0, (void *)&from, &flen); - if (got != 4 || from.sin_port == 0 || flen != sizeof from) { - return fail("echo: server heard the client and its address", got, ntohs(from.sin_port)); - } - sendto(s, "pong!", 5, 0, (void *)&from, flen); - got = recvfrom(c, buf, sizeof buf, 0, (void *)&back, &blen); - close(s); - close(c); - if (got != 5 || back.sin_port != srv.sin_port || memcmp(buf, "pong!", 5)) { - return fail("echo: client heard the server", got, ntohs(back.sin_port)); - } - ok("echo", "4 bytes there, back from the server's port", ntohs(back.sin_port) > 0); - return 0; -} - -static int connected(void) { - struct sockaddr_in a, b, x; - int sa = bound(&a), sb = bound(&b), sx = bound(&x); - char buf[8]; - connect(sb, (void *)&a, sizeof a); - if (send(sb, "to-a", 4, 0) != 4 || recv(sa, buf, 8, 0) != 4) { - return fail("connected: send without an address", 0, errno); - } - // sb keeps only a's datagrams: x's is dropped, a's arrives. - sendto(sx, "noise", 5, 0, (void *)&b, sizeof b); - sendto(sa, "real", 4, 0, (void *)&b, sizeof b); - long got = recv(sb, buf, 8, 0); - long more = recv(sb, buf, 8, MSG_DONTWAIT); - int e = errno; - close(sa); - close(sb); - close(sx); - if (got != 4 || more != -1 || e != EAGAIN) { - return fail("connected: only the peer's datagrams kept", got, more); - } - ok("connected", "a stranger's datagram dropped, peer's bytes", got); - return 0; -} - -#include "cudp_parts.h" +#include "cudp_1.h" +#include "cudp_2.h" +#include "cudp_3.h" int main(void) { int (*const part[])(void) = {echo, connected, cut, refused, mmsg, nameless}; diff --git a/userland/linux_guests/c/cudp_1.h b/userland/linux_guests/c/cudp_1.h new file mode 100644 index 000000000..02ce6c122 --- /dev/null +++ b/userland/linux_guests/c/cudp_1.h @@ -0,0 +1,51 @@ +/* cudp, part 1 of 3: included once, by cudp.c. */ + +static int parts; + +static int fail(const char *what, long a, long b) { + printf("[C] cudp FAIL: %s (%ld, %ld)\n", what, a, b); + fflush(stdout); + return 1; +} + +static void ok(const char *part, const char *detail, long n) { + parts++; + printf("[C] cudp %s ok: %s %ld\n", part, detail, n); + fflush(stdout); +} + +/* A datagram socket bound to 127.0.0.1 at a port the kernel picks. */ +static int bound(struct sockaddr_in *at) { + int s = socket(AF_INET, SOCK_DGRAM, 0); + memset(at, 0, sizeof *at); + at->sin_family = AF_INET; + at->sin_addr.s_addr = htonl(INADDR_LOOPBACK); + socklen_t len = sizeof *at; + if (s < 0 || bind(s, (void *)at, sizeof *at) || getsockname(s, (void *)at, &len)) { + return -1; + } + return s; +} + +static int echo(void) { + struct sockaddr_in srv, from, back; + int s = bound(&srv), c = socket(AF_INET, SOCK_DGRAM, 0); + char buf[32]; + socklen_t flen = sizeof from, blen = sizeof back; + if (s < 0 || sendto(c, "ping", 4, 0, (void *)&srv, sizeof srv) != 4) { + return fail("echo: send", s, errno); + } + long got = recvfrom(s, buf, sizeof buf, 0, (void *)&from, &flen); + if (got != 4 || from.sin_port == 0 || flen != sizeof from) { + return fail("echo: server heard the client and its address", got, ntohs(from.sin_port)); + } + sendto(s, "pong!", 5, 0, (void *)&from, flen); + got = recvfrom(c, buf, sizeof buf, 0, (void *)&back, &blen); + close(s); + close(c); + if (got != 5 || back.sin_port != srv.sin_port || memcmp(buf, "pong!", 5)) { + return fail("echo: client heard the server", got, ntohs(back.sin_port)); + } + ok("echo", "4 bytes there, back from the server's port", ntohs(back.sin_port) > 0); + return 0; +} diff --git a/userland/linux_guests/c/cudp_parts.h b/userland/linux_guests/c/cudp_2.h similarity index 50% rename from userland/linux_guests/c/cudp_parts.h rename to userland/linux_guests/c/cudp_2.h index ab4b1b22b..96d477852 100644 --- a/userland/linux_guests/c/cudp_parts.h +++ b/userland/linux_guests/c/cudp_2.h @@ -1,4 +1,30 @@ -// cudp: truncation, a refused port, several messages at once, and no peer. +/* cudp, part 2 of 3: included once, by cudp.c. */ + +static int connected(void) { + struct sockaddr_in a, b, x; + int sa = bound(&a), sb = bound(&b), sx = bound(&x); + char buf[8]; + connect(sb, (void *)&a, sizeof a); + if (send(sb, "to-a", 4, 0) != 4 || recv(sa, buf, 8, 0) != 4) { + return fail("connected: send without an address", 0, errno); + } + /* sb keeps only a's datagrams: x's is dropped, a's arrives. */ + sendto(sx, "noise", 5, 0, (void *)&b, sizeof b); + sendto(sa, "real", 4, 0, (void *)&b, sizeof b); + long got = recv(sb, buf, 8, 0); + long more = recv(sb, buf, 8, MSG_DONTWAIT); + int e = errno; + close(sa); + close(sb); + close(sx); + if (got != 4 || more != -1 || e != EAGAIN) { + return fail("connected: only the peer's datagrams kept", got, more); + } + ok("connected", "a stranger's datagram dropped, peer's bytes", got); + return 0; +} + +/* cudp: truncation, a refused port, several messages at once, and no peer. */ static int cut(void) { struct sockaddr_in a; @@ -23,7 +49,7 @@ static int cut(void) { return 0; } -// A port with no socket: one is bound and closed to find it. +/* A port with no socket: one is bound and closed to find it. */ static int refused(void) { struct sockaddr_in gone; close(bound(&gone)); @@ -40,43 +66,3 @@ static int refused(void) { ok("refused", "connected to a closed port, errno", e); return 0; } - -static int mmsg(void) { - struct sockaddr_in a; - int s = bound(&a), c = socket(AF_INET, SOCK_DGRAM, 0); - connect(c, (void *)&a, sizeof a); - char out[3][4] = {"one", "two", "six"}, in[3][8]; - struct iovec oi[3], ii[3]; - struct mmsghdr om[3], im[3]; - memset(om, 0, sizeof om); - memset(im, 0, sizeof im); - for (int i = 0; i < 3; i++) { - oi[i] = (struct iovec){out[i], 3}; - ii[i] = (struct iovec){in[i], 8}; - om[i].msg_hdr.msg_iov = &oi[i]; - om[i].msg_hdr.msg_iovlen = 1; - im[i].msg_hdr.msg_iov = &ii[i]; - im[i].msg_hdr.msg_iovlen = 1; - } - int sent = sendmmsg(c, om, 3, 0); - int got = recvmmsg(s, im, 3, MSG_WAITFORONE, 0); - close(s); - close(c); - if (sent != 3 || got != 3 || im[2].msg_len != 3 || memcmp(in[2], "six", 3)) { - return fail("mmsg: three out, three in", sent, got); - } - ok("mmsg", "sendmmsg and recvmmsg moved", got); - return 0; -} - -static int nameless(void) { - int c = socket(AF_INET, SOCK_DGRAM, 0); - long sent = send(c, "x", 1, 0); - int e = errno; - close(c); - if (sent != -1 || e != EDESTADDRREQ) { - return fail("nameless: no peer is EDESTADDRREQ", sent, e); - } - ok("nameless", "errno", e); - return 0; -} diff --git a/userland/linux_guests/c/cudp_3.h b/userland/linux_guests/c/cudp_3.h new file mode 100644 index 000000000..686e9ea28 --- /dev/null +++ b/userland/linux_guests/c/cudp_3.h @@ -0,0 +1,41 @@ +/* cudp, part 3 of 3: included once, by cudp.c. */ + +static int mmsg(void) { + struct sockaddr_in a; + int s = bound(&a), c = socket(AF_INET, SOCK_DGRAM, 0); + connect(c, (void *)&a, sizeof a); + char out[3][4] = {"one", "two", "six"}, in[3][8]; + struct iovec oi[3], ii[3]; + struct mmsghdr om[3], im[3]; + memset(om, 0, sizeof om); + memset(im, 0, sizeof im); + for (int i = 0; i < 3; i++) { + oi[i] = (struct iovec){out[i], 3}; + ii[i] = (struct iovec){in[i], 8}; + om[i].msg_hdr.msg_iov = &oi[i]; + om[i].msg_hdr.msg_iovlen = 1; + im[i].msg_hdr.msg_iov = &ii[i]; + im[i].msg_hdr.msg_iovlen = 1; + } + int sent = sendmmsg(c, om, 3, 0); + int got = recvmmsg(s, im, 3, MSG_WAITFORONE, 0); + close(s); + close(c); + if (sent != 3 || got != 3 || im[2].msg_len != 3 || memcmp(in[2], "six", 3)) { + return fail("mmsg: three out, three in", sent, got); + } + ok("mmsg", "sendmmsg and recvmmsg moved", got); + return 0; +} + +static int nameless(void) { + int c = socket(AF_INET, SOCK_DGRAM, 0); + long sent = send(c, "x", 1, 0); + int e = errno; + close(c); + if (sent != -1 || e != EDESTADDRREQ) { + return fail("nameless: no peer is EDESTADDRREQ", sent, e); + } + ok("nameless", "errno", e); + return 0; +} diff --git a/userland/linux_guests/c/cunix.c b/userland/linux_guests/c/cunix.c index dc6309e99..c333e06ab 100644 --- a/userland/linux_guests/c/cunix.c +++ b/userland/linux_guests/c/cunix.c @@ -1,10 +1,13 @@ -// Unix sockets with names, as Linux has them: a listener on a path, its -// name and its client's, a connect that completes at once; ENOENT for a path -// with nothing there and ECONNREFUSED for one with no listener; a path that -// stays after its socket closes until it is unlinked; abstract names and the -// sender a datagram reports; a connected datagram socket that refuses -// strangers; a name bind chooses; and a connection across fork. Each part -// prints as it passes and every part runs. +/* + * Unix sockets with names, as Linux has them: a listener on a path, its + * name and its client's, a connect that completes at once; ENOENT for a path + * with nothing there and ECONNREFUSED for one with no listener; a path that + * stays after its socket closes until it is unlinked; abstract names and the + * sender a datagram reports; a connected datagram socket that refuses + * strangers; a name bind chooses; and a connection across fork. Each part + * prints as it passes and every part runs. + */ + #define _GNU_SOURCE #include #include @@ -14,83 +17,28 @@ #include #include #include - #define PATH "/tmp/cunix.sock" #define GONE "/tmp/cunix.none" -static int parts; - -static int fail(const char *what, long a, long b) { - printf("[C] cunix FAIL: %s (%ld, %ld)\n", what, a, b); - fflush(stdout); - return 1; -} +#include "cunix_1.h" +#include "cunix_2.h" +#include "cunix_3.h" +#include "cunix_4.h" -static void ok(const char *part, const char *detail, long n) { - parts++; - printf("[C] cunix %s ok: %s %ld\n", part, detail, n); +int main(void) { + int (*const part[])(void) = {path_stream, left_behind, abstract_dgram, + connected_dgram, autobind, across_fork}; + const int count = sizeof part / sizeof part[0]; + int failed = 0; + for (int i = 0; i < count; i++) { + failed += part[i](); + } + if (failed) { + printf("[C] cunix FAIL: %d parts failed, %d passed\n", failed, parts); + fflush(stdout); + return 1; + } + printf("[C] cunix PASS: %d parts\n", parts); fflush(stdout); -} - -// A sockaddr_un for a path, or for an abstract name when `abs` is set. -static socklen_t name(struct sockaddr_un *a, const char *p, int abs) { - memset(a, 0, sizeof *a); - a->sun_family = AF_UNIX; - if (abs) { - memcpy(a->sun_path + 1, p, strlen(p)); - return offsetof(struct sockaddr_un, sun_path) + 1 + strlen(p); - } - strcpy(a->sun_path, p); - return offsetof(struct sockaddr_un, sun_path) + strlen(p) + 1; -} - -static int path_stream(void) { - struct sockaddr_un a, b; - socklen_t l = name(&a, PATH, 0), bl = sizeof b; - unlink(PATH); - int s = socket(AF_UNIX, SOCK_STREAM, 0); - if (listen(s, 1) != -1 || errno != EINVAL) { - return fail("path_stream: listen before bind is EINVAL", errno, EINVAL); - } - if (bind(s, (void *)&a, l) || listen(s, 4)) { - return fail("path_stream: bind and listen", -1, errno); - } - int twice = socket(AF_UNIX, SOCK_STREAM, 0); - if (bind(twice, (void *)&a, l) != -1 || errno != EADDRINUSE) { - return fail("path_stream: a second bind is EADDRINUSE", errno, EADDRINUSE); - } - close(twice); - int c = socket(AF_UNIX, SOCK_STREAM | SOCK_NONBLOCK, 0); - if (connect(c, (void *)&a, l)) { - return fail("path_stream: a non-blocking connect completes at once", -1, errno); - } - int t = accept(s, (void *)&b, &bl); - if (t < 0 || bl != 2) { - return fail("path_stream: accept names an unnamed client with 2 bytes", t, bl); - } - bl = sizeof b; - getsockname(s, (void *)&b, &bl); - if (bl != l || strcmp(b.sun_path, PATH)) { - return fail("path_stream: getsockname", bl, l); - } - bl = sizeof b; - getpeername(c, (void *)&b, &bl); - if (bl != l || strcmp(b.sun_path, PATH)) { - return fail("path_stream: the client's peer is the path", bl, l); - } - char buf[8]; - if (write(c, "unix", 4) != 4 || read(t, buf, 8) != 4) { - return fail("path_stream: bytes", 0, errno); - } - close(c); - long eof = read(t, buf, 8); - close(t); - close(s); - if (eof != 0) { - return fail("path_stream: end of file", eof, 0); - } - ok("path_stream", "bound, named, connected, 4 bytes, eof; name length", l); return 0; } - -#include "cunix_parts.h" diff --git a/userland/linux_guests/c/cunix_1.h b/userland/linux_guests/c/cunix_1.h new file mode 100644 index 000000000..bf03cf563 --- /dev/null +++ b/userland/linux_guests/c/cunix_1.h @@ -0,0 +1,27 @@ +/* cunix, part 1 of 4: included once, by cunix.c. */ + +static int parts; + +static int fail(const char *what, long a, long b) { + printf("[C] cunix FAIL: %s (%ld, %ld)\n", what, a, b); + fflush(stdout); + return 1; +} + +static void ok(const char *part, const char *detail, long n) { + parts++; + printf("[C] cunix %s ok: %s %ld\n", part, detail, n); + fflush(stdout); +} + +/* A sockaddr_un for a path, or for an abstract name when `abs` is set. */ +static socklen_t name(struct sockaddr_un *a, const char *p, int abs) { + memset(a, 0, sizeof *a); + a->sun_family = AF_UNIX; + if (abs) { + memcpy(a->sun_path + 1, p, strlen(p)); + return offsetof(struct sockaddr_un, sun_path) + 1 + strlen(p); + } + strcpy(a->sun_path, p); + return offsetof(struct sockaddr_un, sun_path) + strlen(p) + 1; +} diff --git a/userland/linux_guests/c/cunix_2.h b/userland/linux_guests/c/cunix_2.h new file mode 100644 index 000000000..4f6d31a9f --- /dev/null +++ b/userland/linux_guests/c/cunix_2.h @@ -0,0 +1,50 @@ +/* cunix, part 2 of 4: included once, by cunix.c. */ + +static int path_stream(void) { + struct sockaddr_un a, b; + socklen_t l = name(&a, PATH, 0), bl = sizeof b; + unlink(PATH); + int s = socket(AF_UNIX, SOCK_STREAM, 0); + if (listen(s, 1) != -1 || errno != EINVAL) { + return fail("path_stream: listen before bind is EINVAL", errno, EINVAL); + } + if (bind(s, (void *)&a, l) || listen(s, 4)) { + return fail("path_stream: bind and listen", -1, errno); + } + int twice = socket(AF_UNIX, SOCK_STREAM, 0); + if (bind(twice, (void *)&a, l) != -1 || errno != EADDRINUSE) { + return fail("path_stream: a second bind is EADDRINUSE", errno, EADDRINUSE); + } + close(twice); + int c = socket(AF_UNIX, SOCK_STREAM | SOCK_NONBLOCK, 0); + if (connect(c, (void *)&a, l)) { + return fail("path_stream: a non-blocking connect completes at once", -1, errno); + } + int t = accept(s, (void *)&b, &bl); + if (t < 0 || bl != 2) { + return fail("path_stream: accept names an unnamed client with 2 bytes", t, bl); + } + bl = sizeof b; + getsockname(s, (void *)&b, &bl); + if (bl != l || strcmp(b.sun_path, PATH)) { + return fail("path_stream: getsockname", bl, l); + } + bl = sizeof b; + getpeername(c, (void *)&b, &bl); + if (bl != l || strcmp(b.sun_path, PATH)) { + return fail("path_stream: the client's peer is the path", bl, l); + } + char buf[8]; + if (write(c, "unix", 4) != 4 || read(t, buf, 8) != 4) { + return fail("path_stream: bytes", 0, errno); + } + close(c); + long eof = read(t, buf, 8); + close(t); + close(s); + if (eof != 0) { + return fail("path_stream: end of file", eof, 0); + } + ok("path_stream", "bound, named, connected, 4 bytes, eof; name length", l); + return 0; +} diff --git a/userland/linux_guests/c/cunix_parts.h b/userland/linux_guests/c/cunix_3.h similarity index 95% rename from userland/linux_guests/c/cunix_parts.h rename to userland/linux_guests/c/cunix_3.h index a49728c64..4ea494bc5 100644 --- a/userland/linux_guests/c/cunix_parts.h +++ b/userland/linux_guests/c/cunix_3.h @@ -1,4 +1,6 @@ -// cunix: what a path leaves behind, abstract datagrams, and fork. +/* cunix, part 3 of 4: included once, by cunix.c. */ + +/* cunix: what a path leaves behind, abstract datagrams, and fork. */ static int left_behind(void) { struct sockaddr_un a, g; @@ -62,5 +64,3 @@ static int abstract_dgram(void) { ok("abstract_dgram", "names reported, no peer errno", nopeer); return 0; } - -#include "cunix_parts2.h" diff --git a/userland/linux_guests/c/cunix_parts2.h b/userland/linux_guests/c/cunix_4.h similarity index 73% rename from userland/linux_guests/c/cunix_parts2.h rename to userland/linux_guests/c/cunix_4.h index 7f7a79ed3..c68289a96 100644 --- a/userland/linux_guests/c/cunix_parts2.h +++ b/userland/linux_guests/c/cunix_4.h @@ -1,4 +1,6 @@ -// cunix: a connected datagram socket, a chosen name, and fork. +/* cunix, part 4 of 4: included once, by cunix.c. */ + +/* cunix: a connected datagram socket, a chosen name, and fork. */ static int connected_dgram(void) { struct sockaddr_un r, a; @@ -7,7 +9,7 @@ static int connected_dgram(void) { int stranger = socket(AF_UNIX, SOCK_DGRAM, 0); bind(rs, (void *)&r, rl); bind(as, (void *)&a, al); - // r talks only to a: a stranger's datagram to r is refused. + /* r talks only to a: a stranger's datagram to r is refused. */ connect(rs, (void *)&a, al); int refused = sendto(stranger, "s", 1, 0, (void *)&r, rl) < 0 ? errno : 0; long sent = send(rs, "to-a", 4, 0); @@ -32,7 +34,7 @@ static int autobind(void) { int rc = bind(s, (void *)&a, sizeof(sa_family_t)); getsockname(s, (void *)&b, &bl); close(s); - // Linux chooses a NUL and five hex digits. + /* Linux chooses a NUL and five hex digits. */ if (rc || bl != 8 || b.sun_path[0] != 0) { return fail("autobind: a NUL and five hex digits", rc, bl); } @@ -66,21 +68,3 @@ static int across_fork(void) { ok("across_fork", "bytes from the child", got); return 0; } - -int main(void) { - int (*const part[])(void) = {path_stream, left_behind, abstract_dgram, - connected_dgram, autobind, across_fork}; - const int count = sizeof part / sizeof part[0]; - int failed = 0; - for (int i = 0; i < count; i++) { - failed += part[i](); - } - if (failed) { - printf("[C] cunix FAIL: %d parts failed, %d passed\n", failed, parts); - fflush(stdout); - return 1; - } - printf("[C] cunix PASS: %d parts\n", parts); - fflush(stdout); - return 0; -} diff --git a/userland/linux_guests/go/http/main.go b/userland/linux_guests/go/http/main.go index 139431c3b..ea13e86b2 100644 --- a/userland/linux_guests/go/http/main.go +++ b/userland/linux_guests/go/http/main.go @@ -1,7 +1,9 @@ -// net/http as Linux runs it, inside one guest: a server on 127.0.0.1:0 and a -// client of it. Twenty GETs, each answered 200 with the path it asked for, -// over one kept-alive connection, which the server's own count of new -// connections shows. +/* + * net/http as Linux runs it, inside one guest: a server on 127.0.0.1:0 and a + * client of it. Twenty GETs, each answered 200 with the path it asked for, + * over one kept-alive connection, which the server's own count of new + * connections shows. + */ package main import (