From 1f06c7855347230690b9ec35eeb0709070fc9b4a Mon Sep 17 00:00:00 2001 From: DevomB Date: Mon, 5 Oct 2026 10:57:11 -0700 Subject: [PATCH 01/18] A check that a zone cannot send from another zone's address: personal sends datagrams to the bridge from its own address and, with IPV6_FREEBIND, from untrusted's, and counters in the net zone say which arrived --- build/guest-tests/zones-check.sh | 41 ++++++++++++++++++++++++++++++++ tools/image/zones-test.sh | 2 +- 2 files changed, 42 insertions(+), 1 deletion(-) diff --git a/build/guest-tests/zones-check.sh b/build/guest-tests/zones-check.sh index 7d8569af..63fbfc2f 100755 --- a/build/guest-tests/zones-check.sh +++ b/build/guest-tests/zones-check.sh @@ -83,6 +83,47 @@ PY' printf 'personal-pass\n' > /root/zt/personal.pass; chmod 600 /root/zt/personal.pass "$KD" volume init personal --size 64M --passphrase-file /root/zt/personal.pass > "$LOG/vol-personal.out" 2>&1 \ && pass "volume-init" "personal: $(tail -1 "$LOG/vol-personal.out")" || fail "volume-init" "$(tail -2 "$LOG/vol-personal.out" | tr '\n' ' ')" +# A zone that sends from another zone's address. The rule that opens the +# uplink's own network goes by a packet's source, and nothing visible ties a +# bridge port to the address of the zone behind it. personal sends datagrams +# to the bridge from its own address (the control: they must arrive) and then, +# with IPV6_FREEBIND, from untrusted's; counters put in the net zone for the +# time of the probe say which arrived. +net_init="$(cut -d' ' -f1 /run/kryptik/zones/net/init.pid 2>/dev/null)" +netns() { nsenter -t "${net_init:-0}" -n "$@"; } +own6="fd19::$(printf '%x' "$PER")"; other6="fd19::$(printf '%x' "$UNT")" +{ + netns nft add table inet ztprobe + netns nft add chain inet ztprobe pre '{ type filter hook prerouting priority -300; }' + netns nft add rule inet ztprobe pre ip6 saddr "$own6" ip6 daddr fd19::1 udp dport 9 counter comment '"zt-own"' + netns nft add rule inet ztprobe pre ip6 saddr "$other6" ip6 daddr fd19::1 udp dport 9 counter comment '"zt-other"' +} > "$LOG/source-probe.err" 2>&1 +zrun personal 40 --passphrase-file /root/zt/personal.pass -- python3 -c ' +import socket, sys, time +def send(src): + s = socket.socket(socket.AF_INET6, socket.SOCK_DGRAM) + if src: + s.setsockopt(socket.IPPROTO_IPV6, 78, 1) # IPV6_FREEBIND + s.bind((src, 0)) + for _ in range(3): + s.sendto(b"zt", ("fd19::1", 9)); time.sleep(0.2) +for what, src in (("OWN", None), ("OTHER", sys.argv[1])): + try: + send(src); print(what + "-SENT") + except OSError as e: + print("%s-REFUSED %s" % (what, e)) +' "$other6" +counted() { netns nft list table inet ztprobe 2>/dev/null | sed -n "s/.*counter packets \([0-9]*\) .*\"$1\".*/\1/p" | head -1; } +own_seen="$(counted zt-own)"; other_seen="$(counted zt-other)" +netns nft delete table inet ztprobe 2>/dev/null +said="$(tr '\n' ' ' <<<"$ZOUT")" +if [[ -z "$own_seen" || "$own_seen" -eq 0 ]]; then + fail "zone-source-pinned" "the probe has no path: personal's own datagrams did not reach the net zone (${said}; $(tr '\n' ' ' < "$LOG/source-probe.err"))" +elif [[ "${other_seen:-0}" -eq 0 ]]; then + pass "zone-source-pinned" "${own_seen} datagram(s) from personal's own ${own6} reached the net zone, none from untrusted's ${other6} (${said})" +else + fail "zone-source-pinned" "${other_seen} datagram(s) personal sent from untrusted's address ${other6} reached the net zone (${said})" +fi # personal stays up in the background for the separation and restart checks setsid "$KD" run personal --zones "$Z" --rootfs "$R" --passphrase-file /root/zt/personal.pass -- sh -c 'echo PERSONAL-UP; sleep 600' > "$LOG/personal-bg.out" 2>&1 & PBG=$! diff --git a/tools/image/zones-test.sh b/tools/image/zones-test.sh index 98b59121..3453d52b 100755 --- a/tools/image/zones-test.sh +++ b/tools/image/zones-test.sh @@ -65,7 +65,7 @@ zp="$(sed -n 's/.*passed=\([0-9]*\).*/\1/p' <<<"$summary")"; zf="$(sed -n 's/.*f if [[ -n "$summary" && "${zf:-1}" -eq 0 && "${zp:-0}" -ge 30 ]]; then green "every guest check passed (${zp})"; else red "guest checks: ${zp:-0} passed, ${zf:-?} failed"; fi grep 'ZT FAIL' <<<"$T2" | sed 's/^/ /' # The key verdicts one by one, so a pass is not a single line. -for name in kernel-support net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns routed-ipv6-noglobal zone-separation fail-closed net-restart-ready reattach-after-restart uplink-refused wifi-beyond \ +for name in kernel-support net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns routed-ipv6-noglobal zone-source-pinned zone-separation fail-closed net-restart-ready reattach-after-restart uplink-refused wifi-beyond \ wifi-module wifi-ap wifi-add wifi-associated wifi-lease wifi-egress wifi-forget \ time-floor-ran time-clamp time-claim-stepped time-claim-floor time-claim-consent pids-limit ephemeral-size-bound cpu-max-set \ terminal-terminfo man-page text-browser tls-trust \ From 16bc28f9237f727466726ea6216798fc13da5830 Mon Sep 17 00:00:00 2001 From: DevomB Date: Tue, 6 Oct 2026 21:08:36 -0700 Subject: [PATCH 02/18] The source probe waits out duplicate address detection, binds every source, and counts before and after the net zone's chains The probe's control sent with no bound source while personal's fd19:: address was still tentative, so its datagrams left from the link-local address and matched no counter: run 37352305190 read "own datagrams did not reach the net zone" and could say nothing about the spoofed ones. Now personal waits until eth0 has no tentative address and binds each source, so a send that cannot use its address says so (OWN6-REFUSED EADDRNOTAVAIL). It sends over IPv4 too, where the kernel refuses a source the zone does not hold without IP_TRANSPARENT. The net zone counts each source at prerouting -350, ahead of its own chains, and again at input, after them, and the verdict prints every count. --- build/guest-tests/zones-check.sh | 102 +++++++++++++++++++++---------- 1 file changed, 71 insertions(+), 31 deletions(-) diff --git a/build/guest-tests/zones-check.sh b/build/guest-tests/zones-check.sh index 63fbfc2f..6b976c50 100755 --- a/build/guest-tests/zones-check.sh +++ b/build/guest-tests/zones-check.sh @@ -84,45 +84,85 @@ printf 'personal-pass\n' > /root/zt/personal.pass; chmod 600 /root/zt/personal.p "$KD" volume init personal --size 64M --passphrase-file /root/zt/personal.pass > "$LOG/vol-personal.out" 2>&1 \ && pass "volume-init" "personal: $(tail -1 "$LOG/vol-personal.out")" || fail "volume-init" "$(tail -2 "$LOG/vol-personal.out" | tr '\n' ' ')" # A zone that sends from another zone's address. The rule that opens the -# uplink's own network goes by a packet's source, and nothing visible ties a -# bridge port to the address of the zone behind it. personal sends datagrams -# to the bridge from its own address (the control: they must arrive) and then, -# with IPV6_FREEBIND, from untrusted's; counters put in the net zone for the -# time of the probe say which arrived. +# uplink's own network goes by a packet's source, and IPV6_FREEBIND lets an +# unprivileged socket send from an address its host does not hold. personal +# sends to the bridge from its own addresses (the control) and with FREEBIND +# from untrusted's. Counters in the net zone say what reached it, ahead of its +# own chains, and what it took in after them. net_init="$(cut -d' ' -f1 /run/kryptik/zones/net/init.pid 2>/dev/null)" netns() { nsenter -t "${net_init:-0}" -n "$@"; } +own4="10.19.0.$PER"; other4="10.19.0.$UNT" own6="fd19::$(printf '%x' "$PER")"; other6="fd19::$(printf '%x' "$UNT")" -{ - netns nft add table inet ztprobe - netns nft add chain inet ztprobe pre '{ type filter hook prerouting priority -300; }' - netns nft add rule inet ztprobe pre ip6 saddr "$own6" ip6 daddr fd19::1 udp dport 9 counter comment '"zt-own"' - netns nft add rule inet ztprobe pre ip6 saddr "$other6" ip6 daddr fd19::1 udp dport 9 counter comment '"zt-other"' -} > "$LOG/source-probe.err" 2>&1 +netns nft -f - > "$LOG/source-probe.err" 2>&1 </dev/null | sed -n "s/.*counter packets \([0-9]*\) .*\"$1\".*/\1/p" | head -1; } -own_seen="$(counted zt-own)"; other_seen="$(counted zt-other)" + print("%s-REFUSED %s" % (what, errno.errorcode.get(e.errno, e))) + finally: + s.close() +send("OWN4", socket.AF_INET, own4, "10.19.0.1", False) +send("OTHER4", socket.AF_INET, other4, "10.19.0.1", True) +send("OWN6", socket.AF_INET6, own6, "fd19::1", False) +send("OTHER6", socket.AF_INET6, other6, "fd19::1", True) +# a datagram still waiting on neighbour discovery goes with the namespace +time.sleep(1) +' "$own4" "$other4" "$own6" "$other6" +table="$(netns nft list table inet ztprobe 2>&1)" netns nft delete table inet ztprobe 2>/dev/null -said="$(tr '\n' ' ' <<<"$ZOUT")" -if [[ -z "$own_seen" || "$own_seen" -eq 0 ]]; then - fail "zone-source-pinned" "the probe has no path: personal's own datagrams did not reach the net zone (${said}; $(tr '\n' ' ' < "$LOG/source-probe.err"))" -elif [[ "${other_seen:-0}" -eq 0 ]]; then - pass "zone-source-pinned" "${own_seen} datagram(s) from personal's own ${own6} reached the net zone, none from untrusted's ${other6} (${said})" +declare -A seen +for c in pre-any pre-own4 pre-other4 pre-own6 pre-other6 taken-own4 taken-other4 taken-own6 taken-other6; do + seen[$c]="$(sed -n "s/.*counter packets \([0-9]*\) .*\"$c\".*/\1/p" <<<"$table" | head -1)" +done +counts="reached the net zone (IPv4/IPv6): own ${seen[pre-own4]:--}/${seen[pre-own6]:--}, untrusted's ${seen[pre-other4]:--}/${seen[pre-other6]:--}, any ${seen[pre-any]:--}; taken in: own ${seen[taken-own4]:--}/${seen[taken-own6]:--}, untrusted's ${seen[taken-other4]:--}/${seen[taken-other6]:--}; personal: $(tr '\n' ' ' <<<"$ZOUT")" +probe_err="$(cat "$LOG/source-probe.err"; [[ -n "${seen[pre-any]}" ]] || head -2 <<<"$table")" +if [[ "${seen[taken-own4]:-0}" -eq 0 || "${seen[taken-own6]:-0}" -eq 0 ]]; then + fail "zone-source-pinned" "the probe has no path: personal's own datagrams were not taken in; ${counts}${probe_err:+; nft: $(tr '\n' ' ' <<<"$probe_err" | cut -c1-200)}" +elif [[ "${seen[taken-other4]:-0}" -gt 0 || "${seen[taken-other6]:-0}" -gt 0 ]]; then + fail "zone-source-pinned" "the net zone took in datagrams personal sent from untrusted's addresses; ${counts}" else - fail "zone-source-pinned" "${other_seen} datagram(s) personal sent from untrusted's address ${other6} reached the net zone (${said})" + pass "zone-source-pinned" "personal's own datagrams were taken in and none from untrusted's addresses; ${counts}" fi # personal stays up in the background for the separation and restart checks setsid "$KD" run personal --zones "$Z" --rootfs "$R" --passphrase-file /root/zt/personal.pass -- sh -c 'echo PERSONAL-UP; sleep 600' > "$LOG/personal-bg.out" 2>&1 & From 7b935d44652ce1ff94ab4e607391593dc9503803 Mon Sep 17 00:00:00 2001 From: DevomB Date: Tue, 6 Oct 2026 21:11:45 -0700 Subject: [PATCH 03/18] The net zone takes a zone's address only with that zone's MAC, which kryptikd sets on its eth0 IPV6_FREEBIND needs no capability and IPv6 checks no source on the way out, so a routed zone could send as fd19:: of another zone: personal as untrusted, whose address the forward chain lets onto the uplink's own network, or any zone to the resolver with the answer going to another. The design said a zone cannot send from another address; for IPv6 it could. kryptikd now creates each routed zone's eth0 with the MAC 02:19:00:00:00:, which the zone can neither change nor forge (no CAP_NET_ADMIN, no CAP_NET_RAW, no packet sockets). The net zone's ruleset pairs 10.19.0. and fd19:: with that MAC for every host number and drops, at prerouting ahead of conntrack, a packet from the bridge whose source and MAC are not a pair; from a link-local address only neighbour discovery passes. It is the inet table the net zone already loads, so the kernel gains nothing (no NF_TABLES_BRIDGE), and the read-back after loading now looks for the pin too. netzone-uplink.sh checks every host's pair, the order of the prerouting rules, and that netlink.rs gives the MAC the script expects; the kernel-backed veth tests read the MAC back inside the zone's namespace; zone-source-pinned counts what reaches the net zone ahead of the pin and what it takes in. --- compartments/kryptikd/src/netlink.rs | 41 +++++++++++++++++- compartments/kryptikd/src/netlink/tests.rs | 24 ++++++++--- compartments/kryptikd/src/netzone.rs | 4 +- compartments/kryptikd/src/netzone/tests.rs | 12 ++++-- docs/design/net-zone.md | 50 ++++++++++++++-------- tools/net/netzone-init.sh | 25 +++++++++-- tools/tests/netzone-uplink.sh | 19 ++++++-- 7 files changed, 136 insertions(+), 39 deletions(-) diff --git a/compartments/kryptikd/src/netlink.rs b/compartments/kryptikd/src/netlink.rs index 8d2c08e3..8149c5b3 100755 --- a/compartments/kryptikd/src/netlink.rs +++ b/compartments/kryptikd/src/netlink.rs @@ -36,6 +36,7 @@ const NLM_F_EXCL: u16 = 0x200; const NLM_F_CREATE: u16 = 0x400; const NLA_F_NESTED: u16 = 0x8000; +const IFLA_ADDRESS: u16 = 1; const IFLA_IFNAME: u16 = 3; const IFLA_MASTER: u16 = 10; /* A bridge setlink acks an unknown attribute and ignores it, so a wrong @@ -253,8 +254,9 @@ fn check_name(name: &str) -> io::Result<()> { } /// Create a veth pair `a` <-> `b`. When `peer_ns` is given, `b` is created in -/// that network namespace and never appears in this one. -pub fn create_veth(a: &str, b: &str, peer_ns: Option) -> io::Result<()> { +/// that network namespace and never appears in this one; `peer_mac` is `b`'s +/// MAC, random when None. +pub fn create_veth(a: &str, b: &str, peer_ns: Option, peer_mac: Option<[u8; 6]>) -> io::Result<()> { check_name(a)?; check_name(b)?; let mut m = Msg::new(RTM_NEWLINK, NLM_F_CREATE | NLM_F_EXCL, 1); @@ -269,6 +271,9 @@ pub fn create_veth(a: &str, b: &str, peer_ns: Option) -> io::Result<()> { if let Some(fd) = peer_ns { m.attr_u32(IFLA_NET_NS_FD, fd as u32); } + if let Some(mac) = peer_mac { + m.attr(IFLA_ADDRESS, &mac); + } m.end_nested(peer); m.end_nested(data); m.end_nested(li); @@ -444,6 +449,32 @@ pub fn is_up(dev: &str) -> io::Result { Ok(flags & libc::IFF_UP != 0) } +/// `dev`'s MAC. +#[cfg(test)] +pub fn mac_of(dev: &str) -> io::Result<[u8; 6]> { + let c = CString::new(dev).map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "NUL"))?; + let sock = unsafe { libc::socket(libc::AF_INET, libc::SOCK_DGRAM | libc::SOCK_CLOEXEC, 0) }; + if sock < 0 { + return Err(io::Error::last_os_error()); + } + let mut ifr: libc::ifreq = unsafe { std::mem::zeroed() }; + for (i, b) in c.as_bytes_with_nul().iter().take(libc::IFNAMSIZ).enumerate() { + ifr.ifr_name[i] = *b as libc::c_char; + } + let r = unsafe { libc::ioctl(sock, libc::SIOCGIFHWADDR as _, &mut ifr) }; + let e = io::Error::last_os_error(); + unsafe { libc::close(sock) }; + if r < 0 { + return Err(e); + } + let sa = unsafe { ifr.ifr_ifru.ifru_hwaddr }; + let mut mac = [0u8; 6]; + for (m, b) in mac.iter_mut().zip(sa.sa_data.iter()) { + *m = *b as u8; + } + Ok(mac) +} + /// Open a handle on a process's network namespace. pub fn open_netns_of(pid: libc::pid_t) -> io::Result { let p = CString::new(format!("/proc/{pid}/ns/net")).unwrap(); @@ -489,5 +520,11 @@ pub fn zone_v6(k: u8) -> [u8; 16] { a } +/// Host `k`'s eth0 MAC, 02:19:00:00:00:k (locally administered, unicast). The +/// net zone takes 10.19.0.k and fd19::k only with it (tools/net/netzone-init.sh). +pub fn zone_mac(k: u8) -> [u8; 6] { + [0x02, 0x19, 0, 0, 0, k] +} + #[cfg(test)] pub(crate) mod tests; diff --git a/compartments/kryptikd/src/netlink/tests.rs b/compartments/kryptikd/src/netlink/tests.rs index d9275315..bb6d798b 100644 --- a/compartments/kryptikd/src/netlink/tests.rs +++ b/compartments/kryptikd/src/netlink/tests.rs @@ -175,6 +175,11 @@ fn address_plan_is_fixed() { assert_eq!(zone_v4(2), [10, 19, 0, 2]); assert_eq!(zone_v6(2)[15], 2); assert_eq!(&zone_v6(2)[..2], &[0xfd, 0x19]); + // tools/net/netzone-init.sh pairs 10.19.0.k and fd19::k with this MAC + assert_eq!(zone_mac(2), [0x02, 0x19, 0, 0, 0, 2]); + assert_eq!(zone_mac(249)[5], 249); + // locally administered, unicast + assert_eq!(zone_mac(249)[0] & 0x03, 0x02); assert!(check_name("kv-untrusted").is_ok()); assert!(check_name("kv-averylongzonename").is_err()); assert!(check_name("a/b").is_err()); @@ -273,7 +278,7 @@ fn udp_received(fd: RawFd) -> bool { fn veth_bridge_addresses_routes() { let rc = in_userns_netns(|| { let r: Result<(), i32> = (|| { - step(1, create_veth("va", "vb", None))?; + step(1, create_veth("va", "vb", None, None))?; if index_of("va").is_err() || index_of("vb").is_err() { return Err(2); } @@ -294,7 +299,7 @@ fn veth_bridge_addresses_routes() { step(14, add_addr4("br0", [10, 99, 0, 2], 24))?; step(18, add_default_route4([10, 99, 0, 2], "va"))?; step(19, add_default_route6(zone_v6(2), "va"))?; - if create_veth("va", "vx", None).is_ok() { + if create_veth("va", "vx", None, None).is_ok() { return Err(20); // EXCL: a duplicate name is refused } Ok(()) @@ -318,7 +323,7 @@ fn veth_peer_in_other_namespace() { Ok(v) => v, Err(c) => return c, }; - let r = create_veth("kv-t", "eth0", Some(ns)); + let r = create_veth("kv-t", "eth0", Some(ns), Some(zone_mac(9))); if let Err(e) = r { eprintln!("create_veth into peer ns: {e}"); unsafe { libc::kill(gc, libc::SIGKILL) }; @@ -330,9 +335,16 @@ fn veth_peer_in_other_namespace() { if index_of("eth0").is_ok() { return 36; } - let there = with_netns(ns, || index_of("eth0").map(|_| ())).is_ok(); + let mac = with_netns(ns, || mac_of("eth0")); unsafe { libc::kill(gc, libc::SIGKILL) }; - if there { 0 } else { 38 } + match mac { + Ok(m) if m == zone_mac(9) => 0, + Ok(m) => { + eprintln!("eth0 came up with MAC {m:02x?}, not {:02x?}", zone_mac(9)); + 39 + } + Err(_) => 38, + } }); match rc { 0 => {} @@ -355,7 +367,7 @@ fn isolated_ports_block_zone_to_zone() { step(42, add_addr4("kryptik0", [10, 99, 0, 254], 24))?; for (k, ns) in [(1u8, ns1), (2u8, ns2)] { let port = format!("kv-z{k}"); - step(43, create_veth(&port, "eth0", Some(ns)))?; + step(43, create_veth(&port, "eth0", Some(ns), None))?; step(44, set_master(&port, "kryptik0"))?; step(45, set_port_isolated(&port, true))?; let flag = fs::read_to_string(format!("/sys/class/net/{port}/brport/isolated")).map_err(|_| 49)?; diff --git a/compartments/kryptikd/src/netzone.rs b/compartments/kryptikd/src/netzone.rs index c1156d15..6f6efde8 100755 --- a/compartments/kryptikd/src/netzone.rs +++ b/compartments/kryptikd/src/netzone.rs @@ -6,7 +6,7 @@ //! kryptik0 bridge 10.19.0.1/24, fd19::1/64 //! routed zone k: kv- in net's netns, on kryptik0, ISOLATED //! eth0 in the zone's netns: 10.19.0.k/24, fd19::k/64, -//! default routes via the bridge +//! MAC 02:19:00:00:00:k, default routes via the bridge //! zone 0: keeps lo only once net has taken the NIC //! mode = "none": nothing is ever created //! ``` @@ -393,7 +393,7 @@ fn attach_routed(name: &str, k: u8, nic_ns: i32, zone_ns: i32, host_gid: Option< fn attach_v4(port: &str, k: u8, nic_ns: i32, zone_ns: i32, host_gid: Option) -> Result<(), NetError> { netlink::with_netns(nic_ns, || { - netlink::create_veth(port, "eth0", Some(zone_ns))?; + netlink::create_veth(port, "eth0", Some(zone_ns), Some(netlink::zone_mac(k)))?; netlink::set_master(port, BRIDGE)?; netlink::set_port_isolated(port, true)?; netlink::set_up(port) diff --git a/compartments/kryptikd/src/netzone/tests.rs b/compartments/kryptikd/src/netzone/tests.rs index 6911fb3d..51bad97e 100644 --- a/compartments/kryptikd/src/netzone/tests.rs +++ b/compartments/kryptikd/src/netzone/tests.rs @@ -87,7 +87,7 @@ fn only_bus_devices_count_as_physical() { eprintln!("a fresh namespace already counts {before:?} as physical"); return Err(2); } - step(3, netlink::create_veth("pa", "pb", None))?; + step(3, netlink::create_veth("pa", "pb", None, None))?; step(4, netlink::create_bridge("br-t"))?; if !class_net.join("pa").exists() || !class_net.join("br-t").exists() { eprintln!("the mounted sysfs does not show this namespace's devices; this proved nothing"); @@ -206,17 +206,23 @@ fn ipv4_survives_disabled_ipv6() { eprintln!("attach_routed: {e}"); 5 })?; - let (routes, v6, up) = netlink::with_netns(zone_ns, || { + let (routes, v6, up, mac) = netlink::with_netns(zone_ns, || { Ok(( std::fs::read_to_string("/proc/self/net/route")?, std::fs::read_to_string("/proc/self/net/if_inet6").unwrap_or_default(), netlink::is_up("eth0")?, + netlink::mac_of("eth0")?, )) }) .map_err(|_| 6)?; if !up { return Err(7); } + // The net zone takes the zone's addresses only with this MAC. + if mac != netlink::zone_mac(7) { + eprintln!("eth0 has MAC {mac:02x?}, not {:02x?}", netlink::zone_mac(7)); + return Err(10); + } let rows: Vec<&str> = routes.lines().skip(1).collect(); let default = rows.iter().any(|l| { let mut f = l.split_whitespace(); @@ -261,7 +267,7 @@ fn uplink_config_travels_with_nic() { Err(c) => return c, }; let r: Result<(), i32> = (|| { - step(1, netlink::create_veth("up0", "up1", None))?; + step(1, netlink::create_veth("up0", "up1", None, None))?; step(2, netlink::set_up("up1"))?; step(3, netlink::set_up("up0"))?; step(4, netlink::add_addr4("up0", [10, 77, 0, 5], 24))?; diff --git a/docs/design/net-zone.md b/docs/design/net-zone.md index fea42b14..6044836e 100755 --- a/docs/design/net-zone.md +++ b/docs/design/net-zone.md @@ -40,8 +40,8 @@ query and the [update](update-channel.md) fetcher. Builds on with its peer born in the routed zone as `eth0`, enslaves `kv-` to `kryptik0` and isolates the port (`IFLA_BRPORT_ISOLATED`), so no frame passes between two `kv-*` ports. The zone's addresses, `10.19.0./24` and - `fd19::/64` with default routes via the bridge, follow from its declared - identity (`netzone::host_number`: `uid_base` 131072 is `.2`, 196608 is `.3`, + `fd19::/64` with default routes via the bridge, and its MAC, + `02:19:00:00:00:`, follow from its declared identity (`netzone::host_number`: `uid_base` 131072 is `.2`, 196608 is `.3`, and so on), not from DHCP: one less daemon, no broadcast domain. `accept_ra = 0` is set first, and `ping_group_range` names the zone's host gid so unprivileged ICMP echo works (the sysctl takes host ids, so the @@ -55,9 +55,18 @@ query and the [update](update-channel.md) fetcher. Builds on - **A routed zone owns its namespace, not its port.** Its bounding set is `CAP_NET_BIND_SERVICE`, `policy::check_for_zone` refuses a policy keeping `CAP_NET_ADMIN` or `CAP_NET_RAW`, and seccomp refuses packet sockets. It - cannot change its address or MAC, send from another address, or put a frame - on the wire that the kernel did not build; `ip link set eth0 down` fails - with `EPERM`. + cannot change its address or MAC, or put a frame on the wire that the + kernel did not build; `ip link set eth0 down` fails with `EPERM`. +- **A zone's addresses count only with its MAC.** A routed zone can still + send from an address it does not hold: `IPV6_FREEBIND` needs no capability + and IPv6 checks no source on the way out. (IPv4 refuses such a source + unless the socket is transparent, which needs one of the two capabilities.) + The net zone's ruleset therefore pairs `10.19.0.` and `fd19::` with + `02:19:00:00:00:` for every host number and, at prerouting ahead of + conntrack, drops a packet from the bridge whose source and MAC are not a + pair. From a link-local address only neighbour discovery passes, so one zone + cannot borrow another's address to reach what that one may, or send the net + zone's answers to it. - **Every namespace starts with loopback only.** The kernel builds SIT in, so the launcher sets `net.core.fb_tunnels_only_for_init_net = 1` first, and a privileged launch refuses a namespace holding anything else. @@ -126,8 +135,8 @@ its definition says `[network] local = true`. the verified root, which the net zone shares read-only, and a routed zone's address follows from its `uid_base`. The script puts the addresses of the zones that claim `local` into `local4` and `local6` when it loads the - ruleset. A routed zone cannot change its address, so the address is the - zone. kryptikd refuses the key on a zone that is not routed, and + ruleset. A routed zone cannot change its address, and the net zone takes + an address only with that zone's MAC, so the address is the zone. kryptikd refuses the key on a zone that is not routed, and `kryptikd explain` says which way a zone is set. - **`untrusted` is the one shipped zone that claims it.** A hotel's or café's Wi-Fi asks for a login on a page its gateway serves, and the net zone has @@ -222,18 +231,18 @@ changes. - A routed zone reaches the network an uplink sits on only if its definition says so. The others reach what a gateway carries, and never the gateway itself. -- Addresses are identities: 10.19.0.k follows from `uid_base`, so the broker - or a future policy can name zones by address as safely as by uid, as long - as routed zones cannot change their address. +- Addresses are identities: 10.19.0.k follows from `uid_base` and the net + zone takes it only with that zone's MAC, so the broker or a future policy + there can name zones by address as safely as by uid, as long as routed + zones keep neither network capability. ## Not built -- **MAC/IP pinning of bridge ports** (nftables `bridge` rules dropping frames - with a source that is not the assigned one). A routed zone already cannot - re-address itself or forge frames. Pinning would check that again inside the - hostile net zone, and need `NF_TABLES_BRIDGE` and `BRIDGE_NETFILTER` built - in: more kernel reachable from a hostile zone for no new guarantee. Revisit - if a routed zone may ever keep either network capability. +- **Pinning by bridge port** (nftables `bridge` rules on each `kv-*` port). + The `inet` table pins by MAC instead, which a routed zone can neither change + nor forge, so a port would add nothing, and `NF_TABLES_BRIDGE` and + `BRIDGE_NETFILTER` would put more kernel within reach of the hostile net + zone. Revisit if a routed zone may ever keep either network capability. ## Tests @@ -248,8 +257,9 @@ changes. - `wifi.rs` unit tests and the serve and cli suites cover the credentials file and `kryptik wifi`. - `tools/tests/netzone-uplink.sh`: the zones a definition lets through, by the - address kryptikd derives for each, the gateway sets as nft is fed them, and - the order of the forward rules. + address kryptikd derives for each, every host's addresses pinned to its own + MAC and the MAC the same as `netlink::zone_mac`, the gateway sets as nft is + fed them, and the order of the prerouting and forward rules. - The launcher suite reads a zone's bounding set (exactly `0x400`) and the boundary suite asks for an `AF_PACKET` socket. The launcher suite's routed-networking section runs only with `KRYPTIK_VM_DISPOSABLE=1`, since @@ -257,7 +267,9 @@ changes. - `build/guest-tests/zones-check.sh` on the installed system checks every guarantee above under QEMU user networking: the net zone `READY`, zone 0 offline, a routed zone's address, NAT, ULA-only IPv6 and resolver, zones - separated, `vault` offline, no egress while the net zone is down, + separated, a zone's datagrams sent from another zone's addresses counted + where they reach the net zone and never taken in while its own are, `vault` + offline, no egress while the net zone is down, reattachment after a restart, a zone without `local` refused the VM gateway, and, on two `mac80211_hwsim` radios, the net zone associating, leasing and routing over one while the other is the access point, whose own diff --git a/tools/net/netzone-init.sh b/tools/net/netzone-init.sh index da04c6c9..efc61cb0 100755 --- a/tools/net/netzone-init.sh +++ b/tools/net/netzone-init.sh @@ -76,6 +76,12 @@ LOCAL6="$(local_zones "$ZONES" | awk '{ print $2 }' | tr '\n' ',' | sed 's/,$//' SET4="set local4 { type ipv4_addr; }"; SET6="set local6 { type ipv6_addr; }" [ -z "$LOCAL4" ] || SET4="set local4 { type ipv4_addr; elements = { ${LOCAL4} } }" [ -z "$LOCAL6" ] || SET6="set local6 { type ipv6_addr; elements = { ${LOCAL6} } }" +# Host K's eth0 has the MAC kryptikd gives it (netzone.rs, zone_mac), +# 02:19:00:00:00:K, which a routed zone can neither change nor forge. With +# IPV6_FREEBIND any socket sends from an address it does not hold, so a packet +# from the bridge counts as from 10.19.0.K or fd19::K only with that MAC. +PIN4="$(awk 'BEGIN { for (k = 2; k < 250; k++) printf "%s10.19.0.%d . 02:19:00:00:00:%02x", (k > 2 ? ", " : ""), k, k }')" +PIN6="$(awk 'BEGIN { for (k = 2; k < 250; k++) printf "%sfd19::%x . 02:19:00:00:00:%02x", (k > 2 ? ", " : ""), k, k }')" # A zone goes out by a gateway (gw4, gw6) and never to the gateway itself: # the rest of what an uplink reaches is the network it sits on, open to local4 @@ -86,6 +92,15 @@ RULES="table inet kryptik { set gw6 { type ipv6_addr; } ${SET4} ${SET6} + set pin4 { type ipv4_addr . ether_addr; elements = { ${PIN4} } } + set pin6 { type ipv6_addr . ether_addr; elements = { ${PIN6} } } + chain prerouting { + type filter hook prerouting priority raw; policy accept; + iifname \"${BR}\" ip saddr . ether saddr != @pin4 drop + iifname \"${BR}\" ip6 saddr . ether saddr @pin6 accept + iifname \"${BR}\" ip6 saddr fe80::/10 icmpv6 type { nd-neighbor-solicit, nd-neighbor-advert } accept + iifname \"${BR}\" meta nfproto ipv6 drop + } chain forward { type filter hook forward priority filter; policy drop; ct state established,related accept @@ -115,10 +130,12 @@ load_policy() { nft flush ruleset 2>/dev/null || true printf '%s\n' "$RULES" | nft -f - 2>/tmp/nft.err || { say "nftables: load FAILED: $(tr '\n' ' ' < /tmp/nft.err)"; return 1; } live="$(nft list table inet kryptik 2>/dev/null)" - case "$live" in - *"policy drop"*"masquerade"*) ;; - *) say "nftables: the loaded table is not the policy (missing drop policy or masquerade)"; nft flush ruleset 2>/dev/null; return 1 ;; - esac + for want in "@pin6 accept" "policy drop" "masquerade"; do + case "$live" in + *"$want"*) ;; + *) say "nftables: the loaded table is not the policy (no \"${want}\")"; nft flush ruleset 2>/dev/null; return 1 ;; + esac + done GATEWAYS="" return 0 } diff --git a/tools/tests/netzone-uplink.sh b/tools/tests/netzone-uplink.sh index 6095cd2e..1fbd9c28 100755 --- a/tools/tests/netzone-uplink.sh +++ b/tools/tests/netzone-uplink.sh @@ -1,9 +1,9 @@ #!/usr/bin/env bash # Tests for the net zone's rule on the networks its uplinks sit on # (tools/net/netzone-init.sh): the zones a definition lets through, by the -# address kryptikd gives each; the gateways as the sets take them; and the -# order of the rules. Offline, with stand-ins for ip and nft, under each POSIX -# shell here. +# address kryptikd gives each; each address pinned to its zone's MAC; the +# gateways as the sets take them; and the order of the rules. Offline, with +# stand-ins for ip and nft, under each POSIX shell here. set -uo pipefail ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" SCRIPT="${ROOT}/tools/net/netzone-init.sh" @@ -106,10 +106,23 @@ for sh in sh bash dash; do forward="$(sed -n '/chain forward {/,/^ }$/p' <<<"$rules" | grep -oE 'established,related accept|ip6? saddr @local[46] accept|rt ip6? nexthop @gw[46] ip6? daddr != @gw[46] accept|reject with icmpx type admin-prohibited|oifname "kryptik0" drop' | tr '\n' '|')" same "replies first, then the local zones, then what a gateway carries but the gateway itself, then the refusal" "$forward" \ 'established,related accept|ip saddr @local4 accept|ip6 saddr @local6 accept|rt ip nexthop @gw4 ip daddr != @gw4 accept|rt ip6 nexthop @gw6 ip6 daddr != @gw6 accept|reject with icmpx type admin-prohibited|oifname "kryptik0" drop|' + pairs4="$(grep 'set pin4 ' <<<"$rules" | grep -oE '10\.19\.0\.[0-9]+ \. 02:19:00:00:00:[0-9a-f]{2}' \ + | awk '{ split($1, a, "."); n++; if (sprintf("%02x", a[4]) == substr($3, 16)) ok++ } END { print n + 0, ok + 0 }')" + pairs6="$(grep 'set pin6 ' <<<"$rules" | grep -oE 'fd19::[0-9a-f]+ \. 02:19:00:00:00:[0-9a-f]{2}' \ + | awk '{ h = substr($1, 7); n++; if ((length(h) == 1 ? "0" h : h) == substr($3, 16)) ok++ } END { print n + 0, ok + 0 }')" + same "hosts 2 to 249 each have both addresses pinned to their own MAC" "$pairs4 $pairs6" "248 248 248 248" + pre="$(sed -n '/chain prerouting {/,/^ }$/p' <<<"$rules" | grep -oE 'priority raw|ip saddr \. ether saddr != @pin4 drop|ip6 saddr \. ether saddr @pin6 accept|ip6 saddr fe80::/10 icmpv6 type \{ nd-neighbor-solicit, nd-neighbor-advert \} accept|meta nfproto ipv6 drop' | tr '\n' '|')" + same "from the bridge, ahead of conntrack: IPv4 off its pin dropped, IPv6 on its pin taken, neighbour discovery from a link-local address taken, other IPv6 dropped" "$pre" \ + 'priority raw|ip saddr . ether saddr != @pin4 drop|ip6 saddr . ether saddr @pin6 accept|ip6 saddr fe80::/10 icmpv6 type { nd-neighbor-solicit, nd-neighbor-advert } accept|meta nfproto ipv6 drop|' rules="$(run rules "$T/none")" grep -qF 'set local4 { type ipv4_addr; }' <<<"$rules" && green "with no zone let through the sets are empty, and every zone is refused" || red "the empty local sets" "$(grep 'set local' <<<"$rules" | tr '\n' '|')" done +# The MAC the script pins each host to is the one kryptikd gives its eth0. +grep -qF '[0x02, 0x19, 0, 0, 0, k]' "${ROOT}/compartments/kryptikd/src/netlink.rs" \ + && green "netlink.rs gives host k the MAC 02:19:00:00:00:k the pin expects" \ + || red "netlink.rs's zone_mac is not 02:19:00:00:00:k" "$(grep -A2 'fn zone_mac' "${ROOT}/compartments/kryptikd/src/netlink.rs" | tr '\n' ' ')" + # Where this user may make a network namespace, nft itself reads the ruleset. # A kernel without the pieces is no finding here; a ruleset nft cannot parse is. if command -v nft >/dev/null 2>&1 && unshare -rn true 2>/dev/null; then From 73e148868fba137c8411189505acde0b5486eda6 Mon Sep 17 00:00:00 2001 From: DevomB Date: Tue, 6 Oct 2026 21:19:05 -0700 Subject: [PATCH 04/18] A zone running across a net zone restart is shown going out through the gateway again, and the uplink coming back to zone 0 under its own name reattach-after-restart pinged the bridge once #213 refused personal the VM gateway, so nothing showed a reattached zone going out through the net zone's forwarding and NAT any more. personal keeps that check, worded as what it shows. untrusted, which its definition lets reach the gateway, now runs across a second restart and must reach 10.0.2.2 before it, have no eth0 and no path while the net zone is down, and reach it again once reattached. While the net zone is down the uplink must be back in zone 0 as eth0, down, with no address, and the next start must take it again. A review read the uplink as renamed by the net zone and every later start failing on "create bridge kryptik0: File exists"; nothing renames it, and the kernel names a returning interface dev only when zone 0 already has one by its name. The design doc says so. --- build/guest-tests/zones-check.sh | 46 +++++++++++++++++++++++++++++--- docs/design/net-zone.md | 10 ++++--- tools/image/zones-test.sh | 2 +- 3 files changed, 51 insertions(+), 7 deletions(-) diff --git a/build/guest-tests/zones-check.sh b/build/guest-tests/zones-check.sh index 4de1c784..b90dcb92 100755 --- a/build/guest-tests/zones-check.sh +++ b/build/guest-tests/zones-check.sh @@ -131,9 +131,49 @@ done sleep 2 zrun untrusted 30 -- sh -c 'python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.0.2.2 3 >/dev/null 2>&1 && echo GATEWAY-OK || echo GATEWAY-FAIL' [[ "$ZOUT" == *GATEWAY-OK* ]] && pass "egress-after-restart" "a zone started after the restart has egress" || fail "egress-after-restart" "$ZOUT" -# the running zone was reattached -ppid="$(cat /run/kryptik/zones/personal/init.pid 2>/dev/null | cut -d' ' -f1)" -if [[ -n "$ppid" ]] && nsenter -t "$ppid" -n ping -c1 -W3 10.19.0.1 >/dev/null 2>&1; then pass "reattach-after-restart" "the zone that was running reaches the net zone again"; else fail "reattach-after-restart" "personal (init $ppid) does not reach the bridge after the net restart"; fi +# personal, running across the restart, was reattached. It is refused the VM +# gateway, so the bridge is as far as it can show; untrusted shows egress below. +ppid="$(cut -d' ' -f1 /run/kryptik/zones/personal/init.pid 2>/dev/null)" +if [[ -n "$ppid" ]] && nsenter -t "$ppid" -n ping -c1 -W3 10.19.0.1 >/dev/null 2>&1; then pass "reattach-after-restart" "personal, running across the restart, reaches the net zone again"; else fail "reattach-after-restart" "personal (init ${ppid:-none}) does not reach the bridge after the net restart"; fi +# A zone that may reach the gateway, kept running across a second restart: it +# goes out through the gateway before, has no path while the net zone is down, +# and goes out again once reattached. Meanwhile the uplink is back in zone 0 +# under its own name, down and with no address, and the next start takes it. +setsid "$KD" run untrusted --zones "$Z" --rootfs "$R" -- sh -c 'echo UNTRUSTED-UP; sleep 600' > "$LOG/untrusted-bg.out" 2>&1 & +UBG=$! +for _ in $(seq 1 40); do grep -q UNTRUSTED-UP "$LOG/untrusted-bg.out" 2>/dev/null && break; sleep 0.5; done +upid="$(cut -d' ' -f1 /run/kryptik/zones/untrusted/init.pid 2>/dev/null)" +gateway_echo() { [[ -n "$upid" ]] && nsenter -t "$upid" -n python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.0.2.2 "$1" >/dev/null 2>&1; } +physical() { local d; for d in /sys/class/net/*; do [[ -e "$d/device" ]] && printf '%s ' "${d##*/}"; done; } +out_before=no; gateway_echo 3 && out_before=yes +before="$(grep -hc 'netzone: READY' /run/uncaught-logs/current 2>/dev/null)"; before="${before:-0}" +s6-svc -d /run/service/net-zone +returned="" +for _ in $(seq 1 20); do returned="$(ip -o link show eth0 2>/dev/null)"; [[ -n "$returned" ]] && break; sleep 0.5; done +flags="$(sed -n 's/^[0-9]*: eth0: <\([^>]*\)>.*/\1/p' <<<"$returned")" +held="$(ip -o addr show eth0 2>/dev/null | awk '{ print $4 }' | tr '\n' ' ')" +if [[ -n "$flags" && ",$flags," != *",UP,"* && -z "$held" && "$(physical)" == "eth0 " ]]; then + pass "uplink-returned" "while the net zone is down the uplink is back in zone 0 as eth0, down and with no address" +else + fail "uplink-returned" "zone 0 while the net zone is down: ${returned:-no eth0}; addresses: ${held:-none}; physical interfaces: $(physical)" +fi +eth0_gone=no; [[ -n "$upid" ]] && ! nsenter -t "$upid" -n ip -o link show eth0 >/dev/null 2>&1 && eth0_gone=yes +out_down=no; gateway_echo 2 && out_down=yes +s6-svc -u /run/service/net-zone +ok=0 +for _ in $(seq 1 60); do + after="$(grep -hc 'netzone: READY' /run/uncaught-logs/current 2>/dev/null)"; after="${after:-0}" + [[ "$after" -gt "$before" ]] && { ok=1; break; }; sleep 1 +done +if [[ "$ok" = 1 ]] && ! ip link show eth0 >/dev/null 2>&1; then pass "uplink-retaken" "the next net zone start took eth0 from zone 0 again and came READY"; else fail "uplink-retaken" "READY again: $ok; zone 0 still holds: $(physical)"; fi +sleep 2 +out_after=no; gateway_echo 3 && out_after=yes +if [[ "$out_before" = yes && "$eth0_gone" = yes && "$out_down" = no && "$out_after" = yes ]]; then + pass "reattach-egress" "untrusted, running across a restart, reached the VM gateway before it, had no eth0 and no path while the net zone was down, and reaches the gateway again once reattached" +else + fail "reattach-egress" "untrusted (init ${upid:-none}): gateway before ${out_before}; eth0 gone while down ${eth0_gone}; gateway while down ${out_down}; gateway after ${out_after}; $(tail -2 "$LOG/untrusted-bg.out" | tr '\n' ' ')" +fi +"$KD" stop untrusted >/dev/null 2>&1; wait "$UBG" 2>/dev/null # --- zones: the net zone over a radio ----------------------------------------------- # QEMU has no radio, so mac80211_hwsim makes two. phy1 goes into a network diff --git a/docs/design/net-zone.md b/docs/design/net-zone.md index fea42b14..129b5965 100755 --- a/docs/design/net-zone.md +++ b/docs/design/net-zone.md @@ -206,7 +206,9 @@ changes. one started before any net zone gets a path but no resolver until it restarts; kryptikd does not edit a running zone's sealed root. - The kernel returns the physical interface to the initial namespace, down - and unaddressed; kryptikd leaves it so until the next net zone start. + and unaddressed, under its own name (`dev` only if zone 0 has an + interface by that name by then); kryptikd leaves it so until the next net + zone start, which takes it again whatever its name. ## What this guarantees @@ -257,8 +259,10 @@ changes. - `build/guest-tests/zones-check.sh` on the installed system checks every guarantee above under QEMU user networking: the net zone `READY`, zone 0 offline, a routed zone's address, NAT, ULA-only IPv6 and resolver, zones - separated, `vault` offline, no egress while the net zone is down, - reattachment after a restart, a zone without `local` refused the VM + separated, `vault` offline, no egress while the net zone is down, the + uplink back in zone 0 under its own name, down and with no address until + the next start takes it, a zone running across a restart going out through + the gateway again once reattached, a zone without `local` refused the VM gateway, and, on two `mac80211_hwsim` radios, the net zone associating, leasing and routing over one while the other is the access point, whose own address that zone is refused while it reaches an address the access point diff --git a/tools/image/zones-test.sh b/tools/image/zones-test.sh index 2965a1cf..b6ec9a62 100755 --- a/tools/image/zones-test.sh +++ b/tools/image/zones-test.sh @@ -65,7 +65,7 @@ zp="$(sed -n 's/.*passed=\([0-9]*\).*/\1/p' <<<"$summary")"; zf="$(sed -n 's/.*f if [[ -n "$summary" && "${zf:-1}" -eq 0 && "${zp:-0}" -ge 30 ]]; then green "every guest check passed (${zp})"; else red "guest checks: ${zp:-0} passed, ${zf:-?} failed"; fi grep 'ZT FAIL' <<<"$T2" | sed 's/^/ /' # The key verdicts one by one, so a pass is not a single line. -for name in kernel-support policies net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns routed-ipv6-noglobal zone-separation volume-hidden home-hidden fail-closed net-restart-ready reattach-after-restart uplink-refused wifi-beyond \ +for name in kernel-support policies net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns routed-ipv6-noglobal zone-separation volume-hidden home-hidden fail-closed net-restart-ready reattach-after-restart uplink-returned uplink-retaken reattach-egress uplink-refused wifi-beyond \ wifi-module wifi-ap wifi-add wifi-associated wifi-lease wifi-egress wifi-forget \ time-floor-ran time-clamp time-floor-forged time-claim-stepped time-claim-floor time-claim-consent pids-limit ephemeral-size-bound cpu-max-set lifecycle-repeat lifecycle-registry \ terminal-terminfo man-page text-browser tls-trust \ From a03c1ea7913f79af4f38c80e28d7958cf2142102 Mon Sep 17 00:00:00 2001 From: DevomB Date: Tue, 6 Oct 2026 21:25:35 -0700 Subject: [PATCH 05/18] From the bridge the net zone takes in only what is addressed to the bridge, so no zone reaches it on its uplink addresses MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A zone's packet to the net zone's own address on an uplink (10.0.2.15 under QEMU, a café's 192.168.x.y on a real one) is local delivery, so the forward chain's rule on the uplinks' networks never saw it, and the input chain accepted it: every zone, local or not, reached whatever the net zone listens on there, such as dhcpcd. The input chain now drops what comes in from kryptik0 unless it is addressed to 10.19.0.1, fd19::1, a link-local or a link-scope multicast address. The guest check uplink-address-refused has untrusted reach the VM gateway and be refused the net zone's addresses beside it; netzone-uplink.sh checks the input rules. A review asked the same of a router's admin page on its WAN address. The net zone cannot know that address, so no rule can refuse it; the design doc says so where it lists what the rule does not cover. --- build/guest-tests/zones-check.sh | 14 ++++++++++++++ docs/design/net-zone.md | 22 ++++++++++++++++------ tools/image/zones-test.sh | 2 +- tools/net/netzone-init.sh | 6 +++++- tools/tests/netzone-uplink.sh | 3 +++ 5 files changed, 39 insertions(+), 8 deletions(-) diff --git a/build/guest-tests/zones-check.sh b/build/guest-tests/zones-check.sh index 4de1c784..14c91883 100755 --- a/build/guest-tests/zones-check.sh +++ b/build/guest-tests/zones-check.sh @@ -114,6 +114,20 @@ if [[ -n "$ppid" ]] && nsenter -t "$ppid" -n ping -c1 -W3 10.19.0.1 >/dev/null 2 else fail "uplink-refused" "personal (init ${ppid:-none}) reached the VM gateway, or not even the bridge; $(grep -h 'netzone: nftables: zones go out' /run/uncaught-logs/current 2>/dev/null | tail -1)" fi +# The net zone's own address on the uplink is the net zone, not the network the +# uplink sits on. untrusted, which may reach that network, reaches the VM +# gateway and neither of the net zone's addresses beside it. +nz="$(cut -d' ' -f1 /run/kryptik/zones/net/init.pid 2>/dev/null)" +uplink4="$([[ -n "$nz" ]] && nsenter -t "$nz" -n ip -4 -o addr show eth0 2>/dev/null | awk '{ split($4, a, "/"); print a[1]; exit }')" +uplink6="$([[ -n "$nz" ]] && nsenter -t "$nz" -n ip -6 -o addr show eth0 2>/dev/null | awk '$4 !~ /^fe80:/ { split($4, a, "/"); print a[1]; exit }')" +zrun untrusted 40 -- sh -c "python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.0.2.2 3 >/dev/null 2>&1 && echo GATEWAY-OK +python3 /usr/lib/kryptik/guest-tests/icmp-echo.py ${uplink4:-192.0.2.1} 3 >/dev/null 2>&1 && echo UPLINK4-REACHED || echo UPLINK4-REFUSED +[ -z '${uplink6}' ] || { python3 /usr/lib/kryptik/guest-tests/icmp-echo.py '${uplink6}' 3 >/dev/null 2>&1 && echo UPLINK6-REACHED || echo UPLINK6-REFUSED; }" +if [[ -n "$uplink4" && "$ZOUT" == *GATEWAY-OK* && "$ZOUT" == *UPLINK4-REFUSED* && "$ZOUT" != *UPLINK6-REACHED* ]]; then + pass "uplink-address-refused" "untrusted reaches the VM gateway and not the net zone's own uplink address ${uplink4}${uplink6:+ or ${uplink6}}" +else + fail "uplink-address-refused" "the net zone's uplink addresses: ${uplink4:-none} ${uplink6:-none}; untrusted: $(tr '\n' ' ' <<<"$ZOUT")" +fi # net zone restart: routed zones fail closed while it is down, recover after # Not `|| echo 0`: grep -c prints 0 and also exits 1. diff --git a/docs/design/net-zone.md b/docs/design/net-zone.md index fea42b14..93539394 100755 --- a/docs/design/net-zone.md +++ b/docs/design/net-zone.md @@ -137,11 +137,20 @@ its definition says `[network] local = true`. can still address the router and the machines beside it, as every zone could before; without the key there, a network with a login page is no network at all. +- **The net zone's own address on an uplink is not that network.** It is the + net zone, which a zone needs only for its resolver on the bridge. From the + bridge the input chain takes only what is addressed to `10.19.0.1` or + `fd19::1`, or to a link-local or link-scope multicast address for neighbour + discovery, so no zone, `local` or not, reaches what the net zone listens on + over its uplink addresses, such as dhcpcd. - **What it does not cover.** A network behind the gateway, such as a modem's - own pages on another subnet, is past the gateway and so allowed. An uplink - whose default route names no gateway, a point-to-point link, carries only - the zones that claim `local`. The net zone itself reaches the local - network, as DHCP and the resolver need. + own pages on another subnet, is past the gateway and so allowed. So is the + gateway's address on its far side: a router that answers its admin page on + its WAN address to the machines inside serves it to every zone, and the net + zone cannot know that address to refuse it. An uplink whose default route + names no gateway, a point-to-point link, carries only the zones that claim + `local`. The net zone itself reaches the local network, as DHCP and the + resolver need. ## DNS @@ -249,7 +258,7 @@ changes. file and `kryptik wifi`. - `tools/tests/netzone-uplink.sh`: the zones a definition lets through, by the address kryptikd derives for each, the gateway sets as nft is fed them, and - the order of the forward rules. + the order of the forward and input rules. - The launcher suite reads a zone's bounding set (exactly `0x400`) and the boundary suite asks for an `AF_PACKET` socket. The launcher suite's routed-networking section runs only with `KRYPTIK_VM_DISPOSABLE=1`, since @@ -259,7 +268,8 @@ changes. offline, a routed zone's address, NAT, ULA-only IPv6 and resolver, zones separated, `vault` offline, no egress while the net zone is down, reattachment after a restart, a zone without `local` refused the VM - gateway, and, on two `mac80211_hwsim` radios, the net zone associating, + gateway, `untrusted` refused the net zone's own uplink addresses while it + reaches the gateway, and, on two `mac80211_hwsim` radios, the net zone associating, leasing and routing over one while the other is the access point, whose own address that zone is refused while it reaches an address the access point routes. It pings with an unprivileged ICMP socket diff --git a/tools/image/zones-test.sh b/tools/image/zones-test.sh index 2965a1cf..b53a0058 100755 --- a/tools/image/zones-test.sh +++ b/tools/image/zones-test.sh @@ -65,7 +65,7 @@ zp="$(sed -n 's/.*passed=\([0-9]*\).*/\1/p' <<<"$summary")"; zf="$(sed -n 's/.*f if [[ -n "$summary" && "${zf:-1}" -eq 0 && "${zp:-0}" -ge 30 ]]; then green "every guest check passed (${zp})"; else red "guest checks: ${zp:-0} passed, ${zf:-?} failed"; fi grep 'ZT FAIL' <<<"$T2" | sed 's/^/ /' # The key verdicts one by one, so a pass is not a single line. -for name in kernel-support policies net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns routed-ipv6-noglobal zone-separation volume-hidden home-hidden fail-closed net-restart-ready reattach-after-restart uplink-refused wifi-beyond \ +for name in kernel-support policies net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns routed-ipv6-noglobal zone-separation volume-hidden home-hidden fail-closed net-restart-ready reattach-after-restart uplink-refused uplink-address-refused wifi-beyond \ wifi-module wifi-ap wifi-add wifi-associated wifi-lease wifi-egress wifi-forget \ time-floor-ran time-clamp time-floor-forged time-claim-stepped time-claim-floor time-claim-consent pids-limit ephemeral-size-bound cpu-max-set lifecycle-repeat lifecycle-registry \ terminal-terminfo man-page text-browser tls-trust \ diff --git a/tools/net/netzone-init.sh b/tools/net/netzone-init.sh index da04c6c9..88704c98 100755 --- a/tools/net/netzone-init.sh +++ b/tools/net/netzone-init.sh @@ -80,7 +80,9 @@ SET4="set local4 { type ipv4_addr; }"; SET6="set local6 { type ipv6_addr; }" # A zone goes out by a gateway (gw4, gw6) and never to the gateway itself: # the rest of what an uplink reaches is the network it sits on, open to local4 # and local6 alone. With no gateway in the sets nothing goes out, so a new -# lease opens no way in before sync_gateways has seen it. +# lease opens no way in before sync_gateways has seen it. From the bridge the +# net zone takes in only what is addressed to the bridge: its own address on an +# uplink is the net zone, not the network a local zone may reach. RULES="table inet kryptik { set gw4 { type ipv4_addr; } set gw6 { type ipv6_addr; } @@ -105,6 +107,8 @@ RULES="table inet kryptik { type filter hook input priority filter; policy accept; iifname ${NICSET} ct state new tcp dport 53 drop iifname ${NICSET} ct state new udp dport 53 drop + iifname \"${BR}\" ip daddr != 10.19.0.1 drop + iifname \"${BR}\" ip6 daddr != { fd19::1, fe80::/10, ff02::/16 } drop } }" diff --git a/tools/tests/netzone-uplink.sh b/tools/tests/netzone-uplink.sh index 6095cd2e..d20e49cb 100755 --- a/tools/tests/netzone-uplink.sh +++ b/tools/tests/netzone-uplink.sh @@ -106,6 +106,9 @@ for sh in sh bash dash; do forward="$(sed -n '/chain forward {/,/^ }$/p' <<<"$rules" | grep -oE 'established,related accept|ip6? saddr @local[46] accept|rt ip6? nexthop @gw[46] ip6? daddr != @gw[46] accept|reject with icmpx type admin-prohibited|oifname "kryptik0" drop' | tr '\n' '|')" same "replies first, then the local zones, then what a gateway carries but the gateway itself, then the refusal" "$forward" \ 'established,related accept|ip saddr @local4 accept|ip6 saddr @local6 accept|rt ip nexthop @gw4 ip daddr != @gw4 accept|rt ip6 nexthop @gw6 ip6 daddr != @gw6 accept|reject with icmpx type admin-prohibited|oifname "kryptik0" drop|' + input="$(sed -n '/chain input {/,/^ }$/p' <<<"$rules" | grep -oE 'ct state new (tcp|udp) dport 53 drop|"kryptik0" ip daddr != 10\.19\.0\.1 drop|"kryptik0" ip6 daddr != \{ fd19::1, fe80::/10, ff02::/16 \} drop' | tr '\n' '|')" + same "into the net zone: no resolver for an uplink, and from the bridge only what is addressed to the bridge" "$input" \ + 'ct state new tcp dport 53 drop|ct state new udp dport 53 drop|"kryptik0" ip daddr != 10.19.0.1 drop|"kryptik0" ip6 daddr != { fd19::1, fe80::/10, ff02::/16 } drop|' rules="$(run rules "$T/none")" grep -qF 'set local4 { type ipv4_addr; }' <<<"$rules" && green "with no zone let through the sets are empty, and every zone is refused" || red "the empty local sets" "$(grep 'set local' <<<"$rules" | tr '\n' '|')" done From e9fa8faf5c66dd6f4d90c0d539fb3272a54e9c8d Mon Sep 17 00:00:00 2001 From: DevomB Date: Tue, 6 Oct 2026 21:29:05 -0700 Subject: [PATCH 06/18] zone-source-pinned is listed beside volume-init, where it runs, so the list's first line stays as main has it and the PR merges clean GitHub runs no CI for a pull request that conflicts with main, and this one did, on the verdict list's first line, which main and the DNS branch also change. --- tools/image/zones-test.sh | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tools/image/zones-test.sh b/tools/image/zones-test.sh index 3453d52b..00a822c1 100755 --- a/tools/image/zones-test.sh +++ b/tools/image/zones-test.sh @@ -65,11 +65,11 @@ zp="$(sed -n 's/.*passed=\([0-9]*\).*/\1/p' <<<"$summary")"; zf="$(sed -n 's/.*f if [[ -n "$summary" && "${zf:-1}" -eq 0 && "${zp:-0}" -ge 30 ]]; then green "every guest check passed (${zp})"; else red "guest checks: ${zp:-0} passed, ${zf:-?} failed"; fi grep 'ZT FAIL' <<<"$T2" | sed 's/^/ /' # The key verdicts one by one, so a pass is not a single line. -for name in kernel-support net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns routed-ipv6-noglobal zone-source-pinned zone-separation fail-closed net-restart-ready reattach-after-restart uplink-refused wifi-beyond \ +for name in kernel-support net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns routed-ipv6-noglobal zone-separation fail-closed net-restart-ready reattach-after-restart uplink-refused wifi-beyond \ wifi-module wifi-ap wifi-add wifi-associated wifi-lease wifi-egress wifi-forget \ time-floor-ran time-clamp time-claim-stepped time-claim-floor time-claim-consent pids-limit ephemeral-size-bound cpu-max-set \ terminal-terminfo man-page text-browser tls-trust \ - volume-init encrypted-zone-start stop-closes-volume wrong-passphrase persist-reopen no-mapping-after ephemeral-gone concurrent-start-refused full-volume header-restore volume-destroy vault-offline vault-ping no-passphrase-leak \ + volume-init zone-source-pinned encrypted-zone-start stop-closes-volume wrong-passphrase persist-reopen no-mapping-after ephemeral-gone concurrent-start-refused full-volume header-restore volume-destroy vault-offline vault-ping no-passphrase-leak \ setuid-only-allowed no-file-capabilities sysctls-applied; do grep -q "ZT PASS ${name}" <<<"$T2" && green "guest: ${name}" || red "guest: ${name} (not passed)" done From d9585e008f26e596e7b1337da50aa4ada62f27ea Mon Sep 17 00:00:00 2001 From: DevomB Date: Tue, 6 Oct 2026 22:04:57 -0700 Subject: [PATCH 07/18] The zones suite reads the catch-all log across its archives, so a rotation cannot undo a count of READY lines or hide the newest one s6-log moves /run/uncaught-logs/current aside at about 100 KB. The restart checks counted READY lines in current alone, so a rotation during their wait made the count fall below the one taken before and the check fail with nothing wrong; and the newest READY line was read from current first and the archives after, so tail -1 gave an archived line once one existed. One set of helpers now reads the archives, previous and current, oldest first, and every count, newest line and diagnostic in the suite goes through it. Found by linux-distro-a2's read of #230. --- build/guest-tests/zones-check.sh | 39 +++++++++++++++++--------------- 1 file changed, 21 insertions(+), 18 deletions(-) diff --git a/build/guest-tests/zones-check.sh b/build/guest-tests/zones-check.sh index b90dcb92..04bdd753 100755 --- a/build/guest-tests/zones-check.sh +++ b/build/guest-tests/zones-check.sh @@ -24,6 +24,12 @@ zrun() { # zrun ZONE TIMEOUT [--passphrase-file F] -- CMD... ZOUT="$(cat "$LOG/$zone.out")" } host_of() { local b; b="$(sed -n 's/^uid_base *= *\([0-9]*\).*/\1/p' "$Z/$1.toml")"; echo $(( (b - 131072) / 65536 + 2 )); } +# The catch-all log, oldest first: s6-log moves current aside, through previous +# to @.s, at about 100 KB, so a count read from current alone can go back. +uncaught() { cat /run/uncaught-logs/@* /run/uncaught-logs/previous /run/uncaught-logs/current 2>/dev/null; } +netzone_said() { uncaught | grep -a "netzone: $1"; } +ready_count() { netzone_said READY | grep -c .; } +last_ready() { netzone_said READY | tail -1; } [[ "$(id -u)" = 0 ]] || { fail "root" "this must run as root"; echo "ZT END"; exit 1; } echo "ZT BEGIN $(date -Iseconds 2>/dev/null)" @@ -41,12 +47,12 @@ zones="$("$KD" list --zones "$Z" 2>/dev/null | tr '\n' ' ')" if [[ "$(s6-svstat -o up /run/service/net-zone 2>/dev/null)" = true ]]; then pass "net-zone-up" "supervised and up"; else fail "net-zone-up" "$(s6-svstat /run/service/net-zone 2>&1)"; fi ready="" for _ in $(seq 1 30); do - ready="$(grep -h 'netzone: READY' /run/uncaught-logs/current /run/uncaught-logs/@* 2>/dev/null | tail -1)" + ready="$(last_ready)" [[ -n "$ready" ]] && break; sleep 1 done -if [[ "$ready" == *"nat=yes"* ]]; then pass "net-ready" "$ready"; else fail "net-ready" "no READY line with nat=yes in the catch-all log (last: $(grep -h 'netzone:' /run/uncaught-logs/current 2>/dev/null | tail -1))"; fi +if [[ "$ready" == *"nat=yes"* ]]; then pass "net-ready" "$ready"; else fail "net-ready" "no READY line with nat=yes in the catch-all log (last: $(netzone_said '' | tail -1))"; fi # The routed zones' resolver, named on its own: routed-dns only times out. -if [[ "$ready" == *" dns=yes "* ]]; then pass "net-dns" "dnsmasq is running"; else fail "net-dns" "$(grep -h 'dnsmasq' /run/uncaught-logs/current 2>/dev/null | tail -2 | tr '\n' ' ')"; fi +if [[ "$ready" == *" dns=yes "* ]]; then pass "net-dns" "dnsmasq is running"; else fail "net-dns" "$(uncaught | grep -a 'dnsmasq' | tail -2 | tr '\n' ' ')"; fi if ip link show eth0 >/dev/null 2>&1; then fail "zone0-nic" "eth0 is still in zone 0"; else pass "zone0-nic" "eth0 is not in zone 0 (moved into the net zone)"; fi if [[ -z "$(ip route show default 2>/dev/null)" ]]; then pass "zone0-no-route" "zone 0 has no default route"; else fail "zone0-no-route" "$(ip route show default)"; fi if ping -c1 -W2 10.0.2.2 >/dev/null 2>&1; then fail "zone0-offline" "zone 0 reached the VM gateway"; else pass "zone0-offline" "zone 0 cannot reach the VM gateway"; fi @@ -112,19 +118,18 @@ if [[ -n "$ppid" ]] && nsenter -t "$ppid" -n ping -c1 -W3 10.19.0.1 >/dev/null 2 && ! nsenter -t "$ppid" -n ping -c1 -W3 10.0.2.2 >/dev/null 2>&1; then pass "uplink-refused" "personal reaches the bridge and is refused the VM gateway, which untrusted reached" else - fail "uplink-refused" "personal (init ${ppid:-none}) reached the VM gateway, or not even the bridge; $(grep -h 'netzone: nftables: zones go out' /run/uncaught-logs/current 2>/dev/null | tail -1)" + fail "uplink-refused" "personal (init ${ppid:-none}) reached the VM gateway, or not even the bridge; $(netzone_said 'nftables: zones go out' | tail -1)" fi # net zone restart: routed zones fail closed while it is down, recover after -# Not `|| echo 0`: grep -c prints 0 and also exits 1. -before="$(grep -hc 'netzone: READY' /run/uncaught-logs/current 2>/dev/null)"; before="${before:-0}" +before="$(ready_count)" s6-svc -d /run/service/net-zone; sleep 3 zrun untrusted 20 -- sh -c 'python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.0.2.2 2 >/dev/null 2>&1 && echo EGRESS-WHILE-DOWN || echo CLOSED-WHILE-DOWN; ip -o link show eth0 >/dev/null 2>&1 && echo HAS-ETH0 || echo NO-ETH0' [[ "$ZOUT" == *CLOSED-WHILE-DOWN* ]] && pass "fail-closed" "no egress while the net zone is down ($(grep -o 'HAS-ETH0\|NO-ETH0' "$LOG/untrusted.out" | head -1))" || fail "fail-closed" "$ZOUT" s6-svc -u /run/service/net-zone ok=0 for _ in $(seq 1 60); do - after="$(grep -hc 'netzone: READY' /run/uncaught-logs/current 2>/dev/null)"; after="${after:-0}" + after="$(ready_count)" [[ "$after" -gt "$before" ]] && { ok=1; break; }; sleep 1 done [[ "$ok" = 1 ]] && pass "net-restart-ready" "the net zone came back READY after a restart" || fail "net-restart-ready" "no new READY line ($before -> $after)" @@ -146,7 +151,7 @@ upid="$(cut -d' ' -f1 /run/kryptik/zones/untrusted/init.pid 2>/dev/null)" gateway_echo() { [[ -n "$upid" ]] && nsenter -t "$upid" -n python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.0.2.2 "$1" >/dev/null 2>&1; } physical() { local d; for d in /sys/class/net/*; do [[ -e "$d/device" ]] && printf '%s ' "${d##*/}"; done; } out_before=no; gateway_echo 3 && out_before=yes -before="$(grep -hc 'netzone: READY' /run/uncaught-logs/current 2>/dev/null)"; before="${before:-0}" +before="$(ready_count)" s6-svc -d /run/service/net-zone returned="" for _ in $(seq 1 20); do returned="$(ip -o link show eth0 2>/dev/null)"; [[ -n "$returned" ]] && break; sleep 0.5; done @@ -162,7 +167,7 @@ out_down=no; gateway_echo 2 && out_down=yes s6-svc -u /run/service/net-zone ok=0 for _ in $(seq 1 60); do - after="$(grep -hc 'netzone: READY' /run/uncaught-logs/current 2>/dev/null)"; after="${after:-0}" + after="$(ready_count)" [[ "$after" -gt "$before" ]] && { ok=1; break; }; sleep 1 done if [[ "$ok" = 1 ]] && ! ip link show eth0 >/dev/null 2>&1; then pass "uplink-retaken" "the next net zone start took eth0 from zone 0 again and came READY"; else fail "uplink-retaken" "READY again: $ok; zone 0 still holds: $(physical)"; fi @@ -185,8 +190,6 @@ AP_SSID=kryptik-hwsim; AP_PASS=hwsim-passphrase; AP_ADDR=192.168.77.1 # An address the access point routes to, past the network the radio is on. AP_FAR=198.51.100.1 WIFI_DIR=/var/lib/kryptik/wifi -ready_count() { local n; n="$(grep -hc 'netzone: READY' /run/uncaught-logs/current 2>/dev/null)"; echo "${n:-0}"; } -last_ready() { grep -h 'netzone: READY' /run/uncaught-logs/current /run/uncaught-logs/@* 2>/dev/null | tail -1; } ready_after() { # ready_after COUNT TEXT SECONDS: the newest READY line once there are more than COUNT and it holds TEXT local before="$1" text="$2" n="$3" while [[ "$n" -gt 0 ]]; do @@ -262,7 +265,7 @@ kills="$(dmesg 2>/dev/null | grep -a 'type=1326' | grep -ac 'comm="wpa_supplican if [[ "$line" == *" wifi=$AP_SSID "* && "${kills:-0}" = 0 ]]; then pass "wifi-associated" "the net zone's supplicant joined $AP_SSID over $STA_IF with no filter kill: ${line#*netzone: }" else - fail "wifi-associated" "newest READY line: ${line:-none}; filter kills of wpa_supplicant: ${kills:-0}; $(grep -h 'netzone: wifi' /run/uncaught-logs/current 2>/dev/null | tail -3 | tr '\n' ' ')" + fail "wifi-associated" "newest READY line: ${line:-none}; filter kills of wpa_supplicant: ${kills:-0}; $(netzone_said wifi | tail -3 | tr '\n' ' ')" fi net_init="$(cut -d' ' -f1 /run/kryptik/zones/net/init.pid 2>/dev/null)" lease="" @@ -283,7 +286,7 @@ fi # personal, and what lies past it is not. ppid="$(cut -d' ' -f1 /run/kryptik/zones/personal/init.pid 2>/dev/null)" for _ in $(seq 1 30); do - grep -h 'netzone: nftables: zones go out' /run/uncaught-logs/current 2>/dev/null | tail -1 | grep -q "$AP_ADDR" && break; sleep 1 + netzone_said 'nftables: zones go out' | tail -1 | grep -q "$AP_ADDR" && break; sleep 1 done far=1; near=0 if [[ -n "$ppid" ]]; then @@ -293,7 +296,7 @@ fi if [[ "$far" -eq 0 && "$near" -ne 0 ]]; then pass "wifi-beyond" "personal is refused the access point's own address and reaches $AP_FAR past it" else - fail "wifi-beyond" "personal (init ${ppid:-none}): $AP_FAR rc=$far, $AP_ADDR rc=$near; $(grep -h 'netzone: nftables: zones go out' /run/uncaught-logs/current 2>/dev/null | tail -1); routes: $(nsenter -t "$net_init" -n ip -4 route 2>/dev/null | tr '\n' ';')" + fail "wifi-beyond" "personal (init ${ppid:-none}): $AP_FAR rc=$far, $AP_ADDR rc=$near; $(netzone_said 'nftables: zones go out' | tail -1); routes: $(nsenter -t "$net_init" -n ip -4 route 2>/dev/null | tr '\n' ';')" fi # Back to the wire: the access point, its namespace (whose end returns phy1 # to zone 0) and the radios go first, so the net zone the forget restarts @@ -334,7 +337,7 @@ print(s.recv(4096).decode("utf-8", "replace").strip())' "$1" 2>&1 | head -1 } if grep -q 'time floor' /var/log/kryptik/time.log 2>/dev/null; then pass "time-floor-ran" "$(tail -1 /var/log/kryptik/time.log | cut -c1-160)"; else fail "time-floor-ran" "the boot service left no line in /var/log/kryptik/time.log"; fi -ready_time="$(grep -h 'netzone: READY' /run/uncaught-logs/current 2>/dev/null | tail -1 | grep -o 'time=[^ ]*')" +ready_time="$(last_ready | grep -o 'time=[^ ]*')" info "time-reported ${ready_time:-the readiness line has no time= field} (no answer through this network is reported as that, never as a pass)" if [[ "$FLOOR_S" -gt 0 ]]; then @@ -383,10 +386,10 @@ if [[ "$FLOOR_S" -gt 0 ]]; then case "$ready_time" in time=-[0-9]*|time=[0-9]*) date -u -s "@$(( $(true_now) + 300 ))" >/dev/null 2>&1; rm -f /var/lib/kryptik/time/state - n0="$(grep -hc 'netzone: READY' /run/uncaught-logs/current 2>/dev/null)"; n0="${n0:-0}" + n0="$(ready_count)" s6-svc -r /run/service/net-zone - for _ in $(seq 1 90); do n1="$(grep -hc 'netzone: READY' /run/uncaught-logs/current 2>/dev/null)"; [[ "${n1:-0}" -gt "$n0" ]] && break; sleep 1; done - m="$(grep -h 'netzone: READY' /run/uncaught-logs/current | tail -1 | grep -o 'time=[^ ]*' | cut -d= -f2)" + for _ in $(seq 1 90); do [[ "$(ready_count)" -gt "$n0" ]] && break; sleep 1; done + m="$(last_ready | grep -o 'time=[^ ]*' | cut -d= -f2)" err=$(( $(date +%s) - $(true_now) )) if [[ "$m" == -29[0-9]* || "$m" == -30[0-9]* || "$m" == -31[0-9]* ]] && (( err > -10 && err < 10 )); then pass "time-sign" "a clock 300 s fast was measured as ${m} s and zone 0 put it right (now ${err} s from true)" From 97b431aeee146fd690324000fcf355fb4c61fa9c Mon Sep 17 00:00:00 2001 From: DevomB Date: Tue, 6 Oct 2026 23:49:53 -0700 Subject: [PATCH 08/18] reattach-egress pings the gateway from untrusted's namespace with ping, whose raw socket root may use there Run 37570910486 failed reattach-egress with "gateway before no" while a fresh untrusted reached the gateway: icmp-echo.py opens only an ICMP datagram socket, and kryptikd sets the zone's ping_group_range to the zone's own gid, so root entering the namespace with nsenter was refused before anything was sent. The checks that already ping from a zone's namespace use ping, which falls back to a raw socket as root; this one does too. uplink-returned and uplink-retaken passed in that run. --- build/guest-tests/zones-check.sh | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/build/guest-tests/zones-check.sh b/build/guest-tests/zones-check.sh index 04bdd753..e84bae04 100755 --- a/build/guest-tests/zones-check.sh +++ b/build/guest-tests/zones-check.sh @@ -148,7 +148,9 @@ setsid "$KD" run untrusted --zones "$Z" --rootfs "$R" -- sh -c 'echo UNTRUSTED-U UBG=$! for _ in $(seq 1 40); do grep -q UNTRUSTED-UP "$LOG/untrusted-bg.out" 2>/dev/null && break; sleep 0.5; done upid="$(cut -d' ' -f1 /run/kryptik/zones/untrusted/init.pid 2>/dev/null)" -gateway_echo() { [[ -n "$upid" ]] && nsenter -t "$upid" -n python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.0.2.2 "$1" >/dev/null 2>&1; } +# ping, not icmp-echo.py: the zone's ping_group_range names its own gid, not +# root's, so root in its namespace needs ping's raw socket. +gateway_echo() { [[ -n "$upid" ]] && nsenter -t "$upid" -n ping -c1 -W"$1" 10.0.2.2 >/dev/null 2>&1; } physical() { local d; for d in /sys/class/net/*; do [[ -e "$d/device" ]] && printf '%s ' "${d##*/}"; done; } out_before=no; gateway_echo 3 && out_before=yes before="$(ready_count)" From 505c928d0ebcbd9eca825845f3910b79084627c1 Mon Sep 17 00:00:00 2001 From: DevomB Date: Wed, 7 Oct 2026 01:25:53 -0700 Subject: [PATCH 09/18] uplink-address-refused warms the path to the bridge before its gateway control, and says why the control failed Run 37571430413 refused both of the net zone's uplink addresses as it should (UPLINK4-REFUSED UPLINK6-REFUSED) but failed on its control: the echo to the VM gateway, the fresh zone's first packet, got no answer within 3 s, while routed-egress and egress-after-restart reached the same gateway in that run. The zone now echoes the bridge first, as routed-egress does, gives the gateway 5 s, and on a miss the verdict carries icmp-echo.py's NOPONG reason, the zone's exit status and the end of its stderr. --- build/guest-tests/zones-check.sh | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/build/guest-tests/zones-check.sh b/build/guest-tests/zones-check.sh index dee04fa7..730f32a9 100755 --- a/build/guest-tests/zones-check.sh +++ b/build/guest-tests/zones-check.sh @@ -167,13 +167,16 @@ fi nz="$(cut -d' ' -f1 /run/kryptik/zones/net/init.pid 2>/dev/null)" uplink4="$([[ -n "$nz" ]] && nsenter -t "$nz" -n ip -4 -o addr show eth0 2>/dev/null | awk '{ split($4, a, "/"); print a[1]; exit }')" uplink6="$([[ -n "$nz" ]] && nsenter -t "$nz" -n ip -6 -o addr show eth0 2>/dev/null | awk '$4 !~ /^fe80:/ { split($4, a, "/"); print a[1]; exit }')" -zrun untrusted 40 -- sh -c "python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.0.2.2 3 >/dev/null 2>&1 && echo GATEWAY-OK +# The bridge first, as routed-egress does, so the gateway's echo is not the +# zone's first packet; its NOPONG line says why if it still fails. +zrun untrusted 40 -- sh -c "python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.19.0.1 3 >/dev/null 2>&1 +python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.0.2.2 5 > /tmp/gw.out 2>&1 && echo GATEWAY-OK || echo \"GATEWAY-NO \$(tail -1 /tmp/gw.out)\" python3 /usr/lib/kryptik/guest-tests/icmp-echo.py ${uplink4:-192.0.2.1} 3 >/dev/null 2>&1 && echo UPLINK4-REACHED || echo UPLINK4-REFUSED [ -z '${uplink6}' ] || { python3 /usr/lib/kryptik/guest-tests/icmp-echo.py '${uplink6}' 3 >/dev/null 2>&1 && echo UPLINK6-REACHED || echo UPLINK6-REFUSED; }" if [[ -n "$uplink4" && "$ZOUT" == *GATEWAY-OK* && "$ZOUT" == *UPLINK4-REFUSED* && "$ZOUT" != *UPLINK6-REACHED* ]]; then pass "uplink-address-refused" "untrusted reaches the VM gateway and not the net zone's own uplink address ${uplink4}${uplink6:+ or ${uplink6}}" else - fail "uplink-address-refused" "the net zone's uplink addresses: ${uplink4:-none} ${uplink6:-none}; untrusted: $(tr '\n' ' ' <<<"$ZOUT")" + fail "uplink-address-refused" "the net zone's uplink addresses: ${uplink4:-none} ${uplink6:-none}; untrusted (rc ${ZRC}): $(tr '\n' ' ' <<<"$ZOUT") $(tail -2 "$LOG/untrusted.err" | tr '\n' ' ')" fi # net zone restart: routed zones fail closed while it is down, recover after From 2f87d80c6190b2d5a069c56f6f5f7f5ec3383563 Mon Sep 17 00:00:00 2001 From: DevomB Date: Wed, 7 Oct 2026 08:37:33 -0700 Subject: [PATCH 10/18] The boot and service scripts' comments are shorter, with their reasons kept: cleanup-held's cut of build/recipes/services.sh and five service scripts, redone on main by hand, comments only From 65a6fbb, taken where the comment it shortens is still main's and the reason survives. Changed from it: boot-success's header names the slot booted from outside (#226), esp_committed is read at every boot with no trial on record, not only on a degraded state, and forget_entries keeps that the committed slot's own entry is a second way to it; testctl keeps why a control disk is read only on a medium and only when signed, and its key list is the thirteen keys main reads; time-floor keeps the newest committed release (#181); sysinit keeps a one-line list of the state's three kinds and why the consent directory is setgid. Dropped: the shortened section rulers (churn) and installer-run's shellcheck line, which predates slot_arg and kbd_arg. --- build/recipes/services.sh | 11 ++---- build/service-scripts/boot-success.sh | 35 +++++----------- build/service-scripts/installer-run.sh | 11 ++---- build/service-scripts/sysinit.sh | 55 ++++++++------------------ build/service-scripts/testctl.sh | 19 ++++----- build/service-scripts/time-floor.sh | 7 ++-- 6 files changed, 45 insertions(+), 93 deletions(-) diff --git a/build/recipes/services.sh b/build/recipes/services.sh index af86e01b..71f85edf 100644 --- a/build/recipes/services.sh +++ b/build/recipes/services.sh @@ -1,15 +1,11 @@ #!/usr/bin/env bash -# services: a stage 04 recipe, sourced by build/stages/04-base-system.sh, -# which runs it in the order its list gives. -# The s6-rc database compiled from build/services, the scripts the services run -# (build/service-scripts) and the sysctl fragments, all on the verified root. +# The s6-rc database, the service scripts and the sysctl fragments, all on the verified root. s_services() { local src="${KRYPTIK_ROOT}/build/services" [[ -d "$src" ]] || { echo "no service source tree at ${src}"; return 1; } - # The scripts live outside the s6-rc source tree: s6-rc-compile reads every - # directory there as a service. + # Outside the s6-rc source tree, where s6-rc-compile reads every directory as a service. local scripts="${KRYPTIK_ROOT}/build/service-scripts" install -d -m 0755 /usr/libexec/kryptik install -m 0755 "$scripts"/*.sh /usr/libexec/kryptik/ @@ -47,8 +43,7 @@ s_services() { echo "--- keyboard layouts ---" awk '!/^#/ && NF { printf "%s ", $1 } END { print "" }' "$table" - # s6-rc-compile will not overwrite: build beside and swap, as a half-written - # database does not boot. + # s6-rc-compile will not overwrite; build beside and swap, as a half-written one does not boot. local dbdir=/usr/lib/kryptik/s6-rc local tmpdb="$dbdir/compiled.new" rm -rf "$tmpdb" diff --git a/build/service-scripts/boot-success.sh b/build/service-scripts/boot-success.sh index 95c30d7c..dfa1f8ce 100755 --- a/build/service-scripts/boot-success.sh +++ b/build/service-scripts/boot-success.sh @@ -1,17 +1,9 @@ #!/bin/sh -# A/B boot-success tracking (docs/design/boot-and-updates.md): decides, late in -# boot, whether the slot that booted is one to keep. -# /var/lib/kryptik/boot/trial holds the slot kryptik-update armed, then armed=0 -# (before BootNext was set) or armed=1; last-result holds the last boot's -# outcome, which also stops the updater re-arming a failed payload. -# A trial slot that passes health() is committed (its kernel becomes -# BOOTX64.EFI). One that fails reboots, and with BootNext spent that lands on -# the committed slot. A committed slot is only reported on, never rebooted. -# A slot that runs with no trial on record and is not the committed one was -# booted from outside: it is reported as uncommitted and rebooted from, once. -# The paths are overridable for tools/tests/boot-success.sh only. +# A/B boot success (docs/design/boot-and-updates.md): commit a healthy trial, reboot a failed one. +# A committed slot is only reported on, never rebooted; a slot booted from outside is rebooted from once. set -u say() { echo "boot-success: $*"; } +# The overrides are for tools/tests/boot-success.sh only. RUN="${KRYPTIK_RUN:-/run/kryptik}" B="${KRYPTIK_BOOT_STATE:-/var/lib/kryptik/boot}" SVC="${KRYPTIK_SERVICE_DIR:-/run/service}" @@ -37,11 +29,12 @@ if [ -z "$slot" ]; then exit 0 fi +# trial: the slot kryptik-update armed, then armed=0 (before BootNext was set) or armed=1. trial=""; armed="" if [ -r "$B/trial" ]; then trial="$(sed -n '1p' "$B/trial")" armed="$(sed -n 's/^armed=//p' "$B/trial" | head -1)" - [ -n "$armed" ] || armed=1 # a record from before the armed= line: assume it was + [ -n "$armed" ] || armed=1 # an older record has no armed= line: assume armed fi # --- the essential-readiness check ------------------------------------------ @@ -94,12 +87,8 @@ commit_slot() { # commit_slot : make BOOTX64.EFI this slot's kernel return "$rc" } -# However a trial ends, remove BootNext and both slots' entries, so the -# firmware boots the disk's own entry (BOOTX64.EFI, the committed slot). A -# firmware re-adds that entry at the end of BootOrder when devices change, so a -# leftover entry for the other slot would win every cold boot. The committed -# slot then gets its own entry back: it boots what BOOTX64.EFI boots, and is a -# second way to it should that one file be lost. +# However a trial ends: a stale slot entry would outrank BOOTX64.EFI at every cold boot, +# and the committed slot's own entry is a second way to it should that file be lost. forget_entries() { # forget_entries COMMITTED-SLOT if ! kryptik-efiboot forget >/dev/null 2>&1; then say "the firmware's Kryptik entries could not be removed; its own boot order may not name the committed slot" @@ -110,8 +99,7 @@ forget_entries() { # forget_entries COMMITTED-SLOT return 0 } -# On a degraded state the trial record is unreadable; the ESP still names the -# committed slot, and any other slot is on trial. +# The committed slot as the ESP names it, read whenever no trial is on record. esp_committed() { e="$(kryptik_part kryptik-esp 2>/dev/null)" && [ -n "$e" ] || return 0 mkdir -p "$ESP_MNT" @@ -160,8 +148,7 @@ if [ -n "$trial" ]; then [ ! -f "$B/trial" ] || mv -f "$B/trial" "$B/trial.failed" sync if ! forget_entries "$(other_slot "$slot")" && [ -n "$unrecorded" ]; then - # With no record of this trial, only removing its entries - # stops the next boot from repeating it. + # No trial record: only removing its entries stops the next boot repeating it. say "not rebooting: with its entries still there the firmware could boot this trial again" elif [ "${KRYPTIK_NO_REBOOT:-0}" = 1 ]; then say "not rebooting (KRYPTIK_NO_REBOOT=1)" @@ -174,14 +161,12 @@ if [ -n "$trial" ]; then fi else if [ "$armed" = 1 ]; then - # BootNext is spent and the old slot is running: the trial did not - # come up. The updater will not re-arm this payload without --retry. + # trial.failed stops kryptik-update re-arming this payload without --retry. say "trial slot $trial did NOT boot; running slot $slot again" result "trial-failed $trial" mv -f "$B/trial" "$B/trial.failed" forget_entries "$slot" else - # The updater stopped before setting BootNext: nothing was tried. say "the arming of slot $trial was interrupted before BootNext was set; nothing was tried" result "arming-interrupted $trial" rm -f "$B/trial" diff --git a/build/service-scripts/installer-run.sh b/build/service-scripts/installer-run.sh index 518b6238..b8f9b253 100755 --- a/build/service-scripts/installer-run.sh +++ b/build/service-scripts/installer-run.sh @@ -1,6 +1,5 @@ #!/bin/sh -# Unattended install (or recovery) for the VM tests: only on an install medium -# whose test control disk asks for it (testctl.sh). Users run kryptik-install. +# Unattended install or recovery for the VM tests, when a test control disk asks (testctl.sh). set -u . /usr/libexec/kryptik/testctl.sh . /usr/libexec/kryptik/esp-records.sh @@ -67,8 +66,7 @@ if [ ! -x /usr/sbin/kryptik-install ]; then exit 0 fi -# An account for the test driver: the installer writes it to the new state -# partition, for kryptik-firstboot to consume once. +# The test driver's account, left on the new state partition for kryptik-firstboot to consume. preseed_args="" pu="$(testctl_get preseed_user)"; ph="$(testctl_get preseed_password_hash)" rh="$(testctl_get preseed_root_hash)" @@ -77,7 +75,7 @@ if [ -n "$pu" ] && [ -n "$ph" ]; then printf 'user=%s\npassword_hash=%s\nroot_password_hash=%s\n' "$pu" "$ph" "$rh" > /run/kryptik/firstboot.preseed preseed_args="--preseed /run/kryptik/firstboot.preseed" fi -# Replacing an old Kryptik disk is asked for by name, here as by a user. +# Replacing an old Kryptik disk takes the same explicit flag a user gives. replace_arg="" [ "$(testctl_get install_replace)" = "1" ] && replace_arg="--replace-kryptik" # install_slot_mib=MIB: a slot size, as a user gives one. @@ -100,8 +98,7 @@ sed 's/^/KRYPTIK_INSTALL: /' "$logf" say "rc=${rc}" if [ "$rc" -eq 0 ]; then - # Check the disk independently of the installer's report, finding each - # partition by label as the boot chain will. + # Check the disk apart from the installer's report, finding partitions by label as boot does. say "verify: table=$(sfdisk -l "$target" 2>/dev/null | grep -c "^${target}")" for lbl in kryptik-esp kryptik-a kryptik-b kryptik-state; do dev="$(blkid -t PARTLABEL="$lbl" -o device 2>/dev/null | grep "^${target}" | head -1)" diff --git a/build/service-scripts/sysinit.sh b/build/service-scripts/sysinit.sh index bb6fa6a7..94dbdbc3 100755 --- a/build/service-scripts/sysinit.sh +++ b/build/service-scripts/sysinit.sh @@ -1,16 +1,13 @@ #!/bin/sh -e -# Early boot: filesystems, the state partition, the /etc overlay, sysctls. -# Idempotent, since s6-rc may run it again after a runlevel change. +# Early boot: filesystems, state, the /etc overlay, sysctls; idempotent, as s6-rc may rerun it. -# kryptik-console holds the getty back until this finishes: it may ask for the -# state passphrase, and two readers on one terminal lose keystrokes. +# kryptik-console holds its getty until this ends, so the passphrase prompt gets every keystroke. echo running > /run/kryptik-sysinit trap 'echo finished > /run/kryptik-sysinit' EXIT [ -r /etc/hostname ] && hostname "$(cat /etc/hostname)" || true -# The only names the /etc overlay's upper layer may carry: accounts (with the -# shadow tools' backups and lock), identity, clock and zone 0's resolver. +# The only names the /etc upper layer may carry: accounts, identity, clock, zone 0's resolver. ETC_MUTABLE="passwd shadow group gshadow subuid subgid passwd- shadow- group- gshadow- subuid- subgid- .pwd.lock hostname machine-id localtime adjtime resolv.conf" prune_etc_upper() { # prune_etc_upper UPPER QUARANTINE up="$1"; q="$2"; moved=0 @@ -25,9 +22,9 @@ prune_etc_upper() { # prune_etc_upper UPPER QUARANTINE name="${e##*/}" keep=0 for k in $ETC_MUTABLE; do [ "$name" = "$k" ] && keep=1; done - # Accounts must be regular files, not FIFOs or links into mutable - # state. localtime may point only into the verified zoneinfo tree. + # A kept name must be a regular file, not a FIFO or a link into mutable state. if [ "$keep" = 1 ] && [ -f "$e" ] && [ ! -L "$e" ]; then continue; fi + # localtime may link only into the verified zoneinfo tree. if [ "$name" = localtime ] && [ -L "$e" ] && [ -f "$e" ]; then case "$(realpath -e -- "$e")" in /usr/share/zoneinfo/*) continue ;; esac fi @@ -41,13 +38,12 @@ prune_etc_upper() { # prune_etc_upper UPPER QUARANTINE return 0 } -# Up to three passphrase prompts, on every console (ask.sh). printf is a -# builtin: the passphrase never appears as an argument. . /usr/libexec/kryptik/ask.sh unlock_state() { # unlock_state DEVICE -> /dev/mapper/kryptik-state try=1 while [ "$try" -le 3 ] && [ ! -b /dev/mapper/kryptik-state ]; do pass=$(ask -s 0 "sysinit: passphrase for the state partition (try $try of 3): ") || pass="" + # printf is a builtin: the passphrase never appears as an argument. printf '%s' "$pass" | cryptsetup open --type luks2 --key-file=- "$1" kryptik-state 2>/dev/null || true try=$((try + 1)) done @@ -70,12 +66,10 @@ keyboard_at_boot() { return 0 } -# The kernel mounts devtmpfs (CONFIG_DEVTMPFS_MOUNT=y); stage 2 init may -# already have mounted the rest. +# devtmpfs is the kernel's (CONFIG_DEVTMPFS_MOUNT=y); stage 2 init may have mounted the rest. mountpoint -q /proc || mount -t proc proc /proc -o nosuid,noexec,nodev mountpoint -q /sys || mount -t sysfs sysfs /sys -o nosuid,noexec,nodev -# Neither securityfs nor cgroup2 may abort this `sh -e` script: s6-rc would -# then start no services at all. +# Neither securityfs nor cgroup2 may abort this `sh -e` script, or s6-rc starts no services. if ! mountpoint -q /sys/kernel/security 2>/dev/null; then if mount -t securityfs securityfs /sys/kernel/security \ -o nosuid,noexec,nodev 2>/dev/null; then @@ -101,8 +95,7 @@ fi mkdir -p /dev/pts /dev/shm mountpoint -q /dev/pts || mount -t devpts devpts /dev/pts -o gid=5,mode=620,nosuid,noexec mountpoint -q /dev/shm || mount -t tmpfs tmpfs /dev/shm -o nosuid,nodev -# efivarfs: the A/B trial (boot-success, kryptik-update) uses Boot#### and -# BootNext. boot-success reports a non-UEFI boot. +# efivarfs, for the A/B trial's Boot#### and BootNext (boot-success, kryptik-update). if [ -d /sys/firmware/efi/efivars ] && ! mountpoint -q /sys/firmware/efi/efivars; then mount -t efivarfs efivarfs /sys/firmware/efi/efivars -o nosuid,noexec,nodev 2>/dev/null \ || echo "sysinit: efivarfs did not mount" >&2 @@ -119,14 +112,8 @@ done # A medium keeps the kernel's layout; a second run of this script asks nothing. [ -n "$media" ] || mountpoint -q /var || keyboard_at_boot -# --- persistent state -------------------------------------------------------- -# The root is read-only; what changes lives on the kryptik-state partition of -# the root's own disk (devices.sh), seeded once from the image's /var. -# persistent the state partition is mounted at /var -# tmpfs install medium: nothing persists -# degraded installed, but the state partition cannot be used: /var is a -# tmpfs for repair, and first boot, the session, the update -# commit and the updater refuse (/run/kryptik/state-degraded) +# --- persistent state, on the root disk's kryptik-state partition ----------- +# STATE: persistent (mounted on /var), tmpfs (a medium) or degraded (unusable; /var is a tmpfs). . /usr/libexec/kryptik/devices.sh state_mnt=/run/kryptik/state STATE=""; STATE_REASON=""; state_dev=""; root_disk="" @@ -167,6 +154,7 @@ if ! mountpoint -q /var; then mount -t tmpfs -o nosuid,nodev,mode=0755 tmpfs "$state_mnt" fi if [ "$STATE" = degraded ]; then + # First boot, the session, the update commit and the updater refuse while this exists. printf '%s\n' "$STATE_REASON" > /run/kryptik/state-degraded tell "" \ "sysinit: ******************************************************************" \ @@ -191,9 +179,7 @@ if ! mountpoint -q /var; then chmod 0755 "$state_mnt/home" date -Iseconds > "$state_mnt/.kryptik-state" 2>/dev/null || : > "$state_mnt/.kryptik-state" fi - # State is not authenticated, and root honours files under /etc unasked - # (ld.so.preload, nsswitch.conf, udev rules, login configuration), so the - # upper layer is pruned to ETC_MUTABLE before the overlay is mounted. + # State is unauthenticated, and root obeys /etc unasked (ld.so.preload, nsswitch.conf, udev). prune_etc_upper "$state_mnt/lib/kryptik/etc/upper" "$state_mnt/lib/kryptik/etc/quarantine" || { echo "sysinit: refusing to boot with an unsafe /etc upper layer; recover from the install medium" >&2 exit 1 @@ -204,13 +190,7 @@ else fi rmdir "$state_mnt" 2>/dev/null || true -# /etc as an overlay: the verified root's /etc under the machine's changes on -# state. The upper layer is not authenticated, so nothing deciding privilege or -# trust is read from /etc; those come from the verified root: -# init, services /usr/lib/s6-linux-init, /usr/lib/kryptik/s6-rc -# sysctls /usr/lib/kryptik/sysctl.d -# zones /usr/lib/kryptik/zones -# release anchor /usr/share/kryptik/trust +# Nothing deciding privilege or trust is read from /etc: its upper layer is unauthenticated. if ! mountpoint -q /etc; then mkdir -p /var/lib/kryptik/etc/upper /var/lib/kryptik/etc/work if mount -t overlay overlay \ @@ -230,9 +210,8 @@ mountpoint -q /tmp || mount -t tmpfs -o nosuid,nodev,mode=1777 tmpfs /tmp mkdir -p /run/kryptik /run/lock /var/log/kryptik /var/lib/kryptik/boot chmod 0700 /run/kryptik chmod 0755 /run/lock /var/log/kryptik -# Transfer consent (kryptikd consent.rs): broker questions, answers from the -# desktop session (group kryptik). Not under the 0700 /run/kryptik; no zone has -# a path here. Setgid so the session's answers belong to the group. +# Consent questions (consent.rs) for the session's group kryptik, setgid so its answers stay +# the group's; no zone has a path here. mkdir -p /run/kryptik-consent chown root:kryptik /run/kryptik-consent 2>/dev/null || true chmod 2770 /run/kryptik-consent @@ -242,7 +221,7 @@ printf 'slot=%s\nmedia=%s\nstate=%s\nstate_dev=%s\nroot_disk=%s\n' \ "$slot" "$media" "$STATE" "$state_dev" "$root_disk" > /run/kryptik/boot-identity echo "sysinit: booted slot='${slot}' media='${media}' state=${STATE}${state_dev:+ (${state_dev})}" -# Kernel tunables, from the verified root only. Failures are reported. +# Kernel tunables, from the verified root only. if [ -d /usr/lib/kryptik/sysctl.d ]; then for f in /usr/lib/kryptik/sysctl.d/*.conf; do [ -r "$f" ] || continue diff --git a/build/service-scripts/testctl.sh b/build/service-scripts/testctl.sh index f7bd03e2..df26f3f5 100755 --- a/build/service-scripts/testctl.sh +++ b/build/service-scripts/testctl.sh @@ -1,22 +1,19 @@ #!/bin/sh -# Test control for install media, sourced by boot-time services. A disk -# labelled kryptik-testctl carries a key=value file, read only when booted from -# an install medium (kryptik.media=), so such a disk cannot reinstall or shut -# down an installed system, and only when the kryptik-testctl key the medium's -# anchor lists signed it, so no one else's disk can arm an install either. +# Test control, sourced by boot services: key=value from a disk labelled kryptik-testctl. +# Read only on an install medium, so the disk cannot reinstall or shut down an installed +# system, and only when the anchor's kryptik-testctl key signed it, so no other disk can arm one. # testctl_load 0, with TESTCTL_FILE set, when a control file was read # testctl_get KEY the value, or empty -# Keys: install_target=/dev/vdb smoke_poweroff=1 preseed_user=NAME -# preseed_password_hash=HASH preseed_root_hash=HASH install_wait=SECONDS -# recover_disk=/dev/vda recover_slot=a|b recover_mode=restore|commit|header|status +# Keys: install_target=/dev/vdb install_replace=1 install_slot_mib=MIB install_keyboard=NAME +# install_wait=SECONDS state_passphrase=TEXT preseed_user=NAME preseed_password_hash=HASH +# preseed_root_hash=HASH smoke_poweroff=1 recover_disk=/dev/vda recover_slot=a|b +# recover_mode=restore|commit|header|status TESTCTL_MNT=/run/kryptik/testctl TESTCTL_FILE="" TESTCTL_ANCHOR="${TESTCTL_ANCHOR:-/usr/share/kryptik/trust/release-signers}" -# The file and a signature over it by the kryptik-testctl key the anchor -# lists, in that key's own namespace: the holder of that key alone can arm an -# install on a machine that boots this medium. +# FILE.sig must verify with the anchor's kryptik-testctl key, in that key's own namespace. testctl_signed() { # testctl_signed FILE [ -r "$1" ] && [ -r "$1.sig" ] || return 1 ssh-keygen -Y verify -f "$TESTCTL_ANCHOR" -I kryptik-testctl -n kryptik-testctl \ diff --git a/build/service-scripts/time-floor.sh b/build/service-scripts/time-floor.sh index 69a304ca..539f6daa 100755 --- a/build/service-scripts/time-floor.sh +++ b/build/service-scripts/time-floor.sh @@ -1,12 +1,11 @@ #!/bin/sh -# Clock floor (docs/design/time.md): a clock earlier than the image's build -# date, or the newest release committed to, is raised to it before the -# network is asked. -# Always exits 0, so the net zone that depends on this oneshot still starts. +# Clock floor (docs/design/time.md): a clock behind the build date, or the newest release +# committed to, is raised to it before the network is asked. log=/var/log/kryptik/time.log mkdir -p /var/log/kryptik out="$(/usr/bin/kryptikd time floor 2>&1)"; rc=$? echo "=== time floor $(date -Iseconds 2>/dev/null) rc=${rc} === ${out}" >> "$log" 2>/dev/null # On the console too, so a serial log shows it. echo "time-floor: ${out}" +# Always 0: the net zone depends on this oneshot. exit 0 From 71b439dd472da18b082bdf913b2d8626c9f38886 Mon Sep 17 00:00:00 2001 From: DevomB Date: Wed, 7 Oct 2026 08:37:50 -0700 Subject: [PATCH 11/18] A routed zone started again as its last run ends keeps its path: kryptikd replaces the port that run left in the net zone Run 37593717720 showed why uplink-address-refused's control failed: its untrusted started seconds after the one before it exited, and attach failed with "create veth kv-untrusted/eth0: File exists". A zone's port outlives it until the kernel has torn its network namespace down, so the new run started with loopback only, its ICMP socket got EACCES, and everything it tried was refused for want of a path. The same race let zone-separation pass with no path at all: an echo that cannot leave is never answered. registry::claim lets one instance of a zone run, so a port by its name is stale when the next one attaches: attach deletes it and retries for up to 5 s. zone-separation now needs the bridge to answer as well, and routed-restart-path starts untrusted again as its last run ends and needs its path to the bridge. --- build/guest-tests/zones-check.sh | 15 ++++++++++++--- compartments/kryptikd/src/netzone.rs | 17 ++++++++++++++++- docs/design/net-zone.md | 8 ++++++-- tools/image/zones-test.sh | 2 +- 4 files changed, 35 insertions(+), 7 deletions(-) diff --git a/build/guest-tests/zones-check.sh b/build/guest-tests/zones-check.sh index 730f32a9..9a9b6a39 100755 --- a/build/guest-tests/zones-check.sh +++ b/build/guest-tests/zones-check.sh @@ -143,14 +143,23 @@ pp_seen=0; pp_leak=0 grep -qs 'passphrase-fil[e]' /proc/[0-9]*/cmdline && pp_seen=1 grep -rqs 'personal-pas[s]' /run/kryptik /proc/[0-9]*/cmdline && pp_leak=1 # An echo that gets no answer shows separation only while personal holds the -# address that was tried. +# address that was tried, and untrusted has a path: it reaches the bridge. per_init="$(cut -d' ' -f1 /run/kryptik/zones/personal/init.pid 2>/dev/null)" per_addr="$(nsenter -t "${per_init:-0}" -n ip -4 -o addr show eth0 2>/dev/null | awk '{print $4}' | head -1)" -zrun untrusted 30 -- sh -c "python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.19.0.$PER 2 >/dev/null 2>&1 && echo CROSS-ZONE-REACHED || echo CROSS-ZONE-BLOCKED; test -e /var/lib/kryptik/volumes && echo VOLUMES-VISIBLE || echo VOLUMES-ABSENT; echo \"HOMES=\$(ls /home 2>&1 | tr '\n' ' ')\"" -[[ "$ZOUT" == *CROSS-ZONE-BLOCKED* && "$per_addr" == "10.19.0.$PER/24" ]] && pass "zone-separation" "untrusted cannot reach personal, which holds 10.19.0.$PER on the bridge" || fail "zone-separation" "$ZOUT; personal's address: ${per_addr:-none}" +zrun untrusted 30 -- sh -c "python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.19.0.1 3 >/dev/null 2>&1 && echo BRIDGE-OK; python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.19.0.$PER 2 >/dev/null 2>&1 && echo CROSS-ZONE-REACHED || echo CROSS-ZONE-BLOCKED; test -e /var/lib/kryptik/volumes && echo VOLUMES-VISIBLE || echo VOLUMES-ABSENT; echo \"HOMES=\$(ls /home 2>&1 | tr '\n' ' ')\"" +[[ "$ZOUT" == *BRIDGE-OK* && "$ZOUT" == *CROSS-ZONE-BLOCKED* && "$per_addr" == "10.19.0.$PER/24" ]] && pass "zone-separation" "untrusted reaches the bridge and not personal, which holds 10.19.0.$PER on it" || fail "zone-separation" "$ZOUT; personal's address: ${per_addr:-none}; $(grep -h 'network path' "$LOG/untrusted.err" | tail -1)" [[ "$ZOUT" == *VOLUMES-ABSENT* ]] && pass "volume-hidden" "no /var/lib/kryptik/volumes inside untrusted" || fail "volume-hidden" "the volume directory is visible from untrusted, or the probe did not run: $ZOUT" homes="$(sed -n 's/^HOMES=//p' <<<"$ZOUT")" [[ "$(tr -d ' ' <<<"$homes")" == untrusted ]] && pass "home-hidden" "/home in untrusted holds its own directory and no other zone's" || fail "home-hidden" "/home in untrusted: ${homes:-not listed}" +# untrusted again at once: its last run's port stays in the net zone until the +# kernel has torn that run's namespace down, and the new run must not lose its +# path to it. +zrun untrusted 30 -- sh -c 'python3 /usr/lib/kryptik/guest-tests/icmp-echo.py 10.19.0.1 3 >/dev/null 2>&1 && echo BRIDGE-OK' +if [[ "$ZOUT" == *BRIDGE-OK* ]] && ! grep -q 'has no network path' "$LOG/untrusted.err"; then + pass "routed-restart-path" "untrusted, started again as its last run ended, reaches the bridge" +else + fail "routed-restart-path" "rc ${ZRC}: $(tr '\n' ' ' <<<"$ZOUT") $(grep -h 'network path' "$LOG/untrusted.err" | tail -1)" +fi # The network the uplink sits on: untrusted's definition opens it ([network] # local) and personal's does not. The bridge answers personal, so the refusal # is the rule's and not a dead path. diff --git a/compartments/kryptikd/src/netzone.rs b/compartments/kryptikd/src/netzone.rs index dcd9838b..0302935d 100755 --- a/compartments/kryptikd/src/netzone.rs +++ b/compartments/kryptikd/src/netzone.rs @@ -370,9 +370,24 @@ fn attach_routed(name: &str, k: u8, nic_ns: i32, zone_ns: i32, host_gid: Option< Ok(()) } +/// How long a port a zone left behind may take to go: the kernel tears a dead zone's namespace +/// down, and its end of the pair with it, after the zone has exited. +const STALE_PORT_WAIT: std::time::Duration = std::time::Duration::from_secs(5); + fn attach_v4(port: &str, k: u8, nic_ns: i32, zone_ns: i32, host_gid: Option) -> Result<(), NetError> { netlink::with_netns(nic_ns, || { - netlink::create_veth(port, "eth0", Some(zone_ns))?; + /* One instance of a zone runs at a time (registry::claim), so a port by this name is the + * last instance's, still waiting on its namespace's teardown: delete it, or wait. */ + let since = std::time::Instant::now(); + loop { + match netlink::create_veth(port, "eth0", Some(zone_ns)) { + Err(e) if e.kind() == io::ErrorKind::AlreadyExists && since.elapsed() < STALE_PORT_WAIT => { + let _ = netlink::delete_link(port); + std::thread::sleep(std::time::Duration::from_millis(100)); + } + r => break r?, + } + } netlink::set_master(port, BRIDGE)?; netlink::set_port_isolated(port, true)?; netlink::set_up(port) diff --git a/docs/design/net-zone.md b/docs/design/net-zone.md index 12843c7a..500c9ef1 100755 --- a/docs/design/net-zone.md +++ b/docs/design/net-zone.md @@ -39,7 +39,10 @@ query and the [update](update-channel.md) fetcher. Builds on zone waits at its handshake, kryptikd creates `kv-` in the net zone with its peer born in the routed zone as `eth0`, enslaves `kv-` to `kryptik0` and isolates the port (`IFLA_BRPORT_ISOLATED`), so no frame - passes between two `kv-*` ports. The zone's addresses, `10.19.0./24` and + passes between two `kv-*` ports. A zone's last run can still hold that name + for a moment, until the kernel has torn its namespace down; one instance of + a zone runs at a time, so kryptikd deletes the stale port and waits up to + 5 s for the name. The zone's addresses, `10.19.0./24` and `fd19::/64` with default routes via the bridge, follow from its declared identity (`netzone::host_number`: `uid_base` 131072 is `.2`, 196608 is `.3`, and so on), not from DHCP: one less daemon, no broadcast domain. @@ -279,7 +282,8 @@ as `SIGSYS` in the zone's log and a `wifi=connecting` that never changes. - `build/guest-tests/zones-check.sh` on the installed system checks every guarantee above under QEMU user networking: the net zone `READY`, zone 0 offline, a routed zone's address, NAT, ULA-only IPv6 and resolver, zones - separated, `vault` offline, no egress while the net zone is down, + separated while each reaches the bridge, a routed zone started again as its + last run ends keeping its path, `vault` offline, no egress while the net zone is down, reattachment after a restart, a zone without `local` refused the VM gateway, `untrusted` refused the net zone's own uplink addresses while it reaches the gateway, and, on two `mac80211_hwsim` radios, the net zone associating, diff --git a/tools/image/zones-test.sh b/tools/image/zones-test.sh index a2d5b4e1..b454d8f3 100755 --- a/tools/image/zones-test.sh +++ b/tools/image/zones-test.sh @@ -65,7 +65,7 @@ zp="$(sed -n 's/.*passed=\([0-9]*\).*/\1/p' <<<"$summary")"; zf="$(sed -n 's/.*f if [[ -n "$summary" && "${zf:-1}" -eq 0 && "${zp:-0}" -ge 30 ]]; then green "every guest check passed (${zp})"; else red "guest checks: ${zp:-0} passed, ${zf:-?} failed"; fi grep 'ZT FAIL' <<<"$T2" | sed 's/^/ /' # The key verdicts one by one, so a pass is not a single line. -for name in kernel-support policies net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns net-lease-names-resolver dns-follows-lease dns-after-reload routed-ipv6-noglobal zone-separation volume-hidden home-hidden fail-closed net-restart-ready reattach-after-restart uplink-refused uplink-address-refused wifi-beyond \ +for name in kernel-support policies net-ready net-dns zone0-nic zone0-no-route zone0-offline routed-egress routed-ping routed-ping6 routed-dns net-lease-names-resolver dns-follows-lease dns-after-reload routed-ipv6-noglobal zone-separation volume-hidden home-hidden routed-restart-path fail-closed net-restart-ready reattach-after-restart uplink-refused uplink-address-refused wifi-beyond \ wifi-module wifi-ap wifi-add wifi-associated wifi-lease wifi-egress wifi-forget \ time-floor-ran time-clamp time-floor-forged time-claim-stepped time-claim-floor time-claim-consent pids-limit ephemeral-size-bound cpu-max-set lifecycle-repeat lifecycle-registry \ terminal-terminfo man-page text-browser tls-trust \ From de78fb862497b0e6f1650a2ffbb58a80ccfcea23 Mon Sep 17 00:00:00 2001 From: DevomB Date: Wed, 7 Oct 2026 08:43:37 -0700 Subject: [PATCH 12/18] The signature, provenance and zone tools' comments are shorter, with their reasons kept: cleanup-held's cut of tools/kryptik, tools/provenance-inventory.sh, tools/verify-signatures.sh and three of their tests, redone on main by hand, comments only Only comment regions are taken from 65a6fbb; its message, help-range and test-label changes are left out. Where the cut dropped a reason, a shorter one is kept: kryptik has no transfer, clipboard or mount command because from zone 0 those would bypass the chrome, --ask passes the passphrase never as an argument, a Wi-Fi passphrase goes on stdin; the inventory's || true and its passthrough of publisher-named classes; the GNU keyring vouches only for whoever it names until checked out of band, a pinned key is one its project publishes on its own origin, which CPython releases the python key signs, netfilter's confirmation of its current key, anchored_import's return codes and kernel.org's signed projects. tools/key-provenance.tsv is left as it is: its header says what each confirmation route establishes and where it stops, which a summary loses. --- tools/kryptik | 41 +++----- tools/provenance-inventory.sh | 55 ++++------ tools/tests/boot-success.sh | 5 +- tools/tests/update-manifest-snapshot.sh | 29 ++--- tools/tests/verify-signatures.sh | 43 +++----- tools/verify-signatures.sh | 134 ++++++++---------------- 6 files changed, 99 insertions(+), 208 deletions(-) diff --git a/tools/kryptik b/tools/kryptik index 064d053b..aafaa51b 100755 --- a/tools/kryptik +++ b/tools/kryptik @@ -1,11 +1,6 @@ #!/usr/bin/env bash -# kryptik: the user's command for zones. Fills in kryptikd's paths from -# kryptik.conf and never offers a way past kryptikd's refusals. -# -# When not root and the session's launch service (kryptik-launch) answers, -# zones start through it; otherwise kryptikd runs them directly. There is no -# transfer, clipboard or mount command: from zone 0 those would bypass the -# chrome, where the user approves each move. +# kryptik: the user's zone command; fills in kryptikd's paths and offers no way past its refusals. +# No transfer, clipboard or mount command: from zone 0 those would bypass the chrome's approval. set -uo pipefail PROG="${0##*/}" @@ -82,8 +77,7 @@ $PROG has no commands for either, on purpose. USAGE } -# True when not root and the session's launch service answers. --runtime-dir -# is its cheapest request and fails fast rather than hanging. +# Not root and the launch service answers; --runtime-dir is its cheapest request and fails fast. LAUNCH="${KRYPTIK_LAUNCH:-kryptik-launch}" via_launch() { [[ "$(id -u)" -ne 0 ]] || return 1 @@ -120,17 +114,14 @@ have_kryptikd() { arguments. Install it, or set KRYPTIKD=/path/to/kryptikd." } -# The flags that map a zone to an unprivileged host user. kryptikd takes them -# only as root, and then requires them unless the zone declares its own. -# Sets an array rather than printing for `< <(...)`: process substitution -# needs /dev/fd, which a minimal system may lack. +# An array, not printed for `< <(...)`: process substitution needs /dev/fd, which may be missing. IDENTITY_ARGS=() +# The flags mapping a zone to a host user: kryptikd takes them only as root, then requires them. set_identity_args() { # zone name IDENTITY_ARGS=() [[ "$(id -u)" -eq 0 ]] || return 0 - # kryptikd refuses the flags for a zone with [identity] uid_base. A - # presence test is enough: kryptikd checks the value itself. + # Not for a zone with [identity] uid_base; kryptikd checks the value, so presence is enough. if grep -qE '^[[:space:]]*uid_base[[:space:]]*=' "$ZONES_DIR/$1.toml" 2>/dev/null; then return 0 fi @@ -175,8 +166,7 @@ require_zone() { } # --- running state ---------------------------------------------------------- -# A word, not a colour: it must survive a pipe, a monochrome terminal and a -# screen reader. +# A word, not a colour: it must survive a pipe, a monochrome terminal and a screen reader. running_state() { local out out="$("$KRYPTIKD" status "$1" 2>/dev/null)" || { printf 'unknown'; return; } @@ -252,13 +242,13 @@ cmd_stop() { # --- starting --------------------------------------------------------------- -# The one place that runs a zone, for both shell and run. +# Shared by shell and run. start_zone() { local name="$1"; shift if via_launch; then [[ -f "$ZONES_DIR/$name.toml" ]] || die "no zone named $name in $ZONES_DIR" - # --ask reads an encrypted zone's passphrase on this terminal and passes - # it on a descriptor, never as an argument. + # --ask reads an encrypted zone's passphrase on this terminal and passes it on a + # descriptor, never as an argument. "$LAUNCH" --ask --no-display "$name" -- "$@" local rc=$? (( rc == 0 )) || printf '\n%s: %s did not start through the launch service (exit %d).\n' "$PROG" "$name" "$rc" >&2 @@ -272,8 +262,7 @@ start_zone() { local rc set_identity_args "$name" - # Not captured: a shell needs the terminal, and kryptikd's warnings and - # refusals must reach the user unchanged. + # Not captured: a shell needs the terminal, and the user must see kryptikd's refusals unchanged. "$KRYPTIKD" run "$name" --zones "$ZONES_DIR" \ --rootfs "$ROOTFS" "${IDENTITY_ARGS[@]}" -- "$@" rc=$? @@ -299,9 +288,8 @@ cmd_run() { start_zone "$name" "$@" } -# --- the net zone's Wi-Fi networks ------------------------------------------ -# kryptikd keeps the credentials in zone 0 and restarts the net zone after a -# change (docs/design/net-zone.md). The passphrase always goes on stdin. +# --- the net zone's Wi-Fi networks, kept by kryptikd in zone 0 -------------- +# The passphrase always goes on stdin; kryptikd restarts the net zone after a change. # Read one line into PASSPHRASE: from the terminal with echo off, or a pipe. PASSPHRASE="" @@ -341,8 +329,7 @@ cmd_wifi() { esac } -# Installed systems only: the update state is zone 0's, reached through the -# launch service. +# Installed systems only: the update state is zone 0's, reached through the launch service. cmd_update() { case "${1:-}" in status|fetch|apply) [[ $# -eq 1 ]] || die "$PROG update $1 takes no argument" 2 ;; diff --git a/tools/provenance-inventory.sh b/tools/provenance-inventory.sh index 896c8b52..376a1b42 100755 --- a/tools/provenance-inventory.sh +++ b/tools/provenance-inventory.sh @@ -3,8 +3,7 @@ # # ./tools/provenance-inventory.sh [options] # --offline lock integrity only -# --identity also check signer identity against kernel.org's -# published developer keys +# --identity also check signers against kernel.org's developer keys # --md markdown table # --json JSON document # --licences also collect licence evidence (slow) @@ -63,9 +62,7 @@ done < "$MANIFEST" # --- recorded provenance caveats -------------------------------------------- -# Facts no check can see (a recipe that rewrites upstream files, a signer -# upstream never designated): caveats, never classes. They describe the tree -# being inventoried, so they come from KRYPTIK_ROOT. Optional unless --notes. +# Caveats describe the inventoried tree, so they come from KRYPTIK_ROOT; optional unless --notes. NOTESF="${KRYPTIK_ROOT}/tools/source-notes.tsv" if [[ -n "${NOTES_ARG:-}" ]]; then NOTESF="$NOTES_ARG" @@ -124,8 +121,7 @@ elif [[ "$OFFLINE" -eq 1 ]]; then warn "--offline: signature and publisher evidence will not be collected." warn "Every source will therefore show only what sources.lock establishes." else - # The evidence is the verifiers' --report output. This is not a gate, so - # their exit status is ignored. + # Not a gate: the verifiers' --report output is the evidence, and their exit status is ignored. dim " running tools/verify-signatures.sh" "${KRYPTIK_ROOT}/tools/verify-signatures.sh" --report="$SIGREP" \ > "${WORK}/signatures.log" 2>&1 || true @@ -136,9 +132,7 @@ fi # --- signer identity -------------------------------------------------------- -# Looks up each keys.manifest key in kernel.org's pgpkeys.git (developer keys -# with trust paths to Torvalds, one per long key id under keys/). A match means -# kernel.org publishes that key; its trust root is TLS to git.kernel.org. +# Looks up each keys.manifest key in kernel.org's pgpkeys.git; its trust root is TLS to kernel.org. IDREP="${WORK}/identity.tsv" # Keep what the test hook copied in. [[ -n "${IDREP_PRESET:-}" ]] || : > "$IDREP" @@ -154,8 +148,7 @@ collect_identity() { local home="${WORK}/gnupg" mkdir -p "$home"; chmod 700 "$home" - # Reference keys. The kernel tarball is already verified against the stable - # key, so a certification by it is no new trust decision. + # Reference keys: the kernel is verified against the stable key, so they add no new trust. local ref rid for ref in "79BE3E4300411886:Torvalds" "38DBBDC86092693E:Kroah-Hartman" \ "E63EDCA9329DD07E:Ryabitsev"; do @@ -218,20 +211,17 @@ fi # --- licence evidence and built artefacts ----------------------------------- -# Off by default: listing a tarball means decompressing all of it. Uncollected -# licences read not-collected, never unknown. +# Off by default, as listing a tarball decompresses all of it; uncollected reads not-collected. LICREP="${WORK}/licences.tsv" : > "$LICREP" if [[ "$LICENCES" -eq 1 ]]; then log "Collecting licence evidence" - # Part of this tool, so found beside it rather than under KRYPTIK_ROOT. + # Part of this tool, so found beside it, not under KRYPTIK_ROOT. "$(dirname "${BASH_SOURCE[0]}")/scan-licenses.sh" > "$LICREP" 2>/dev/null || true dim " $(grep -c . "$LICREP" || true) source(s) scanned" fi -# A built tree's identity: path, file count, size, BUILD_ID and the sha256 of -# the artifact-manifest.txt beside it. Re-hashing the files is left to -# `make verify-manifest`. +# A built tree's identity; `make verify-manifest` re-hashes its files. ARTREP="${WORK}/artifacts.tsv" : > "$ARTREP" if [[ -n "$ARTIFACTS" ]]; then @@ -240,9 +230,8 @@ if [[ -n "$ARTIFACTS" ]]; then printf 'sysroot\t%s\tabsent\t-\t-\t-\t-\n' "$ARTIFACTS" >> "$ARTREP" else log "Recording built artefact identity" - # `|| true`: find and du fail on a chroot-built tree's root-owned - # directories yet still count, and under pipefail common.sh's ERR trap - # would abort. Such counts are recorded as partial. + # || true: find and du fail on a chroot-built tree's root-owned directories but still + # count, and common.sh's ERR trap would abort; such counts are recorded as partial. art_files="$(find "$ARTIFACTS" -type f 2>/dev/null | wc -l || true)" art_readable=yes find "$ARTIFACTS" -type d >/dev/null 2>&1 || art_readable=partial @@ -262,8 +251,7 @@ fi if [[ "$MD" -eq 1 || "$JSON" -eq 1 ]]; then exec 1>&3 3>&-; fi -# The keyring the counts were measured against: one warmed by an earlier -# --fetch-unknown-keys run moves many sources out of lock-only. +# A keyring warmed by --fetch-unknown-keys moves sources out of lock-only, so the report names it. KEYSTATE="unknown" if [[ -f "${WORK}/signatures.log" ]]; then KEYSTATE="$(grep -oE '(keyring ready \([0-9]+ public keys\)|using cached keyring \([0-9]+ keys\))' \ @@ -311,10 +299,8 @@ def rows(path): return -# ---- assurance classes, strongest first ------------------------------------ -# -# The order is the point. Each entry is (key, one-line statement of what it is -# worth). Nothing is ever summed across two of these. +# ---- assurance classes, strongest first, never summed ---------------------- + CLASSES = [ ("signed-tree-pinned-key", "signed tag by a key pinned in-tree, and the archive reproduces that tag's tree"), @@ -408,10 +394,8 @@ def classify(name): if s and s[0] == "signature-pinned-key": return "signature-pinned-key", s[1] - # verify-signatures.sh emits these directly when tools/key-provenance.tsv - # names a publisher for the key. Without this passthrough they would fall - # off the end of the chain and be reported as "unverified", which would be - # a worse answer than the one the verifier actually gave. + # verify-signatures.sh emits these when tools/key-provenance.tsv names the key's publisher; + # passed through, as falling off the chain would report them unverified. if s and s[0] in ("signature-korg-published-key", "signature-korg-certified-key", "signature-savannah-published-key", @@ -469,8 +453,7 @@ for r in rows(lic_path): lic[r[0]] = {"spdx": r[2], "multiple": r[3] == "yes", "files": r[4], "method": r[5]} -# recorded caveats, keyed by source name. Validated in the shell above, so a -# row reaching here is well formed and names a source in the manifest. +# recorded caveats, keyed by source name and already validated by the shell above notes = {} for r in rows(notes_path): if len(r) >= 4: @@ -485,8 +468,7 @@ for r in rows(art_path): "files": int(r[3]) if r[3].isdigit() else None, "size": r[4], "build_id": r[5], "build_manifest_sha256": r[6] if len(r) > 6 else "-", - # A tree built through a chroot has root-owned directories this - # process cannot descend, so the count is a floor, not a fact. + # A chroot-built tree has root-owned directories, so a partial count is a floor. "count_complete": not r[2].endswith("partial"), }) @@ -546,8 +528,7 @@ if os.environ.get("JSON") == "1": doc["per_licence_counts"] = lic_counts doc["source_count"] = len(doc["sources"]) - # Counted separately and deliberately never folded into per_class_counts: - # a caveat is not a weaker class, and a class is not a caveat. + # Never folded into per_class_counts: a caveat is not a weaker class. note_counts = {} for src in doc["sources"]: for n in src.get("notes", []): diff --git a/tools/tests/boot-success.sh b/tools/tests/boot-success.sh index 45488d7e..f9352cae 100755 --- a/tools/tests/boot-success.sh +++ b/tools/tests/boot-success.sh @@ -1,7 +1,6 @@ #!/usr/bin/env bash -# Test boot-success.sh's decision table with stand-ins: a fake /run/kryptik, -# service directory and ESP, and fake s6-svstat, kryptikd, kryptik-efiboot, -# reboot and mount that record what they were asked. +# boot-success.sh's decision table, against a fake /run/kryptik, service directory and ESP, and +# fake s6-svstat, kryptikd, kryptik-efiboot, reboot and mount that record what they were asked. set -uo pipefail ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" SCRIPT="$ROOT/build/service-scripts/boot-success.sh" diff --git a/tools/tests/update-manifest-snapshot.sh b/tools/tests/update-manifest-snapshot.sh index 9631c04e..f4298ea0 100755 --- a/tools/tests/update-manifest-snapshot.sh +++ b/tools/tests/update-manifest-snapshot.sh @@ -52,8 +52,7 @@ mkpayload replacement 3 ssh-keygen -q -t ed25519 -N '' -f key >/dev/null 2>&1 || { echo "cannot make a key"; exit 77; } ssh-keygen -q -t ed25519 -N '' -f latestkey >/dev/null 2>&1 || { echo "cannot make a key"; exit 77; } -# The trust anchor as stage 04 installs it: each key limited to its own -# namespace (release: manifests; latest: statements of what is current). +# The trust anchor as stage 04 installs it: each key limited to its own namespace. { printf 'kryptik-release namespaces="kryptik-release" %s\n' "$(cut -d' ' -f1,2 key.pub)" printf 'kryptik-latest namespaces="kryptik-latest" %s\n' "$(cut -d' ' -f1,2 latestkey.pub)" @@ -61,8 +60,7 @@ ssh-keygen -q -t ed25519 -N '' -f latestkey >/dev/null 2>&1 || { echo "cannot ma ssh-keygen -Y sign -f key -n kryptik-release signed/manifest >/dev/null 2>&1 || { echo "cannot sign"; exit 77; } printf 'development\n' > role -# The tool's functions, verbatim, with the environment they expect. die exits, -# so each case runs in its own bash. +# The tool's functions verbatim, with their environment; die exits, so each case runs alone. { echo 'NAMESPACE=kryptik-release' echo 'MAGIC=KRYPTIK-MANIFEST-1' @@ -80,9 +78,7 @@ grep -q '^verify_payload() {' verify.sh || { echo "could not extract verify_payl grep -q '^pin() {' verify.sh || { echo "could not extract pin from $TOOL"; exit 1; } -# run_case NAME WHAT: verify_payload over a fresh copy of the signed payload, -# with WHAT changed the moment ssh-keygen verifies. Prints the tool's output, -# plus VERSION= if it accepted. +# run_case NAME WHAT: verify_payload on a copy of the signed payload, WHAT changed as it verifies. run_case() { local name="$1" what="$2" rm -rf "$T/payload" "$T/snap-$name"; cp -a "$T/signed" "$T/payload"; mkdir -p "$T/snap-$name" @@ -117,8 +113,7 @@ EOF out="$(run_case plain none)" if [[ "$out" == *"VERSION=2"* ]]; then ok "the signed payload verifies and reports version 2"; else bad "the signed payload did not verify: $(tail -2 <<<"$out" | tr '\n' ' ')"; fi -# Only the manifest replaced: the kept copy still matches the files, so the -# answer is version 2, never 3. +# Only the manifest replaced: the kept copy still matches the files, so the answer stays 2. out="$(run_case manifest manifest)" if [[ "$out" == *"VERSION=3"* ]]; then bad "the replaced, unsigned manifest was read after the signature check (version 3)" @@ -157,9 +152,7 @@ else ok "control: the real verifier rejects the replacement manifest" fi -# --- the update channel's two checks ----------------------------------------- -# Zone 0 runs these before it believes a manifest or a statement of what is -# current (docs/design/update-channel.md). +# --- check-manifest and check-pointer, run by zone 0 before it believes either ----- check() { # check FUNCTION ARGS... -> the tool's output, REFUSED: on a refusal { echo "source $T/verify.sh"; printf 'SNAP=%q\n' "$(mktemp -d "$T/snap.XXXXXX")"; printf '%q ' "$@"; echo; } > "$T/check.sh" bash "$T/check.sh" 2>&1 @@ -177,8 +170,7 @@ if [[ "$out" == *"version: 2"* && "$out" == *"sha256: $want_sha"* && "$(grep -c else bad "check-manifest on a signed manifest: $(tail -3 <<<"$out" | tr '\n' ' ')" fi -# Zone 0 reads only stdout and wants the version on its first line -# (compartments/kryptikd/src/update.rs); the case above searches both streams. +# Zone 0 (kryptikd's update.rs) reads the version from stdout's first line; above, both streams. staged stage check cmd_check_manifest "$T/stage" > /dev/null first="$(bash "$T/check.sh" 2>/dev/null | head -1)" @@ -186,8 +178,7 @@ first="$(bash "$T/check.sh" 2>/dev/null | head -1)" && ok "check-manifest: the first line of its standard output is the version, as zone 0 reads it" \ || bad "check-manifest: the first line zone 0 reads is '${first}', not 'version: 2'" -# Signed by the right key in the pointer's namespace: a pointer's signature -# must never pass for a manifest's. +# The right key in the pointer's namespace: a pointer's signature must not pass for a manifest's. staged crossed; rm -f "$T/crossed/manifest.sig" ssh-keygen -Y sign -f key -n kryptik-latest "$T/crossed/manifest" >/dev/null 2>&1 out="$(check cmd_check_manifest "$T/crossed")" @@ -203,8 +194,7 @@ out="$(check cmd_check_manifest "$T/stranger")" && ok "check-manifest: a manifest signed by a key that is not enrolled is refused" \ || bad "check-manifest accepted a stranger's key: $(tail -2 <<<"$out" | tr '\n' ' ')" -# apply's rules too, as it is the same function: the role, and no downgrade -# (a network delivery is never a recovery). +# apply's rules too (the same function): the role, and no downgrade, as a download is no recovery. resigned() { # resigned NAME SED-EXPRESSION -> the signed manifest, edited, signed again rm -rf "${T:?}/$1"; mkdir -p "$T/$1" sed "$2" "$T/signed/manifest" > "$T/$1/manifest" @@ -256,8 +246,7 @@ out="$(check cmd_check_pointer "$T/ptr/stranger" "$T/ptr/stranger.sig")" && ok "check-pointer: a pointer signed by a key that is not enrolled is refused" \ || bad "check-pointer accepted a stranger's key: $(tail -2 <<<"$out" | tr '\n' ' ')" -# Each key is enrolled for one namespace only, both ways round; this is what -# lets the statement key live where a timer can reach it. +# Each key holds one namespace, both ways round, so the statement key can live where a timer runs. cp "$T/ptr/latest" "$T/ptr/by-release-key" ssh-keygen -Y sign -f key -n kryptik-latest "$T/ptr/by-release-key" >/dev/null 2>&1 out="$(check cmd_check_pointer "$T/ptr/by-release-key" "$T/ptr/by-release-key.sig")" diff --git a/tools/tests/verify-signatures.sh b/tools/tests/verify-signatures.sh index 01dd0f56..0702ef64 100755 --- a/tools/tests/verify-signatures.sh +++ b/tools/tests/verify-signatures.sh @@ -1,12 +1,9 @@ #!/usr/bin/env bash -# Tests for tools/verify-signatures.sh. Offline, with real GnuPG: per-run keys -# produce GOODSIG, EXPKEYSIG, REVKEYSIG, BADSIG and NO_PUBKEY. Each case runs -# the tool against a throwaway KRYPTIK_ROOT. +# Tests for tools/verify-signatures.sh: offline, real GnuPG, per-run keys, a throwaway KRYPTIK_ROOT. set -uo pipefail -# common.sh prefers these over paths derived from KRYPTIK_ROOT, so an exported -# one would point the tool at the real tree. +# common.sh prefers these, when exported, to paths derived from KRYPTIK_ROOT. unset KRYPTIK_SOURCES KRYPTIK_WORK KRYPTIK_LOCK KRYPTIK_OUT KRYPTIK_ROOT ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" @@ -78,8 +75,7 @@ printf 'TAMPERED AFTER SIGNING\n' >> "${SRC}/bad.tar.gz" # Upstream publishes nothing alongside this one. printf 'no signature is published for this\n' > "${SRC}/nosig.tar.gz" -# The tool's keyring: good, expired and revoked (revocation applied). GnuPG -# prefixes its revocation certificates with ':' against accidental import. +# The tool's keyring: good, expired and revoked; sed strips the ':' guarding gpg's revocation. REVFPR="$(GNUPGHOME="$FIXG" gpg --batch --list-keys --with-colons revoked@example.test \ | awk -F: '$1=="fpr"{print $10; exit}')" GNUPGHOME="$FIXG" gpg --batch --quiet \ @@ -90,8 +86,7 @@ sed 's/^://' "${FIXG}/openpgp-revocs.d/${REVFPR}.rev" \ | GNUPGHOME="$BUILDG" gpg --batch --quiet --import >/dev/null 2>&1 GNUPGHOME="$BUILDG" gpg --batch --quiet --export > "${W}/keyring.gpg" -# The unknown key is reachable only by the id its signature names, as a long -# key id or a fingerprint (gpg may report either). +# The unknown key is served only under the id its signature names: long key id or fingerprint. UNKFPR="$(GNUPGHOME="$FIXG" gpg --batch --list-keys --with-colons unknown@example.test \ | awk -F: '$1=="fpr"{print $10; exit}')" UNKID="${UNKFPR: -16}" @@ -139,8 +134,7 @@ echo # --- harness ---------------------------------------------------------------- -# write_manifest ...: rows shaped like fetch-sources.sh --list, each -# declaring probe: whichever of .sig, .asc and .sign is published. +# write_manifest NAME...: rows shaped like fetch-sources.sh --list, each declaring probe. write_manifest() { : > "${W}/manifest" local n @@ -277,8 +271,7 @@ write_manifest good fresh_root; run --strict "--notes=${W}/notes-stale.tsv" expect_fail "a no-usable-key note for a source whose key is held fails --strict" "stale note" -# A noted source whose signature could not be fetched this run: unverifiable, -# and the note untried, not stale. +# A noted source whose signature could not be fetched: unverifiable, its note untried, not stale. printf 'fixture payload for unreached\n' > "${SRC}/unreached.tar.gz" rm -f "${SRC}/unreached.tar.gz.sig" "${SRC}/.signatures/unreached.tar.gz.sig" printf 'unreached no-usable-key https://example.invalid/ No route to the key was found. Checked 2026-09-28.\n' > "${W}/notes-unreached.tsv" @@ -375,8 +368,7 @@ printf 'fixture payload for kern, altered\n' | gzip -c > "${SRC}/kern.tar.gz" fresh_root; run expect_fail "a kernel row whose tar is not the signed one fails" "BAD SIGNATURE" -# verify-provenance.sh checks these, so nothing is fetched for them, even -# where a signature exists. +# verify-provenance.sh checks these, so nothing is fetched for them even where a signature exists. : > "${W}/manifest"; add_row good sha256; add_row expired tag; add_row nosig none fresh_root; run --report="${W}/declared.tsv" if [[ "$RC" -eq 0 ]] \ @@ -428,8 +420,7 @@ fresh_root; run expect_pass "a sha256.txt row is left to verify-provenance.sh" \ "good: no OpenPGP signature upstream; the publisher's .sha256.txt is verify-provenance's" -# A signature over a checksum list beside the file, like cmake's (sha256) -# and pixman's (sha512). The list must give the file's digest. +# A signed checksum list beside the file, like cmake's (sha256) and pixman's (sha512). printf 'fixture payload for summed\n' > "${SRC}/summed.tar.gz" { printf '%s other.tar.gz\n' "$(printf other | sha256sum | cut -d' ' -f1)" printf '%s summed.tar.gz\n' "$(sha256sum "${SRC}/summed.tar.gz" | cut -d' ' -f1)"; } \ @@ -541,8 +532,7 @@ else sed 's/^/ /' "${FAKE}/keys.manifest" fi -# Known limit: keys.manifest is the only record of how a cached key got there, -# so without it the key counts as verified. The remedy, --refresh, is next. +# Known limit: a cached key without its keys.manifest entry counts as verified; --refresh is next. rm -f "${FAKE}/keys.manifest" run if grep -qF "verified: 1" "$OUT"; then @@ -572,8 +562,7 @@ KEYRING="${W}/keyring.gpg" write_manifest good expired revoked bad nosig unknown fresh_root; run --fetch-unknown-keys -# good + expired verified; unknown unaudited; nosig unverifiable; -# revoked + bad fatal. +# good and expired verified, unknown unaudited, nosig unsigned, revoked and bad fatal. for want in "verified: 2" "unaudited: 1" "unsigned: 1" \ "REVOKED KEYS: 1" "FAILED: 2"; do if grep -qF "$want" "$OUT"; then @@ -799,8 +788,7 @@ else red "an unresolvable wkd locator warns and leaves the source unverified (exit ${RC})"; show fi -# A key already held is still merged from where it is published: good is in -# the tool's keyring, and its published copy carries its revocation. +# good is held, but its published copy carries a revocation, which the merge must pick up. GOODFPR="$(GNUPGHOME="$FIXG" gpg --batch --list-keys --with-colons good@example.test \ | awk -F: '$1=="fpr"{print $10; exit}')" REVG="${W}/gnupg-revoke" @@ -821,8 +809,7 @@ else red "a held key is merged from its locator, so a revocation published there is seen (exit ${RC})"; show fi -# A held key whose locator now serves another key: the held copy must not -# carry the source through. +# A held key whose locator serves another key: the held copy must not carry the source. fresh_root write_manifest good prov_table "${GOODFPR} korg file://${PROV}/unknown.asc 2026-09-11 good good fixture, whose locator now serves another key" @@ -834,8 +821,7 @@ else red "a held key refused at its locator fails the source it signs (exit ${RC}, got $(klass_of "${W}/r6.tsv" good))"; show fi -# A locator that serves the recorded key and another: only the recorded one is -# taken, so a source signed by the other stays unverified. +# A locator serving the recorded key and another: only the recorded one is taken. fixgpg --quick-generate-key "extra fixture " ed25519 sign never >/dev/null 2>&1 printf 'fixture payload for extra\n' > "${SRC}/extra.tar.gz" fixgpg --yes --local-user extra@example.test \ @@ -927,8 +913,7 @@ bad_prov "${UNKFPR} korg file://${PROV}/unknown.asc 2026-09-11 unknown" \ # --- the platform-published kind -------------------------------------------- -# github: GitHub publishes the key of the account that published the pinned -# release. Its own class, since holding that account defeats both at once. +# github: the release publisher's account key; its own class, as holding that account defeats both. fresh_root write_manifest unknown prov_table "${UNKFPR} github file://${PROV}/unknown.asc 2026-09-11 unknown unknown fixture; fixture/repo v1.0 was published by nobody" diff --git a/tools/verify-signatures.sh b/tools/verify-signatures.sh index 115f0b34..116ca815 100755 --- a/tools/verify-signatures.sh +++ b/tools/verify-signatures.sh @@ -4,12 +4,11 @@ # ./tools/verify-signatures.sh [--strict] [--refresh] [--fetch-unknown-keys] # [--report=FILE] [--notes=FILE] # --strict release gate: anything unverified or unaudited fails, -# except a signer no publisher states, when -# tools/source-notes.tsv says so (no-usable-key) +# except an unheld key the notes record as no-usable-key # --refresh discard cached keys and re-import # --fetch-unknown-keys import keys the signatures name (unaudited) # --report=FILE per-source results to FILE -# --notes=FILE the caveats, not tools/source-notes.tsv +# --notes=FILE caveats from FILE, not tools/source-notes.tsv source "$(dirname "${BASH_SOURCE[0]}")/../build/lib/common.sh" load_config @@ -18,11 +17,10 @@ have gpg || die "gpg not found. Install gnupg." KEYDIR="${KRYPTIK_ROOT}/build/work/keys" SIGDIR="${KRYPTIK_SOURCES}/.signatures" -# With the sources, so a cache of them carries it (tools/fetch-sources.sh). +# Kept with the sources so their cache carries it; tools/fetch-sources.sh fetches it too. GNU_KEYRING="${KRYPTIK_SOURCES}/.keys/gnu-keyring.gpg" -# A private GNUPGHOME, not --keyring: GnuPG 2.4 with keyboxd silently ignores -# --keyring and verifies against the user's own store. +# Not --keyring: GnuPG 2.4 with keyboxd ignores it and verifies against the user's own store. export GNUPGHOME="${KEYDIR}/gnupg" FETCH_UNKNOWN=0 @@ -42,9 +40,7 @@ for a in "$@"; do done [[ -n "$REPORT" ]] && : > "$REPORT" -# A signer no publisher states, accepted deliberately: the note names the -# routes that were tried. Such a source is not held against --strict; a note -# for a source whose key is held is stale, and --strict fails on it. +# no-usable-key notes let their sources pass --strict; --strict fails a note whose key is held. declare -A NOTED_NO_KEY=() if [[ -f "$NOTES" ]]; then while read -r n_pkg n_kind _; do @@ -56,28 +52,23 @@ NOTED_LIST=() declare -A NOTE_USED=() declare -A SEEN_SOURCE=() -# report -# One tab-separated line per source, for provenance-inventory.sh. The class is -# an assurance class (which kind of key verified it), not a pass/fail. +# report SOURCE CLASS DETAIL for the inventory; CLASS is an assurance class, not pass or fail. report() { [[ -n "$REPORT" ]] || return 0 printf '%s\t%s\t%s\n' "$1" "$2" "${3//$'\t'/ }" >> "$REPORT" } -# Keys accepted without audit, awaiting out-of-band confirmation. Read on every -# run, not only with --fetch-unknown-keys (see UNAUDITED_FPRS). +# Unaudited keys awaiting out-of-band confirmation, read on every run (see UNAUDITED_FPRS). KEYS_MANIFEST="${KRYPTIK_ROOT}/keys.manifest" mkdir -p "$KEYDIR" "$SIGDIR" "$GNUPGHOME" "$(dirname "$GNU_KEYRING")" chmod 700 "$GNUPGHOME" IMPORTED_MARK="${GNUPGHOME}/.kryptik-imported" -# This GNUPGHOME has the GNU keyring in it: kept beside the keys, not beside -# the keyring file, which outlives any one checkout's keys. +# Beside the keys, not the keyring file, which outlives any one checkout's keys. GNU_IMPORTED="${GNUPGHOME}/.gnu-keyring-imported" -# A host that throttles (freedesktop.org answers 418 to a busy runner, others -# 429 or 503) is asked again after a pause; a 404 is an answer. +# Throttling answers (418 from freedesktop.org, 429, 503) are retried after a pause; a 404 is final. quiet_fetch() { # quiet_fetch URL OUT local code try for try in 1 2 3; do @@ -112,7 +103,7 @@ manifest_source() { "${KRYPTIK_ROOT}/tools/fetch-sources.sh" --list } -# recv_key : from a keyserver, or KRYPTIK_SIGCHECK_KEYSOURCE in tests. +# recv_key KEYID: from a keyserver, or KRYPTIK_SIGCHECK_KEYSOURCE in tests. recv_key() { local keyid="$1" f if [[ -n "${KRYPTIK_SIGCHECK_KEYSOURCE:-}" ]]; then @@ -152,10 +143,8 @@ import_keys() { return 0 fi - # Fetched over the network, so it gives "signed by whoever the keyring - # says" unless its keys are checked out of band (docs/supply-chain.md). - # The keyring counts as imported only when the GNU keyring and every - # pinned key are in it; what a run could not fetch, the next fetches again. + # Fetched over the network: it vouches for whoever the keyring says until checked out of band + # (docs/supply-chain.md). It counts as imported once it and every pinned key are in. local count log "fetching GNU keyring" if [[ ! -s "$GNU_KEYRING" ]]; then @@ -167,8 +156,6 @@ import_keys() { count="$(gpg --batch --list-keys 2>/dev/null | grep -c '^pub' || true)" [[ "$count" =~ ^[0-9]+$ && "$count" -ge 100 ]] && : > "$GNU_IMPORTED" fi - # Without it nothing a GNU maintainer signed can be checked: a run that - # could not fetch it is not a pass. if [[ ! -f "$GNU_IMPORTED" ]]; then rm -f "$IMPORTED_MARK" if [[ "$STRICT" -eq 1 ]]; then @@ -180,9 +167,7 @@ Run again: what was fetched is kept." warn "the GNU keyring could not be fetched: what GNU maintainers signed is unverifiable this run" fi - # Safe from a keyserver: it cannot serve another key under a full - # fingerprint. One it does not serve is fetched again next run, and the - # sources that key signs are unverifiable until then. + # Safe from a keyserver, which cannot serve another key under a full fingerprint. log "fetching pinned maintainer keys (${#PINNED_FPRS[@]})" local fpr missing=0 for fpr in "${PINNED_FPRS[@]}"; do @@ -222,10 +207,7 @@ mark_unverifiable() { UNVERIFIABLE_LIST+=("$1") } -# Upstream signs with nothing OpenPGP, by the manifest's own declaration or -# by a listing that holds none: the lock pins the file, and -# tools/verify-provenance.sh checks whatever else upstream publishes. A -# declared signature that is missing or is not a signature stays unverifiable. +# No OpenPGP signature upstream: the lock pins the file and verify-provenance.sh checks the rest. UNSIGNED=0 UNSIGNED_LIST=() mark_unsigned() { @@ -233,8 +215,7 @@ mark_unsigned() { UNSIGNED_LIST+=("$1") } -# Unaudited keys, from keys.manifest. Once cached, such a key gives a plain -# GOODSIG, so the manifest (read on every run) is what keeps it marked. +# keys.manifest's keys: a cached one gives a plain GOODSIG, so only this list keeps it unaudited. declare -a UNAUDITED_FPRS=() if [[ -f "$KEYS_MANIFEST" ]]; then while read -r _pkg fpr _rest; do @@ -242,16 +223,14 @@ if [[ -f "$KEYS_MANIFEST" ]]; then done < <(grep -v '^[[:space:]]*#' "$KEYS_MANIFEST" || true) fi -# Keys trusted by fingerprint in the tree, reported as a stronger class than the -# fetched GNU keyring. Each must be a fingerprint the project publishes on its -# own origin, with that source noted here so it can be rechecked. +# Pinned keys outrank the GNU keyring. Each is a fingerprint its project publishes on its own +# origin, noted with it so it can be rechecked. PINNED_FPRS=( # kernel.org mainline and stable: pgpkeys.git and kernel.org's WKD serve both (2026-10-02; key-provenance.tsv). "ABAF11C65A2970B130ABE3C479BE3E4300411886" # Linus Torvalds, mainline "647F28654894E3BD457199BE38DBBDC86092693E" # Greg Kroah-Hartman, stable - # Signs CPython 3.12.x and 3.13.x, per - # https://www.python.org/downloads/metadata/pgp/ (retrieved 2026-09-11). + # CPython 3.12.x and 3.13.x, per https://www.python.org/downloads/metadata/pgp/ (retrieved 2026-09-11) "7169605F62C751356D054A26A821E680E5FA6305" # Thomas Wouters, CPython 3.12/3.13 # From https://openssl-library.org/source/ (retrieved 2026-09-11): the page @@ -265,31 +244,21 @@ PINNED_FPRS=( # release statement, so openssh is on the update path. "7168B983815A5EEF59A4ADFD2A3F414E736060BA" # Damien Miller, OpenSSH - # From https://www.greenwoodsoftware.com/less/pubkey.asc (retrieved - # 2026-09-27), linked from the download page beside each release's .sig. - # DSA-1024 signing with SHA-1: weak, as docs/supply-chain.md says. - "AE27252BD6846E7D6EAE1DD6F153A7C833235259" # Mark Nudelman, less + # https://www.greenwoodsoftware.com/less/pubkey.asc (retrieved 2026-09-27) + "AE27252BD6846E7D6EAE1DD6F153A7C833235259" # Mark Nudelman, less (DSA-1024 with SHA-1: weak) - # From https://www.netfilter.org/files/coreteam-gpg-key-0xD70D1A666ACF2B21.txt - # (retrieved 2026-09-27), the "key" linked beside each release on the - # download pages. https://www.netfilter.org/about.html names it the current - # key, valid until 2028-10-12, and the older keys revoked. + # https://www.netfilter.org/files/coreteam-gpg-key-0xD70D1A666ACF2B21.txt (retrieved 2026-09-27); + # https://www.netfilter.org/about.html names it the current key, valid until 2028-10-12. "8C5F7146A1757A65E2422A94D70D1A666ACF2B21" # Netfilter Core Team, libnftnl and nftables - # From https://cmake.org/download/ (retrieved 2026-09-27): beside each - # release's SHA-256.txt.asc the page names the signer 2D2CEF1034921684 and - # links it to the keyserver's lookup of this primary, whose signing - # subkey that is. + # https://cmake.org/download/ (retrieved 2026-09-27) names its signing subkey 2D2CEF1034921684 "CBA23971357C2E6590D9EFD3EC8FEF3A7BFB4EDA" # Brad King, cmake checksum lists - # libexpat names no release signer. This is the key gentoo.org's WKD serves - # for sping@gentoo.org, and the pin means only that; tools/source-notes.tsv - # carries the undesignated-signer caveat. + # libexpat names no signer: gentoo.org's WKD serves this for sping@gentoo.org (source-notes.tsv) "3176EF7DB2367F1FCA4F306B1F9B0E909AF37285" # Sebastian Pipping, expat ) -# Primary and subkey fingerprints. GOODSIG names the signing subkey, while the -# pins and keys.manifest record primaries. +# Primaries and subkeys: GOODSIG names a subkey, while pins and keys.manifest hold primaries. key_fingerprints() { gpg --batch --with-colons --fingerprint --fingerprint "$1" 2>/dev/null \ | awk -F: '$1=="fpr"{print $10}' @@ -317,8 +286,7 @@ key_is_pinned() { _key_in "$1" "${PINNED_FPRS[@]}"; } # --- published key provenance ---------------------------------------------- -# Per key, where a publisher states its fingerprint by a route independent of -# the signature. Tool data, not tree data, so it is found via BASH_SOURCE. +# Where publishers state key fingerprints; tool data, not tree data, so found via BASH_SOURCE. KEY_PROVENANCE="${KRYPTIK_SIGCHECK_PROVENANCE:-$(dirname "${BASH_SOURCE[0]}")/key-provenance.tsv}" declare -a PROV_FPR=() PROV_KIND=() PROV_LOC=() PROV_SIGNS=() @@ -356,8 +324,7 @@ load_key_provenance() { else why="a github locator must be https://github.com/.gpg" fi - # Without the release-author tie the row says only - # "GitHub hosts this key". + # Without the release-author tie the row says only "GitHub hosts this key". [[ "$rest" == *published\ by* ]] \ || why="a github row must record which account published the release" ;; @@ -399,11 +366,8 @@ primaries_in() { | awk -F: '$1 == "pub" { p = 1; next } p && $1 == "fpr" { print toupper($10); p = 0 }' || true } -# anchored_import FPR FILE: merge the key FPR from FILE into the keyring, with -# the revocations, subkeys and signatures FILE carries for it, and nothing else -# FILE holds. A keyring of its own picks it out, and the export is checked to -# be that key alone. Prints FILE's primary fingerprints. Returns 1 when FILE -# lacks FPR, and 2 when FPR is there but cannot be taken alone. +# anchored_import FPR FILE: merge key FPR alone from FILE and print FILE's primary fingerprints; +# returns 1 when FILE lacks FPR, and 2 when FPR is there but cannot be taken alone. anchored_import() { local fpr="$1" file="$2" home found rc=1 found="$(primaries_in "$file" | tr '\n' ' ')" @@ -524,11 +488,7 @@ has_provenance_row() { load_key_provenance -# check_sig [how]: classify gpg's status output. -# An empty datafile checks a signed message, which carries its own data. -# EXPKEYSIG counts as verified: the signature is valid and only the keyring's -# copy of the key has expired (maintainers extend expiry; the keyring lags). -# how, when given, ends each report detail. +# check_sig NAME SIG DATA [HOW]: classify gpg's status; an empty DATA checks a signed message. check_sig() { local name="$1" sigfile="$2" datafile="$3" how="${4:+; $4}" local out signer keyid @@ -549,6 +509,7 @@ check_sig() { out="$(gpg --batch --status-fd 1 --verify "${signed[@]}" 2>/dev/null || true)" + # EXPKEYSIG counts: the signature is valid and only the keyring's copy of the key has expired. if printf '%s' "$out" | grep -qE "^\[GNUPG:\] (GOODSIG|EXPKEYSIG)"; then local kind kind="$(printf '%s' "$out" | sed -n 's/^\[GNUPG:\] \(GOODSIG\|EXPKEYSIG\) .*/\1/p' | head -1)" @@ -617,9 +578,7 @@ check_sig() { if printf '%s' "$out" | grep -q "^\[GNUPG:\] NO_PUBKEY"; then keyid="$(printf '%s' "$out" | sed -n 's/^\[GNUPG:\] NO_PUBKEY //p' | head -1)" - # The key the signature names is circular trust: it proves only who - # signed. So it goes to keys.manifest for an out-of-band audit and is - # never counted as verified. + # A key the signature names proves only who signed: recorded for audit, never verified. if [[ "$FETCH_UNKNOWN" -eq 1 ]]; then if recv_key "$keyid"; then out="$(gpg --batch --status-fd 1 --verify "${signed[@]}" 2>/dev/null || true)" @@ -694,7 +653,6 @@ verify_gnu() { check_sig "$name" "$sig" "${KRYPTIK_SOURCES}/${file}" || true } -# A suffix is not a format: python.org's .sig is Sigstore, its .asc OpenPGP. # A signed message carries its data; a detached signature does not. is_signed_message() { local packets @@ -702,13 +660,13 @@ is_signed_message() { grep -q ':literal data packet:' <<< "$packets" } +# A suffix is not a format: python.org's .sig is Sigstore, its .asc OpenPGP. is_pgp_signature() { [[ -s "$1" ]] || return 1 gpg --batch --list-packets "$1" 2>/dev/null | grep -q ':signature packet:' } -# Try .sig, .asc and .sign; one that is not OpenPGP passes the turn on. The -# report names the suffix found, which the manifest can then declare. +# Try .sig, .asc and .sign, skipping any that is not OpenPGP; the report names the suffix found. verify_any() { local name="$1" url="$2" file="$3" local suffix sig @@ -739,8 +697,7 @@ verify_any() { report "$name" no-signature-upstream "none of .sig/.asc/.sign is published" } -# verify_detached [how]: the detached signature at -# sigurl over the file data, cached under its own name. +# verify_detached NAME DATA SIGURL [HOW]: the signature at SIGURL over DATA, cached under its name. verify_detached() { local name="$1" data="$2" sigurl="$3" how="${4:-}" local sig="${SIGDIR}/${sigurl##*/}" suffix=".${sigurl##*.}" @@ -752,8 +709,7 @@ verify_detached() { report "$name" no-signature-upstream "no ${suffix} published beside the tarball" return fi - # A host can answer a busy runner with a page in place of the file: what - # came back is not kept, and the file is asked for once more. + # A busy host can answer with a page instead of the file: drop it and ask once more. if ! is_pgp_signature "$sig"; then rm -f "$sig" [[ "${KRYPTIK_SIGCHECK_SELFTEST:-0}" == "1" ]] || sleep 5 @@ -766,8 +722,7 @@ verify_detached() { report "$name" signature-not-openpgp "published ${suffix} is not an OpenPGP signature" return fi - # A signed message carries its data: it vouches for this data only if - # what it carries is exactly this data. + # A signed message vouches for DATA only if it carries DATA byte for byte. if is_signed_message "$sig"; then local carried="${sig}.carried" gpg --batch --quiet --yes --output "$carried" --decrypt "$sig" >/dev/null 2>&1 || true @@ -785,8 +740,7 @@ verify_detached() { check_sig "$name" "$sig" "$data" "$how" || true } -# The digest LIST gives FILE: the first field of the line naming it (a -# leading * marks binary mode), or of a list that is one bare digest. +# FILE's digest in LIST: from the line naming it (* marks binary mode), or a lone bare digest. listed_digest() { # listed_digest LIST FILE awk -v f="$2" ' NF >= 2 { n = $NF; sub(/^\*/, "", n); if (n == f) { print tolower($1); hit = 1; exit } } @@ -794,10 +748,7 @@ listed_digest() { # listed_digest LIST FILE END { if (!hit && NR == 1 && bare != "") print bare }' "$1" } -# verify_sums : a detached signature beside -# the file over a checksum list, named for the signature without its -# suffix. The file must match its digest in the list, and the signature the -# list. +# verify_sums NAME URL FILE SIG: FILE must match the list SIG signs (named SIG minus its suffix). verify_sums() { local name="$1" url="$2" file="$3" signame="$4" local listname="${signame%.*}" @@ -825,8 +776,8 @@ verify_sums() { verify_detached "$name" "$list" "${url%/*}/${signame}" "signs ${listname}" } -# kernel.org signs the uncompressed tar (.tar.sign), for the kernel and -# for util-linux, kbd, kmod, iproute2, libcap and e2fsprogs. +# kernel.org signs the uncompressed tar (.tar.sign): the kernel, util-linux, kbd, kmod, +# iproute2, libcap and e2fsprogs. verify_kernel() { local name="$1" url="$2" file="$3" local sign="${SIGDIR}/${file%.xz}.sign" @@ -976,8 +927,7 @@ if [[ "$NOTED" -gt 0 ]]; then warn "${NOTED} source(s) signed by a key no publisher states, accepted by note (${NOTES#"$KRYPTIK_ROOT"/}):" printf ' - %s\n' "${NOTED_LIST[@]}" fi -# A note that no unheld key needed: the key is held now, or the source went. -# A signature that could not be checked this run tried no note. +# An unused note is stale, unless its source is outside the manifest or went unchecked this run. untried_this_run() { # untried_this_run NAME local u for u in "${UNVERIFIABLE_LIST[@]}"; do [[ "$u" == "$1 ("* ]] && return 0; done From d37c459a40e4ccc61a59472ab6a6d28b1085f1aa Mon Sep 17 00:00:00 2001 From: DevomB Date: Wed, 7 Oct 2026 08:46:23 -0700 Subject: [PATCH 13/18] The VM suites' comments are shorter, with their reasons kept: cleanup-held's cut of install-test, state-test, update-test, suite-lib and vm-drive.py, redone on main by hand, comments only Only comment regions are taken from 65a6fbb. Kept where it corrects main: suite-lib's header names what install_disk reads. Changed from it: update-test's header keeps the broken trial, and its channel comment keeps what the not-a-pointer control is; suite-lib keeps the production pair's testctl key. Dropped with the cut, as untrue or repeated: update-test's 'plain http is for development images only' (the suite writes an http channel on either role) and the request-order comment its own pass and fail lines state. --- tools/image/install-test.sh | 18 +++------ tools/image/state-test.sh | 14 ++----- tools/image/suite-lib.sh | 21 ++++------ tools/image/update-test.sh | 80 +++++++++++-------------------------- tools/image/vm-drive.py | 12 ++---- 5 files changed, 44 insertions(+), 101 deletions(-) diff --git a/tools/image/install-test.sh b/tools/image/install-test.sh index 53a0981c..94f3ad1c 100755 --- a/tools/image/install-test.sh +++ b/tools/image/install-test.sh @@ -1,12 +1,10 @@ #!/usr/bin/env bash -# Install from the medium, boot the result from firmware alone (medium gone, -# variables reset), and check that the installer's refusals refuse. +# Install from the medium, boot the result from firmware alone, and check the installer's refusals. # # tools/image/install-test.sh --usb IMG [--disk FILE] [--size 12G] # [--vars clean|enrolled] [--timeout N] [--quick] # -# step 1 install unattended onto a blank disk; check the transcript and, -# from the host, the partition table +# step 1 install onto a blank disk; check the transcript and the partition table # step 2 boot the disk alone: first boot, login, reboot, login, poweroff # step 3 cold boot it again (not with --quick) # step 4 a disk too small, a read-only disk, and an I/O error in the root @@ -208,11 +206,9 @@ EOF refusal_case ioerror "$SIZE" 'writing the root image failed' --blkdebug "${VMDIR}/blkdebug.conf" fi -# The runner always passes --yes, so none of these refusals is the ERASE -# prompt waiting. +# The runner always passes --yes, so no refusal here is the ERASE prompt waiting. -# The disk this system runs from: the medium, the one USB disk in the guest. -# No flag opens it. +# The medium, the guest's one USB disk, is the disk this system runs from: no flag opens it. ctl="${VMDIR}/testctl-medium.img" "${SELF}/mk-testctl.sh" --out "$ctl" --key "$TESTCTL_KEY" install_target=/dev/sda install_replace=1 smoke_poweroff=1 install_wait=5 > /dev/null d="${VMDIR}/refuse-medium.img"; rm -f "$d"; truncate -s "$SIZE" "$d" @@ -222,9 +218,7 @@ want "$t" 'KRYPTIK_INSTALL: BEGIN target=/dev/sda' "medium: want "$t" 'KRYPTIK_INSTALL: .*is the disk this system is running from' "medium: refused as the disk this system runs from, even with --replace-kryptik" deny "$t" 'KRYPTIK_INSTALL: rc=0' "medium: never reported success" -# A control disk signed by some other key: the medium honours the -# kryptik-testctl key its anchor lists and no other, so nothing is armed. The -# disk's poweroff is ignored with the rest, so this boot runs to its timeout. +# The medium ignores a control disk its testctl key did not sign, poweroff too: this boot times out. ssh-keygen -q -t ed25519 -N '' -C stranger -f "${VMDIR}/stranger-key" < /dev/null ctl="${VMDIR}/testctl-stranger.img" "${SELF}/mk-testctl.sh" --out "$ctl" --key "${VMDIR}/stranger-key" install_target=/dev/vda smoke_poweroff=1 install_wait=5 > /dev/null @@ -241,7 +235,7 @@ state_uuid() { [[ -n "$start" ]] && blkid -p -O "$(( start * 512 ))" -s UUID -o value "$1" 2>/dev/null } -# An old installation: refused without the flag, and left exactly as it was. +# An old installation is refused without the flag, and left as it was. old="${VMDIR}/old-install.img"; rm -f "$old"; cp --sparse=always "$DISK" "$old" ctl="${VMDIR}/testctl-oldinstall.img" "${SELF}/mk-testctl.sh" --out "$ctl" --key "$TESTCTL_KEY" install_target=/dev/vda smoke_poweroff=1 install_wait=5 > /dev/null diff --git a/tools/image/state-test.sh b/tools/image/state-test.sh index cec7b316..8ee2b1fb 100755 --- a/tools/image/state-test.sh +++ b/tools/image/state-test.sh @@ -1,7 +1,5 @@ #!/usr/bin/env bash -# The installed system's state partition: found on the system's own disk, and -# a boot that says so when it cannot use it (docs/design/boot-and-updates.md, -# sysinit.sh). +# The installed system's state partition: found on its own disk, and a degraded boot when unusable. # # tools/image/state-test.sh --usb IMG [--disk FILE] [--timeout N] # @@ -16,8 +14,7 @@ # step 8 root changes the passphrase with `kryptik state passphrase`: the # old one no longer unlocks, the new one does # -# A degraded boot has no accounts, so only its console is checked; after each -# repair a login must find the user's file. Every disk is a file made here. +# After each repair a login must find the user's file. Every disk is a file made here. set -uo pipefail SELF="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # shellcheck source=/dev/null @@ -46,9 +43,7 @@ source "${SELF}/suite-lib.sh" VARSF="${VMDIR}/state-vars.fd"; cp /usr/share/OVMF/OVMF_VARS_4M.fd "$VARSF" -# A boot that must come up degraded. Nobody can log in to power it off, so it -# is stopped once its report ends, and only its transcript is checked: every -# line below comes before the report's END. +# Nobody can log in to a degraded boot: stop it at its report's END and check the transcript. degraded_boot() { # degraded_boot NAME REASON-REGEX start_vm "$1" DRIVE_TIMEOUT=150 drive "expect:KRYPTIK_SMOKE: END" > /dev/null @@ -144,10 +139,9 @@ sfdisk --part-label "$DISK" 4 kryptik-state >/dev/null 2>&1 || die "relabel back normal_boot state-p5b # ----------------------------------------------------------------- step 6 -- -# Stop the feeder rather than kill it (s6 would restart it): to the timer that -# is a hung userspace, and only the watchdog's reset can bring a second boot. step "step 6: nothing feeds the watchdog: the machine resets itself and comes back with its data" start_vm state-p6 +# The feeder stopped, not killed (s6 would restart it): a hung userspace only the watchdog resets. p6=( "expect:KRYPTIK_SMOKE: END" "login:${TUSER}:${TPASS}" "$(ROOTSH 's6-svc -p /run/service/watchdog && echo FEEDER-STOPPED')" "expect:FEEDER-STOPPED" "expect:KRYPTIK_SMOKE: BEGIN" "expect:KRYPTIK_SMOKE: END" diff --git a/tools/image/suite-lib.sh b/tools/image/suite-lib.sh index 63815322..e1f7e1f5 100755 --- a/tools/image/suite-lib.sh +++ b/tools/image/suite-lib.sh @@ -1,7 +1,6 @@ #!/usr/bin/env bash -# Shared by the installed-system suites: the verdict, the preseeded accounts, -# and wrappers for run-ovmf.sh and vm-drive.py. Source after common.sh with -# SELF set; start_vm reads DISK and VARSF, drive reads DRIVE_TIMEOUT. +# Shared by the installed-system suites; source it after common.sh, with SELF set. +# start_vm reads DISK and VARSF, install_disk TIMEOUT and VMDIR, drive DRIVE_TIMEOUT. # shellcheck disable=SC2034 # read by the suite that sources this PASS=0; FAIL=0 @@ -12,21 +11,17 @@ step() { printf '\n==> %s\n' "$*"; } # The plaintext exists only in the harness; the hashes are what lands on disk. TUSER=tester; TPASS=tester-pw; RPASS=root-pw TUSER_HASH="$(openssl passwd -6 "$TPASS")"; ROOT_HASH="$(openssl passwd -6 "$RPASS")" -# The state passphrase: the installer reads it from the control disk, and -# vm-drive.py (hence the export) answers sysinit with it at every boot. +# The installer reads it from the control disk; vm-drive.py answers sysinit with it (hence export). export KRYPTIK_STATE_PASSPHRASE=state-pw PRESEED=( "preseed_user=${TUSER}" "preseed_password_hash=${TUSER_HASH}" "preseed_root_hash=${ROOT_HASH}" "state_passphrase=${KRYPTIK_STATE_PASSPHRASE}" ) -# The control disks are signed with the kryptik-testctl key the medium's -# anchor lists: the developer keys' for a development medium, and for the -# production pair the one tools/production-pair.sh keeps beside it. +# The testctl key the medium's anchor lists signs the control disks: the developer one, or the +# production pair's (KRYPTIK_TESTCTL_KEY, tools/production-pair.sh). TESTCTL_KEY="${KRYPTIK_TESTCTL_KEY:-${KRYPTIK_WORK}/keys/release/kryptik-testctl}" DRV="${SELF}/vm-drive.py" -# A smoke boot with a transcript of its own. run-ovmf.sh repoints the -# ovmf-serial.latest.log link at every boot on this host, another suite's -# included, so a suite never reads through it. +# A smoke boot with its own transcript: ovmf-serial.latest.log may be another suite's boot. BOOTS=0 smoke() { # smoke NAME [run-ovmf args] -> BOOTLOG; run-ovmf.sh's status BOOTS=$((BOOTS + 1)) @@ -35,9 +30,7 @@ smoke() { # smoke NAME [run-ovmf args] -> BOOTLOG; run-ovmf.sh's status } boot_txt() { tr -d '\r' < "$BOOTLOG"; } -# Every suite starts from a fresh install: a DISK sized from the medium, not a -# constant (see test-disk-size.sh), then the medium's installer run onto it -# with the preseeded accounts. +# Every suite starts from a fresh install onto a DISK sized from the medium (test-disk-size.sh). fresh_disk() { # fresh_disk MEDIUM [test-disk-size.sh args] local size; size="$("${SELF}/test-disk-size.sh" --medium "$@")" || die "could not size the test disk from the medium" rm -f "$DISK"; truncate -s "$size" "$DISK" diff --git a/tools/image/update-test.sh b/tools/image/update-test.sh index dc19d168..28059205 100755 --- a/tools/image/update-test.sh +++ b/tools/image/update-test.sh @@ -1,20 +1,16 @@ #!/usr/bin/env bash -# OS updates on an installed system: install release A, update to B, reboot -# into it, roll back, and check refusals, interruptions and a broken trial. +# OS updates on an installed system: install A, update to B, roll back; refusals, interruptions +# and a broken trial. # # tools/image/update-test.sh --usb-a IMG_A --payload-a DIR_A --payload-b DIR_B # [--foreign DIR] [--disk FILE] [--timeout N] # [--vars clean|enrolled | --vars-file FILE] # -# --foreign DIR a payload signed by a key B does not trust, which B must -# refuse: the other role's build, or another medium's; -# required when B is a production release -# --vars-file FILE the firmware variable store to start from, for media -# signed with a key other than this build's own +# --foreign DIR a payload signed by a key B does not trust (the other role's build, +# or another medium's); required when B is a production release +# --vars-file FILE the variable store to start from, for media another key signed # -# A and B are two stage 06 releases of this tree (make media KRYPTIK_VERSION=... -# twice). Steps 2-7 give the guest the payload on an ext4 disk image; step 8 -# has the net zone fetch it (docs/design/update-channel.md). +# A and B: stage 06 releases (make media KRYPTIK_VERSION=... twice); B's manifest says the role. # # step 1 install A, boot, create a zone volume and a home file # step 2 apply B, reboot: slot b committed, data intact @@ -24,8 +20,8 @@ # manifest for the other role, signed by B's own key # step 4 apply A with --recovery, reboot: slot a # step 5 rollback: slot b again -# step 6 the VM killed mid-write (the slot then named by nothing on the -# ESP, so rollback refuses it), then after arming: both recover +# step 6 the VM killed mid-write (rollback then refuses the unnamed slot), +# then after arming: both recover # step 7 a corrupt trial falls back to slot a, is recorded, needs --retry # step 8 automatic fetching brings B from a loopback release host; applied. # A production image fetches nothing over plain http, and says so @@ -76,11 +72,8 @@ elif [[ "$B_ROLE" == production ]]; then die "a production B must refuse a build it does not trust: name one with --foreign DIR" fi [[ -z "$VARS_FILE" || -f "$VARS_FILE" ]] || die "--vars-file ${VARS_FILE} is not a file" -# Stage 06 publishes B into a channel beside its payload with -# tools/release-channel.sh (this job has no private key): the signed statement -# that B is current, and B under ${VB}/. not-a-pointer, its control, is the -# same text signed by the release key in the manifest's namespace; only a -# development build makes it. +# Stage 06's channel beside B (tools/release-channel.sh). Its control, not-a-pointer, is the +# statement signed in the manifest's namespace, which only a development build makes. CHAN_B="$(dirname "$PAY_B")/channel-${VB}" CHAN_FILES=(latest latest.sig "${VB}/manifest") [[ "$B_ROLE" == development ]] && CHAN_FILES+=(not-a-pointer not-a-pointer.sig) @@ -102,8 +95,7 @@ if [[ -n "$VARS_FILE" ]]; then elif [[ "$VARS" == "enrolled" ]]; then cp "${KRYPTIK_WORK}/keys/sb/vars/enrolled.fd" "$VARSF" else cp /usr/share/OVMF/OVMF_VARS_4M.fd "$VARSF"; fi -# A payload as a plain ext4 disk image, which the guest mounts read-only under -# /run (its root is read-only, so no mount point can be made under /mnt). +# A payload as an ext4 disk image, mounted under /run: the read-only root takes no new mount point. payload_disk() { # payload_disk OUT DIR rm -f "$1"; local bytes; bytes="$(du -sb "$2" | cut -f1)" truncate -s $(( bytes + bytes / 10 + 64 * 1024 * 1024 )) "$1" @@ -112,12 +104,9 @@ payload_disk() { # payload_disk OUT DIR PA="${VMDIR}/payload-a.img"; PB="${VMDIR}/payload-b.img" payload_disk "$PA" "$PAY_A"; payload_disk "$PB" "$PAY_B" -# Variants of A for the refusals, applied on B with --recovery, which admits -# the older version so the check under test is reached (a variant of the -# running version would stop at "nothing to apply"). +# Variants of A, applied with --recovery so the older version reaches the check under test. BAD="${VMDIR}/bad"; rm -rf "$BAD"; mkdir -p "$BAD" -# Payload A hard-linked, and a real copy of each FILE the variant changes in -# place: a root image is gigabytes, and mkfs -d stores a link once. +# A hard-linked, with copies of the FILEs changed in place: mkfs -d stores a link once. mk_variant() { # mk_variant NAME [FILE...] rm -rf "${BAD:?}/$1" cp -al "$PAY_A" "$BAD/$1" 2>/dev/null || cp -a --sparse=always "$PAY_A" "$BAD/$1" @@ -125,15 +114,13 @@ mk_variant() { # mk_variant NAME [FILE...] rm -f "$BAD/$1/$f"; cp --sparse=always "$PAY_A/$f" "$BAD/$1/$f" done } -# The byte at OFFSET, inverted: a fixed value can land on a byte that already -# holds it, and the "modified" payload is then A itself. +# Inverted, not overwritten: a fixed value may already be there and leave the file as it was. flip() { # flip FILE OFFSET local b; b="$(od -An -tu1 -j "$2" -N1 "$1" 2>/dev/null | tr -d ' ')" || b="" [[ -n "$b" ]] || die "flip: ${1} has no byte at offset ${2}" printf '%b' "\\x$(printf '%02x' $(( b ^ 255 )))" | dd of="$1" bs=1 seek="$2" conv=notrunc status=none } -# A variant's FILE must not be the one A's manifest lists, or its refusal -# check could pass on a genuine payload. +# A variant's FILE must differ from A's, or its refusal check could pass on a genuine payload. changed() { # changed NAME FILE local want; want="$(awk -v f="$2" '$3 == f { print $1; exit }' "$PAY_A/manifest")" [[ -n "$want" && "$(sha256sum "$BAD/$1/$2" | cut -c1-64)" != "$want" ]] \ @@ -149,10 +136,7 @@ mk_variant hidden; mkdir -p "$BAD/hidden/lost+found"; echo "ride along" > "$BAD/ # The statement and its control, for step 3 to check offline against the real anchor. mkdir -p "$BAD/statement"; cp "$CHAN_B/latest" "$CHAN_B/latest.sig" "$BAD/statement/" [[ "$B_ROLE" == development ]] && cp "$CHAN_B/not-a-pointer" "$CHAN_B/not-a-pointer.sig" "$BAD/statement/" -# The signature and the role are checked before any file, so these two need -# only a manifest and its signature: the other role's build, and, beside a -# development B, stage 06's role control (B's manifest for production, signed -# by B's own key). +# Signature and role are checked before any file: the foreign and role cases need only a manifest. [[ -n "$FOREIGN" ]] && { mkdir -p "$BAD/foreign"; cp "$FOREIGN/manifest" "$FOREIGN/manifest.sig" "$BAD/foreign/"; } if [[ "$B_ROLE" == development ]]; then ROLECTL="$(dirname "$PAY_B")/role-control-${VB}" @@ -173,9 +157,7 @@ stop_unless_ok() { # stop_unless_ok RC WHAT # ----------------------------------------------------------------- step 1 -- step "step 1: install ${VA}, boot it, create zone data" -# Sized from the medium, with room for one payload: the release step 8 fetches -# is staged on kryptik-state, and every other step applies from the payload -# disk, mounted read-only. +# Room for one payload: the release step 8 fetches is staged on kryptik-state. fresh_disk "$USB_A" --payloads 1 install_disk update-install "$USB_A" "${INSTALL_VARS[@]}" && green "A installed" || { red "A did not install"; exit 1; } @@ -363,16 +345,14 @@ drive "expect:BdsDxe: starting Boot" \ rc=$?; stop_vm [[ "$rc" -eq 0 ]] && green "broken trial: verity panic, fallback to a, trial-failed recorded, refused without --retry, rewritten and committed with it; data intact" || red "step 7 drive failed" txt | grep -q 'boot-success: trial slot b did NOT boot' && green "boot-success named the failed trial" || red "boot-success did not record the failed trial" -# loglevel=4 keeps the kernel banner off the console (every "Linux version" is -# boot-smoke's, from userspace), so boots are counted by the firmware's line. +# loglevel=4 hides the kernel banner ("Linux version" is boot-smoke's): count the firmware's starts. starts="$(txt | grep -c 'BdsDxe: starting Boot')"; ups="$(txt | grep -c 'KRYPTIK_SMOKE: END')"; panics="$(txt | grep -c 'Kernel panic')" if [[ "$starts" -ge 3 && "$ups" -ge 2 && "$panics" -ge 1 ]]; then green "three boots in one session: the corrupt trial (panicked), the fallback and the retried trial (both reached userspace)"; else red "expected three boots: firmware starts=${starts}, userspace ends=${ups}, panics=${panics}"; fi # ----------------------------------------------------------------- step 8 -- if [[ "$B_ROLE" == development ]]; then step "step 8: ${VB} once more, fetched by the net zone and staged by zone 0" else step "step 8: ${VB}'s statement over plain http, by which a production image fetches nothing"; fi -# Step 7 leaves B committed with nothing newer to fetch, so roll back to A -# first. +# Step 7 leaves B committed with nothing newer to fetch, so roll back to A first. start_vm update-p8 drive "expect:KRYPTIK_SMOKE: END" "login:${TUSER}:${TPASS}" \ "$(ROOTSH 'kryptik-update rollback && echo RB8-OK')" "expect:armed: the next boot tries slot a" "expect:RB8-OK" \ @@ -384,9 +364,7 @@ rc=$?; stop_vm [[ "$rc" -eq 0 ]] && green "back on slot a (${VA}) by rollback, with room for one staged release" || red "step 8: the rollback to slot a failed" stop_unless_ok "$rc" "step 8 rollback" -# The release host serves the channel stage 06 published, as it stands, on -# loopback (10.0.2.2 to the guest); plain http is for development images only. -# The trap stops the host however the suite ends. +# The release host serves stage 06's channel on loopback; the trap stops it however the suite ends. CHAN_LOG="${VMDIR}/channel-requests.log"; : > "$CHAN_LOG"; rm -f "${VMDIR}/channel.port" python3 "${SELF}/release-host.py" "$CHAN_B" "${VMDIR}/channel.port" "$CHAN_LOG" > "${VMDIR}/channel-host.err" 2>&1 & CHAN_PID=$! @@ -395,21 +373,13 @@ for _ in $(seq 50); do [[ -s "${VMDIR}/channel.port" ]] && break; sleep 0.1; don CHAN_PORT="$(cat "${VMDIR}/channel.port" 2>/dev/null)" [[ -n "$CHAN_PORT" ]] || die "the release host did not start: $(cat "${VMDIR}/channel-host.err")" -# Restart the net zone so it reads the new update.conf, as zones-check.sh -# does, and wait for a new "netzone: READY" line. A ROOTSH command may hold no -# single quote (su -c wraps it in them) and must not exit the shell (the -# driver's marker must still print), hence the subshell. +# A ROOTSH command holds no single quote (su -c adds them) and must not exit: hence the subshell. RESTART_NET='before=$(grep -hc "netzone: READY" /run/uncaught-logs/current 2>/dev/null); before=${before:-0}; s6-svc -d /run/service/net-zone; sleep 3; s6-svc -u /run/service/net-zone; (i=0; until [ "$(grep -hc "netzone: READY" /run/uncaught-logs/current 2>/dev/null || true)" -gt "$before" ]; do i=$((i+1)); [ $i -lt 90 ] || exit 1; sleep 1; done) && echo NET-RESTARTED || echo NET-NOT-READY' -# Waits on progress, not a clock: done at "complete", failed after 100 s with -# no change (the net zone polls once a minute), and after six minutes of -# arrival it passes, for the next wait to take over. +# Ends at "complete"; fails after 100 s unchanged (polls are a minute apart); hands on after 6 min. wait_arrival() { printf '%s' '(prev=; same=0; i=0; while [ $i -lt 72 ]; do s="$(kryptik update status | sed -n "s/^staged *//p")"; case "$s" in *"bytes, complete"*) echo ARRIVED-WHOLE; exit 0 ;; esac; if [ "$s" = "$prev" ]; then same=$((same+1)); else same=0; prev="$s"; fi; [ $same -lt 20 ] || { echo "STALLED at: $s"; exit 1; }; i=$((i+1)); sleep 5; done; echo "still arriving: $s")'; } -# Waits until status shows $1, then prints $2 for the driver to expect (echo is -# off, so only the output shows it). Giving up fails the run: step. +# Prints $2 once status shows $1 (echo is off, so only output shows it); giving up fails the step. wait_status() { printf '(i=0; until kryptik update status | grep -q "%s"; do i=$((i+1)); [ $i -lt 72 ] || exit 1; sleep 5; done) && echo %s || { kryptik update status; false; }' "$1" "$2"; } -# The step assumes slot a; check, as the firmware's own boot order can still -# name the last slot tried. The statement is signed, so it is taken over plain -# http on either role. +# The firmware may boot the last slot tried, so check for a; a signed statement is safe over http. UP_TO_STATEMENT=("expect:KRYPTIK_SMOKE: END" "login:${TUSER}:${TPASS}" \ "$(ROOTSH 'echo P8B-BOOTED-$(sed -n "s/^slot=//p" /run/kryptik/boot-identity | head -1)')" "expect:P8B-BOOTED-a" \ "$(ROOTSH "mkdir -p /etc/kryptik && printf \"channel = http://10.0.2.2:${CHAN_PORT}/\\n\" > /etc/kryptik/update.conf && echo CONF-OK")" "expect:CONF-OK" \ @@ -462,8 +432,6 @@ else last_boot="$(txt | sed -n 's/^KRYPTIK_SMOKE: os_id=.* version_id=//p' | tail -1)" [[ "$first_boot" == "$VA" && "$last_boot" == "$VB" ]] && green "the guest booted ${VA} and, after the fetched update, reports ${VB}" \ || red "the guest's boots in this step: first ${first_boot:-none}, last ${last_boot:-none}; wanted ${VA} then ${VB}" - # What the release host was asked for: the statement, then the manifest and - # its signature before anything large. first="$(awk '{sub("^/", "", $1); if (!seen[$1]++) print $1}' "$CHAN_LOG" | head -5 | tr '\n' ' ')" if [[ "$first" == "latest latest.sig ${VB}/manifest ${VB}/manifest.sig "* ]]; then green "the release host was asked for the statement, then the manifest and its signature, before any image"; else red "the release host was asked in another order: ${first}"; fi fi diff --git a/tools/image/vm-drive.py b/tools/image/vm-drive.py index 78d866b9..5f1017c6 100755 --- a/tools/image/vm-drive.py +++ b/tools/image/vm-drive.py @@ -70,8 +70,7 @@ def _read(self): if self.log: self.log.write(d); self.log.flush() if self.passphrase: - # From the last answer, or just before this read: a prompt may - # straddle two reads, and none is answered twice. + # A prompt may straddle reads: search back a little, but never before the last answer. for m in UNLOCK.finditer(self.all, max(self.answered, len(self.all) - len(d) - 80)): self.answered = m.end() self.send_secret(self.passphrase) @@ -155,9 +154,7 @@ def knock(self, regex, timeout=None, every=5): self._read() def login(self, user, password): - # agetty (built with AGETTY_RELOAD) prints "login:" only after input, - # and flushes input that arrives within a second of starting or waking, - # so one Enter can be lost: knock until the prompt appears. + # agetty prints "login:" only after input and may drop an early Enter, so keep knocking. self.drain(1) self.knock(r"login: ?$", self.timeout) self.send(user) @@ -192,10 +189,7 @@ def su(self, password, cmd): return self.finish(tag, cmd) def finish(self, tag, cmd): - # Wait for the exit marker but leave the command's output for the - # steps that follow; only the marker is dropped. A shutdown message - # that matches instead stays, and is returned in place of a status. - # The status is read once its line has ended, as line_value's is. + # Only the marker goes, once its line ends; a shutdown message stays, returned in its place. rx = re.compile(rf"{tag}=(\d+)(?=\r?\n)|Power down|reboot: Restarting|Restarting system".encode(), re.M) deadline = time.time() + self.timeout while True: From 548301239243797897eb0a8a6355549749390b37 Mon Sep 17 00:00:00 2001 From: DevomB Date: Wed, 7 Oct 2026 08:47:29 -0700 Subject: [PATCH 14/18] The session's and the launcher's comments are shorter, with their reasons kept: cleanup-held's cut of tools/desktop/kryptik-launch.c and kryptik-session, redone on main by hand, comments only Only comment regions are taken from 65a6fbb. Changed from it: the launcher's header keeps that it is the daemon's one client for the keys and the menu, and that the proxy socket is bound into the zone so a zone never sees the compositor's own; its update verbs keep that apply answers once the slot is written; the session keeps that anything needing root goes to the launch daemon. --- tools/desktop/kryptik-launch.c | 37 ++++++++++++---------------------- tools/desktop/kryptik-session | 17 ++++++---------- 2 files changed, 19 insertions(+), 35 deletions(-) diff --git a/tools/desktop/kryptik-launch.c b/tools/desktop/kryptik-launch.c index 25360142..e72d0b92 100644 --- a/tools/desktop/kryptik-launch.c +++ b/tools/desktop/kryptik-launch.c @@ -1,6 +1,5 @@ -/* kryptik-launch: start a program in a zone from the desktop session. The - * client of kryptikd's launch daemon (a root-owned socket open to group - * `kryptik`); the compositor's keybindings and the chrome's menu run only this. +/* kryptik-launch: start a program in a zone, as a client of kryptikd's launch daemon (a + * root-owned socket open to group kryptik); the compositor's keys and the chrome's menu run only this. * * kryptik-launch [--ask | --passphrase-fd N] [--no-display] ZONE -- COMMAND [ARG...] * kryptik-launch --stop ZONE @@ -13,14 +12,10 @@ * kryptik-launch --update status|fetch|apply show, fetch or install a release * kryptik-launch --update auto on|off fetch each release as it is announced, or when asked * - * With a display, the zone's kryptik-wlproxy is started if needed at - * $XDG_RUNTIME_DIR/kryptik/ZONE/wayland-0; the daemon binds that socket into - * the zone, which never sees the compositor's own. - * - * --ask: for an encrypted zone, read the passphrase on the controlling terminal, - * or without one hand over to kryptik-chrome --prompt, which calls back with - * --passphrase-fd. The passphrase travels as a descriptor (SCM_RIGHTS), never - * in argv or environ. + * A display goes through the zone's own proxy at $XDG_RUNTIME_DIR/kryptik/ZONE/wayland-0, which the + * daemon binds into the zone: a zone never sees the compositor's own socket. + * --ask reads an encrypted zone's passphrase on the controlling terminal, or hands over to + * kryptik-chrome --prompt; it travels as a descriptor, never in argv or environ. */ #define _GNU_SOURCE #include @@ -189,8 +184,7 @@ static const char *ensure_proxy(const char *zone) if (stat(upstream, &st) != 0 || !S_ISSOCK(st.st_mode)) die("no compositor at %s", upstream); - /* The log is in /run, which is RAM, and each proxy bounds only its own - * lines: a new proxy starts a new file, and the last one is kept. */ + /* The log is in RAM and each proxy bounds only its own lines: keep one old file. */ (void)rename(logfile, oldlog); pid_t pid = fork(); if (pid < 0) @@ -201,8 +195,7 @@ static const char *ensure_proxy(const char *zone) int log = open(logfile, O_WRONLY | O_CREAT | O_TRUNC | O_APPEND, 0600); if (null < 0 || log < 0 || dup2(null, 0) < 0 || dup2(log, 1) < 0 || dup2(log, 2) < 0) _exit(127); - /* The proxy gets stdio only, never the passphrase fd or other session - * descriptors. */ + /* The proxy gets stdio only, never the passphrase fd or any other descriptor. */ if (close_range(3, ~0U, 0) < 0) _exit(127); execl(PROXY_BIN, "kryptik-wlproxy", "--zone", zone, "--listen", sock, "--upstream", upstream, @@ -344,8 +337,7 @@ static void secret_from_stdin(const char *prompt, char *buf, size_t size) die("empty passphrase"); } -/* The daemon's wifi verbs (kryptikd's wifi.rs): it validates, writes the net - * zone's file and restarts it. The passphrase goes in the request body. */ +/* The daemon's wifi verbs (kryptikd's wifi.rs); the passphrase goes in the request body. */ static int wifi_main(int argc, char **argv) { const char *mode = argv[1]; @@ -393,9 +385,8 @@ static int wifi_main(int argc, char **argv) return ok ? 0 : 1; } -/* The daemon's update verbs (kryptikd's update.rs). The reply is an `ok` line - * and text for the user, or one `error:` line; `apply` answers once - * kryptik-update has written the slot. */ +/* The daemon's update verbs (kryptikd's update.rs): an `ok` line and text, or one `error:` line; + * `apply` answers once kryptik-update has written the slot. */ static int update_main(int argc, char **argv) { int automatic = argc == 4 && strcmp(argv[2], "auto") == 0 && (strcmp(argv[3], "on") == 0 || strcmp(argv[3], "off") == 0); @@ -498,16 +489,14 @@ int main(int argc, char **argv) const char *wl = no_display ? NULL : ensure_proxy(zone); - /* Every string in the request is counted, the socket path included (up to - * PATH_MAX); 1024 covers the fixed words. */ + /* Every string is counted, the socket path included; 1024 covers the fixed words. */ size_t cap = 1024 + strlen(zone) + (wl ? strlen(wl) : 0); for (i = 0; i < ncmd; i++) cap += strlen(cmd[i]) + 8; char *req = malloc(cap); if (!req) die("out of memory"); - /* snprintf returns the length it wanted: check each piece fitted before - * advancing, or a short buffer becomes a write past its end. */ + /* snprintf returns the length it wanted: unchecked, len would run past the buffer. */ size_t len = 0; #define PUT(...) do { \ int n_ = snprintf(req + len, cap - len, __VA_ARGS__); \ diff --git a/tools/desktop/kryptik-session b/tools/desktop/kryptik-session index a5d90ae0..c7c2821b 100755 --- a/tools/desktop/kryptik-session +++ b/tools/desktop/kryptik-session @@ -1,8 +1,7 @@ #!/bin/sh -# kryptik-session: the zoned desktop, exec'd by /etc/profile.d for a tty1 login. -# Runs as that user, never with privilege: anything that needs root, starting -# zones included, goes to the launch daemon through kryptik-launch. The user -# needs groups seat (seatd) and kryptik (the daemon); kryptik-firstboot grants both. +# kryptik-session: the zoned desktop, exec'd by /etc/profile.d for a tty1 login, never as root. +# Anything that needs root, starting zones included, goes to the launch daemon (kryptik-launch). +# Needs groups seat (seatd) and kryptik (the launch daemon); kryptik-firstboot grants both. umask 077 @@ -10,8 +9,7 @@ if [ "$(id -u)" -eq 0 ]; then echo "kryptik-session: the desktop is an ordinary user's session, not root's" >&2 exit 1 fi -# A degraded state partition (sysinit.sh) could keep nothing: say so and exit, -# which ends the login. +# A degraded state partition (sysinit.sh) keeps nothing: exit, which ends the login. if [ -r /run/kryptik/state-degraded ]; then echo "kryptik-session: NOT starting the desktop: state is degraded ($(cat /run/kryptik/state-degraded))." >&2 echo "kryptik-session: nothing would persist. Repair the state partition, or boot the install medium." >&2 @@ -32,8 +30,7 @@ esac export XDG_RUNTIME_DIR="$rt" export XDG_SESSION_TYPE=wayland export XDG_CONFIG_HOME="${XDG_CONFIG_HOME:-$HOME/.config}" -# No WAYLAND_DISPLAY: wlroots would take it as a display to nest in. dwl names -# its own socket and exports it to the chrome. +# wlroots would nest in a set WAYLAND_DISPLAY; dwl names its own socket for the chrome. unset WAYLAND_DISPLAY WAYLAND_SOCKET DISPLAY # The layout sysinit loaded on the console, for dwl's xkb keymap; every zone's # clients are handed the compositor's. @@ -54,7 +51,5 @@ log="$rt/kryptik/session.log" done } >> "$log" -# dwl draws the zone borders (build/desktop/dwl-config.h) and pipes its status -# to the chrome, which shows the focused zone and is the launcher. When dwl -# exits, the login ends and tty1 returns to the login prompt. +# dwl pipes its status to the chrome; when dwl exits, the login ends. exec /usr/bin/dwl -s /usr/bin/kryptik-chrome 2>> "$log" From 0386d23870bcc21f4512f705ec77b2c28850c8cb Mon Sep 17 00:00:00 2001 From: DevomB Date: Wed, 7 Oct 2026 09:12:47 -0700 Subject: [PATCH 15/18] kryptikd's broker, launch daemon, update, zone and command comments are shorter, with their reasons kept: cleanup-held's cut of kryptikd's sources, probes and cli suite, redone on main by hand, comments only Only comment regions are taken from 65a6fbb. Where the cut dropped a reason a shorter one is kept: the broker's peer uid is the kernel's, a zone reaches only its own clipboard and a move crosses once, None refuses every transfer, the destination is looked up again as it may have restarted, the identity switch is restored on every path; update.rs never resolves under what the net zone reports, never stages past the signed size, and its channel file is the image's; serve.rs's socket directory admits the group alone; stop signals a launcher only while its pid and start time match; and the suites' charters and orderings. Kept where the cut corrects main: the broker's request table gains transfer and clipboard-move, cli.sh's header its exit codes. --- compartments/kryptikd/Cargo.toml | 3 +- .../kryptikd/probes/boundary-checks.sh | 49 +++----- compartments/kryptikd/probes/fixtures.sh | 32 ++---- compartments/kryptikd/src/broker.rs | 105 +++++++----------- compartments/kryptikd/src/broker/tests.rs | 50 +++------ compartments/kryptikd/src/main.rs | 81 +++++--------- compartments/kryptikd/src/serve.rs | 83 +++++--------- compartments/kryptikd/src/update.rs | 77 +++++-------- compartments/kryptikd/src/update/tests.rs | 2 +- compartments/kryptikd/src/zone.rs | 63 ++++------- compartments/tests/cli.sh | 40 +++---- 11 files changed, 209 insertions(+), 376 deletions(-) diff --git a/compartments/kryptikd/Cargo.toml b/compartments/kryptikd/Cargo.toml index c3220108..44f40443 100644 --- a/compartments/kryptikd/Cargo.toml +++ b/compartments/kryptikd/Cargo.toml @@ -5,8 +5,7 @@ edition = "2021" description = "Kryptik compartment manager: zone lifecycle and isolation" license = "GPL-2.0-or-later" -# kryptikd is the most privileged process on the system, so it depends on -# `libc` alone (ADR-010); anything else, TOML parsing included, is written here. +# The most privileged process on the system depends on libc alone; even TOML parsing is ours (ADR-010). [dependencies] libc = "0.2" diff --git a/compartments/kryptikd/probes/boundary-checks.sh b/compartments/kryptikd/probes/boundary-checks.sh index d1912d59..980df64f 100755 --- a/compartments/kryptikd/probes/boundary-checks.sh +++ b/compartments/kryptikd/probes/boundary-checks.sh @@ -1,10 +1,8 @@ #!/usr/bin/env bash -# Regression probes for the kryptikd zone boundary, through the real launcher -# (`kryptikd run`). Exit status is the number of failures. +# Regression probes for the zone boundary, through `kryptikd run`; the exit status counts failures. +# Unprivileged, what needs root or the target kernel is reported as not run; zones-test.sh also +# runs it as root on the installed system. # compartments/kryptikd/probes/boundary-checks.sh [path/to/kryptikd] -# Unprivileged on a developer host, where what needs root and the target kernel -# is reported as not run; also run as root on the installed system -# (tools/image/zones-test.sh). set -u HERE="$(cd "$(dirname "$0")" && pwd)" @@ -15,9 +13,7 @@ K="$(cd "$(dirname "$K")" && pwd)/$(basename "$K")" # shellcheck source=fixtures.sh source "$HERE/fixtures.sh" -# As root, zones need an explicit identity (kryptikd will not map a zone's root -# to uid 0), and the fixture roots must belong to it: setup drops to that uid -# before it opens the data directory. +# As root, zones need an identity other than uid 0, and setup opens their data as that uid. IDFLAGS=() if [[ "$(id -u)" -eq 0 ]]; then IDFLAGS=(--zone-uid 100000 --zone-gid 100000) @@ -33,8 +29,7 @@ fail() { echo "FAIL $1"; shift; [[ $# -gt 0 ]] && printf '%s\n' "$1" | sed 's/^ skip() { echo "SKIP $1"; SKIPS=$((SKIPS+1)); } head_() { echo; echo "== $1"; } -# Noise every launch prints that no check is about: the experimental hint, -# the swap caveat, inherited groups an unprivileged launcher cannot drop. +# Lines every launch prints that no check is about. denoise() { grep -v 'KRYPTIK_EXPERIMENTAL=1 -\|supplementary host group\|not secure erasure\|auto-approve-transfers: every\|NOT encrypted at rest' } @@ -43,9 +38,7 @@ denoise() { zrun() { local zone="$1"; shift [[ "${1:-}" == "--" ]] && shift - # No pipe, so ZRC is the launcher's status. stdout and stderr are captured - # apart and joined after, zone output first: on one pipe they interleave - # mid-line. + # Files, not a pipe: ZRC stays the launcher's status, and the two streams cannot interleave. local out="$F/zrun.out" err="$F/zrun.err" timeout 60 "$K" run "$zone" "${ZFLAGS[@]}" "${IDFLAGS[@]}" --zones "$F/zones" --rootfs "$F/roots" -- "$@" > "$out" 2> "$err" ZRC=$? @@ -134,8 +127,7 @@ import socket for n,f,t in [('AF_VSOCK',40,1),('AF_ALG',38,5),('AF_PACKET',17,2)]: try: socket.socket(f,t); print(n,'OPENED') except OSError as e: print(n,'refused',e.errno)" -# The kernel runs a family's create code before it refuses a pair (EOPNOTSUPP, -# 95), so the filter refuses every family but AF_UNIX first (97). +# The kernel runs a family's create code before refusing a pair, so the filter refuses first (97). MATCH="^unix-pair inet 97$" check "socketpair(2) makes AF_UNIX pairs and refuses AF_INET by family" 0 /usr/bin/python3 -c " import socket a,b=socket.socketpair(); a.close(); b.close() @@ -162,8 +154,7 @@ MATCH="owns the NIC" checkz routedraw "a zone that does not own the NIC may not MATCH="policy" checkz nopolicy "a policy file that does not exist is a refusal, not a fallback" 1 /bin/sh -c "echo RAN-ANYWAY" MATCH="ptrace" checkz badpolicy "a policy may not re-allow something on the denied list" 1 /bin/sh -c "echo RAN-ANYWAY" -# Landlock: a zone policy is a second layer, and layers intersect, so it can -# only take access away. +# A Landlock policy is a second layer, and layers intersect, so it can only take access away. MATCH="^ok$" checkz narrowed "a zone with a Landlock policy still runs and can read its root" 0 /bin/sh -c "test -r /usr/bin/env && echo ok" MATCH="^ok$" checkz narrowed "... and writes where the policy grants write" 0 /bin/sh -c "echo x > /tmp/f && echo x > /dev/null && echo ok" MATCH="^denied$" checkz narrowed "... but NOT its own HOME, which the base rules alone would allow" 0 /bin/sh -c "echo x > \$HOME/f 2>/dev/null && echo WROTE || echo denied" @@ -177,8 +168,7 @@ MATCH="symbolic link" checkz swapped "... which its next start refuses rather th # --------------------------------------------------------------------------- head_ "F. Guarantees a build cannot give are refused, not implied" -# Unprivileged, the refusal is that a LUKS2 volume needs a root launch; as -# root, that the passphrase is missing. Both say "is encrypted:". +# Unprivileged it needs a root launch, as root a passphrase; both refusals say "is encrypted:". MATCH="is encrypted:" checkz sealed "a zone declaring encrypted storage does not start on a plain directory" 1 /bin/sh -c "echo RAN-ANYWAY" if "$K" run capped "${ZFLAGS[@]}" "${IDFLAGS[@]}" --zones "$F/zones" --rootfs "$F/roots" -- /bin/sh -c "echo LIMITS-RAN" 2>&1 | grep -q LIMITS-RAN; then pass "[limits] is enforced here: cgroups are creatable and the zone ran" @@ -214,16 +204,13 @@ MATCH="^kryptik-broker 1 zone=probe$" check "a zone reaches its own broker and i " MATCH="^error: unknown verb$" check "an unknown verb is refused" 0 /usr/bin/python3 -c "$BRK" "steal " -# The clock (docs/design/time.md): only the zone that holds the network may -# claim a time. A malformed claim is refused at parse time, before the caller -# is checked. +# Only the zone holding the network may claim a time (docs/design/time.md); parsing comes first. MATCH="does not hold the network" check "the clock's verb is refused from a zone that does not hold the network" 0 /usr/bin/python3 -c "$BRK" "time-offset 5 4 " MATCH="is not an offset in seconds" check "a time claim outside the grammar is refused at parse time" 0 /usr/bin/python3 -c "$BRK" "time-offset 1e9 4 " -# The update channel (docs/design/update-channel.md): only the zone that holds -# the network may bring a release. Other zones are refused before any payload -# is read, and a malformed request at parse time, before that. +# Only the zone holding the network may bring a release (docs/design/update-channel.md); other +# zones are refused before any payload is read. MATCH="does not hold the network" check "a statement of what is current is refused from a zone that does not hold the network" 0 /usr/bin/python3 -c "$BRK" "update-latest 5 3 helloabc" MATCH="does not hold the network" check "asking whether a release is wanted is refused from a zone that does not hold the network" 0 /usr/bin/python3 -c "$BRK" "update-poll @@ -255,10 +242,7 @@ s=socket.socket(socket.AF_UNIX); s.connect("/run/kryptik/broker") s.sendmsg([("transfer %s %s\n"%(dest,name)).encode()],[(socket.SOL_SOCKET,socket.SCM_RIGHTS,array.array("i",[fd]))]) print(s.recv(300).decode().strip())' -# A real transfer between two running zones: packet announces itself, waits -# for the file and reports it; probe then sends one. Waits are bounded polls, -# not sleeps, as zone start-up time varies widely. The broker creates the file -# under its final name before filling it, so an empty file is still landing. +# packet polls for a non-empty file: the broker creates it under its final name before filling it. "$K" run packet "${ZFLAGS[@]}" "${IDFLAGS[@]}" --zones "$F/zones" --rootfs "$F/roots" -- /usr/bin/python3 -u -c " import os,time print('PACKET-UP') @@ -293,8 +277,7 @@ MATCH="not on the zone" check "a file from the zone's tmpfs, not its data mount, MATCH="transfer limit is 16" check "a file over the destination's [transfer] max_bytes is refused" 0 /bin/sh -c "printf '%017d' 0 > /home/probe/big && python3 -c '$TX' packet big /home/probe/big 0" MATCH="not running" check "a destination that is not running is refused" 0 /bin/sh -c "echo x > /home/probe/f && python3 -c '$TX' packet f /home/probe/f 0" ZFLAGS=() -# Consent is asked only for a running destination. Nobody answers here, so a -# short deadline makes that a refusal rather than a timeout. +# Consent is asked only for a running destination; a short deadline makes no answer a refusal. "$K" run packet "${IDFLAGS[@]}" --zones "$F/zones" --rootfs "$F/roots" -- /usr/bin/python3 -u -c " import time print('PACKET-UP') @@ -309,9 +292,7 @@ MATCH="single path component" check "a name carrying a path separator is refused # --------------------------------------------------------------------------- head_ "H. The zone dies with its launcher" -# MARK is argv[0] of the zone's command, so `pgrep -f "^$MARK"` matches it and -# not the launcher, whose command line carries it further along. The zone must -# be up before its launcher is signalled; both waits are bounded polls. +# MARK is argv[0] of the zone's command, so "^$MARK" matches it and not the launcher. MARK="kryptik-probe-sleep-$$" zone_up() { for _ in $(seq 1 300); do pgrep -f "^$MARK" >/dev/null && return 0; kill -0 "$1" 2>/dev/null || return 1; sleep 0.1; done; return 1; } zone_gone() { for _ in $(seq 1 100); do pgrep -f "^$MARK" >/dev/null || return 0; sleep 0.1; done; return 1; } diff --git a/compartments/kryptikd/probes/fixtures.sh b/compartments/kryptikd/probes/fixtures.sh index 39369cd0..0899775d 100755 --- a/compartments/kryptikd/probes/fixtures.sh +++ b/compartments/kryptikd/probes/fixtures.sh @@ -1,7 +1,5 @@ #!/usr/bin/env bash -# Synthetic zones for the kryptikd boundary probes, sourced by boundary-checks.sh. -# Exports $F, a temporary root (removed on exit) holding zones/, zones/policy/ -# and roots/. Nothing here is secret; payloads are the strings a check greps for. +# Synthetic zones for boundary-checks.sh, under $F (zones/, zones/policy/, roots/; removed on exit). set -u F="$(mktemp -d -t kryptik-probe-XXXXXX)" @@ -17,8 +15,7 @@ zone() { # zone NAME BODY... mkdir -p "$F/roots/$name" } -# probe: the ordinary zone most checks run in. No policy file: base seccomp -# allowlist and CAP_NET_BIND_SERVICE only. +# probe: where most checks run, with no policy file: the base allowlist and CAP_NET_BIND_SERVICE. zone probe \ '[zone]' 'name = "probe"' \ '[network]' 'mode = "routed"' \ @@ -26,8 +23,7 @@ zone probe \ '[transfer]' 'to = "packet"' \ '[ui]' 'border_color = "#123456"' -# packet: the transfer destination. Its policy allows AF_PACKET but keeps no -# capability, so the socket passes seccomp and the kernel refuses it. +# packet: the transfer destination, allowed AF_PACKET but no capability, so the kernel refuses it. zone packet \ '[zone]' 'name = "packet"' \ '[network]' 'mode = "none"' \ @@ -39,8 +35,7 @@ printf '%s\n' \ '# Synthetic: the socket family is allowed, no capability is kept.' \ 'allow-socket AF_PACKET' > "$F/zones/policy/packet.seccomp" -# capped: declares [limits]. Where cgroups cannot be created it must be -# refused, naming [limits]. +# capped: declares [limits], which must be refused where cgroups cannot be created. zone capped \ '[zone]' 'name = "capped"' \ '[network]' 'mode = "none"' \ @@ -55,16 +50,14 @@ zone keeper \ '[storage]' 'mode = "persistent"' \ '[ui]' 'border_color = "#fedcba"' -# sealed: encrypted storage must refuse to start, never fall back to a plain -# directory. +# sealed: encrypted storage must refuse to start, never fall back to a plain directory. zone sealed \ '[zone]' 'name = "sealed"' \ '[network]' 'mode = "none"' \ '[storage]' 'mode = "encrypted"' 'volume = "/dev/null"' \ '[ui]' 'border_color = "#0f0f0f"' -# nicholder: the NIC owner every zone set needs, and the zone NIC-only policy -# rules apply to. It names no interface, so none is moved. +# nicholder: the NIC owner every zone set needs; it names no interface, so none is moved. zone nicholder \ '[zone]' 'name = "nicholder"' \ '[network]' 'mode = "nic"' \ @@ -77,8 +70,7 @@ printf '%s\n' \ 'keep-capability CAP_NET_RAW' \ 'keep-capability CAP_NET_ADMIN' > "$F/zones/policy/nic.seccomp" -# routedraw: a zone that does not own the NIC may not keep CAP_NET_RAW, -# whatever its policy says. +# routedraw: does not own the NIC, so it may not keep CAP_NET_RAW whatever its policy says. zone routedraw \ '[zone]' 'name = "routedraw"' \ '[network]' 'mode = "routed"' \ @@ -86,8 +78,7 @@ zone routedraw \ '[policy]' 'seccomp = "policy/nic.seccomp"' \ '[ui]' 'border_color = "#ff0088"' -# nopolicy: a missing policy file is a refusal, never a fallback to the base -# rules. +# nopolicy: a missing policy file is a refusal, never a fallback to the base rules. zone nopolicy \ '[zone]' 'name = "nopolicy"' \ '[network]' 'mode = "none"' \ @@ -95,8 +86,7 @@ zone nopolicy \ '[policy]' 'seccomp = "policy/absent.seccomp"' \ '[ui]' 'border_color = "#333333"' -# narrowed: Landlock grants read everywhere and write only in /tmp and /dev, -# so its HOME becomes read-only; the second layer can only subtract. +# narrowed: its Landlock policy grants write only in /tmp and /dev, so its HOME is read-only. zone narrowed \ '[zone]' 'name = "narrowed"' \ '[network]' 'mode = "none"' \ @@ -109,9 +99,7 @@ printf '%s\n' \ 'read-write /tmp' \ 'read-write /dev' > "$F/zones/policy/narrowed.landlock" -# swapped: persistent, and its policy keeps HOME read-only but for work, with -# exec only in work/bin. Write in work lets it swap work/bin for a link to -# HOME, which a rule must refuse at the next start, not follow. +# swapped: can replace work/bin with a link to HOME, which its next start must refuse, not follow. zone swapped \ '[zone]' 'name = "swapped"' \ '[network]' 'mode = "none"' \ diff --git a/compartments/kryptikd/src/broker.rs b/compartments/kryptikd/src/broker.rs index 2caf58d6..eb3c2c37 100755 --- a/compartments/kryptikd/src/broker.rs +++ b/compartments/kryptikd/src/broker.rs @@ -1,9 +1,7 @@ -//! The zone broker (docs/design/broker.md): each zone's one socket into zone 0, -//! serving the clipboard, transfer, time and update verbs. -//! -//! A peer is known by its `SO_PEERCRED` uid, which the kernel asserts and the -//! zone cannot choose; each zone has its own `[identity]` uid range. The pid -//! is never used, since it may be reused. +//! The zone broker (docs/design/broker.md): each zone's socket into zone 0, for the clipboard, +//! transfer, time and update verbs. A peer is known by its `SO_PEERCRED` uid, which the kernel +//! asserts and the zone cannot choose, each zone having its own uid range; never by its pid, +//! which may be reused. use std::cell::Cell; use std::ffi::CString; @@ -19,8 +17,7 @@ use crate::zone::{NetworkMode, Zone}; /// The socket's name in a registry entry, and in the zone's /run/kryptik. pub const SOCKET_NAME: &str = "broker"; -/// Listen on an AF_UNIX socket at `path` that only the zone identity can -/// connect to. A stale file is removed first: this launcher owns the entry. +/// Listen at `path` for the zone identity alone, replacing a stale file: the entry is ours. pub fn listen_at(path: &std::path::Path, uid: u32, gid: u32) -> io::Result { let _ = std::fs::remove_file(path); let c = std::ffi::CString::new(path.as_os_str().as_encoded_bytes()) @@ -60,9 +57,8 @@ pub fn listen_at(path: &std::path::Path, uid: u32, gid: u32) -> io::Result kryptik-broker 1 zone=NAME\n /// clipboard-set \n -> ok\n /// clipboard-get\n -> ok \n | empty\n -/// time-offset \n -> ok ignored | slewed | stepped | stepped after consent\n -/// (from the zone that holds the network, and no other) -/// update-latest \n -> ok current | ok available \n +/// transfer \n (+1 fd) -> ok \n +/// clipboard-move ... -> error: clipboard-move is a zone 0 act, not a zone verb\n +/// time-offset \n -> ok ignored | slewed | stepped | stepped after consent\n +/// update-latest \n +/// -> ok current | ok available \n /// update-poll\n -> idle | fetch need ...\n -/// update-put \n -> ok / | ok complete\n -/// (the same zone, and no other) +/// update-put \n +/// -> ok / | ok complete\n /// anything else -> error: \n /// ``` +/// +/// The time and update verbs are taken from the nic zone alone. #[derive(Debug, PartialEq)] pub enum Request { Version, @@ -107,8 +106,6 @@ pub enum Request { /// The nic zone's claim of the clock's offset from network time (docs/design/time.md). TimeOffset(crate::time::Claim), /// The update channel (docs/design/update-channel.md), nic zone only. - /// `update-latest` is followed by the pointer and its signature, - /// `update-put` by `len` bytes of the named file. UpdateLatest { plen: usize, slen: usize }, UpdatePoll, UpdatePut { name: String, offset: u64, len: usize }, @@ -141,8 +138,7 @@ fn check_zone_name(s: &str) -> Result<(), String> { Ok(()) } -/// A transferred file's name: one path component of 1 to 255 printable ASCII -/// bytes, with no leading dot, so a zone cannot plant dotfiles. +/// A transferred file's name: one component of 1 to 255 printable ASCII bytes, not a dotfile. pub fn check_transfer_name(n: &str) -> Result<(), String> { if n.is_empty() || n.len() > 255 { return Err("name must be 1 to 255 bytes".into()); @@ -215,8 +211,7 @@ pub fn parse_request(line: &str) -> Result { } } -/// `time-offset`, taken only from the nic zone: no other zone has a network of -/// its own to measure with. `time::consider` decides what to believe. +/// `time-offset`, nic zone only: no other zone has a network of its own to measure with. fn handle_time_offset(zone: &Zone, claim: &crate::time::Claim, asking: &dyn Fn() -> bool) -> crate::time::Outcome { time_offset_in( zone, @@ -228,8 +223,7 @@ fn handle_time_offset(zone: &Zone, claim: &crate::time::Claim, asking: &dyn Fn() ) } -/// Why `zone` may not bring an update: only the nic zone can have fetched one. -/// What it sends is still untrusted; `update.rs` judges it. +/// Only the nic zone can have fetched an update; what it sends is still judged by update.rs. fn update_refusal(zone: &Zone) -> Option { (zone.network != NetworkMode::Nic) .then(|| format!("zone {:?} does not hold the network; only the zone that does may bring an update", zone.name)) @@ -304,8 +298,8 @@ pub struct Served<'a> { pub entry: &'a Path, /// The zone directory, to load a destination's zone file. pub zones_dir: &'a Path, - /// st_dev of the zone's /home/, looked up per request because the - /// zone's root is built after its pid 1 starts. None refuses every transfer. + /// st_dev of the zone's /home/, per request: its root is built after pid 1 starts. + /// None refuses every transfer. pub home_dev: &'a dyn Fn() -> Option, /// The development stand-in for the zone 0 prompt. pub auto_approve: bool, @@ -313,21 +307,18 @@ pub struct Served<'a> { pub max_bytes: u64, /// Finds a running destination's root and identity: the registry, or a test directory. pub resolve_dest: &'a dyn Fn(&str) -> Result, - /// Called ten times a second while the user is being asked; the launcher - /// pumps zone output there, and `false` withdraws the question. + /// Called ten times a second while asking: the launcher pumps zone output; `false` withdraws. pub asking: &'a dyn Fn() -> bool, /// Where the broker's lines go; the launcher bounds them per launch. pub log: &'a dyn Fn(&str), - /// Until when this launch asks nothing more after a refusal: every - /// question takes focus in zone 0, so a zone may not raise them in a loop. + /// After a refusal, no new question until then: each one takes focus in zone 0. pub refused_until: &'a Cell>, } /// How long a refused zone waits before it may ask again. pub const REFUSAL_PAUSE: Duration = Duration::from_secs(60); -/// The launcher's `resolve_dest`: the registry says whether the zone runs and -/// as whom, and its pid 1's root leads into its mount namespace. +/// The launcher's `resolve_dest`, from the registry and the zone's pid 1 root. pub fn registry_target(dest: &str) -> Result { let st = match registry::state(dest) { Ok(registry::State::Running { init: Some(st), .. }) if st.still_alive() => st, @@ -448,10 +439,9 @@ fn handle_transfer(s: &Served, dest: &str, name: &str, fds: &[OwnedFd]) -> Resul if st.st_size as u64 > limit { return Err(format!("file is {} bytes; the transfer limit is {limit}{whose}", st.st_size)); } - /* Ask last, once even the destination is known to be running: a question - * whose answer changes nothing teaches people to say yes. Look it up again - * afterwards; holding its root through the wait would pin its mounts, and - * the zone may have restarted meanwhile. */ + /* Ask last, once the destination is known to run: a pointless question teaches people to say + * yes. Then look it up again: holding its root through the wait would pin its mounts, and + * the zone may have restarted. */ if !s.auto_approve { drop((s.resolve_dest)(dest)?); if s.refused_until.get().is_some_and(|t| Instant::now() < t) { @@ -470,8 +460,7 @@ fn handle_transfer(s: &Served, dest: &str, name: &str, fds: &[OwnedFd]) -> Resul deliver(&target, name, src, st.st_size as u64) } -/// Copy into the destination's `incoming/`. Every path is resolved from its -/// root with openat2 and no symlinks, so nothing written leaves its tree. +/// Copy into the destination's `incoming/`, every path resolved in its root without symlinks. fn deliver(target: &Target, name: &str, src: RawFd, size: u64) -> Result<(String, u64), String> { let home = openat2( target.root_fd.as_raw_fd(), @@ -481,10 +470,9 @@ fn deliver(target: &Target, name: &str, src: RawFd, size: u64) -> Result<(String RESOLVE_IN_ROOT | RESOLVE_NO_SYMLINKS | RESOLVE_NO_MAGICLINKS, ) .map_err(|e| format!("destination home is not reachable: {e}"))?; - /* Create as the destination identity: an ephemeral zone's home is a tmpfs - * mounted in its user namespace, which refuses (EOVERFLOW) to create an - * inode for an unmapped uid such as host root. Restored on every path; - * the broker serves one request at a time. */ + /* As the destination identity: an ephemeral home is a tmpfs in the zone's user namespace, + * which refuses (EOVERFLOW) an inode for an unmapped uid such as host root. Restored on + * every path; the broker serves one request at a time. */ let switched = unsafe { libc::geteuid() } == 0; if switched { unsafe { @@ -532,8 +520,7 @@ fn deliver_into(home: RawFd, target: &Target, name: &str, src: RawFd, size: u64) if st.st_uid != target.uid { return Err("incoming/ is not owned by the destination zone".into()); } - /* O_EXCL picks the name: a collision, or a planted symlink (also EEXIST), - * moves on to the next number. */ + // O_EXCL picks the name: a collision or a planted symlink (EEXIST) moves to the next number. for i in 1..=100u32 { let cand = if i == 1 { name.to_string() } else { format!("{name}-{i}") }; match openat2( @@ -575,10 +562,8 @@ fn fill(out: RawFd, src: RawFd, size: u64, target: &Target) -> Result Result { let mut total: u64 = 0; let mut fallback = false; @@ -646,8 +631,7 @@ fn write_all(fd: RawFd, mut data: &[u8]) -> io::Result<()> { Ok(()) } -/// Accept one connection and answer one request. Returns the request's name -/// (Request::name), for logging. +/// Accept one connection and answer its request; returns the request's name for the log. pub fn serve_one(listen_fd: RawFd, s: &Served) -> io::Result> { let fd = unsafe { libc::accept4(listen_fd, std::ptr::null_mut(), std::ptr::null_mut(), libc::SOCK_CLOEXEC) }; if fd < 0 { @@ -662,8 +646,7 @@ pub fn serve_one(listen_fd: RawFd, s: &Served) -> io::Result io::Result> { let zone = s.zone.name.as_str(); let entry = s.entry; @@ -793,8 +776,7 @@ fn recv_first(fd: RawFd, buf: &mut Vec, started: Instant) -> io::Result<(boo } } -/// Receive straight into `buf` until it holds `want` bytes (a length the -/// header was checked against) or the peer reaches EOF, within the deadline. +/// Receive into `buf` until it holds `want` bytes (a checked length) or EOF, by the deadline. fn read_more(fd: RawFd, buf: &mut Vec, want: usize, started: Instant) -> io::Result<()> { let mut filled = buf.len(); if filled >= want { @@ -900,8 +882,7 @@ pub fn clipboard_read(entry: &Path) -> io::Result)>> { Ok(Some((mime, bytes))) } -/// Replace the zone's payload whole, 0600; a planted symlink is replaced, -/// never followed. +/// Replace the zone's payload whole, 0600; a planted symlink is replaced, never followed. pub fn clipboard_write(entry: &Path, mime: &str, bytes: &[u8]) -> io::Result<()> { if !MIME_TYPES.contains(&mime) { return Err(io::Error::new(io::ErrorKind::InvalidInput, format!("unsupported MIME type {mime:?}"))); @@ -912,9 +893,8 @@ pub fn clipboard_write(entry: &Path, mime: &str, bytes: &[u8]) -> io::Result<()> crate::files::write_atomic(&entry.join(CLIPBOARD_FILE), &[mime.as_bytes(), b"\n", bytes], 0o600, None) } -/// `kryptikd clipboard move`: `to` takes `from`'s payload, which leaves -/// `from`: one payload crosses, once. Both are registry entries of running -/// zones. Returns what moved. +/// Move `from`'s payload to `to`, leaving `from` empty: one payload crosses, once. Both are +/// registry entries of running zones; returns what moved. pub fn clipboard_move(from: &Path, to: &Path) -> io::Result<(String, usize)> { let Some((mime, bytes)) = clipboard_read(from)? else { return Err(io::Error::new(io::ErrorKind::NotFound, "nothing on the source zone's clipboard")); @@ -931,8 +911,7 @@ pub fn clipboard_move(from: &Path, to: &Path) -> io::Result<(String, usize)> { Ok((mime, bytes.len())) } -/// The zone 0 gesture, for `kryptikd clipboard move` and serve alike: both -/// zones must be running. Ok is the line to show, Err why it was refused. +/// The zone 0 gesture, for `kryptikd clipboard move` and serve: Ok is the line to show. pub fn move_between(from: &str, to: &str) -> Result { use crate::registry::{self, State}; for z in [from, to] { diff --git a/compartments/kryptikd/src/broker/tests.rs b/compartments/kryptikd/src/broker/tests.rs index a5222b51..a8003e67 100644 --- a/compartments/kryptikd/src/broker/tests.rs +++ b/compartments/kryptikd/src/broker/tests.rs @@ -258,8 +258,7 @@ struct Lab { dev: u64, } -/// Zones a (sender), b and c (plain) and n (nic); a destination root with -/// homes for b and c; the lab directory's filesystem as the data mount. +/// Zones a (sender), b, c and n (nic); homes for b and c; the lab's filesystem as data mount. fn lab(tag: &str, to: &str) -> Lab { use std::os::unix::fs::MetadataExt; let dir = entry(&format!("transfer-{tag}")); @@ -302,8 +301,7 @@ fn resolver(root: std::path::PathBuf) -> impl Fn(&str) -> Result } } -/// How many of this process's descriptors point at `p`; tests run in -/// parallel, so a total count would be noise. +/// How many of our descriptors point at `p` (tests run in parallel: a total would be noise). fn fds_pointing_at(p: &Path) -> usize { std::fs::read_dir("/proc/self/fd") .unwrap() @@ -381,8 +379,7 @@ fn transfer_lands_in_incoming() { let _ = std::fs::remove_dir_all(&lab.dir); } -/// One transfer of notes.txt ("hello transfer", 14 bytes) from a descriptor -/// the sender left at byte 6; returns the reply. +/// Transfer notes.txt (14 bytes) from a descriptor the sender left at byte 6; returns the reply. fn send_notes(sv: &Served, file: &Path) -> String { std::fs::write(file, b"hello transfer").unwrap(); let src = open_flags(file, libc::O_RDONLY); @@ -496,8 +493,7 @@ fn transfer_refusals_precede_copy() { let text = String::from_utf8_lossy(&r); assert!(text.starts_with("error: ") && text.contains(want), "request {req:?}: got {text:?}, wanted {want:?}"); } - /* Every refusal closed what it was handed. Checked after the loop, - * since the cases open all their descriptors up front. */ + // Every refusal closed what it was handed; checked here, as the cases open theirs up front. let left = fds_pointing_at(&file) + fds_pointing_at(&lab.dir) + fds_pointing_at(&big); assert_eq!(left, 0, "a refusal leaked a descriptor in the broker"); // A file on another filesystem than the zone's data mount. @@ -511,16 +507,15 @@ fn transfer_refusals_precede_copy() { } else { eprintln!("/dev/shm is on the same filesystem as the lab; the st_dev case is not exercised here"); } - /* Consent and an unknown data mount each refuse on their own. This sets - * the variable consent's tests set, so it takes their lock. */ + // Consent and an unknown data mount each refuse on their own. This sets the variable + // consent's tests set, so it takes their lock. sv.auto_approve = false; { let _env = crate::consent::ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner()); std::env::set_var("KRYPTIK_CONSENT_DIR", "/nonexistent/kryptik-consent"); let (_, r) = ask_with(&sv, "transfer b f.txt\n", &[ro()], false); assert!(String::from_utf8_lossy(&r).contains("no consent channel"), "{}", String::from_utf8_lossy(&r)); - /* A destination that is not running is refused before any question: - * even with no consent channel, the error is about the destination. */ + // A destination not running is refused before any question, even with no consent channel. sv.resolve_dest = ¬_running; let (_, r) = ask_with(&sv, "transfer b f.txt\n", &[ro()], false); let text = String::from_utf8_lossy(&r); @@ -584,8 +579,7 @@ fn zone_limits_bound_transfer() { #[test] fn dest_resolved_after_consent() { - /* The first lookup finds the zone under `before`; by the answer it runs - * under the lab root, as a zone restarted during the wait would. */ + // Found under `before` first, then under the lab root: a zone restarted during the wait. let lab = lab("again", "b"); let entry_dir = lab.dir.join("entry"); let before = lab.dir.join("before"); @@ -655,8 +649,7 @@ fn dest_resolved_after_consent() { #[test] fn refusal_pauses_questions() { - /* Every question takes focus in zone 0: after a refusal the same launch - * may raise no other for a minute, and is told so without one. */ + // After a refusal the launch raises no question for a minute, and is told so without one. let lab = lab("pause", "b"); let entry_dir = lab.dir.join("entry"); std::fs::create_dir_all(&entry_dir).unwrap(); @@ -719,9 +712,8 @@ fn refusal_pauses_questions() { let second = send(); done.store(true, std::sync::atomic::Ordering::SeqCst); let asked = person.join().unwrap(); - /* Once the pause is over the next request is asked again; with no - * channel, it says so. That asked nobody, so the one right after it is - * not paused: it reaches the channel check again. */ + /* After the pause the next request is asked again (no channel, so it says so); that asked + * nobody, so the one after it is not paused either. */ refused.set(Some(Instant::now())); std::env::set_var("KRYPTIK_CONSENT_DIR", "/nonexistent/kryptik-consent"); let third = send(); @@ -814,8 +806,7 @@ fn parse_request_wire_format() { } } -/// From the nic zone the claim reaches the decision, which refuses for want -/// of a floor; no clock is touched either way. +/// The nic zone's claim reaches the decision (refused: no floor); no clock is touched. #[test] fn only_nic_zone_reports_time() { let claim = crate::time::Claim { offset: 2.0, sources: 3 }; @@ -874,11 +865,8 @@ fn only_nic_zone_brings_update() { assert_eq!(update_refusal(&zone_of("nic", "")), None); } -/// Requests from fuzz-corpus/broker-requests (add any that ever breaks the -/// broker), damaged by a fixed-seed generator and sent down a real -/// connection, each with a fresh descriptor of `attach` when given. No -/// panic, no overrun of the deadline, one well-formed reply, and nothing -/// accepted outside the grammar. Returns (sent, accepted). +/// Serve each fuzz-corpus/broker-requests line damaged by a seeded generator, with a fresh +/// `attach` fd if given: no panic or overrun, one well-formed reply. Returns (sent, accepted). fn fuzz_pass(s: &Served, hello: &str, attach: Option<&Path>) -> (u32, u32) { const CORPUS: &str = include_str!("../../fuzz-corpus/broker-requests"); // xorshift64*: small, seeded, the same sequence everywhere. @@ -961,9 +949,7 @@ fn broker_survives_any_request() { assert!(sent > 2000 && accepted > 50 && accepted < sent, "sent {sent}, accepted {accepted}"); } -/// The same damage past the transfer path's descriptor check: a sender -/// whose policy names b, its data mount the lab's, b running under the lab -/// root, and every request carrying a file of that mount. +/// The same damage on the transfer path: b allowed and running, and a file on every request. #[test] fn broker_survives_any_transfer() { let lab = lab("fuzz", "b"); @@ -997,10 +983,8 @@ fn broker_survives_any_transfer() { assert!(sent > 2000, "sent {sent}"); } -/// The nic zone reaches time-offset's judgement and the update verbs' -/// payloads, and through them this host's clock and update state. As root -/// that would be the real thing, so it runs only unprivileged, where every -/// such write is refused, and with no consent channel for a question. +/// The nic zone's verbs reach this host's clock and update state, so this runs unprivileged +/// only, where every such write is refused, and with no consent channel. #[test] fn broker_survives_any_request_from_the_nic_zone() { if unsafe { libc::geteuid() } == 0 { diff --git a/compartments/kryptikd/src/main.rs b/compartments/kryptikd/src/main.rs index 490bafd9..464c9ebd 100644 --- a/compartments/kryptikd/src/main.rs +++ b/compartments/kryptikd/src/main.rs @@ -1,8 +1,6 @@ -//! kryptikd, the Kryptik compartment manager. -//! -//! Owns zone lifecycle. Runs privileged in zone 0 (ADR-003) and is the only -//! process that creates zones or moves data between them. Anything not built -//! is refused with an error, never a silent no-op. +//! kryptikd, the Kryptik compartment manager: runs privileged in zone 0 (ADR-003) and is the +//! only process that creates zones or moves data between them. Anything not built is refused +//! with an error, never a silent no-op. // cargo fuzz builds this file with cfg(fuzzing), and libFuzzer brings its own main. #![cfg_attr(fuzzing, no_main, allow(dead_code))] @@ -133,9 +131,8 @@ fn main() -> ExitCode { ExitCode::from(2) } }, - /* confine-test ROOTFS TARGET: Landlock-confine to ROOTFS, then read - * TARGET. Exit 0: read succeeded (confinement failed); 4: blocked; - * 1: could not confine. Used by the isolation exit test. */ + /* confine-test ROOTFS TARGET: confine to ROOTFS with Landlock, then read TARGET. Exit 0: + * the read worked, so confinement failed; 4: blocked; 1: could not confine. */ "confine-test" => { let Some(root) = args.get(1) else { eprintln!("confine-test: expected ROOTFS TARGET"); @@ -171,10 +168,7 @@ fn main() -> ExitCode { } } } - /* seccomp-test SYSCALL|PROBE: a forked child installs the zone filter, - * then makes the call (probes: cmd_seccomp_probe). Exit 0: completed; - * 5: SIGSYS; 6: another signal; 7: refused with the intended errno; - * 1: filter not installed. */ + // seccomp-test SYSCALL|PROBE: try it under the zone filter (exit codes: under_zone_filter). "seccomp-test" => { let Some(name) = args.get(1) else { eprintln!("seccomp-test: expected a syscall name"); @@ -197,9 +191,6 @@ fn main() -> ExitCode { } }, "run" => cmd_run(&zone_dir, &args), - /* seccomp-trace [--zone NAME] -- CMD: run CMD under the base filter, or - * NAME's widened by its policy file, and name every call it refuses - * (cmd_seccomp_trace). For writing a policy file. */ "seccomp-trace" => { let Some(sep) = args.iter().position(|a| a == "--") else { eprintln!("seccomp-trace: expected `-- COMMAND`"); @@ -262,9 +253,8 @@ fn main() -> ExitCode { } } -/// Stop a running zone by signalling its launcher, which forwards the signal -/// to pid 1 and escalates to SIGKILL after 5 s. The launcher is signalled only -/// while its pid and start time both still match: pids are reused. +/// Stop a zone; its launcher forwards the signal to pid 1 and escalates to SIGKILL after 5 s. +/// The launcher is signalled only while its pid and start time both match: pids are reused. fn cmd_stop(name: &str, now: bool, base: &str) -> ExitCode { let mut st = match registry::state(name) { Ok(s) => s, @@ -274,8 +264,7 @@ fn cmd_stop(name: &str, now: bool, base: &str) -> ExitCode { } }; - /* "Still starting" lasts a few ms, between claim() and the fork that - * records the launcher pid; `run &` then `stop` often lands in it. */ + // "Still starting" lasts a few ms after claim(); `run &` then `stop` often lands in it. if matches!(st, registry::State::Running { launcher: None, .. }) { std::thread::sleep(std::time::Duration::from_millis(100)); st = match registry::state(name) { @@ -317,10 +306,8 @@ fn cmd_stop(name: &str, now: bool, base: &str) -> ExitCode { println!("zone {name:?} exited while stopping it"); return ExitCode::SUCCESS; } - /* --now kills the zone's pid 1, which takes the pid namespace with - * it, not the launcher: the launcher outlives its zone to unmount - * and close the zone's volume. During setup pid 1 may not yet be - * recorded; signal the launcher gracefully in that case. */ + /* --now kills pid 1 and so the pid namespace, not the launcher, which outlives the + * zone to close its volume. Before pid 1 is recorded the launcher gets SIGTERM. */ let (pid, sig) = match init.filter(|i| now && i.still_alive()) { Some(i) => (i.pid, libc::SIGKILL), None => (l.pid, libc::SIGTERM), @@ -357,9 +344,7 @@ fn cmd_stop(name: &str, now: bool, base: &str) -> ExitCode { } } -/// Close the volume of a zone whose launcher died without closing it, as -/// `gc` does for every zone: otherwise the plaintext stays mounted and its -/// key in the kernel after the zone is reported stopped. +/// Close a volume a dead launcher left open, as `gc` does, so no plaintext outlives a stop. fn close_left_volume(name: &str, base: &str) { if !volume::mappings().iter().any(|z| z == name) || matches!(registry::state(name), Ok(registry::State::Running { .. })) { return; @@ -371,9 +356,8 @@ fn close_left_volume(name: &str, base: &str) { } } -/// Cross-zone paste, a zone 0 gesture: no zone's socket has a verb to fetch -/// another zone's payload. Both zones must be running; the payload lives in -/// one zone's registry entry at a time. +/// Cross-zone paste, a zone 0 gesture: no zone's socket has a verb to fetch another's payload, +/// which lives in one zone's registry entry at a time. fn cmd_clipboard(args: &[String]) -> ExitCode { let (from, to) = match (args.get(1).map(|s| s.as_str()), args.get(2), args.get(3)) { (Some("move"), Some(f), Some(t)) if !f.starts_with("--") && !t.starts_with("--") => (f.as_str(), t.as_str()), @@ -413,8 +397,8 @@ fn cmd_status(name: &str) -> ExitCode { ExitCode::SUCCESS } Ok(registry::State::Running { launcher, init, cgroup, started }) => { - /* core-sched is asked of the kernel (nothing in /proc shows it): own, - * none, no-smt (no sibling thread online) or unavailable. */ + // core-sched is asked of the kernel, as nothing in /proc shows it: own, none, no-smt or + // unavailable. println!( "{name} running launcher {} init {} since {}{}{}", launcher.map(|l| l.pid.to_string()).unwrap_or_else(|| "starting".into()), @@ -542,8 +526,7 @@ fn cmd_check(dir: &Path, target: bool) -> ExitCode { None => println!(" landlock NO"), } - /* Install the filter in a child and make an allowed call. A self-test that - * cannot run is a failure: this command gates a kernel as fit for zones. */ + // A self-test that cannot run is a failure: this command gates a kernel as fit for zones. let self_test = std::env::current_exe() .map_err(|e| format!("current_exe: {e}")) .and_then(|exe| { @@ -566,8 +549,7 @@ fn cmd_check(dir: &Path, target: bool) -> ExitCode { } } - /* On the target only CAP_SYS_ADMIN in the initial namespace may create a - * user namespace. Read the knob, then prove it by trying. */ + // On the target only CAP_SYS_ADMIN may create a user namespace: read the knob, then try. let knob = isolate::userns_restriction_sysctl(); match isolate::probe_userns_restriction() { Ok(true) => println!( @@ -650,8 +632,7 @@ fn cmd_check(dir: &Path, target: bool) -> ExitCode { } } } - /* Parsed as the launcher parses it (spawn.rs), so a file that - * would stop a launch fails the check. */ + // Parsed as spawn.rs does, so a file that would stop a launch fails the check. if let Some(rel) = &z.landlock { let path = policy::resolve(dir, rel); match std::fs::read_to_string(&path) @@ -787,8 +768,7 @@ fn cmd_explain(dir: &Path, name: &str, args: &[String]) -> ExitCode { ExitCode::SUCCESS } -/// Argument-rule and soft-refusal probes for `seccomp-test`; None when `name` -/// is not a probe. +/// Argument-rule and soft-refusal probes for `seccomp-test`; None if `name` is not one. fn cmd_seccomp_probe(name: &str) -> Option { // Runs in the filtered child: 7 if refused with the intended errno, else 0. let probe: fn() -> i32 = match name { @@ -874,8 +854,8 @@ fn run_options_from(args: &[String]) -> Result { }) } -/// An encrypted zone's LUKS2 container, outside any launch -/// (docs/design/encrypted-volumes.md). init and passwd need root. +/// An encrypted zone's LUKS2 container, outside any launch (docs/design/encrypted-volumes.md). +/// init, passwd and destroy need root. /// /// volume init NAME [--size 512M] --passphrase-file F [--zone-uid N --zone-gid N] /// volume passwd NAME --passphrase-file OLD --new-passphrase-file NEW @@ -1058,9 +1038,7 @@ fn cmd_time(args: &[String]) -> ExitCode { } } -/// `kryptikd wifi list | add SSID | forget SSID`: the net zone's Wi-Fi networks, -/// for root and the tests (the session goes through the launch daemon). The -/// passphrase is one line on stdin, never an argument. +/// The net zone's Wi-Fi networks, for root and the tests; a passphrase comes on stdin, never argv. fn cmd_wifi(zones_dir: &Path, args: &[String]) -> ExitCode { let dir = wifi_dir_from(args); let sub = args.get(1).map(String::as_str).unwrap_or(""); @@ -1183,10 +1161,8 @@ fn trace_filter(dir: &Path, name: &str) -> Result<(Vec, seccomp::S Ok((seccomp::widened(&p.extra_syscalls).map_err(|e| e.to_string())?, p.sockets)) } -/* Run CMD under a zone filter and name every call it refuses. Refused calls - * come here by seccomp user notification (Kryptik forbids ptrace) and fail - * with ENOSYS, so one run lists them all. The child shares our descriptor - * table until exec: the zone filter has no sendmsg to pass its listener. */ +/// Run CMD under a zone filter, naming every call it refuses. Refusals arrive by user +/// notification (Kryptik forbids ptrace) and fail with ENOSYS, so one run lists them all. fn cmd_seccomp_trace(cmd: &[String], allow: &[libc::c_long], sockets: &seccomp::SocketPolicy) -> ExitCode { use std::ffi::CString; @@ -1205,7 +1181,7 @@ fn cmd_seccomp_trace(cmd: &[String], allow: &[libc::c_long], sockets: &seccomp:: return ExitCode::FAILURE; } - // A fork that shares the descriptor table; the child's exec unshares it. + // Shares the descriptor table until exec: the zone filter has no sendmsg to pass the listener. let pid = unsafe { libc::syscall(libc::SYS_clone, libc::CLONE_FILES | libc::SIGCHLD, 0, 0, 0, 0) } as libc::pid_t; if pid < 0 { eprintln!("seccomp-trace: clone: {}", std::io::Error::last_os_error()); @@ -1291,9 +1267,8 @@ fn cmd_seccomp_test(name: &str, nr: libc::c_long) -> ExitCode { }) } -/// Run `probe` in a child under the zone filter and say how it ended. Exit 5: -/// killed by SIGSYS; 6: another signal; 7: refused with the intended errno; -/// 1: no filter; 0: completed. +/// Run `probe` in a child under the zone filter. Exit 5: killed by SIGSYS; 6: another signal; +/// 7: refused with the intended errno; 1: no filter; 0: completed. fn under_zone_filter(name: &str, probe: impl FnOnce() -> i32) -> ExitCode { // SAFETY: fork in a program that does no threading before this point. let pid = unsafe { libc::fork() }; diff --git a/compartments/kryptikd/src/serve.rs b/compartments/kryptikd/src/serve.rs index 6495b834..bab41eb0 100644 --- a/compartments/kryptikd/src/serve.rs +++ b/compartments/kryptikd/src/serve.rs @@ -1,8 +1,5 @@ -//! `kryptikd serve`: the launch daemon the desktop session talks to. -//! -//! Zones are created by root and the session is an ordinary user; this socket -//! is the one door between them. The daemon checks the peer (SO_PEERCRED) and -//! the request, then runs `kryptikd run`, whose every refusal still applies. +//! `kryptikd serve`: the root daemon the unprivileged desktop session asks to launch zones. +//! It checks the peer (SO_PEERCRED) and the request; `kryptikd run` still applies its refusals. //! //! socket /run/kryptik-launch/launch.sock (root:kryptik 0660) //! request one per connection, NUL-free text lines: @@ -59,8 +56,7 @@ fn clear_cloexec(fd: RawFd) { } } -/// One recvmsg into `buf`, owning the descriptors that came with it. Control -/// data cut short is refused, and what did arrive is closed. +/// One recvmsg into `buf`, owning any descriptors; truncated control data is refused. pub fn recv_with_fds(fd: RawFd, buf: &mut [u8], flags: libc::c_int) -> std::io::Result<(usize, Vec)> { // Aligned for cmsghdr, with room for twelve, so an excess is refused by count. #[repr(C, align(8))] @@ -117,8 +113,7 @@ fn gid_of_uid(uid: u32) -> Option { } } -/// Is `uid` root, or a member (primary or supplementary) of `group`? Member -/// names are copied out first in case getpwuid(3) reuses getgrnam(3)'s buffer. +/// Is `uid` root, or a member (primary or supplementary) of `group`? fn in_group(uid: u32, group: &str) -> bool { if uid == 0 { return true; @@ -129,6 +124,7 @@ fn in_group(uid: u32, group: &str) -> bool { return false; } let gid = unsafe { (*g).gr_gid }; + // Copied out first: getpwuid(3) may reuse getgrnam(3)'s buffer. let mut members: Vec = Vec::new(); let mut mem = unsafe { (*g).gr_mem }; unsafe { @@ -161,8 +157,7 @@ pub fn peer_cred(fd: RawFd) -> std::io::Result { // --- the request ------------------------------------------------------------- -/// Whether the bytes so far are a whole request. `run`, `wifi-add` and -/// `wifi-forget` end at an `end` line, other verbs at a newline. +/// `run`, `wifi-add` and `wifi-forget` end at an `end` line, other verbs at a newline. fn request_complete(text: &[u8]) -> bool { if !text.ends_with(b"\n") { return false; @@ -174,8 +169,7 @@ fn request_complete(text: &[u8]) -> bool { } } -/// Read the request text and any descriptor that came with it, within -/// `deadline`. At most one descriptor, and only with the first bytes. +/// Read the request, and at most one descriptor (with its first bytes), within `deadline`. fn recv_request(fd: RawFd, deadline: Instant) -> Result<(Vec, Option), String> { let mut text = Vec::new(); let mut carried: Option = None; @@ -294,8 +288,7 @@ impl std::fmt::Debug for WifiRequest { } } -/// Parse `wifi-add` or `wifi-forget`. A value is all after the first space; a -/// refusal never repeats a line, which may hold the passphrase. +/// Parse `wifi-add` or `wifi-forget`. A refusal never repeats a line: it may hold the passphrase. fn parse_wifi(text: &str) -> Result { let mut lines = text.lines(); let verb = lines.next().unwrap_or(""); @@ -354,9 +347,7 @@ fn openat_component(dir: RawFd, name: &str, flags: libc::c_int) -> Result Result { if !path.is_absolute() { return Err(format!("{}: not an absolute path", path.display())); @@ -430,9 +420,8 @@ pub fn open_nofollow(path: &Path, want_socket: bool) -> Result Ok(dir) } -/// Accept only `/run/user//kryptik//wayland-0`, walked without -/// following symlinks. `/`, `kryptik/`, `/` and the socket must be -/// the session's, the last two directories private, and the listener its proxy. +/// Accept only `/run/user//kryptik//wayland-0`, walked without following symlinks: +/// the session's from `/` down, `kryptik/` and `/` private, its proxy listening. fn verify_proxy_socket(p: &Path, uid: u32, zone: &str, proxy_exe: Option<&Path>) -> Result { let want = PathBuf::from(format!("/run/user/{uid}/kryptik/{zone}/wayland-0")); if p != want { @@ -467,14 +456,10 @@ fn verify_proxy_socket(p: &Path, uid: u32, zone: &str, proxy_exe: Option<&Path>) Ok(ProxySocket { _fd: sock, path: want, inode: InodeId::of(&st) }) } -/// Ask the kernel who listens on the socket's inode. SO_PEERCRED names the -/// caller of listen(), which must be the session's uid running kryptik-wlproxy -/// for this zone (the probe shows in its log as a client disconnect). A process -/// handed the socket later goes unseen, but only that uid could hand it over, -/// and it reaches the compositor directly anyway. +/// SO_PEERCRED names whoever called listen(), which must be the session's kryptik-wlproxy for +/// this zone. Only the session could pass the socket on, and it reaches the compositor anyway. fn verify_proxy_listener(sock: &OwnedFd, uid: u32, zone: &str, proxy_exe: Option<&Path>) -> Result<(), String> { - /* Non-blocking: the listener is the session's and may never accept; a - * full backlog must refuse at once, not block the root daemon. */ + // Non-blocking: a full backlog on the session's listener must not block the root daemon. let s = unsafe { libc::socket(libc::AF_UNIX, libc::SOCK_STREAM | libc::SOCK_CLOEXEC | libc::SOCK_NONBLOCK, 0) }; if s < 0 { return Err(format!("socket: {}", std::io::Error::last_os_error())); @@ -524,8 +509,7 @@ struct Launch { log: PathBuf, } -/// Start `kryptikd run` for the request; returns the readiness pipe's read end. -/// The passphrase descriptor goes to the child; ours all close on every path. +/// Start `kryptikd run` for the request; our copies of its descriptors close on every path. fn spawn_launcher( req: &Request, cfg: &ServeConfig, @@ -546,9 +530,8 @@ fn spawn_launcher( "--ready-fd".into(), ready_w.as_raw_fd().to_string(), ]; - /* The path, not a descriptor: one opened in this mount namespace cannot be - * bind-mounted from the zone's (EINVAL). The launcher reopens the path and - * refuses any other inode (spawn.rs, StagedSocket). */ + /* The path, not a descriptor: one opened in this mount namespace cannot be bind-mounted + * from the zone's (EINVAL). The launcher refuses any other inode (spawn.rs, StagedSocket). */ if let Some(w) = &wayland { args.push("--wayland-socket".into()); args.push(w.path.display().to_string()); @@ -569,9 +552,7 @@ fn spawn_launcher( .chain(args.iter().map(|a| CString::new(a.as_str()).unwrap_or_else(|_| CString::new("?").unwrap()))) .collect(); let path = CString::new("PATH=/usr/bin:/usr/sbin").unwrap(); - /* A developer instance's registry is under XDG_RUNTIME_DIR (registry::base), - * and its launcher must use the same one or `status` and `stop` would not - * find the zone. Nothing else of the environment crosses. */ + // Of our environment only XDG_RUNTIME_DIR crosses: a developer launcher's registry is under it. let runtime_dir = if unsafe { libc::geteuid() } != 0 { std::env::var("XDG_RUNTIME_DIR").ok().and_then(|v| CString::new(format!("XDG_RUNTIME_DIR={v}")).ok()) } else { @@ -691,9 +672,8 @@ impl Pending { } } -/// A slow request (`stop`, `update-apply`), answered when `reap` sees its -/// command end so the daemon keeps serving meanwhile. Not a thread: `reap` -/// collects every child and would take the status a thread waited for. +/// A slow request (`stop`, `update-apply`), answered once `reap` sees its command end. Not a +/// thread: `reap` collects every child and would take the status a thread waited for. struct Job { conn: UnixStream, pid: libc::pid_t, @@ -731,7 +711,6 @@ fn start_job(conn: UnixStream, what: JobKind, cmd: &mut std::process::Command) - _ => return Err((conn, "could not hold the command's output".into())), }; match cmd.stdin(std::process::Stdio::null()).stdout(o2).stderr(e2).spawn() { - // The Child is dropped without a wait: `reap` collects it. Ok(child) => Ok(Job { conn, pid: child.id() as libc::pid_t, what, out, err, exited: None }), Err(e) => Err((conn, e.to_string())), } @@ -789,8 +768,7 @@ fn reply(mut c: &UnixStream, text: &str) { let _ = c.flush(); } -/// The write end of the pipe SIGCHLD writes to, so a launcher or job that -/// ends is reaped at once rather than at the next request. +/// Write end of the SIGCHLD wake pipe, so a launcher or job that ends is reaped at once. static CHILD_WAKE: std::sync::atomic::AtomicI32 = std::sync::atomic::AtomicI32::new(-1); extern "C" fn on_child(_sig: libc::c_int) { @@ -823,8 +801,7 @@ fn reap(pending: &mut [Pending], jobs: &mut [Job]) { } } -/// Wait up to 500 ms for a launcher's status once its pipe has closed: the -/// close can reach us before the exit does. +/// Wait up to 500 ms for a launcher's status after its pipe closed: the close can come first. fn wait_exit(p: &mut Pending) -> Option { if p.exited.is_some() { return p.exited; @@ -925,8 +902,8 @@ fn bind(cfg: &ServeConfig) -> Result { } let gid = gid_of_group(&cfg.group).ok_or_else(|| format!("no group {:?}; nobody could connect", cfg.group))?; let dir = cfg.socket.parent().map(Path::to_path_buf).unwrap_or_else(|| PathBuf::from(SOCKET_DIR)); - /* Only the default directory is made root:kryptik 0750, so the group alone - * reaches the socket; a directory named with --socket is left as found. */ + // Only the default directory becomes root:kryptik 0750, so the group alone reaches the socket; + // a --socket one is left as found. if dir == Path::new(SOCKET_DIR) { let _ = std::fs::create_dir_all(&dir); let _ = std::fs::set_permissions(&dir, std::fs::Permissions::from_mode(0o750)); @@ -942,8 +919,7 @@ fn bind(cfg: &ServeConfig) -> Result { Ok(l) } -/// Create the session's /run/user/ (0700, the user's). With no logind, -/// the daemon is the one root process the session can ask. +/// Create the session's /run/user/ (0700, the user's): with no logind, nothing else can. fn runtime_dir(cfg: &ServeConfig, uid: u32) -> Result { let dir = if cfg.developer { cfg.log_dir.join(format!("run-user-{uid}")) @@ -986,8 +962,7 @@ fn zone_running(name: &str) -> bool { matches!(crate::registry::state(name), Ok(crate::registry::State::Running { .. })) } -/// Serve one connection. A started launch is returned to be watched and a -/// slow command pushed to `jobs`; anything else is answered here. +/// Serve one connection; a started launch is returned and a slow command pushed to `jobs`. fn handle(cfg: &ServeConfig, conn: UnixStream, jobs: &mut Vec) -> Option { let fd = conn.as_raw_fd(); let Ok(peer) = peer_cred(fd) else { diff --git a/compartments/kryptikd/src/update.rs b/compartments/kryptikd/src/update.rs index 82ca7c8e..f25f2059 100644 --- a/compartments/kryptikd/src/update.rs +++ b/compartments/kryptikd/src/update.rs @@ -1,10 +1,5 @@ -//! The update channel's rules (docs/design/update-channel.md): which pointers -//! zone 0 accepts, and which bytes it takes from the net zone for a release -//! the user asked for. -//! -//! Signatures are checked by `kryptik-update` (`Checks`), and only verified -//! text reaches the parsers. Decided here: role, replay, staleness, and that -//! each staged byte is one the signed manifest provides for, at its place. +//! Zone 0's side of the update channel (docs/design/update-channel.md): which pointers it accepts +//! and which bytes it stages; signatures are checked first, by `kryptik-update` (`Checks`). use std::cmp::Ordering; use std::io::Write as _; @@ -12,14 +7,13 @@ use std::os::unix::fs::OpenOptionsExt; use std::path::{Path, PathBuf}; pub const POINTER_MAGIC: &str = "KRYPTIK-LATEST-1"; -/// The pointer and its signature, each. +/// Byte limit for the pointer and for its signature. pub const POINTER_MAX: usize = 8 * 1024; -/// The manifest and its signature, each. +/// Byte limit for the manifest and for its signature. pub const MANIFEST_MAX: u64 = 64 * 1024; /// Past this age a pointer is reported stale: it is re-issued on a schedule. pub const STALE_AFTER_SECS: i64 = 30 * 86400; -/// How far ahead of this machine's clock a statement may be dated, allowing -/// for skew. One dated later would make every genuine statement a replay. +/// Skew allowed in a statement's date; one dated later would make every genuine one a replay. pub const MAX_AHEAD_SECS: i64 = 86400; /// One pointer is considered per hour; the rest are refused unread. pub const POINTER_INTERVAL_SECS: u64 = 3600; @@ -39,8 +33,7 @@ fn is_version(s: &str) -> bool { !s.is_empty() && s.len() <= 32 && s.bytes().all(|b| b.is_ascii_alphanumeric() || matches!(b, b'.' | b'+' | b'-' | b'~')) } -/// The pointer's text: the magic line, then each key exactly once. An unknown -/// key is refused, so an older system never half-understands a newer format. +/// The magic line, then each key once; an unknown key is refused, never half-understood. pub fn parse_pointer(text: &str) -> Result { let mut lines = text.lines(); if lines.next() != Some(POINTER_MAGIC) { @@ -77,9 +70,8 @@ pub fn parse_pointer(text: &str) -> Result { Ok(Pointer { role, version, issued, manifest_sha256: sha, base }) } -/// Where a release's files are fetched from: an absolute base as is, a relative -/// one under the `CONF` channel address, never under what the net zone reports. -/// Plain http only for a development image, a privacy matter: the hash is signed. +/// A release's URL: an absolute base as is, a relative one under the `CONF` channel, never under +/// what the net zone reports. pub fn resolve_base(channel: &str, base: &str, role: &str) -> Result { let mut url = if base.contains("://") { base.to_string() @@ -94,6 +86,7 @@ pub fn resolve_base(channel: &str, base: &str, role: &str) -> Result Ok(url), + // Plain http costs privacy, not integrity: the hash is signed. Some("http") if role == "development" => Ok(url), Some("http") => Err("a production image does not fetch over plain http".into()), _ => Err(format!("{url:?} is neither https nor http")), @@ -126,17 +119,15 @@ pub fn version_cmp(a: &str, b: &str) -> Ordering { (a.len() - i).cmp(&(b.len() - j)) } -/// What an accepted pointer says about this machine. +/// Where this machine stands against an accepted pointer. #[derive(Debug, PartialEq)] pub enum Standing { - /// It names the running release, or an older one. + /// The pointer's release is the running one, or older. Current, Available(String), } -/// Whether zone 0 accepts a verified pointer: it must be for this image's role, -/// not issued before the newest already accepted (a replay), and not dated more -/// than `MAX_AHEAD_SECS` after `now`. +/// Accept a verified pointer for this image's role unless it is a replay or dated too far ahead. pub fn accept_pointer(p: &Pointer, required_role: &str, running: &str, newest_issued: Option, now: i64) -> Result { if p.role != required_role { return Err(format!("the pointer's role is '{}'; this image requires '{required_role}'", p.role)); @@ -152,8 +143,7 @@ pub fn accept_pointer(p: &Pointer, required_role: &str, running: &str, newest_is Ok(if version_cmp(&p.version, running) == Ordering::Greater { Standing::Available(p.version.clone()) } else { Standing::Current }) } -/// How many whole days old the newest accepted pointer is, and whether that -/// is past the bound. A clock behind the pointer reads as zero days. +/// Whole days since `issued`, and whether that is stale; a clock behind it reads as zero. pub fn staleness(now: i64, issued: i64) -> (i64, bool) { let age = (now - issued).max(0); (age / 86400, age > STALE_AFTER_SECS) @@ -166,8 +156,7 @@ pub struct Entry { pub size: u64, } -/// Parses `check-manifest`'s output: `file ` lines, others -/// ignored. Names must pass the broker's transfer-name check. +/// Parse `check-manifest`'s `file ` lines; names must pass the broker's check. pub fn parse_file_list(text: &str) -> Result, String> { let mut out: Vec = Vec::new(); for line in text.lines() { @@ -190,10 +179,9 @@ pub fn total_bytes(files: &[Entry]) -> u64 { files.iter().fold(0, |sum, e| sum.saturating_add(e.size)) } -/// Whether `len` bytes for `name` at `offset` may be written, `held` being -/// there already. Until the manifest verifies (`files` is `None`) only it and -/// its signature are taken, whole and small; then only listed names, at exactly -/// `held` (resumable, never out of order), never past the signed size. +/// Whether `len` bytes of `name` may be written at `offset`, with `held` already there. +/// Until the manifest verifies, only it and its signature, whole and small; then listed files, +/// appended in order, never past the signed size. pub fn may_put(files: Option<&[Entry]>, name: &str, offset: u64, len: u64, held: u64) -> Result<(), String> { if len == 0 { return Err("nothing to put".into()); @@ -219,7 +207,7 @@ pub fn may_put(files: Option<&[Entry]>, name: &str, offset: u64, len: u64, held: Ok(()) } -/// What is still missing and from which byte: the answer to `update-poll`. +/// Each missing file and the byte it resumes at, for `update-poll`. pub fn still_needed(files: &[Entry], held: impl Fn(&str) -> u64) -> Vec<(String, u64)> { files.iter().filter_map(|e| { let h = held(&e.name); (h < e.size).then(|| (e.name.clone(), h)) }).collect() } @@ -236,20 +224,17 @@ pub fn still_needed(files: &[Entry], held: impl Fn(&str) -> u64) -> Vec<(String, * * Functions take the directory and the checks, so tests supply their own. */ - - pub const STATE_DIR: &str = "/var/lib/kryptik/update"; pub const TOOL: &str = "/usr/sbin/kryptik-update"; pub const ROLE_FILE: &str = "/usr/share/kryptik/trust/required-role"; pub const CONF: &str = "/etc/kryptik/update.conf"; -/// The most one `update-put` carries. The launcher answers each between two -/// looks at its zone, so supervision is never more than one piece away. +/// Largest `update-put`, so the launcher never goes long without checking its zone. pub const PUT_MAX: usize = 1 << 20; -/// The signature checks, as functions so a test can stand in for -/// `kryptik-update`. `manifest` returns what `check-manifest` printed. +/// The signature checks, as functions so a test can stand in for `kryptik-update`. pub struct Checks<'a> { pub pointer: &'a dyn Fn(&Path, &Path) -> Result<(), String>, + /// Returns what `check-manifest` printed. pub manifest: &'a dyn Fn(&Path) -> Result, } @@ -268,8 +253,7 @@ fn run_tool(args: &[&std::ffi::OsStr]) -> Result { Err(err.lines().last().unwrap_or("refused").trim_start_matches("kryptik-update: ").to_string()) } -/// The checks the installed system uses: `kryptik-update`, with the trust -/// anchor on the verified root and nothing from this process's environment. +/// The installed system's checks: `kryptik-update`, run with a cleared environment. pub fn tool_checks() -> Checks<'static> { Checks { pointer: &|p, s| run_tool(&["check-pointer".as_ref(), p.as_os_str(), s.as_os_str()]).map(|_| ()), @@ -300,9 +284,8 @@ pub fn running_version() -> String { text.lines().find_map(|l| l.strip_prefix("VERSION_ID=")).map(|v| v.trim_matches('"').to_string()).unwrap_or_default() } -/// `channel =
` from `CONF`. A copy written under /etc does not -/// survive the next boot (sysinit.sh, `prune_etc_upper`), and the address is -/// only where to ask: answers are believed on the trust anchor's signature. +/// `channel =
` from `CONF`, the image's own: sysinit prunes a copy written under /etc. +/// The address is only where to ask; answers stand on their signatures. pub fn channel_from(conf: &str) -> Option { conf.lines().find_map(|l| { let (k, v) = l.split_once('=')?; @@ -341,9 +324,7 @@ fn verified_files(dir: &Path, version: &str) -> Option> { (text.lines().next() == Some(&format!("version: {version}"))).then(|| parse_file_list(&text).ok()).flatten() } -/// `update-latest`: a pointer and its signature from the net zone. One is -/// looked at per interval, whatever its fate, so a hostile zone cannot keep -/// zone 0 verifying signatures. +/// `update-latest`: one pointer per interval, so a hostile zone cannot keep zone 0 verifying. pub fn latest(dir: &Path, checks: &Checks, now: i64, role: &str, running: &str, pointer: &[u8], sig: &[u8]) -> Result { private_dir(dir)?; let last: Option = std::fs::read_to_string(dir.join("considered")).ok().and_then(|s| s.trim().parse().ok()); @@ -377,6 +358,7 @@ pub fn want(dir: &Path, channel: Option<&str>, role: &str, running: &str) -> Res return Err(format!("{} is the newest release known, and this machine runs {running}", p.version)); } let channel = channel.ok_or_else(|| format!("this image names no update channel ({CONF})"))?; + // Refused here if this image would not fetch it: the poll would only say `idle`. resolve_base(channel, &p.base, role).map_err(|e| format!("{} cannot be fetched: {e}", p.version))?; if wanted(dir).as_deref() != Some(p.version.as_str()) { let _ = std::fs::remove_file(dir.join("files")); @@ -460,9 +442,7 @@ fn free_bytes(path: &Path) -> Option { (unsafe { libc::statvfs(c.as_ptr(), &mut st) } == 0).then(|| st.f_bavail as u64 * st.f_frsize as u64) } -/// `update-put`: bytes for the wanted release, under `may_put`'s rule. Once -/// both manifest and signature are in, they must verify, match the pointer's -/// hash and fit the free space before any file they list is accepted. +/// `update-put`: bytes for the wanted release, under `may_put`'s rule. pub fn put(dir: &Path, checks: &Checks, now: i64, name: &str, offset: u64, bytes: &[u8]) -> Result { let version = wanted(dir).ok_or("no release has been asked for")?; let p = stored_pointer(dir).filter(|p| p.version == version).ok_or("the release asked for is not the one the newest statement names")?; @@ -487,6 +467,7 @@ pub fn put(dir: &Path, checks: &Checks, now: i64, name: &str, offset: u64, bytes if held(&stage, "manifest") == 0 || held(&stage, "manifest.sig") == 0 { return Ok(format!("{name} complete")); } + // The pair is in: it must verify, match the pointer's hash and fit the free space. let refuse = |why: String| -> Result { let _ = std::fs::remove_dir_all(&stage); Err(why) @@ -579,7 +560,7 @@ pub fn complete_stage(dir: &Path) -> Result { } } -/// Once the machine runs what was staged, the staging area has no job. +/// Clear the staged release once the machine runs it or a newer one. pub fn forget_if_installed(dir: &Path, running: &str) { if wanted(dir).is_some_and(|v| version_cmp(&v, running) != Ordering::Greater) { for f in ["wanted", "files"] { diff --git a/compartments/kryptikd/src/update/tests.rs b/compartments/kryptikd/src/update/tests.rs index 91c76ce4..c7a59354 100644 --- a/compartments/kryptikd/src/update/tests.rs +++ b/compartments/kryptikd/src/update/tests.rs @@ -133,7 +133,7 @@ fn poll_names_missing_files_and_offsets() { assert!(still_needed(&files(), |_| u64::MAX).is_empty()); } -// --- the state, against a directory of the test's own --- +// --- the state, in a scratch directory per test --- fn scratch(tag: &str) -> PathBuf { let d = std::env::temp_dir().join(format!("kryptik-update-test-{}-{tag}", std::process::id())); diff --git a/compartments/kryptikd/src/zone.rs b/compartments/kryptikd/src/zone.rs index 0acd60e9..7c489938 100644 --- a/compartments/kryptikd/src/zone.rs +++ b/compartments/kryptikd/src/zone.rs @@ -1,7 +1,5 @@ -//! Zone definitions: parsing and validation. -//! -//! A hand-written parser rather than a TOML crate, to keep dependencies to -//! `libc` (ADR-010). A malformed file is an error, never a weaker zone. +//! Zone definitions, parsed by hand to keep dependencies to `libc` (ADR-010). +//! A malformed file is an error, never a weaker zone. use std::collections::HashMap; use std::fmt; @@ -42,8 +40,7 @@ pub enum StorageMode { Encrypted, /// tmpfs overlay, destroyed at teardown. Ephemeral, - /// A plain directory on the host filesystem, kept between launches. Not - /// encrypted at rest, and every report of this mode says so. + /// A plain host directory kept between launches; not encrypted at rest, and reports say so. Persistent, } @@ -67,8 +64,7 @@ pub struct Zone { pub name: String, pub description: String, pub network: NetworkMode, - /// Interface a `nic` zone takes (`nic = "eth0"`), or `"*"` for every - /// physical one (`netzone::physical_interfaces`). Refused for other modes. + /// The `nic` zone's interface (`nic = "eth0"`), or `"*"` for every physical one. pub nic: Option, /// `[network] local = true`: the nic zone lets this routed zone reach the /// networks its uplinks sit on; every other routed zone is refused them @@ -78,31 +74,25 @@ pub struct Zone { pub volume: Option, pub seccomp: Option, pub landlock: Option, - /// Upper bound on an ephemeral zone's tmpfs: required for ephemeral, - /// refused otherwise. Unbounded, a zone could fill host memory with files. + /// Required bound on an ephemeral zone's tmpfs, which could otherwise fill host memory. pub size: Option, pub memory_max: Option, pub pids_max: Option, - /// A share of CPU time as a percentage of one CPU (`"150%"`), enforced by - /// cgroup cpu.max; and bytes per second each way on the zone's volume, by - /// io.max, so only an encrypted zone may set it. + /// A percentage of one CPU (`"150%"`), enforced by cgroup cpu.max. pub cpu_max: Option, + /// Bytes per second each way on the zone's volume (io.max), so encrypted zones only. pub io_max: Option, pub border_color: String, - /// Identity without colour: `glyph` and `label` name the zone in the - /// chrome's menu, whose f names the last zone window's. Nothing draws - /// `border_pattern`; it is only checked here, and `zoneid audit` gives it - /// no weight. + /// Never drawn: only checked here, and `zoneid audit` gives it no weight. pub border_pattern: Option, + /// Glyph and label identify the zone without colour, in the chrome menu. pub glyph: Option, pub label: Option, - /// Host identity range, `[identity] uid_base = N`: a privileged launch maps - /// root to N and nobody to N + 65534. Declared, not derived from zone order, - /// so adding a zone never changes who owns another's files. `None`: a root - /// launch needs `--zone-uid/--zone-gid`, and `check --target` refuses it. + /// Host uid the zone's root maps to (nobody to it + 65534); declared, not derived from zone + /// order, so adding a zone never changes who owns another's files. None: a root launch needs + /// `--zone-uid/--zone-gid`, and `check --target` refuses it. pub uid_base: Option, - /// `[transfer] to = "work personal"`: zones this one may send files to via - /// the broker. Never the nic zone (`check_invariants`). + /// `[transfer] to = "work personal"`: zones this one may send files to; never the nic zone. pub transfer_to: Vec, /// `[transfer] max_bytes = N`: the largest file this zone sends or /// receives through the broker, at most its cap. None: the cap alone. @@ -148,8 +138,7 @@ pub const KNOWN_KEYS: &[&str] = &[ "ui.border_color", "ui.border_pattern", "ui.glyph", "ui.label", ]; -/// A CPU limit as a zone file and cgroup cpu.max take it: a percentage of one -/// CPU, digits then `%`, at least 1. None for anything else. +/// A percentage of one CPU, digits then `%`, at least 1; None for anything else. pub fn parse_cpu_max(s: &str) -> Option { let digits = s.strip_suffix('%')?; if digits.is_empty() || !digits.bytes().all(|b| b.is_ascii_digit()) { @@ -158,9 +147,7 @@ pub fn parse_cpu_max(s: &str) -> Option { digits.parse::().ok().filter(|n| *n > 0) } -/// A byte size as a zone file, cgroup memory.max and `volume init --size` -/// take it: digits with an optional K, M, G or T. None for zero, for -/// anything else (so not "max": omit the key), and on overflow. +/// Bytes as digits with an optional K, M, G or T; None for zero, "max", other text or overflow. pub fn parse_size(s: &str) -> Option { let (digits, shift) = match s.as_bytes().last()? { b'K' | b'k' => (&s[..s.len() - 1], 10), @@ -175,8 +162,8 @@ pub fn parse_size(s: &str) -> Option { digits.parse::().ok().filter(|&n| n > 0)?.checked_mul(1 << shift) } -/// Minimal TOML reader: `[section]`, `key = value` (quoted string, integer or -/// boolean) and `#` comments. Arrays and nested tables are errors, not ignored. +/// Flat TOML: `[section]`, `key = value` (quoted string, integer or boolean) and `#` comments; +/// arrays and nested tables are errors, not ignored. fn parse_flat_toml(text: &str) -> Result, ZoneError> { let mut out = HashMap::new(); let mut section = String::new(); @@ -321,7 +308,6 @@ impl Zone { if parse_size(v).is_none() { return Err(bad("limits.io_max", v, "bytes per second such as 20M")); } - // The limit is on the volume's device; a zone without one has nothing to bound. if storage != StorageMode::Encrypted { return Err(ZoneError::Invalid(format!( "zone {name:?}: limits.io_max bounds reads and writes of the zone's volume, \ @@ -373,7 +359,6 @@ impl Zone { } } StorageMode::Persistent => { - // Nothing would enforce it: a host directory with no quota. if kv.contains_key("storage.size") { return Err(ZoneError::Invalid(format!( "zone {:?}: storage.size is only meaningful for storage.mode = \ @@ -538,7 +523,6 @@ impl Zone { self.name ))); } - // A persistent zone is a directory and opens no volume. if self.storage == StorageMode::Persistent && self.volume.is_some() { return Err(ZoneError::Invalid(format!( "zone {:?}: storage.volume is only meaningful for storage.mode = \ @@ -558,8 +542,7 @@ impl Zone { self.name, self.border_color ))); } - /* Shape only, as zoneid checks it. A printable-ASCII label rules out - * bidi controls and homographs. */ + // Shape only, as zoneid checks it. if let Some(p) = &self.border_pattern { const PATTERNS: [&str; 6] = ["solid", "dashed", "dotted", "double", "dash-dot", "notched"]; if !PATTERNS.contains(&p.as_str()) { @@ -577,6 +560,7 @@ impl Zone { ))); } } + // Printable ASCII rules out bidi controls and homographs. if let Some(l) = &self.label { if l.is_empty() || l.len() > 12 || !l.chars().all(|c| matches!(c, ' '..='~')) || l.starts_with(' ') || l.ends_with(' ') { return Err(ZoneError::Invalid(format!( @@ -609,8 +593,7 @@ pub fn load_all(dir: &Path) -> Result, ZoneError> { continue; } let zone = Zone::from_file(&path)?; - /* The broker and the net zone open a zone as .toml, so a file - * named otherwise would launch but never receive a transfer. */ + // The broker and the net zone open .toml; a file named otherwise gets no transfer. if path.file_stem().and_then(|s| s.to_str()) != Some(zone.name.as_str()) { return Err(ZoneError::Invalid(format!( "{}: holds zone {:?}; a zone's file is named {}.toml", @@ -629,8 +612,7 @@ pub fn load_all(dir: &Path) -> Result, ZoneError> { /// Invariants that hold across the whole zone set, not within one file. pub fn check_invariants(zones: &[Zone]) -> Result<(), ZoneError> { - /* [transfer] to must name configured zones, never the nic zone. First - * occurrence wins; duplicate names are refused below. */ + // Transfer targets must be configured zones, never the nic zone; duplicates are refused below. let mut by_name: HashMap<&str, &Zone> = HashMap::with_capacity(zones.len()); for z in zones { by_name.entry(z.name.as_str()).or_insert(z); @@ -688,8 +670,7 @@ pub fn check_invariants(zones: &[Zone]) -> Result<(), ZoneError> { } } - /* Zones sharing a range could not be told apart by uid (broker, compositor - * proxy). Bases are stride-aligned, so distinct bases never overlap. */ + // The broker and compositor proxy tell zones apart by uid; aligned bases never overlap. let mut bases: HashMap = HashMap::new(); for z in zones { if let Some(b) = z.uid_base { diff --git a/compartments/tests/cli.sh b/compartments/tests/cli.sh index 195211d3..6a869886 100755 --- a/compartments/tests/cli.sh +++ b/compartments/tests/cli.sh @@ -1,12 +1,11 @@ #!/usr/bin/env bash -# Tests for `kryptik`, the user-facing command: mostly that convenience has not -# cost safety. No zone starts with its guarantees unmet, the config file is -# never executed, and nothing is offered that does not work. +# Tests for `kryptik`, the user-facing command: its conveniences must not cost safety. No zone +# starts with its guarantees unmet, the config file never runs, and nothing is offered that fails. +# Exit 1 if a check failed, 2 if kryptik or kryptikd is missing. set -uo pipefail REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" -# tools/kryptik in a checkout; on PATH in an image, where this suite lives -# under /usr/lib/kryptik. +# tools/kryptik in a checkout; on PATH in an image, which installs this suite in /usr/lib/kryptik. KRYPTIK="${KRYPTIK:-}" if [[ -z "$KRYPTIK" ]]; then if [[ -x "$REPO/tools/kryptik" ]]; then KRYPTIK="$REPO/tools/kryptik" @@ -43,9 +42,8 @@ trap cleanup EXIT ZONES="$WORK/zones"; ROOTFS="$WORK/data"; mkdir -p "$ZONES" "$ROOTFS" -# As root a zone maps to host uid 100000 and must traverse into its data -# directory, which `mktemp -d` makes 0700 and root-owned. Otherwise every -# launch fails with a misleading "mount(root tmpfs)...: Permission denied". +# As root, zones map to uid 100000, which must traverse mktemp's 0700 directory to its data; +# otherwise every launch fails with a misleading "Permission denied" from mount. if [[ "$(id -u)" -eq 0 ]]; then chmod 0755 "$WORK" "$ZONES" "$ROOTFS" chown -R 100000:100000 "$ROOTFS" @@ -70,8 +68,7 @@ mkzone() { # name storage-mode colour [network-mode] printf '[ui]\nborder_color = "%s"\n' "$3" } > "$ZONES/$1.toml" } -# kryptikd refuses a zone set in which no zone holds the physical NIC, hence -# `carrier`. +# kryptikd refuses a zone set in which no zone holds the NIC, hence `carrier`. mkzone plain ephemeral "#101010" mkzone sealed encrypted "#202020" mkzone carrier ephemeral "#303030" nic @@ -124,9 +121,6 @@ else fi # --- an encrypted zone needs its passphrase; there is no way to skip that ---- -# It starts only with its passphrase (a descriptor from the launch daemon, or a -# root-owned file). Without one the command never runs, and no flag may fall -# back to a plain directory. out="$(K run sealed -- /bin/sh -c 'echo CLI_STARTED' 2>&1)"; rc=$? if [[ "$out" == *CLI_STARTED* ]]; then fail "a zone claiming ENCRYPTED storage ran without its passphrase" @@ -162,9 +156,8 @@ else info "output: $(printf '%s' "$out" | tr '\n' '|' | cut -c1-240)" fi -# It must run in the zone, not on the host: the zone's hostname is its name. -# Read from a sentinel line, since an error such as "no zone named plain" also -# contains the name. +# It must run in the zone, not on the host: the zone's hostname is its name. A sentinel line, +# since an error such as "no zone named plain" also contains the name. out="$(K run plain -- /bin/sh -c 'echo "D2HOST=$(hostname)"' 2>&1)" got="$(printf '%s' "$out" | sed -n 's/^D2HOST=//p' | head -1)" if [[ "$got" == "plain" ]]; then @@ -204,9 +197,9 @@ else fi # --- no transfer or clipboard commands --------------------------------------- -# Files cross zones only through the sending zone's broker after the user's yes, -# and the clipboard only by the chrome's gesture. From zone 0 either command -# would go around the user, so both are refused, naming the real path. +# Files cross zones only through the sending zone's broker after the user's yes, and the +# clipboard only by the chrome's gesture: from zone 0 either would go around the user, so both +# are refused, naming the real path. for c in transfer clipboard; do out="$(K "$c" 2>&1)"; rc=$? if (( rc != 0 )) && [[ "$out" == *"not a"*"command"* && "$out" == *"broker"* && "$out" == *"chrome"* ]]; then @@ -246,9 +239,8 @@ else fi # --- wifi: the net zone's credentials ---------------------------------------- -# A passphrase never goes on a command line, where any process could read it: -# `kryptik wifi add` reads it (terminal with echo off, or a pipe) and passes it -# on stdin to kryptik-launch or, as here, to kryptikd. +# A passphrase never goes on a command line, where any process could read it: `kryptik wifi add` +# takes it from the terminal or a pipe and passes it on stdin. out="$(K wifi 2>&1)"; rc=$? if (( rc != 0 )) && [[ "$out" == *"list, add SSID or forget SSID"* ]]; then pass "wifi without a subcommand fails and names the subcommands" @@ -336,9 +328,7 @@ sys.stdout.write(out.decode("utf-8", "replace")) PY cat > "$WORK/launch-shim" < Date: Wed, 7 Oct 2026 09:24:21 -0700 Subject: [PATCH 16/18] The decision records, the broker design and the supply-chain notes are shorter, with their claims kept: cleanup-held's cut of docs/decisions.md, docs/design/broker.md and docs/supply-chain.md, redone on main by hand Where main changed a passage after the cut, main's text stands: the records include proposed ones, nosmt's cookies stay because root can turn SMT back on, the root image is about 1.7 GB, fuzz.yml exists, and the signature gate reads key-provenance.tsv. supply-chain.md carries two corrections the cut made: Intel's microcode comes from Intel's own unsigned archive, not linux-firmware, and cmake-bin is held to Kitware's signed SHA-256 list like the source tarball. --- docs/decisions.md | 54 +++++++++++------------ docs/design/broker.md | 99 ++++++++++++++++++++++--------------------- docs/supply-chain.md | 65 ++++++++++++++-------------- 3 files changed, 111 insertions(+), 107 deletions(-) diff --git a/docs/decisions.md b/docs/decisions.md index c8549cca..188b9acd 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -34,9 +34,9 @@ containers. X11 lets any client log every other client's keystrokes and capture the whole screen, which ADR-003 cannot allow. -**Cost:** X11 programs are not supported. No Xwayland is built, and a zone -runs Wayland clients alone; if the graphical applications of Version 2 need -one, it runs inside the zone, where it can leak only that zone. +**Cost:** X11 programs are not supported. No Xwayland is built, and zones run +only Wayland clients. If Version 2's graphical applications need Xwayland, it +runs inside the zone, where it can leak only that zone. ## ADR-005: hardened_malloc as the system allocator @@ -54,9 +54,9 @@ Socket activation and journald do not justify a large privileged PID 1 in a system that assumes a local attacker looking for privileged code. **Cost:** off the LFS path, so every service definition is written from -scratch; seatd instead of logind; the `net` zone runs its own DHCP client, and -no other zone touches a real interface; logging is s6-log per service, with no -aggregation. +scratch. Seats come from seatd, not logind. The `net` zone runs its own DHCP +client, and no other zone touches a real interface. Logging is s6-log per +service, with no aggregation. **Revisit if** writing service definitions becomes the main cost of the base system. @@ -119,18 +119,19 @@ process would contradict that. **Cost:** -- rustc is not built from source: that needs an existing rustc, or mrustc, a - project of its own. The shipped kryptikd and kryptik-wlproxy are built by - Rust's release tarballs, held to the hashes in `build/config/rust.lock` - (checked against the Rust release key when pinned): a trust anchor - [supply-chain.md](supply-chain.md) otherwise avoids. +- rustc is not built from source, since that needs an existing rustc or + mrustc, a project of its own. The shipped kryptikd and kryptik-wlproxy are + built with Rust's release tarballs, held to the hashes in + `build/config/rust.lock` (checked against the Rust release key when + pinned). That is a trust anchor [supply-chain.md](supply-chain.md) + otherwise avoids. - kryptikd depends on `libc` only; every new crate is a supply-chain decision justified in review. - A Rust toolchain is a lot to carry for one daemon. -**Rejected:** C (smallest bootstrap, but see above); Go, whose runtime and -scheduler fight `clone()`, `unshare()` and per-thread namespace state; shell, -unsuitable for holding privilege and parsing untrusted zone state. +**Rejected:** C (smallest bootstrap, but not memory-safe); Go, whose runtime +and scheduler fight `clone()`, `unshare()` and per-thread namespace state; +shell, unsuitable for holding privilege and parsing untrusted zone state. Rust is for kryptikd and Kryptik's own tools, not a distribution-wide rule: coreutils stays coreutils. @@ -149,9 +150,9 @@ it. **Cost:** half the logical CPUs on an SMT machine, roughly 15 to 30 percent of parallel throughput. Single-threaded performance is unchanged. -**Decided:** `nosmt` stays. Each zone asks for its own core-scheduling -cookie at launch ([privileged launch](design/privileged-launch.md#core-scheduling)), -which keeps two zones off the two threads of one core; it cannot keep a zone +**Decided:** `nosmt` stays. Each zone asks for its own core-scheduling cookie +at launch ([privileged launch](design/privileged-launch.md#core-scheduling)), +which keeps two zones off the two threads of one core. It cannot keep a zone off the thread beside the kernel, since the kernel's own execution carries no cookie, and that is the leak the mitigations exist for. Closing it with SMT on means a flush on every kernel entry, which costs more than the threads give. @@ -175,10 +176,9 @@ Wi-Fi. These are vendor binaries, not built from source as [supply-chain.md](supply-chain.md) otherwise requires, and run by the device's own processor under the kernel's control of the bus (IOMMU on and strict). -Kryptik establishes that the tarball is the one kernel.org signed, its hash is -pinned, each file's licence is the one `WHENCE` records, and the copy the -kernel loads is on the verified root, so replacing it means re-signing the -kernel. +Kryptik checks that the tarball is the one kernel.org signed, pins its hash +and checks each file's licence against `WHENCE`. The kernel loads the +firmware from the verified root, so replacing it means re-signing the kernel. **Left out:** NVIDIA (nouveau needs tens of megabytes of GSP firmware per generation, and the firmware framebuffer gives those machines a display); @@ -222,7 +222,7 @@ on old microcode, and nobody would notice. The firmware loads Kryptik's kernel as the UEFI application, and nothing else runs before the verified root: no shim, no boot loader, no initramfs. The -command line is compiled in and names the root slot and its dm-verity root +command line is compiled in and carries the root slot and its dm-verity root hash, so the firmware's one signature check covers the code and the hash of everything it will run ([boot and updates](design/boot-and-updates.md)). A machine trusts that signature once Kryptik's certificate is in its firmware's @@ -230,13 +230,13 @@ database ([release keys](release-keys.md)). **Why:** every stage between the firmware and the root is a file to sign, a parser to attack and a place for an unmeasured change. A shim chains from -Microsoft's key, which Kryptik does not use; a boot loader chooses and edits -what boots, which the compiled-in command line forbids on purpose; an -initramfs finds the root, which `dm-mod.create=` does inside the kernel. +Microsoft's key, which Kryptik does not use. A boot loader chooses and edits +what boots, which the compiled-in command line forbids. An initramfs finds the +root, which `dm-mod.create=` does inside the kernel. **Cost:** the certificate is enrolled by hand on every machine, and firmware -that carries only Microsoft's keys refuses the media; the controller the root -sits on is built into the kernel (ADR-013); A/B updates and recovery are the +that carries only Microsoft's keys refuses the media. The controller the root +sits on is built into the kernel (ADR-013). A/B updates and recovery are the firmware's boot entries and a judged trial, not a loader's menu. **Rejected:** shim and a loader, two more signed stages and a configuration diff --git a/docs/design/broker.md b/docs/design/broker.md index 428cb7e0..d1a79240 100755 --- a/docs/design/broker.md +++ b/docs/design/broker.md @@ -13,7 +13,7 @@ started from the desktop gets its own Wayland proxy. Builds on Every zone has its own host uid range (`[identity] uid_base`). A connection accepted in zone 0 carries `SO_PEERCRED`, whose uid the kernel asserts and a zone cannot choose, so no token, handshake or crypto is needed. Each zone's -broker accepts exactly the uid its launcher mapped the zone to; anything else +broker accepts only the uid its launcher mapped the zone to; any other peer gets `error: unidentified peer` and nothing more. The peer pid is never used (it is a zone 0 pid and may be reused). On an unprivileged developer launch every zone maps to the same user, so identity distinguishes nothing there. @@ -22,11 +22,11 @@ every zone maps to the same user, so identity distinguishes nothing there. The launcher binds one socket per zone in its registry entry (`/run/kryptik/zones//broker`, 0600, owned by the zone's identity) and -bind-mounts that one file into the zone at `/run/kryptik/broker`; the +bind-mounts that file into the zone at `/run/kryptik/broker`. The intermediate opens it (`O_PATH`) in its own mount namespace, before the id -switch. A zone reaches only its own endpoint, and the broker socket plus, for -a desktop launch, the proxy socket are all it has under `/run/kryptik`. Zones -inherit only fds 0 to 2 and reach the broker with `connect(2)`. +switch. A zone reaches only its own endpoint: under `/run/kryptik` it has the +broker socket and, for a desktop launch, the proxy socket, and nothing else. +Zones inherit only fds 0 to 2 and reach the broker with `connect(2)`. `kryptikd run` serves one request per connection between `waitpid` polls: one header line, one reply, a 5 s deadline, every refusal decided from the @@ -47,7 +47,7 @@ anything else -> error: \n ## File transfer The zone sends `transfer ` with one `SCM_RIGHTS` descriptor it -opened `O_RDONLY`. A descriptor rather than a path means kryptikd never +opened `O_RDONLY`. Because it sends a descriptor, not a path, kryptikd never resolves a path the zone controls. Checks, in order, stopping at the first failure: @@ -57,20 +57,21 @@ failure: 2. Exactly one descriptor, and `dest` is not the sender. 3. The sender's `[transfer] to` lists `dest`, a configured zone (absent means no transfers). The zone holding the NIC receives nothing: `kryptikd check` - refuses a `[transfer]` list that names it, and the broker refuses it again. - The only shipped policy is `dev`'s `to = "work"`. + refuses a `[transfer]` list that includes it, and the broker refuses it + again. The only shipped policy is `dev`'s `to = "work"`. 4. The descriptor is a regular file, `O_RDONLY` and not `O_PATH`, on the sender's data mount (`st_dev` of `/home/`, read through the zone's pid 1 root at request time), and within the size limit (below). `/proc/self/fd/N` is never consulted. 5. The destination is running. A refusal here tells the sender whether a - zone its own `[transfer] to` names is running, which timing would tell it + zone in its own `[transfer] to` is running, which timing would tell it anyway. -6. The user consents (below). Asked last, so a request that would be refused - anyway never becomes a question. After a question the user saw and did not - allow, the same launch asks nothing for a minute: every question takes - focus in zone 0, so a zone may not raise them in a loop. A refusal that - showed nothing (no channel, nobody watching) does not pause. +6. The user consents ([consent](#consent)). Asked last, so a request that + would be refused anyway never becomes a question. After a question the + user saw and did not allow, the same launch asks nothing for a minute: + every question takes focus in zone 0, so a zone may not raise them in a + loop. A refusal that showed nothing (no channel, nobody watching) does not + pause. No file larger than 1 GiB is carried, and a zone may lower that for itself with `[transfer] max_bytes = N`, a whole number of bytes from 1 to 1073741824; @@ -94,13 +95,13 @@ tree. As root, the copy runs with the destination's filesystem uid and gid, which is also what lets it create files in an ephemeral zone's tmpfs home. The name is taken with `O_CREAT|O_EXCL|O_NOFOLLOW`; a collision or planted link moves on to `-2`, `-3`, and so on, with no stat-then-create, temporary -file or `rename`. The file is 0600, owned by the destination. Exactly the -size the `fstat` found, the size the question showed, is copied, from the -file's first byte whatever the descriptor's position (`copy_file_range` with -its own offset and a running count). A file that grew or shrank since is -refused, and any failure removes the partial file. The sender still owns the -file, so it can change bytes within that size; only the size is fixed. The -sender learns only the outcome and the final name. +file or `rename`. The file is 0600, owned by the destination. The copy is the +size `fstat` found, which the question showed, read from the file's first +byte whatever the descriptor's position (`copy_file_range` with its own +offset and a running count). A file that grew or shrank since is refused, and +any failure removes the partial file. The sender still owns the file, so it +can change bytes within that size; only the size is fixed. The sender learns +only the outcome and the final name. ## Consent @@ -121,25 +122,24 @@ at most 60 s; silence, a malformed answer or a missing directory is a refusal, and a question whose sender goes away is withdrawn. The window takes focus when it maps, so a zone could time a request to land -under keys the user meant for the zone's own window. So the window drops -whatever was typed in its first second, half-typed lines too, and then asks -for a two-digit code drawn for that question alone: only the code, typed +under keys the user meant for the zone's own window. The window therefore +drops whatever was typed in its first second, half-typed lines included, then +asks for a two-digit code drawn for that question alone. Only the code, typed after it shows, answers yes. No zone sees a zone 0 window, so none can type -the code. It is kept beside the question for zone 0's tests, which grants -nothing: whatever can read the directory could write the answer. +the code. The code is kept beside the question for zone 0's tests, which +grants nothing: whatever can read the directory could write the answer. The session's group can write in that directory, so nothing found there is trusted. The broker opens the directory once and uses names relative to it -without following links; the question goes to an `O_EXCL` temporary name with +without following links. The question goes to an `O_EXCL` temporary name with a random nonce that is also in its id, so no answer can be planted in -advance; an answer that is not a plain file (a link, a FIFO) is a refusal. -The chrome's own writes there only create, never follow or clobber +advance, and an answer that is not a plain file (a link, a FIFO) is a +refusal. The chrome's own writes there only create, never follow or clobber (`set -C`), and its watcher removes a `.dialog`, `.code` or `.answer` that -outlived its question. -The watcher holds `watcher.lock` exclusively while it runs; a broker that can -take the lock shared knows nobody is watching and refuses at once. -`--auto-approve-transfers` approves everything for tests without a session, -and the launcher warns about it. +outlived its question. The watcher holds `watcher.lock` exclusively while it +runs, so a broker that can take the lock shared knows nobody is watching and +refuses at once. `--auto-approve-transfers` approves everything for tests +without a session, and the launcher warns about it. ## Clipboard @@ -153,11 +153,11 @@ channel. A cross-zone paste is a user gesture in zone 0: the chrome's `m` (move zone N's clipboard to zone M) runs `kryptik-launch --clipboard-move FROM TO`, -which the launch daemon carries out; as root it is `kryptikd clipboard move -FROM TO`. Both zones must be running. The payload leaves the source and -replaces the destination's: one payload crosses, once, and a second gesture -finds nothing to move. No zone can fetch another's payload or trigger a -move: the verb does not exist on the zone-facing socket. Zone 0 has no +which the launch daemon carries out (as root, `kryptikd clipboard move FROM +TO` does the same). Both zones must be running. The payload leaves the source +and replaces the destination's: one payload crosses, once, and a second +gesture finds nothing to move. No zone can fetch another's payload or trigger +a move: the verb does not exist on the zone-facing socket. Zone 0 has no transfer command either; the `kryptik` tool refuses `transfer`, `clipboard` and `mount` and points at the broker. @@ -172,8 +172,8 @@ serves exactly one zone. The proxy advertises only `wl_compositor`, `wl_subcompositor`, `wl_shm`, `wl_seat`, `wl_output`, `xdg_wm_base`, `zxdg_decoration_manager_v1` and `wp_viewporter`, disconnects a client that binds anything else, bounds objects, pending bytes and descriptors per -client, and rewrites every window's app_id to `kryptik..` and -title to `[zone] ...`, from which the compositor draws the zone's border. +client, and rewrites every window's title to `[zone] ...` and its app_id to +`kryptik..`, from which the compositor draws the zone's border. ## Tests @@ -188,7 +188,7 @@ title to `[zone] ...`, from which the compositor draws the zone's border. - `tools/tests/chrome-confirm.py` runs the chrome's real question window on a pty: the code shown allows, a plain `y` refuses, and keys or a half line typed before the question showed are dropped, for the clock question too. -- The launcher suite's broker section: `version` names the zone, an unknown +- The launcher suite's broker section: `version` reports the zone, an unknown verb is refused, the socket is 0600, a foreign peer is refused on a privileged launch. - `build/guest-tests/gui-check.sh` on the installed desktop: the proxy hides @@ -197,17 +197,18 @@ title to `[zone] ...`, from which the compositor draws the zone's border. refused without a question; the code typed delivers byte-identical, a plain `y` refuses, and no question is left behind. - The two hand-written parsers of zone bytes have seeded mutation tests, so a - failure repeats on every machine. The broker's damages each request in + failure repeats on every machine. The broker's test damages each request in `compartments/kryptikd/fuzz-corpus/broker-requests` a hundred-odd ways and sends it over a real connection: no panic, no overrun of the deadline, one well-formed reply. It does so for a plain zone, for a sender whose policy and data mount let transfers through with a descriptor on every request, - and, unprivileged only, for the nic zone, whose time and update verbs would - otherwise reach the host's clock and update state. The proxy's (`protocol.rs`, `session.rs`) damages a - valid body of every message in the generated tables (no panic, no read past - the body, only exact parses accepted) and feeds a damaged opening through a - live session in arbitrary fragments (only whole messages reach the - compositor). Inputs that ever break a parser join the corpus. + and, only when unprivileged, for the nic zone, whose time and update verbs + would otherwise reach the host's clock and update state. The proxy's tests + (`protocol.rs`, `session.rs`) damage a valid body of every message in the + generated tables (no panic, no read past the body, only exact parses + accepted) and feed a damaged opening through a live session in arbitrary + fragments (only whole messages reach the compositor). Inputs that ever + break a parser join the corpus. - `.github/workflows/fuzz.yml` fuzzes both parsers under libFuzzer weekly, twenty minutes each, on a nightly pinned by date. `compartments/kryptikd/fuzz` takes the request parser, seeded from that corpus, and holds every request diff --git a/docs/supply-chain.md b/docs/supply-chain.md index 2589c59f..ecec9518 100644 --- a/docs/supply-chain.md +++ b/docs/supply-chain.md @@ -4,23 +4,25 @@ Kryptik builds what it ships from source, which turns "do I trust this distribution's build servers" into "do I trust these tarballs". The exceptions: -- **Device firmware and CPU microcode** (ADR-012), selected by - `build/config/firmware.list` from the pinned `linux-firmware` release. The - tarball is signed by its kernel.org maintainer and verified like the - kernel's, and each file's licence is the one its `WHENCE` records. That - shows where the bytes came from, not what they do: they run on the device's - own processor, behind the IOMMU (strict by default). +- **Device firmware and CPU microcode** (ADR-012). Device firmware, selected + by `build/config/firmware.list`, and AMD's microcode come from the pinned + `linux-firmware` release. The tarball is signed by its kernel.org + maintainer and verified like the kernel's, and each file's licence is the + one its `WHENCE` records. That shows where the bytes came from, not what + they do: device firmware runs on the device's own processor, behind the + IOMMU (strict by default). Intel's microcode comes from Intel's own + archive, which is unsigned ([assurance per source](#assurance-per-source)). - **The Rust compiler**: kryptikd and kryptik-wlproxy are built from source by an upstream toolchain pinned by version, not one built here (ADR-010). Two build tools that do not ship are special cases: - **cmake** (the `cmake-bin` manifest row): Kitware's Linux binary generates - json-c's build files in the stage 04 chroot, since compiling cmake cost a - quarter of stage 04 for one package. Its hash in `sources.lock` was checked - against Kitware's published SHA-256 list; it is unpacked under the build - tree, never installed, and excluded from the image by stage 06. The source - tarball stays in the manifest as the fallback. + json-c's build files in the stage 04 chroot, since compiling cmake costs a + quarter of stage 04 for one package. Like the source tarball, it is held to + Kitware's signed SHA-256 list. It is unpacked under the build tree, never + installed, and excluded from the image by stage 06. The source tarball + stays in the manifest as the fallback. - **kernel-hardening-checker** runs in stage 05 and CI on the resolved kernel configuration. Its upstream tags are lightweight and unsigned, so the hash of its GitHub archive in `sources.lock` is its only provenance, and @@ -63,23 +65,24 @@ a compliance scanner. ### Expired keys are not tampering Some sources (glibc, gmp, mpc, patch and ncurses among them) are signed with -keys the keyring believes expired. The signatures are valid; the keyring's +keys the keyring believes expired. The signatures are valid: the keyring's copy predates the maintainer extending the key. `verify-signatures.sh` counts them as verified and lists them separately, because a tool that cries -tampering at routine expiry gets ignored. `BADSIG` (the file does not match -its signature) and `REVKEYSIG` (the key was revoked, possibly compromised) -always fail; `--strict`, the gate CI runs on every push, also fails on a -signature that could not be checked or a signer never established: a key -taken from the signature itself, a key not held, a file not downloaded, a -key whose published copy (`tools/key-provenance.tsv`) could not be read that -run. A row there speaks only for the sources it names. A -key that no publisher states anywhere passes it only while -`tools/source-notes.tsv` records the routes that were tried -(`no-usable-key`); such a note for a key that is held fails it as stale, -while a signature the run could not fetch leaves its note untried. A -source that publishes no OpenPGP signature is not the gate's: the lock pins -it, and `tools/verify-provenance.sh --strict` checks whatever else its -publisher states. +tampering at routine expiry gets ignored. + +`BADSIG` (the file does not match its signature) and `REVKEYSIG` (the key was +revoked, possibly compromised) always fail. `--strict`, the gate CI runs on +every push, also fails on a signature that could not be checked and on a +signer never established: a key taken from the signature itself, a key not +held, a file not downloaded, a key whose published copy +(`tools/key-provenance.tsv`) could not be read that run. A row there speaks +only for the sources it names. A key that no publisher states anywhere +passes the gate only while `tools/source-notes.tsv` records the routes that +were tried (`no-usable-key`). Such a note for a key that is held fails the +gate as stale, while a signature the run could not fetch leaves its note +untried. A source that publishes no OpenPGP signature is outside this gate: +the lock pins it, and `tools/verify-provenance.sh --strict` checks whatever +else its publisher states. ### Signature strength varies @@ -94,7 +97,7 @@ publisher states. DSA-1024 over SHA-1 is below what should be relied on, so the signature on `less` is weaker evidence than the rest. Some sources publish a signature that no usable key checks, file's among them; `tools/source-notes.tsv` -names each, with the routes to a key that were tried. +lists each, with the routes to a key that were tried. ### xz @@ -124,8 +127,9 @@ declare a signed tag or a publisher's `.sha256`: - **iana-etc**. GitHub serves a `.sha256` beside the release tarball, from the same platform as the tarball itself. -Neither is a signature over the artifact, and the tool says so. Under -`--strict`, which CI uses on pushes, a check that could not run fails. +Neither a signed tag nor a publisher's checksum is a signature over the +artifact, and the tool says so. Under `--strict`, which CI uses on pushes, a +check that could not run fails. ## Assurance per source @@ -133,8 +137,7 @@ Neither is a signature over the artifact, and the tool says so. Under class, strongest first, from a key pinned in the tree down to `sources.lock` alone. It counts per class and prints no total: a maintainer signature, a signed tag and a publisher checksum are different strengths of evidence, and -one fraction would hide the weakest links. For the same reason this document -quotes no coverage figure. +one coverage figure would hide the weakest links. Some sources end at `sources.lock` alone because upstream signs nothing Kryptik could check. One of them is guarded further along the chain: From 39fccb82c76b4009defeb1975c8193a66aab01dd Mon Sep 17 00:00:00 2001 From: DevomB Date: Wed, 7 Oct 2026 09:29:49 -0700 Subject: [PATCH 17/18] The clock's, the boot check's and the integrity and installer suites' comments are shorter, with their reasons kept: cleanup-held's cut of time.rs, docs/design/time.md, boot-check.sh, integrity-test.sh and tests/installer.sh, redone on main by hand, comments only Only comment regions are taken from 65a6fbb. Main's text stays wherever the cut would drop what main now relies on: the floor may be a later committed release; integrity-test.sh's header is its --help text, step 5 points at sysinit.sh's trust boundary and says why the attacker's payload is signed; the regulatory database is compressed like all of /lib/firmware. time.rs's module doc keeps why the time is a claim: only zone 0 sets the clock, and it has no network. --- build/recipes/boot-check.sh | 2 -- compartments/kryptikd/src/time.rs | 36 +++++++++++-------------------- docs/design/time.md | 9 ++++---- tools/image/integrity-test.sh | 21 ++++++------------ tools/tests/installer.sh | 3 +-- 5 files changed, 23 insertions(+), 48 deletions(-) diff --git a/build/recipes/boot-check.sh b/build/recipes/boot-check.sh index 05a3d61e..04010f6d 100644 --- a/build/recipes/boot-check.sh +++ b/build/recipes/boot-check.sh @@ -1,6 +1,4 @@ #!/usr/bin/env bash -# boot-check: a stage 04 recipe, sourced by build/stages/04-base-system.sh, -# which runs it in the order its list gives. # Everything a boot needs, checked from the target's point of view. s_boot_check() { diff --git a/compartments/kryptikd/src/time.rs b/compartments/kryptikd/src/time.rs index ddd24045..c5949d65 100644 --- a/compartments/kryptikd/src/time.rs +++ b/compartments/kryptikd/src/time.rs @@ -1,30 +1,23 @@ -//! Wall-clock policy for zone 0 (docs/design/time.md). -//! -//! Only zone 0 may set the clock and it has no network, so the time arrives -//! as a claim from the untrusted net zone. A claim may not cross the floor, -//! and past a bound on what is believed unasked, the user decides. -//! `decide` is pure: it reads no clock, file or environment. +//! Wall-clock policy for zone 0 (docs/design/time.md). Only zone 0 sets the clock and it has no +//! network, so the time is a claim from the untrusted net zone: no claim crosses the floor, and +//! past the bound the user decides. /// Corrections below this many seconds are slewed, so the clock never runs backwards. pub const SLEW_BELOW_SECS: f64 = 1.0; -/// Offsets below this many seconds are ignored. pub const IGNORE_BELOW_SECS: f64 = 0.005; /// Seconds believed without asking, per claim and in total, either direction. pub const DEFAULT_BOUND_SECS: i64 = 3600; -/// One claim is considered per interval; others are refused unread, so a -/// hostile zone cannot flood the user with consent prompts. +/// One claim is considered per interval, so a hostile zone cannot flood the user with prompts. pub const CLAIM_INTERVAL_SECS: u64 = 600; -/// The net zone's claim: add `offset` seconds, the median of what `sources` -/// time servers answered. Untrusted, like the zone. +/// The net zone's untrusted claim: add `offset` seconds, the median of `sources` time servers. #[derive(Debug, Clone, Copy, PartialEq)] pub struct Claim { pub offset: f64, pub sources: u8, } -/// Parse ` `: optional sign, up to ten integer and six -/// fractional digits, and 1 to 16 sources. No exponent, infinity or NaN. +/// Parse ` `: plain decimal seconds (no exponent, inf or NaN), 1 to 16 sources. pub fn parse_claim(args: &str) -> Result { let mut it = args.split(' '); let (Some(secs), Some(sources), None) = (it.next(), it.next(), it.next()) else { @@ -54,8 +47,7 @@ pub struct Knowledge { pub floor: i64, /// What is believed without asking. pub bound: i64, - /// Unasked corrections since the clock was last anchored (by consent or - /// the floor). The bound applies to this sum, so small lies cannot add up. + /// Unasked corrections since the last consent or floor, bounded so small lies cannot add up. pub moved_unasked: f64, } @@ -65,13 +57,13 @@ pub enum Decision { Ignore, /// Apply gradually; the clock never steps backwards. Slew { offset: f64 }, - /// Set the clock to `to`. Step { to: f64 }, /// Past the bound: the user decides, shown both times. Ask { to: f64 }, Refuse(String), } +/// Pure: reads no clock, file or environment. pub fn decide(k: &Knowledge, c: &Claim) -> Decision { if !c.offset.is_finite() || !k.now.is_finite() { return Decision::Refuse("the offset is not a number".into()); @@ -99,8 +91,7 @@ pub fn decide(k: &Knowledge, c: &Claim) -> Decision { } } -/// `moved_unasked` once `d` is carried out: an unasked correction adds to it, -/// and consent resets it. +/// `moved_unasked` once `d` is carried out: unasked corrections add to it, consent resets it. pub fn moved_after(k: &Knowledge, c: &Claim, d: &Decision, consented: bool) -> f64 { match d { Decision::Slew { .. } | Decision::Step { .. } => k.moved_unasked + c.offset.abs(), @@ -109,14 +100,12 @@ pub fn moved_after(k: &Knowledge, c: &Claim, d: &Decision, consented: bool) -> f } } -/// The time to set a clock that reads below the floor (say, after a dead RTC -/// battery), or None. Needs no network: the system cannot predate its build. +/// The floor if `now` is below it, as after a dead RTC battery: no system predates its build. pub fn clamp_to_floor(now: f64, floor: i64) -> Option { (now < floor as f64).then_some(floor as f64) } -/// `built_at` from the image record, as written by `date -Iseconds`, in epoch -/// seconds. Anything else is None: no floor known, never zero. +/// The image record's `built_at` (from `date -Iseconds`) in epoch seconds, or None: never zero. pub fn floor_from_image_json(text: &str) -> Option { let at = text.find("\"built_at\"")?; let rest = &text[at + "\"built_at\"".len()..]; @@ -365,8 +354,7 @@ impl Outcome { } } -/// Decide on a claim, ask the user if needed, and carry it out. -/// `ask(now, proposed, sources)` is only called with a proposal at or above the floor. +/// Decide on a claim and carry it out, calling `ask(now, to, sources)` only at or above the floor. pub fn consider( clock: &mut dyn Clock, dir: &Path, diff --git a/docs/design/time.md b/docs/design/time.md index 4201b378..9f61ecb0 100644 --- a/docs/design/time.md +++ b/docs/design/time.md @@ -42,11 +42,10 @@ reported, so one liar among three is outvoted. The servers come from `/etc/kryptik/time.conf` on the verified root (`server HOST` or `pool HOST`, a pool giving up to four addresses; the public pool without the file). The zone reports an offset, not a time: it reads the same `CLOCK_REALTIME` as -zone 0, so nothing is lost to the delay before zone 0 acts. It is a script -rather than an NTP daemon because a daemon is a whole package for one number, -and the script can be tested against a real server on loopback. NTS is not -used: the image has no gnutls, and it would not authenticate a compromised -net zone. +zone 0, so nothing is lost to the delay before zone 0 acts. It is a script, +not an NTP daemon, because a daemon is a whole package for one number and a +script can be tested against a real server on loopback. NTS is not used: the +image has no gnutls, and it would not authenticate a compromised net zone. **The claim.** `time-offset `: a signed decimal with at most 10 integer and 6 fractional digits, and the number of servers (1 to 16) diff --git a/tools/image/integrity-test.sh b/tools/image/integrity-test.sh index fe6e995a..fa709aea 100755 --- a/tools/image/integrity-test.sh +++ b/tools/image/integrity-test.sh @@ -68,8 +68,7 @@ rc=$?; sleep 1; [[ -f "$PIDF" ]] && kill "$(cat "$PIDF")" 2>/dev/null [[ "$rc" -eq 0 ]] && green "installed system boots with Secure Boot enforced (SecureBoot=1 inside the guest)" || red "step 1 drive failed" T1="$(tr -d '\r' < "$LOG1")" grep -q 'KRYPTIK_SMOKE: verity_root=0 [0-9]* verity V' <<<"$T1" && green "dm-verity reports the root valid" || red "no valid verity root reported" -# Lockdown in confidentiality mode, and module signing both ways: stage 05's -# unsigned copy of a driver is refused with the kernel's reason, the signed one loads. +# Stage 05's unsigned copy of a driver is refused with the kernel's reason; the signed one loads. grep -q 'LOCKDOWN=confidentiality' <<<"$T1" && green "lockdown reports confidentiality" || red "lockdown is not in confidentiality mode" if grep -q 'UNSIGNED=refused' <<<"$T1" && grep -q 'Key was rejected by service\|Required key not available' <<<"$T1"; then green "an unsigned module is refused (Key was rejected by service)" @@ -136,15 +135,12 @@ rm -rf "$TMPK" # ----------------------------------------------------------------- step 3 -- step "step 3: a tampered root is refused by dm-verity before userspace" A_OFF=$(( $(part_start "$DISK" 2) * 512 )) -# Flip a byte of the ext4 superblock (the volume name, 1024 + 0x78): mounting -# the root reads it first, so dm-verity fails before userspace. A block that -# nothing reads at boot would go unnoticed. +# The superblock's volume name (1024 + 0x78): the root mount reads it first, so verity fails early. printf '\xa5' | dd of="$DISK" bs=1 seek=$(( A_OFF + 1024 + 0x78 )) conv=notrunc status=none cp "$ENROLLED" "$VARSF" smoke integ-p3 --no-media --disk "$DISK" --vars-file "$VARSF" --timeout 300 > /dev/null T3="$(boot_txt)" -# loglevel=4 hides the KERN_NOTICE banner, so the kernel's timestamped console -# lines are the proof it started. +# loglevel=4 hides the KERN_NOTICE banner; timestamped console lines show the kernel started. grep -qE '^\[ *[0-9]+\.[0-9]+\] |Linux version' <<<"$T3" && green "the (untampered) kernel still starts" || red "the kernel did not start after the root tamper" # The kernel's own message, not the command line's "panic_on_corruption". grep -qE 'device-mapper: verity:.*(corrupt|mismatch|error)|dm-verity device corrupted' <<<"$T3" && green "dm-verity named the corruption" || red "no dm-verity corruption report" @@ -193,8 +189,7 @@ fi rm -rf "$ALT" "$ALTUSB" smoke integ-p4 --usb "$USB" --disk "$DISK" --testctl "$CTLR" --vars enrolled --timeout "$TIMEOUT" > /dev/null boot_txt | grep -q 'KRYPTIK_RECOVER: rc=0' && green "kryptik-recover --restore-slot a succeeded from the medium" || { red "recovery did not report success"; boot_txt | grep 'KRYPTIK_RECOVER' | tail -5 | sed 's/^/ /'; } -# The records recovery wrote on the ESP, read from the host: whole, and -# nothing written through a .new left behind. +# Recovery's records on the ESP, read from the host: whole, and no .new file left behind. dd if="$DISK" of="$ESPIMG" bs=1M iflag=skip_bytes,count_bytes skip="$ESP_OFF" count=$((512*1024*1024)) status=none MVER="$(basename "$USB")"; MVER="${MVER#kryptik-}"; MVER="${MVER%-usb.img}" CSLOT="$(mtype -i "$ESPIMG" ::/kryptik/committed-slot 2>/dev/null)"; CVER="$(mtype -i "$ESPIMG" ::/kryptik/version-a 2>/dev/null)" @@ -240,8 +235,7 @@ uid_base = 1310720 border_color = "#000001" EOF printf 'kernel.kptr_restrict = 0\n' > "$up/sysctl.d/99-evil.conf" - # Also a preload library and a udev rule run as root, both pointing at the - # state partition, and one allowed change (a subuid line) as a control. + # A preload library and a root udev rule on the state partition; a subuid line is the control. mkdir -p "$up/udev/rules.d" "$MNT/lib/kryptik" printf '/var/lib/kryptik/evil.so\n' > "$up/ld.so.preload" printf 'ACTION=="add", RUN+="/var/lib/kryptik/evil.sh"\n' > "$up/udev/rules.d/99-evil.rules" @@ -269,10 +263,7 @@ else fi cp "$ENROLLED" "$VARSF" start_vm integ-p5 "${EXTRA[@]}"; LOG5="$LOG" -# The planted kryptik/ directory must be quarantined and gone from /etc, and -# the updater's anchor on the verified root must still name the release key. -# The root has an ld.so.preload of its own (the allocator), so the planted -# library is looked for by name. +# The root has its own ld.so.preload (the allocator), so the planted library is looked for by name. python3 "$DRV" --serial "$SER" --timeout 300 \ "expect:KRYPTIK_SMOKE: END" "login:${TUSER}:${TPASS}" \ "grab:overlay:ls /etc/kryptik/ /var/lib/kryptik/etc/quarantine/ 2>&1 | head -12" \ diff --git a/tools/tests/installer.sh b/tools/tests/installer.sh index 2aeb1643..a1584c91 100755 --- a/tools/tests/installer.sh +++ b/tools/tests/installer.sh @@ -84,8 +84,7 @@ grep -q 'testctl_get install_replace' "$RUNNER" \ echo echo "-- the runner reports the installer's exit status, not something else's" -# After `cmd | sed`, $? is sed's. Match a call at the start of a line, not the -# word: comments and kryptik-install.json are not calls. +# After `cmd | sed`, $? is sed's. A call starts its line, unlike comments or kryptik-install.json. piped="$(grep -nE '^[[:space:]]*(/usr/sbin/)?kryptik-install[^|#]*\|' "$RUNNER" || true)" if [[ -n "$piped" ]]; then red "the installer is still piped; rc would be the pipeline's last element" From 0a4ac1c555d5ff38ec674a948751087de896210f Mon Sep 17 00:00:00 2001 From: DevomB Date: Wed, 7 Oct 2026 10:55:14 -0700 Subject: [PATCH 18/18] libevdev 1.14.0 reviewed: two force-feedback functions and two documentation fixes, nothing libinput calls; and the roadmap's core-scheduling row gives ADR-011's reason for the cookies, that root can turn SMT back on --- docs/roadmap.md | 3 ++- tools/pin-reviews.tsv | 1 + 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/docs/roadmap.md b/docs/roadmap.md index 7415daf0..d0cca4a9 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -93,7 +93,8 @@ work in zones from a terminal and a text browser, and keep it up to date. - [x] **Core scheduling per zone, and the SMT decision.** ADR-011 decides: `nosmt` stays, since a zone's cookie cannot keep it off the thread beside the kernel, whose execution carries no cookie; each zone still - takes a cookie, for a machine with SMT it cannot turn off. + takes a cookie, since root can turn SMT back on through + `/sys/devices/system/cpu/smt/control`. - [x] **The net zone's remaining hardening.** Decided, with the reasons, in the [net zone design](design/net-zone.md). - [x] **Someone else has attacked it.** The broker protocol and diff --git a/tools/pin-reviews.tsv b/tools/pin-reviews.tsv index 3ea4b937..45fe6943 100644 --- a/tools/pin-reviews.tsv +++ b/tools/pin-reviews.tsv @@ -44,3 +44,4 @@ libxkbcommon 1.13.2 1.14.0 fine 2026-09-19 NEWS for 1.14.0: no security f libdrm 2.4.129 2.4.134 fine 2026-09-19 The 27 commits between them: none is a security fix. libinput 1.30.4 1.32.0 fine 2026-09-19 CVE-2026-35093, -35094 and -50292 are all fixed in the pinned 1.30.4, the newest 1.30.x. 1.31.3 and 1.32.0 add hardening against malicious uinput devices that was not backported; wlroots 0.19.3 has not been built against 1.32. dwl 0.8 0.9 fine 2026-09-27 v0.8..v0.9, 55 commits. The decoration use-after-free fix (f4dfdab, 04279f2) is carried as build/patches/dwl-0.8, and the idle-inhibitor one (4847f97) needs a protocol the proxy never offers a zone (see that README); 433c325, d16036e and 7ba3350 fix IME and pointer code new in 0.9. The Xwayland fixes do not apply (built with XWAYLAND=). 0.8 holds a monitor's frames while a tiled window's resize is pending only when its client is dwl's own child, and a zone window's client is the proxy, which kryptikd starts. f6800e9 stops a key binding's release from reaching the client focused after it, which learns only which binding key was let go. 0.9 needs wlroots 0.20; its new data-control, image-capture and foreign-toplevel globals are off the proxy's allowlist. +libevdev 1.13.7 1.14.0 fine 2026-10-07 Four commits since 1.13.7 (tagged 2026-10-07): two new force-feedback functions, libevdev_upload_ff_effect and libevdev_remove_ff_effect, and two documentation fixes; no security fix, and libinput calls neither.