From ec74767f800a8939a040e47e4762202a43ce50cb Mon Sep 17 00:00:00 2001 From: DevomB Date: Fri, 2 Oct 2026 03:10:13 -0700 Subject: [PATCH] unshare and a namespace clone fail with EPERM in a zone instead of killing it, so Firefox, Chromium and bubblewrap can probe for user namespaces and start Firefox, Chromium, Electron programs and bubblewrap probe for user namespaces at start with clone(CLONE_NEWUSER) or unshare(CLONE_NEWUSER) and carry on when the answer is EPERM: the kernel's answer to an unprivileged caller, and the one Docker's and Podman's default profiles give. A kill left Firefox unable to start in a zone at all. unshare stays on the denied list, so no policy can allow it; setns is killed and clone3 gets ENOSYS as before. seccomp-trace names both calls as soft refusals and answers EPERM. The launcher, adversarial and boundary suites expect the refusal and that no namespace is made, with an unshare-newuser probe beside clone-newuser. --- .../kryptikd/probes/boundary-checks.sh | 7 ++- compartments/kryptikd/src/main.rs | 8 ++- compartments/kryptikd/src/seccomp.rs | 36 ++++++++--- compartments/kryptikd/src/seccomp/tests.rs | 61 +++++++++++++++++-- compartments/tests/adversarial.sh | 24 +++++++- compartments/tests/launcher.sh | 27 ++++++-- docs/design/zone-policy-files.md | 11 +++- docs/hardening.md | 16 ++--- 8 files changed, 157 insertions(+), 33 deletions(-) diff --git a/compartments/kryptikd/probes/boundary-checks.sh b/compartments/kryptikd/probes/boundary-checks.sh index 3d1416d1..727718b9 100755 --- a/compartments/kryptikd/probes/boundary-checks.sh +++ b/compartments/kryptikd/probes/boundary-checks.sh @@ -126,8 +126,9 @@ MATCH="^0 1 2 3$" check "the only descriptors are 0, 1, 2 and the lister's own" # --------------------------------------------------------------------------- head_ "D. Seccomp" -# unshare(1) makes the call (python here has no ctypes); 159 is 128+SIGSYS. -check "clone(CLONE_NEWUSER) from inside the zone is killed, not refused" 159 /usr/bin/unshare -U /bin/true +# unshare(1) makes the call, which fails with EPERM as the kernel tells an +# unprivileged caller: it exits 1, not 159 (SIGSYS), and never runs true. +MATCH="Operation not permitted" check "unshare(2) from inside the zone fails with EPERM, is not killed, and makes no namespace" 1 /usr/bin/unshare -U /bin/true MATCH="AF_VSOCK refused 97" check "AF_VSOCK, AF_ALG and AF_PACKET are refused by family, with EAFNOSUPPORT" 0 /usr/bin/python3 -c " import socket for n,f,t in [('AF_VSOCK',40,1),('AF_ALG',38,5),('AF_PACKET',17,2)]: @@ -140,7 +141,7 @@ import socket a,b=socket.socketpair(); a.close(); b.close() try: socket.socketpair(socket.AF_INET); print('unix-pair inet PAIRED') except OSError as e: print('unix-pair inet', e.errno)" -for p in "clone-newuser 5" "clone3 7" "inotify 7" "setfsuid 7" "socket-vsock 7" "socket-netlink-nf 7" "socket-inet 0" "ioctl-tiocsti 5" "setns 5" "unshare 5" "mount 5" "getpid 0"; do +for p in "clone-newuser 7" "unshare-newuser 7" "clone3 7" "inotify 7" "setfsuid 7" "socket-vsock 7" "socket-netlink-nf 7" "socket-inet 0" "ioctl-tiocsti 5" "setns 5" "mount 5" "getpid 0"; do set -- $p "$K" seccomp-test "$1" >/dev/null 2>&1; rc=$? if [[ "$rc" == "$2" ]]; then pass "seccomp-test $1 -> $rc"; else fail "seccomp-test $1 -> $rc (want $2)"; fi diff --git a/compartments/kryptikd/src/main.rs b/compartments/kryptikd/src/main.rs index d2f54db7..fe640953 100644 --- a/compartments/kryptikd/src/main.rs +++ b/compartments/kryptikd/src/main.rs @@ -793,7 +793,11 @@ fn cmd_seccomp_probe(name: &str) -> Option { if r > 0 { libc::waitpid(r as libc::pid_t, std::ptr::null_mut(), 0); } - 0 + if r < 0 && *libc::__errno_location() == libc::EPERM { 7 } else { 0 } + }, + "unshare-newuser" => || unsafe { + let r = libc::unshare(libc::CLONE_NEWUSER); + if r < 0 && *libc::__errno_location() == libc::EPERM { 7 } else { 0 } }, "clone3" => || unsafe { let r = libc::syscall(libc::SYS_clone3, std::ptr::null::(), 0usize); @@ -1219,7 +1223,7 @@ fn cmd_seccomp_trace(cmd: &[String], allow: &[libc::c_long], sockets: &seccomp:: let nr = libc::c_long::from(req.data.nr); let name = seccomp::name_of(nr).unwrap_or(""); // A soft refusal gets the errno a zone gets, and is marked. - let soft = seccomp::REFUSED_SOFTLY.iter().find(|(n, _)| *n == nr).map(|&(_, e)| e as libc::c_int); + let soft = seccomp::soft_errno(&req.data).map(|e| e as libc::c_int); eprintln!("KRYPTIK_SECCOMP_DENIED {nr} {name}{}", if soft.is_some() { " soft" } else { "" }); refused += 1; let mut resp: libc::seccomp_notif_resp = unsafe { std::mem::zeroed() }; diff --git a/compartments/kryptikd/src/seccomp.rs b/compartments/kryptikd/src/seccomp.rs index 6223b102..2a64a688 100644 --- a/compartments/kryptikd/src/seccomp.rs +++ b/compartments/kryptikd/src/seccomp.rs @@ -28,7 +28,8 @@ const BPF_RET: u16 = 0x06; const SECCOMP_RET_KILL_PROCESS: u32 = 0x8000_0000; const SECCOMP_RET_ALLOW: u32 = 0x7fff_0000; /* Fail the call instead of killing, for programs that probe for a feature and - * must hear "no": clone3, unwanted socket families, `REFUSED_SOFTLY`. */ + * must hear "no": clone3, a namespace clone, unwanted socket families, + * `REFUSED_SOFTLY`. */ const SECCOMP_RET_ERRNO: u32 = 0x0005_0000; // Offsets into struct seccomp_data. @@ -373,7 +374,7 @@ denied! { /// allowlist, so a syscall named here is decided here. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum ArgRule { - /// clone(2): kill if any CLONE_NEW* flag is set in args[0]. + /// clone(2): EPERM if any CLONE_NEW* flag is set in args[0], like unshare(2). CloneNoNamespaces, /// clone3(2): ENOSYS, so libc falls back to clone(2), which the filter can read. Clone3Enosys, @@ -411,6 +412,9 @@ impl ArgRule { /// any of them starts; a zone policy may allow it. The id and capability calls /// never succeed, but ncurses brackets every terminfo open with setfsuid and /// setfsgid, and sudo, su and privilege-dropping daemons call the rest. +/// unshare: Firefox, Chromium and bubblewrap probe for user namespaces and must +/// hear no, as Kryptik's kernel tells an unprivileged caller; a namespace clone +/// gets the same (`ArgRule::CloneNoNamespaces`). pub const REFUSED_SOFTLY: &[(libc::c_long, u32)] = &[ (libc::SYS_inotify_init, ENOSYS), (libc::SYS_inotify_init1, ENOSYS), @@ -424,12 +428,33 @@ pub const REFUSED_SOFTLY: &[(libc::c_long, u32)] = &[ (libc::SYS_setresgid, EPERM), (libc::SYS_setgroups, EPERM), (libc::SYS_capset, EPERM), + (libc::SYS_unshare, EPERM), ]; const fn errno_action(e: u32) -> u32 { SECCOMP_RET_ERRNO | (e & 0xffff) } +/// A soft refusal's action. seccomp-trace is notified instead and answers with +/// the same errno, so the program runs as it would in a zone and the call is +/// still named. +const fn soft_refusal(e: u32, deny_action: u32) -> u32 { + if deny_action == libc::SECCOMP_RET_USER_NOTIF { deny_action } else { errno_action(e) } +} + +/// The errno a zone gets for a call refused without a kill, None for one it is +/// killed for: what `seccomp-trace` answers a call it is notified of. +pub fn soft_errno(call: &libc::seccomp_data) -> Option { + let nr = libc::c_long::from(call.nr); + if call.arch != AUDIT_ARCH_X86_64 { + return None; + } + if nr == libc::SYS_clone && (call.args[0] as u32) & CLONE_NS_MASK != 0 { + return Some(EPERM); + } + REFUSED_SOFTLY.iter().find(|(n, _)| *n == nr).map(|&(_, e)| e) +} + /// Emit one argument rule. Entered with the syscall number in the accumulator; /// a non-matching number skips the block with the accumulator intact, and every /// path through a matched block ends in `ret`. @@ -438,7 +463,7 @@ fn emit_arg_rule(p: &mut Vec, rule: ArgRule, deny_action: u32, socke ArgRule::CloneNoNamespaces => vec![ stmt(BPF_LD | BPF_W | BPF_ABS, arg_lo(0)), jump(BPF_JMP | BPF_JSET | BPF_K, CLONE_NS_MASK, 0, 1), - stmt(BPF_RET | BPF_K, deny_action), + stmt(BPF_RET | BPF_K, soft_refusal(EPERM, deny_action)), stmt(BPF_RET | BPF_K, SECCOMP_RET_ALLOW), ], ArgRule::Clone3Enosys => vec![stmt(BPF_RET | BPF_K, errno_action(ENOSYS))], @@ -564,11 +589,8 @@ fn build_program_full( for &(nr, e) in REFUSED_SOFTLY { // Allowed by the list, it is allowed below like any other. if !allow.contains(&nr) { - /* seccomp-trace answers these with the same errno, so the program - * runs as it would in a zone and the call is still named. */ - let refuse = if deny_action == libc::SECCOMP_RET_USER_NOTIF { deny_action } else { errno_action(e) }; p.push(jump(BPF_JMP | BPF_JEQ | BPF_K, nr as u32, 0, 1)); - p.push(stmt(BPF_RET | BPF_K, refuse)); + p.push(stmt(BPF_RET | BPF_K, soft_refusal(e, deny_action))); } } diff --git a/compartments/kryptikd/src/seccomp/tests.rs b/compartments/kryptikd/src/seccomp/tests.rs index ad9a4f52..59d61a7c 100644 --- a/compartments/kryptikd/src/seccomp/tests.rs +++ b/compartments/kryptikd/src/seccomp/tests.rs @@ -269,7 +269,8 @@ fn clone_without_namespace_flags_is_allowed() { } #[test] -fn clone_with_any_namespace_flag_is_killed() { +fn namespace_clone_fails_with_eperm() { + // As unshare(2) does: a program probing for user namespaces lives on. let p = build_program(BASE_ALLOWLIST).unwrap(); for f in [ libc::CLONE_NEWUSER, libc::CLONE_NEWNS, libc::CLONE_NEWPID, @@ -279,17 +280,69 @@ fn clone_with_any_namespace_flag_is_killed() { let flags = (f as u32 | libc::SIGCHLD as u32) as u64; assert_eq!( evaluate_args(&p, X86, libc::SYS_clone as u32, with_arg(0, flags)), - SECCOMP_RET_KILL_PROCESS, - "clone flags {flags:#x} must be killed" + errno_action(EPERM), + "clone flags {flags:#x} must fail with EPERM" ); // Legacy clone ignores the high word, and so does the filter. assert_eq!( evaluate_args(&p, X86, libc::SYS_clone as u32, with_arg(0, flags | (1 << 40))), - SECCOMP_RET_KILL_PROCESS + errno_action(EPERM) ); } } +#[test] +fn unshare_fails_setns_killed() { + // Both stay on the denied list: no policy may allow either. + let p = build_program(BASE_ALLOWLIST).unwrap(); + for f in [0, libc::CLONE_NEWUSER as u64, CLONE_NS_MASK as u64] { + let r = evaluate_args(&p, X86, libc::SYS_unshare as u32, with_arg(0, f)); + assert_eq!(r, errno_action(EPERM), "unshare flags {f:#x}"); + } + let r = evaluate_args(&p, X86, libc::SYS_setns as u32, with_arg(1, libc::CLONE_NEWUSER as u64)); + assert_eq!(r, SECCOMP_RET_KILL_PROCESS); + for nr in [libc::SYS_unshare, libc::SYS_setns] { + assert!(is_denied(nr) && widened(&[nr]).is_err(), "syscall {nr} can be allowed"); + } +} + +#[test] +fn trace_names_namespace_refusals() { + // Named like any soft refusal, while a plain clone runs. + let p = build_program_with(BASE_ALLOWLIST, libc::SECCOMP_RET_USER_NOTIF).unwrap(); + let ns = with_arg(0, (libc::CLONE_NEWUSER | libc::SIGCHLD) as u64); + assert_eq!(evaluate_args(&p, X86, libc::SYS_clone as u32, ns), libc::SECCOMP_RET_USER_NOTIF); + assert_eq!(evaluate_args(&p, X86, libc::SYS_unshare as u32, ns), libc::SECCOMP_RET_USER_NOTIF); + let fork = with_arg(0, libc::SIGCHLD as u64); + assert_eq!(evaluate_args(&p, X86, libc::SYS_clone as u32, fork), SECCOMP_RET_ALLOW); +} + +#[test] +fn trace_answers_as_a_zone() { + /* seccomp-trace hears of every call the zone filter refuses, and answers + * each as a zone hears it: a soft refusal's errno, or ENOSYS where a zone + * is killed. */ + let zone = build_program(BASE_ALLOWLIST).unwrap(); + let trace = build_program_with(BASE_ALLOWLIST, libc::SECCOMP_RET_USER_NOTIF).unwrap(); + let ns = with_arg(0, (libc::CLONE_NEWUSER | libc::SIGCHLD) as u64); + let mut calls: Vec<(libc::c_long, [u64; 6])> = DENIED_RATIONALE + .iter() + .map(|&(nr, _)| nr) + .chain(REFUSED_SOFTLY.iter().map(|&(nr, _)| nr)) + .map(|nr| (nr, [0; 6])) + .collect(); + calls.extend([(libc::SYS_clone, ns), (libc::SYS_unshare, ns), (libc::SYS_ioctl, with_arg(1, TIOCSTI as u64))]); + for (nr, args) in calls { + assert_eq!(evaluate_args(&trace, X86, nr as u32, args), libc::SECCOMP_RET_USER_NOTIF, "syscall {nr} is not named"); + let call = libc::seccomp_data { nr: nr as i32, arch: X86, instruction_pointer: 0, args }; + let heard = soft_errno(&call).map_or(SECCOMP_RET_KILL_PROCESS, errno_action); + assert_eq!(evaluate_args(&zone, X86, nr as u32, args), heard, "syscall {nr}"); + } + // Another architecture is killed in a zone whatever the number. + let i386 = libc::seccomp_data { nr: libc::SYS_unshare as i32, arch: 0x4000_0003, instruction_pointer: 0, args: ns }; + assert_eq!(soft_errno(&i386), None); +} + #[test] fn clone3_gets_enosys() { let p = build_program(BASE_ALLOWLIST).unwrap(); diff --git a/compartments/tests/adversarial.sh b/compartments/tests/adversarial.sh index 98fdaba5..80a42ab6 100755 --- a/compartments/tests/adversarial.sh +++ b/compartments/tests/adversarial.sh @@ -290,7 +290,7 @@ else # setns alone would step into another zone's namespaces, defeating 1-4. leaked=0 - for sc in setns ptrace unshare mount bpf perf_event_open userfaultfd \ + for sc in setns ptrace mount bpf perf_event_open userfaultfd \ keyctl init_module kexec_load process_vm_readv pivot_root chroot; do "$KRYPTIKD" seccomp-test "$sc" >/dev/null 2>&1 rc=$? @@ -301,10 +301,30 @@ else done if [[ "$leaked" -eq 0 ]]; then - pass "all 13 dangerous syscalls killed by SIGSYS (setns among them)" + pass "all 12 dangerous syscalls killed by SIGSYS (setns among them)" else fail "${leaked} dangerous syscall(s) reachable from inside a zone" fi + + # unshare(2) and a namespace clone fail with EPERM instead: programs probe + # for user namespaces and must hear no, as the kernel tells an unprivileged + # caller. This host lets the suite make one (the preconditions), so EPERM + # here is the filter's, and exit 7 means no namespace was made. + nested=0 + for p in unshare-newuser clone-newuser; do + "$KRYPTIKD" seccomp-test "$p" >/dev/null 2>&1 + rc=$? + if [[ "$rc" -ne 7 ]]; then + echo " ${p} was NOT refused with EPERM (rc=${rc})" + nested=$((nested + 1)) + fi + done + + if [[ "$nested" -eq 0 ]]; then + pass "unshare(2) and clone(CLONE_NEWUSER) fail with EPERM and make no namespace" + else + fail "${nested} namespace-creating call(s) not refused with EPERM" + fi fi # --- consistency with kryptikd ---------------------------------------------- diff --git a/compartments/tests/launcher.sh b/compartments/tests/launcher.sh index 7994867d..ee66fdf4 100755 --- a/compartments/tests/launcher.sh +++ b/compartments/tests/launcher.sh @@ -856,8 +856,8 @@ probe "I1 a zone process reports seccomp mode 2 (filtered)" "2" zrun alpha -- /bin/sh -c "$PRO echo PROBE=\$(grep '^Seccomp_filters:' /proc/self/status | awk '{print \$2}')" probe "I2 exactly one filter is installed, not zero and not a stack" "1" -# Each is in seccomp.rs::DENIED_RATIONALE. A denied syscall kills the zone -# (SIGSYS, exit 159) rather than returning an error it could ignore. +# Each is in seccomp.rs::DENIED_RATIONALE and kills the zone (SIGSYS, exit +# 159) rather than returning an error it could ignore. seccomp_kill() { # desc shell-command local desc="$1" cmd="$2" zrun alpha -- /bin/sh -c "$cmd" @@ -874,8 +874,12 @@ seccomp_kill "I3 mount(2) kills the zone (re-mount under Landlock)" \ '/bin/mount -t tmpfs none /tmp' seccomp_kill "I4 chroot(2) kills the zone (double-chroot escape)" \ '/usr/sbin/chroot / /bin/true' -seccomp_kill "I5 unshare(2) kills the zone (nested namespace LPE surface)" \ - '/usr/bin/unshare -U /bin/true' +# unshare(2) is denied too, but fails with EPERM: Firefox, Chromium and +# bubblewrap probe for user namespaces and must hear no, as the kernel tells an +# unprivileged caller. unshare(1) lives to say so, and readlink never runs in a +# namespace of its own. +zrun alpha -- /bin/sh -c "$PRO o=\$(LC_ALL=C /usr/bin/unshare -U readlink /proc/self/ns/user 2>&1); rc=\$?; case \"\$rc:\$o\" in 159:*) echo PROBE=SIGSYS;; 0:user:*) echo PROBE=CREATED;; 1:*'Operation not permitted'*) echo PROBE=eperm;; *) echo \"PROBE=\$rc:\$o\";; esac" +probe "I5 unshare(2) fails with EPERM, not SIGSYS, and makes no namespace" "eperm" # mknod(2) is allowed so mkfifo works; a device node is refused by Landlock # (MAKE_CHAR/MAKE_BLOCK granted nowhere) and by nodev on every mount. zrun alpha -- /bin/sh -c "$PRO /usr/bin/mknod /tmp/n c 1 3 2>/dev/null; if [ -e /tmp/n ]; then echo PROBE=CREATED; else echo PROBE=refused; fi" @@ -1236,7 +1240,7 @@ filter_probe() { # desc probe expected fi } -filter_probe "L3 clone(CLONE_NEWUSER) is killed (nested user namespace)" clone-newuser 5 +filter_probe "L3 clone(CLONE_NEWUSER) fails with EPERM rather than killing (no nested user namespace)" clone-newuser 7 filter_probe "L4 clone3 returns ENOSYS rather than killing (glibc falls back)" clone3 7 filter_probe "L5 socket(AF_VSOCK) is refused with an errno" socket-vsock 7 filter_probe "L6 socket(AF_NETLINK/NETFILTER) is refused with an errno" socket-netlink-nf 7 @@ -1256,6 +1260,19 @@ else fail "L9 seccomp-trace did not report reboot(2) and inotify_init1(2) [$(tr '\n' ' ' <<<"$out")]" fi +# clone(CLONE_NEWUSER) and unshare(2) are named too, marked soft, and fail with +# EPERM (1) as in a zone. A kernel may answer EPERM as well, so the names are +# what show that the filter refuses them. +out="$(timeout "$TIMEOUT" "$KRYPTIKD" seccomp-trace -- python3 -c \ + 'import ctypes; c = ctypes.CDLL(None, use_errno=True); [print(c.syscall(*a), ctypes.get_errno()) for a in ((56, 0x10000011, 0, 0, 0, 0), (272, 0x10000000))]' 2>&1)" +if grep -qx 'KRYPTIK_SECCOMP_DENIED 56 clone soft' <<<"$out" \ + && grep -qx 'KRYPTIK_SECCOMP_DENIED 272 unshare soft' <<<"$out" \ + && [[ "$(grep -cx -e '-1 1' <<<"$out")" == 2 ]]; then + pass "L9b seccomp-trace names a namespace clone and unshare(2) as soft refusals, and both fail with EPERM" +else + fail "L9b seccomp-trace did not refuse clone(CLONE_NEWUSER) and unshare(2) with EPERM [$(tr '\n' ' ' <<<"$out")]" +fi + # ncurses brackets each terminfo open with setfsuid and setfsgid. The filter # answers them with EPERM; a kill would take every terminal program with it # (tput would end with 159, SIGSYS). diff --git a/docs/design/zone-policy-files.md b/docs/design/zone-policy-files.md index 3fd4351c..2473d457 100755 --- a/docs/design/zone-policy-files.md +++ b/docs/design/zone-policy-files.md @@ -21,9 +21,9 @@ keep-capability CAP_NET_RAW # left in the bounding set - `allow-syscall`: a name from `seccomp::ADDABLE`. A syscall on the base denied list (`DENIED_RATIONALE`: `ptrace`, `mount`, `setns`, `bpf`, ...) cannot be re-allowed, one on the base allowlist is reported as already - allowed, and any other name is an error. The id and capability calls on - the denied list fail with EPERM instead of killing the caller; the trace - paragraph below says why. + allowed, and any other name is an error. The id and capability calls and + `unshare` on the denied list fail with EPERM instead of killing the caller; + the trace paragraph below says why. - `allow-socket`: `AF_PACKET`, `AF_KEY`, `AF_ALG`, `AF_VSOCK`, `AF_BLUETOOTH`, `AF_CAN`, `AF_RDS`, `AF_TIPC` or `AF_XDP`; `AF_NETLINK` drops the netlink protocol check. `socketpair(2)` stays `AF_UNIX` only whatever the file @@ -55,6 +55,11 @@ after its name and gets the same errno here: `setfsgid`, and `sudo`, `su` and daemons that drop privilege as root call the rest, so a kill would take them down unexplained. They stay on the denied list, and no id or capability changes either way. +- `unshare`, and `clone` with namespace flags, fail with EPERM: Firefox, + Chromium and bubblewrap probe for user namespaces at start and must hear + no, as the kernel tells an unprivileged caller. `unshare` stays on the + denied list, and `setns`, which would enter another zone's namespaces, is + killed. A printed name can go on an `allow-syscall` line unless it is on the denied list; a call refused for its arguments (namespace flags to `clone`, diff --git a/docs/hardening.md b/docs/hardening.md index 281fcc80..b73ce9a1 100644 --- a/docs/hardening.md +++ b/docs/hardening.md @@ -167,11 +167,13 @@ every sysctl reads back as `build/config/sysctl.d` says. Every zoned process runs under a default-deny seccomp-bpf filter (`compartments/kryptikd/src/seccomp.rs`) allowing about 200 syscalls; anything -else is `SECCOMP_RET_KILL_PROCESS`. `clone` with namespace flags is killed, -`clone3` fails with `ENOSYS` so libc falls back to `clone`, the `TIOCSTI` and -`TIOCLINUX` ioctls are killed, and `socket` is limited to `AF_UNIX`, -`AF_INET`, `AF_INET6` and `NETLINK_ROUTE`. A zone policy file can widen this -in named ways but never re-allow a denied syscall +else is `SECCOMP_RET_KILL_PROCESS`. `unshare` and `clone` with namespace flags +fail with `EPERM` instead: Firefox, Chromium and bubblewrap probe for user +namespaces at start and must hear no, as the kernel tells an unprivileged +caller. `clone3` fails with `ENOSYS` so libc falls back to `clone`, the +`TIOCSTI` and `TIOCLINUX` ioctls are killed, and `socket` is limited to +`AF_UNIX`, `AF_INET`, `AF_INET6` and `NETLINK_ROUTE`. A zone policy file can +widen this in named ways but never re-allow a denied syscall ([zone policy files](design/zone-policy-files.md)). | Denied | Why | @@ -186,8 +188,8 @@ in named ways but never re-allow a denied syscall | `init_module`, `finit_module`, `kexec_load` | load kernel code | | `io_uring_*` | does I/O without syscalls, past the filter | -`compartments/tests/adversarial.sh` makes 13 of these calls in a real process -and expects SIGSYS. +`compartments/tests/adversarial.sh` makes 12 of these calls in a real process +and expects SIGSYS; from `unshare` and a namespace `clone` it expects `EPERM`. Other architectures are refused, and x32 calls (x86-64 numbers with bit 30 set) are killed before the allowlist. Each allowed syscall is a compare