diff --git a/compartments/kryptikd/probes/boundary-checks.sh b/compartments/kryptikd/probes/boundary-checks.sh index 3d1416d1..727718b9 100755 --- a/compartments/kryptikd/probes/boundary-checks.sh +++ b/compartments/kryptikd/probes/boundary-checks.sh @@ -126,8 +126,9 @@ MATCH="^0 1 2 3$" check "the only descriptors are 0, 1, 2 and the lister's own" # --------------------------------------------------------------------------- head_ "D. Seccomp" -# unshare(1) makes the call (python here has no ctypes); 159 is 128+SIGSYS. -check "clone(CLONE_NEWUSER) from inside the zone is killed, not refused" 159 /usr/bin/unshare -U /bin/true +# unshare(1) makes the call, which fails with EPERM as the kernel tells an +# unprivileged caller: it exits 1, not 159 (SIGSYS), and never runs true. +MATCH="Operation not permitted" check "unshare(2) from inside the zone fails with EPERM, is not killed, and makes no namespace" 1 /usr/bin/unshare -U /bin/true MATCH="AF_VSOCK refused 97" check "AF_VSOCK, AF_ALG and AF_PACKET are refused by family, with EAFNOSUPPORT" 0 /usr/bin/python3 -c " import socket for n,f,t in [('AF_VSOCK',40,1),('AF_ALG',38,5),('AF_PACKET',17,2)]: @@ -140,7 +141,7 @@ import socket a,b=socket.socketpair(); a.close(); b.close() try: socket.socketpair(socket.AF_INET); print('unix-pair inet PAIRED') except OSError as e: print('unix-pair inet', e.errno)" -for p in "clone-newuser 5" "clone3 7" "inotify 7" "setfsuid 7" "socket-vsock 7" "socket-netlink-nf 7" "socket-inet 0" "ioctl-tiocsti 5" "setns 5" "unshare 5" "mount 5" "getpid 0"; do +for p in "clone-newuser 7" "unshare-newuser 7" "clone3 7" "inotify 7" "setfsuid 7" "socket-vsock 7" "socket-netlink-nf 7" "socket-inet 0" "ioctl-tiocsti 5" "setns 5" "mount 5" "getpid 0"; do set -- $p "$K" seccomp-test "$1" >/dev/null 2>&1; rc=$? if [[ "$rc" == "$2" ]]; then pass "seccomp-test $1 -> $rc"; else fail "seccomp-test $1 -> $rc (want $2)"; fi diff --git a/compartments/kryptikd/src/main.rs b/compartments/kryptikd/src/main.rs index d2f54db7..fe640953 100644 --- a/compartments/kryptikd/src/main.rs +++ b/compartments/kryptikd/src/main.rs @@ -793,7 +793,11 @@ fn cmd_seccomp_probe(name: &str) -> Option { if r > 0 { libc::waitpid(r as libc::pid_t, std::ptr::null_mut(), 0); } - 0 + if r < 0 && *libc::__errno_location() == libc::EPERM { 7 } else { 0 } + }, + "unshare-newuser" => || unsafe { + let r = libc::unshare(libc::CLONE_NEWUSER); + if r < 0 && *libc::__errno_location() == libc::EPERM { 7 } else { 0 } }, "clone3" => || unsafe { let r = libc::syscall(libc::SYS_clone3, std::ptr::null::(), 0usize); @@ -1219,7 +1223,7 @@ fn cmd_seccomp_trace(cmd: &[String], allow: &[libc::c_long], sockets: &seccomp:: let nr = libc::c_long::from(req.data.nr); let name = seccomp::name_of(nr).unwrap_or(""); // A soft refusal gets the errno a zone gets, and is marked. - let soft = seccomp::REFUSED_SOFTLY.iter().find(|(n, _)| *n == nr).map(|&(_, e)| e as libc::c_int); + let soft = seccomp::soft_errno(&req.data).map(|e| e as libc::c_int); eprintln!("KRYPTIK_SECCOMP_DENIED {nr} {name}{}", if soft.is_some() { " soft" } else { "" }); refused += 1; let mut resp: libc::seccomp_notif_resp = unsafe { std::mem::zeroed() }; diff --git a/compartments/kryptikd/src/seccomp.rs b/compartments/kryptikd/src/seccomp.rs index 6223b102..2a64a688 100644 --- a/compartments/kryptikd/src/seccomp.rs +++ b/compartments/kryptikd/src/seccomp.rs @@ -28,7 +28,8 @@ const BPF_RET: u16 = 0x06; const SECCOMP_RET_KILL_PROCESS: u32 = 0x8000_0000; const SECCOMP_RET_ALLOW: u32 = 0x7fff_0000; /* Fail the call instead of killing, for programs that probe for a feature and - * must hear "no": clone3, unwanted socket families, `REFUSED_SOFTLY`. */ + * must hear "no": clone3, a namespace clone, unwanted socket families, + * `REFUSED_SOFTLY`. */ const SECCOMP_RET_ERRNO: u32 = 0x0005_0000; // Offsets into struct seccomp_data. @@ -373,7 +374,7 @@ denied! { /// allowlist, so a syscall named here is decided here. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum ArgRule { - /// clone(2): kill if any CLONE_NEW* flag is set in args[0]. + /// clone(2): EPERM if any CLONE_NEW* flag is set in args[0], like unshare(2). CloneNoNamespaces, /// clone3(2): ENOSYS, so libc falls back to clone(2), which the filter can read. Clone3Enosys, @@ -411,6 +412,9 @@ impl ArgRule { /// any of them starts; a zone policy may allow it. The id and capability calls /// never succeed, but ncurses brackets every terminfo open with setfsuid and /// setfsgid, and sudo, su and privilege-dropping daemons call the rest. +/// unshare: Firefox, Chromium and bubblewrap probe for user namespaces and must +/// hear no, as Kryptik's kernel tells an unprivileged caller; a namespace clone +/// gets the same (`ArgRule::CloneNoNamespaces`). pub const REFUSED_SOFTLY: &[(libc::c_long, u32)] = &[ (libc::SYS_inotify_init, ENOSYS), (libc::SYS_inotify_init1, ENOSYS), @@ -424,12 +428,33 @@ pub const REFUSED_SOFTLY: &[(libc::c_long, u32)] = &[ (libc::SYS_setresgid, EPERM), (libc::SYS_setgroups, EPERM), (libc::SYS_capset, EPERM), + (libc::SYS_unshare, EPERM), ]; const fn errno_action(e: u32) -> u32 { SECCOMP_RET_ERRNO | (e & 0xffff) } +/// A soft refusal's action. seccomp-trace is notified instead and answers with +/// the same errno, so the program runs as it would in a zone and the call is +/// still named. +const fn soft_refusal(e: u32, deny_action: u32) -> u32 { + if deny_action == libc::SECCOMP_RET_USER_NOTIF { deny_action } else { errno_action(e) } +} + +/// The errno a zone gets for a call refused without a kill, None for one it is +/// killed for: what `seccomp-trace` answers a call it is notified of. +pub fn soft_errno(call: &libc::seccomp_data) -> Option { + let nr = libc::c_long::from(call.nr); + if call.arch != AUDIT_ARCH_X86_64 { + return None; + } + if nr == libc::SYS_clone && (call.args[0] as u32) & CLONE_NS_MASK != 0 { + return Some(EPERM); + } + REFUSED_SOFTLY.iter().find(|(n, _)| *n == nr).map(|&(_, e)| e) +} + /// Emit one argument rule. Entered with the syscall number in the accumulator; /// a non-matching number skips the block with the accumulator intact, and every /// path through a matched block ends in `ret`. @@ -438,7 +463,7 @@ fn emit_arg_rule(p: &mut Vec, rule: ArgRule, deny_action: u32, socke ArgRule::CloneNoNamespaces => vec![ stmt(BPF_LD | BPF_W | BPF_ABS, arg_lo(0)), jump(BPF_JMP | BPF_JSET | BPF_K, CLONE_NS_MASK, 0, 1), - stmt(BPF_RET | BPF_K, deny_action), + stmt(BPF_RET | BPF_K, soft_refusal(EPERM, deny_action)), stmt(BPF_RET | BPF_K, SECCOMP_RET_ALLOW), ], ArgRule::Clone3Enosys => vec![stmt(BPF_RET | BPF_K, errno_action(ENOSYS))], @@ -564,11 +589,8 @@ fn build_program_full( for &(nr, e) in REFUSED_SOFTLY { // Allowed by the list, it is allowed below like any other. if !allow.contains(&nr) { - /* seccomp-trace answers these with the same errno, so the program - * runs as it would in a zone and the call is still named. */ - let refuse = if deny_action == libc::SECCOMP_RET_USER_NOTIF { deny_action } else { errno_action(e) }; p.push(jump(BPF_JMP | BPF_JEQ | BPF_K, nr as u32, 0, 1)); - p.push(stmt(BPF_RET | BPF_K, refuse)); + p.push(stmt(BPF_RET | BPF_K, soft_refusal(e, deny_action))); } } diff --git a/compartments/kryptikd/src/seccomp/tests.rs b/compartments/kryptikd/src/seccomp/tests.rs index ad9a4f52..59d61a7c 100644 --- a/compartments/kryptikd/src/seccomp/tests.rs +++ b/compartments/kryptikd/src/seccomp/tests.rs @@ -269,7 +269,8 @@ fn clone_without_namespace_flags_is_allowed() { } #[test] -fn clone_with_any_namespace_flag_is_killed() { +fn namespace_clone_fails_with_eperm() { + // As unshare(2) does: a program probing for user namespaces lives on. let p = build_program(BASE_ALLOWLIST).unwrap(); for f in [ libc::CLONE_NEWUSER, libc::CLONE_NEWNS, libc::CLONE_NEWPID, @@ -279,17 +280,69 @@ fn clone_with_any_namespace_flag_is_killed() { let flags = (f as u32 | libc::SIGCHLD as u32) as u64; assert_eq!( evaluate_args(&p, X86, libc::SYS_clone as u32, with_arg(0, flags)), - SECCOMP_RET_KILL_PROCESS, - "clone flags {flags:#x} must be killed" + errno_action(EPERM), + "clone flags {flags:#x} must fail with EPERM" ); // Legacy clone ignores the high word, and so does the filter. assert_eq!( evaluate_args(&p, X86, libc::SYS_clone as u32, with_arg(0, flags | (1 << 40))), - SECCOMP_RET_KILL_PROCESS + errno_action(EPERM) ); } } +#[test] +fn unshare_fails_setns_killed() { + // Both stay on the denied list: no policy may allow either. + let p = build_program(BASE_ALLOWLIST).unwrap(); + for f in [0, libc::CLONE_NEWUSER as u64, CLONE_NS_MASK as u64] { + let r = evaluate_args(&p, X86, libc::SYS_unshare as u32, with_arg(0, f)); + assert_eq!(r, errno_action(EPERM), "unshare flags {f:#x}"); + } + let r = evaluate_args(&p, X86, libc::SYS_setns as u32, with_arg(1, libc::CLONE_NEWUSER as u64)); + assert_eq!(r, SECCOMP_RET_KILL_PROCESS); + for nr in [libc::SYS_unshare, libc::SYS_setns] { + assert!(is_denied(nr) && widened(&[nr]).is_err(), "syscall {nr} can be allowed"); + } +} + +#[test] +fn trace_names_namespace_refusals() { + // Named like any soft refusal, while a plain clone runs. + let p = build_program_with(BASE_ALLOWLIST, libc::SECCOMP_RET_USER_NOTIF).unwrap(); + let ns = with_arg(0, (libc::CLONE_NEWUSER | libc::SIGCHLD) as u64); + assert_eq!(evaluate_args(&p, X86, libc::SYS_clone as u32, ns), libc::SECCOMP_RET_USER_NOTIF); + assert_eq!(evaluate_args(&p, X86, libc::SYS_unshare as u32, ns), libc::SECCOMP_RET_USER_NOTIF); + let fork = with_arg(0, libc::SIGCHLD as u64); + assert_eq!(evaluate_args(&p, X86, libc::SYS_clone as u32, fork), SECCOMP_RET_ALLOW); +} + +#[test] +fn trace_answers_as_a_zone() { + /* seccomp-trace hears of every call the zone filter refuses, and answers + * each as a zone hears it: a soft refusal's errno, or ENOSYS where a zone + * is killed. */ + let zone = build_program(BASE_ALLOWLIST).unwrap(); + let trace = build_program_with(BASE_ALLOWLIST, libc::SECCOMP_RET_USER_NOTIF).unwrap(); + let ns = with_arg(0, (libc::CLONE_NEWUSER | libc::SIGCHLD) as u64); + let mut calls: Vec<(libc::c_long, [u64; 6])> = DENIED_RATIONALE + .iter() + .map(|&(nr, _)| nr) + .chain(REFUSED_SOFTLY.iter().map(|&(nr, _)| nr)) + .map(|nr| (nr, [0; 6])) + .collect(); + calls.extend([(libc::SYS_clone, ns), (libc::SYS_unshare, ns), (libc::SYS_ioctl, with_arg(1, TIOCSTI as u64))]); + for (nr, args) in calls { + assert_eq!(evaluate_args(&trace, X86, nr as u32, args), libc::SECCOMP_RET_USER_NOTIF, "syscall {nr} is not named"); + let call = libc::seccomp_data { nr: nr as i32, arch: X86, instruction_pointer: 0, args }; + let heard = soft_errno(&call).map_or(SECCOMP_RET_KILL_PROCESS, errno_action); + assert_eq!(evaluate_args(&zone, X86, nr as u32, args), heard, "syscall {nr}"); + } + // Another architecture is killed in a zone whatever the number. + let i386 = libc::seccomp_data { nr: libc::SYS_unshare as i32, arch: 0x4000_0003, instruction_pointer: 0, args: ns }; + assert_eq!(soft_errno(&i386), None); +} + #[test] fn clone3_gets_enosys() { let p = build_program(BASE_ALLOWLIST).unwrap(); diff --git a/compartments/tests/adversarial.sh b/compartments/tests/adversarial.sh index 98fdaba5..80a42ab6 100755 --- a/compartments/tests/adversarial.sh +++ b/compartments/tests/adversarial.sh @@ -290,7 +290,7 @@ else # setns alone would step into another zone's namespaces, defeating 1-4. leaked=0 - for sc in setns ptrace unshare mount bpf perf_event_open userfaultfd \ + for sc in setns ptrace mount bpf perf_event_open userfaultfd \ keyctl init_module kexec_load process_vm_readv pivot_root chroot; do "$KRYPTIKD" seccomp-test "$sc" >/dev/null 2>&1 rc=$? @@ -301,10 +301,30 @@ else done if [[ "$leaked" -eq 0 ]]; then - pass "all 13 dangerous syscalls killed by SIGSYS (setns among them)" + pass "all 12 dangerous syscalls killed by SIGSYS (setns among them)" else fail "${leaked} dangerous syscall(s) reachable from inside a zone" fi + + # unshare(2) and a namespace clone fail with EPERM instead: programs probe + # for user namespaces and must hear no, as the kernel tells an unprivileged + # caller. This host lets the suite make one (the preconditions), so EPERM + # here is the filter's, and exit 7 means no namespace was made. + nested=0 + for p in unshare-newuser clone-newuser; do + "$KRYPTIKD" seccomp-test "$p" >/dev/null 2>&1 + rc=$? + if [[ "$rc" -ne 7 ]]; then + echo " ${p} was NOT refused with EPERM (rc=${rc})" + nested=$((nested + 1)) + fi + done + + if [[ "$nested" -eq 0 ]]; then + pass "unshare(2) and clone(CLONE_NEWUSER) fail with EPERM and make no namespace" + else + fail "${nested} namespace-creating call(s) not refused with EPERM" + fi fi # --- consistency with kryptikd ---------------------------------------------- diff --git a/compartments/tests/launcher.sh b/compartments/tests/launcher.sh index 7994867d..ee66fdf4 100755 --- a/compartments/tests/launcher.sh +++ b/compartments/tests/launcher.sh @@ -856,8 +856,8 @@ probe "I1 a zone process reports seccomp mode 2 (filtered)" "2" zrun alpha -- /bin/sh -c "$PRO echo PROBE=\$(grep '^Seccomp_filters:' /proc/self/status | awk '{print \$2}')" probe "I2 exactly one filter is installed, not zero and not a stack" "1" -# Each is in seccomp.rs::DENIED_RATIONALE. A denied syscall kills the zone -# (SIGSYS, exit 159) rather than returning an error it could ignore. +# Each is in seccomp.rs::DENIED_RATIONALE and kills the zone (SIGSYS, exit +# 159) rather than returning an error it could ignore. seccomp_kill() { # desc shell-command local desc="$1" cmd="$2" zrun alpha -- /bin/sh -c "$cmd" @@ -874,8 +874,12 @@ seccomp_kill "I3 mount(2) kills the zone (re-mount under Landlock)" \ '/bin/mount -t tmpfs none /tmp' seccomp_kill "I4 chroot(2) kills the zone (double-chroot escape)" \ '/usr/sbin/chroot / /bin/true' -seccomp_kill "I5 unshare(2) kills the zone (nested namespace LPE surface)" \ - '/usr/bin/unshare -U /bin/true' +# unshare(2) is denied too, but fails with EPERM: Firefox, Chromium and +# bubblewrap probe for user namespaces and must hear no, as the kernel tells an +# unprivileged caller. unshare(1) lives to say so, and readlink never runs in a +# namespace of its own. +zrun alpha -- /bin/sh -c "$PRO o=\$(LC_ALL=C /usr/bin/unshare -U readlink /proc/self/ns/user 2>&1); rc=\$?; case \"\$rc:\$o\" in 159:*) echo PROBE=SIGSYS;; 0:user:*) echo PROBE=CREATED;; 1:*'Operation not permitted'*) echo PROBE=eperm;; *) echo \"PROBE=\$rc:\$o\";; esac" +probe "I5 unshare(2) fails with EPERM, not SIGSYS, and makes no namespace" "eperm" # mknod(2) is allowed so mkfifo works; a device node is refused by Landlock # (MAKE_CHAR/MAKE_BLOCK granted nowhere) and by nodev on every mount. zrun alpha -- /bin/sh -c "$PRO /usr/bin/mknod /tmp/n c 1 3 2>/dev/null; if [ -e /tmp/n ]; then echo PROBE=CREATED; else echo PROBE=refused; fi" @@ -1236,7 +1240,7 @@ filter_probe() { # desc probe expected fi } -filter_probe "L3 clone(CLONE_NEWUSER) is killed (nested user namespace)" clone-newuser 5 +filter_probe "L3 clone(CLONE_NEWUSER) fails with EPERM rather than killing (no nested user namespace)" clone-newuser 7 filter_probe "L4 clone3 returns ENOSYS rather than killing (glibc falls back)" clone3 7 filter_probe "L5 socket(AF_VSOCK) is refused with an errno" socket-vsock 7 filter_probe "L6 socket(AF_NETLINK/NETFILTER) is refused with an errno" socket-netlink-nf 7 @@ -1256,6 +1260,19 @@ else fail "L9 seccomp-trace did not report reboot(2) and inotify_init1(2) [$(tr '\n' ' ' <<<"$out")]" fi +# clone(CLONE_NEWUSER) and unshare(2) are named too, marked soft, and fail with +# EPERM (1) as in a zone. A kernel may answer EPERM as well, so the names are +# what show that the filter refuses them. +out="$(timeout "$TIMEOUT" "$KRYPTIKD" seccomp-trace -- python3 -c \ + 'import ctypes; c = ctypes.CDLL(None, use_errno=True); [print(c.syscall(*a), ctypes.get_errno()) for a in ((56, 0x10000011, 0, 0, 0, 0), (272, 0x10000000))]' 2>&1)" +if grep -qx 'KRYPTIK_SECCOMP_DENIED 56 clone soft' <<<"$out" \ + && grep -qx 'KRYPTIK_SECCOMP_DENIED 272 unshare soft' <<<"$out" \ + && [[ "$(grep -cx -e '-1 1' <<<"$out")" == 2 ]]; then + pass "L9b seccomp-trace names a namespace clone and unshare(2) as soft refusals, and both fail with EPERM" +else + fail "L9b seccomp-trace did not refuse clone(CLONE_NEWUSER) and unshare(2) with EPERM [$(tr '\n' ' ' <<<"$out")]" +fi + # ncurses brackets each terminfo open with setfsuid and setfsgid. The filter # answers them with EPERM; a kill would take every terminal program with it # (tput would end with 159, SIGSYS). diff --git a/docs/design/zone-policy-files.md b/docs/design/zone-policy-files.md index 3fd4351c..2473d457 100755 --- a/docs/design/zone-policy-files.md +++ b/docs/design/zone-policy-files.md @@ -21,9 +21,9 @@ keep-capability CAP_NET_RAW # left in the bounding set - `allow-syscall`: a name from `seccomp::ADDABLE`. A syscall on the base denied list (`DENIED_RATIONALE`: `ptrace`, `mount`, `setns`, `bpf`, ...) cannot be re-allowed, one on the base allowlist is reported as already - allowed, and any other name is an error. The id and capability calls on - the denied list fail with EPERM instead of killing the caller; the trace - paragraph below says why. + allowed, and any other name is an error. The id and capability calls and + `unshare` on the denied list fail with EPERM instead of killing the caller; + the trace paragraph below says why. - `allow-socket`: `AF_PACKET`, `AF_KEY`, `AF_ALG`, `AF_VSOCK`, `AF_BLUETOOTH`, `AF_CAN`, `AF_RDS`, `AF_TIPC` or `AF_XDP`; `AF_NETLINK` drops the netlink protocol check. `socketpair(2)` stays `AF_UNIX` only whatever the file @@ -55,6 +55,11 @@ after its name and gets the same errno here: `setfsgid`, and `sudo`, `su` and daemons that drop privilege as root call the rest, so a kill would take them down unexplained. They stay on the denied list, and no id or capability changes either way. +- `unshare`, and `clone` with namespace flags, fail with EPERM: Firefox, + Chromium and bubblewrap probe for user namespaces at start and must hear + no, as the kernel tells an unprivileged caller. `unshare` stays on the + denied list, and `setns`, which would enter another zone's namespaces, is + killed. A printed name can go on an `allow-syscall` line unless it is on the denied list; a call refused for its arguments (namespace flags to `clone`, diff --git a/docs/hardening.md b/docs/hardening.md index 281fcc80..b73ce9a1 100644 --- a/docs/hardening.md +++ b/docs/hardening.md @@ -167,11 +167,13 @@ every sysctl reads back as `build/config/sysctl.d` says. Every zoned process runs under a default-deny seccomp-bpf filter (`compartments/kryptikd/src/seccomp.rs`) allowing about 200 syscalls; anything -else is `SECCOMP_RET_KILL_PROCESS`. `clone` with namespace flags is killed, -`clone3` fails with `ENOSYS` so libc falls back to `clone`, the `TIOCSTI` and -`TIOCLINUX` ioctls are killed, and `socket` is limited to `AF_UNIX`, -`AF_INET`, `AF_INET6` and `NETLINK_ROUTE`. A zone policy file can widen this -in named ways but never re-allow a denied syscall +else is `SECCOMP_RET_KILL_PROCESS`. `unshare` and `clone` with namespace flags +fail with `EPERM` instead: Firefox, Chromium and bubblewrap probe for user +namespaces at start and must hear no, as the kernel tells an unprivileged +caller. `clone3` fails with `ENOSYS` so libc falls back to `clone`, the +`TIOCSTI` and `TIOCLINUX` ioctls are killed, and `socket` is limited to +`AF_UNIX`, `AF_INET`, `AF_INET6` and `NETLINK_ROUTE`. A zone policy file can +widen this in named ways but never re-allow a denied syscall ([zone policy files](design/zone-policy-files.md)). | Denied | Why | @@ -186,8 +188,8 @@ in named ways but never re-allow a denied syscall | `init_module`, `finit_module`, `kexec_load` | load kernel code | | `io_uring_*` | does I/O without syscalls, past the filter | -`compartments/tests/adversarial.sh` makes 13 of these calls in a real process -and expects SIGSYS. +`compartments/tests/adversarial.sh` makes 12 of these calls in a real process +and expects SIGSYS; from `unshare` and a namespace `clone` it expects `EPERM`. Other architectures are refused, and x32 calls (x86-64 numbers with bit 30 set) are killed before the allowlist. Each allowed syscall is a compare