Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 4 additions & 3 deletions compartments/kryptikd/probes/boundary-checks.sh
Original file line number Diff line number Diff line change
Expand Up @@ -126,8 +126,9 @@ MATCH="^0 1 2 3$" check "the only descriptors are 0, 1, 2 and the lister's own"
# ---------------------------------------------------------------------------
head_ "D. Seccomp"

# unshare(1) makes the call (python here has no ctypes); 159 is 128+SIGSYS.
check "clone(CLONE_NEWUSER) from inside the zone is killed, not refused" 159 /usr/bin/unshare -U /bin/true
# unshare(1) makes the call, which fails with EPERM as the kernel tells an
# unprivileged caller: it exits 1, not 159 (SIGSYS), and never runs true.
MATCH="Operation not permitted" check "unshare(2) from inside the zone fails with EPERM, is not killed, and makes no namespace" 1 /usr/bin/unshare -U /bin/true
MATCH="AF_VSOCK refused 97" check "AF_VSOCK, AF_ALG and AF_PACKET are refused by family, with EAFNOSUPPORT" 0 /usr/bin/python3 -c "
import socket
for n,f,t in [('AF_VSOCK',40,1),('AF_ALG',38,5),('AF_PACKET',17,2)]:
Expand All @@ -140,7 +141,7 @@ import socket
a,b=socket.socketpair(); a.close(); b.close()
try: socket.socketpair(socket.AF_INET); print('unix-pair inet PAIRED')
except OSError as e: print('unix-pair inet', e.errno)"
for p in "clone-newuser 5" "clone3 7" "inotify 7" "setfsuid 7" "socket-vsock 7" "socket-netlink-nf 7" "socket-inet 0" "ioctl-tiocsti 5" "setns 5" "unshare 5" "mount 5" "getpid 0"; do
for p in "clone-newuser 7" "unshare-newuser 7" "clone3 7" "inotify 7" "setfsuid 7" "socket-vsock 7" "socket-netlink-nf 7" "socket-inet 0" "ioctl-tiocsti 5" "setns 5" "mount 5" "getpid 0"; do
set -- $p
"$K" seccomp-test "$1" >/dev/null 2>&1; rc=$?
if [[ "$rc" == "$2" ]]; then pass "seccomp-test $1 -> $rc"; else fail "seccomp-test $1 -> $rc (want $2)"; fi
Expand Down
8 changes: 6 additions & 2 deletions compartments/kryptikd/src/main.rs
Original file line number Diff line number Diff line change
Expand Up @@ -793,7 +793,11 @@ fn cmd_seccomp_probe(name: &str) -> Option<ExitCode> {
if r > 0 {
libc::waitpid(r as libc::pid_t, std::ptr::null_mut(), 0);
}
0
if r < 0 && *libc::__errno_location() == libc::EPERM { 7 } else { 0 }
},
"unshare-newuser" => || unsafe {
let r = libc::unshare(libc::CLONE_NEWUSER);
if r < 0 && *libc::__errno_location() == libc::EPERM { 7 } else { 0 }
},
"clone3" => || unsafe {
let r = libc::syscall(libc::SYS_clone3, std::ptr::null::<u8>(), 0usize);
Expand Down Expand Up @@ -1219,7 +1223,7 @@ fn cmd_seccomp_trace(cmd: &[String], allow: &[libc::c_long], sockets: &seccomp::
let nr = libc::c_long::from(req.data.nr);
let name = seccomp::name_of(nr).unwrap_or("");
// A soft refusal gets the errno a zone gets, and is marked.
let soft = seccomp::REFUSED_SOFTLY.iter().find(|(n, _)| *n == nr).map(|&(_, e)| e as libc::c_int);
let soft = seccomp::soft_errno(&req.data).map(|e| e as libc::c_int);
eprintln!("KRYPTIK_SECCOMP_DENIED {nr} {name}{}", if soft.is_some() { " soft" } else { "" });
refused += 1;
let mut resp: libc::seccomp_notif_resp = unsafe { std::mem::zeroed() };
Expand Down
36 changes: 29 additions & 7 deletions compartments/kryptikd/src/seccomp.rs
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,8 @@ const BPF_RET: u16 = 0x06;
const SECCOMP_RET_KILL_PROCESS: u32 = 0x8000_0000;
const SECCOMP_RET_ALLOW: u32 = 0x7fff_0000;
/* Fail the call instead of killing, for programs that probe for a feature and
* must hear "no": clone3, unwanted socket families, `REFUSED_SOFTLY`. */
* must hear "no": clone3, a namespace clone, unwanted socket families,
* `REFUSED_SOFTLY`. */
const SECCOMP_RET_ERRNO: u32 = 0x0005_0000;

// Offsets into struct seccomp_data.
Expand Down Expand Up @@ -373,7 +374,7 @@ denied! {
/// allowlist, so a syscall named here is decided here.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum ArgRule {
/// clone(2): kill if any CLONE_NEW* flag is set in args[0].
/// clone(2): EPERM if any CLONE_NEW* flag is set in args[0], like unshare(2).
CloneNoNamespaces,
/// clone3(2): ENOSYS, so libc falls back to clone(2), which the filter can read.
Clone3Enosys,
Expand Down Expand Up @@ -411,6 +412,9 @@ impl ArgRule {
/// any of them starts; a zone policy may allow it. The id and capability calls
/// never succeed, but ncurses brackets every terminfo open with setfsuid and
/// setfsgid, and sudo, su and privilege-dropping daemons call the rest.
/// unshare: Firefox, Chromium and bubblewrap probe for user namespaces and must
/// hear no, as Kryptik's kernel tells an unprivileged caller; a namespace clone
/// gets the same (`ArgRule::CloneNoNamespaces`).
pub const REFUSED_SOFTLY: &[(libc::c_long, u32)] = &[
(libc::SYS_inotify_init, ENOSYS),
(libc::SYS_inotify_init1, ENOSYS),
Expand All @@ -424,12 +428,33 @@ pub const REFUSED_SOFTLY: &[(libc::c_long, u32)] = &[
(libc::SYS_setresgid, EPERM),
(libc::SYS_setgroups, EPERM),
(libc::SYS_capset, EPERM),
(libc::SYS_unshare, EPERM),
];

const fn errno_action(e: u32) -> u32 {
SECCOMP_RET_ERRNO | (e & 0xffff)
}

/// A soft refusal's action. seccomp-trace is notified instead and answers with
/// the same errno, so the program runs as it would in a zone and the call is
/// still named.
const fn soft_refusal(e: u32, deny_action: u32) -> u32 {
if deny_action == libc::SECCOMP_RET_USER_NOTIF { deny_action } else { errno_action(e) }
}

/// The errno a zone gets for a call refused without a kill, None for one it is
/// killed for: what `seccomp-trace` answers a call it is notified of.
pub fn soft_errno(call: &libc::seccomp_data) -> Option<u32> {
let nr = libc::c_long::from(call.nr);
if call.arch != AUDIT_ARCH_X86_64 {
return None;
}
if nr == libc::SYS_clone && (call.args[0] as u32) & CLONE_NS_MASK != 0 {
return Some(EPERM);
}
REFUSED_SOFTLY.iter().find(|(n, _)| *n == nr).map(|&(_, e)| e)
}

/// Emit one argument rule. Entered with the syscall number in the accumulator;
/// a non-matching number skips the block with the accumulator intact, and every
/// path through a matched block ends in `ret`.
Expand All @@ -438,7 +463,7 @@ fn emit_arg_rule(p: &mut Vec<SockFilter>, rule: ArgRule, deny_action: u32, socke
ArgRule::CloneNoNamespaces => vec![
stmt(BPF_LD | BPF_W | BPF_ABS, arg_lo(0)),
jump(BPF_JMP | BPF_JSET | BPF_K, CLONE_NS_MASK, 0, 1),
stmt(BPF_RET | BPF_K, deny_action),
stmt(BPF_RET | BPF_K, soft_refusal(EPERM, deny_action)),
stmt(BPF_RET | BPF_K, SECCOMP_RET_ALLOW),
],
ArgRule::Clone3Enosys => vec![stmt(BPF_RET | BPF_K, errno_action(ENOSYS))],
Expand Down Expand Up @@ -564,11 +589,8 @@ fn build_program_full(
for &(nr, e) in REFUSED_SOFTLY {
// Allowed by the list, it is allowed below like any other.
if !allow.contains(&nr) {
/* seccomp-trace answers these with the same errno, so the program
* runs as it would in a zone and the call is still named. */
let refuse = if deny_action == libc::SECCOMP_RET_USER_NOTIF { deny_action } else { errno_action(e) };
p.push(jump(BPF_JMP | BPF_JEQ | BPF_K, nr as u32, 0, 1));
p.push(stmt(BPF_RET | BPF_K, refuse));
p.push(stmt(BPF_RET | BPF_K, soft_refusal(e, deny_action)));
}
}

Expand Down
61 changes: 57 additions & 4 deletions compartments/kryptikd/src/seccomp/tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -269,7 +269,8 @@ fn clone_without_namespace_flags_is_allowed() {
}

#[test]
fn clone_with_any_namespace_flag_is_killed() {
fn namespace_clone_fails_with_eperm() {
// As unshare(2) does: a program probing for user namespaces lives on.
let p = build_program(BASE_ALLOWLIST).unwrap();
for f in [
libc::CLONE_NEWUSER, libc::CLONE_NEWNS, libc::CLONE_NEWPID,
Expand All @@ -279,17 +280,69 @@ fn clone_with_any_namespace_flag_is_killed() {
let flags = (f as u32 | libc::SIGCHLD as u32) as u64;
assert_eq!(
evaluate_args(&p, X86, libc::SYS_clone as u32, with_arg(0, flags)),
SECCOMP_RET_KILL_PROCESS,
"clone flags {flags:#x} must be killed"
errno_action(EPERM),
"clone flags {flags:#x} must fail with EPERM"
);
// Legacy clone ignores the high word, and so does the filter.
assert_eq!(
evaluate_args(&p, X86, libc::SYS_clone as u32, with_arg(0, flags | (1 << 40))),
SECCOMP_RET_KILL_PROCESS
errno_action(EPERM)
);
}
}

#[test]
fn unshare_fails_setns_killed() {
// Both stay on the denied list: no policy may allow either.
let p = build_program(BASE_ALLOWLIST).unwrap();
for f in [0, libc::CLONE_NEWUSER as u64, CLONE_NS_MASK as u64] {
let r = evaluate_args(&p, X86, libc::SYS_unshare as u32, with_arg(0, f));
assert_eq!(r, errno_action(EPERM), "unshare flags {f:#x}");
}
let r = evaluate_args(&p, X86, libc::SYS_setns as u32, with_arg(1, libc::CLONE_NEWUSER as u64));
assert_eq!(r, SECCOMP_RET_KILL_PROCESS);
for nr in [libc::SYS_unshare, libc::SYS_setns] {
assert!(is_denied(nr) && widened(&[nr]).is_err(), "syscall {nr} can be allowed");
}
}

#[test]
fn trace_names_namespace_refusals() {
// Named like any soft refusal, while a plain clone runs.
let p = build_program_with(BASE_ALLOWLIST, libc::SECCOMP_RET_USER_NOTIF).unwrap();
let ns = with_arg(0, (libc::CLONE_NEWUSER | libc::SIGCHLD) as u64);
assert_eq!(evaluate_args(&p, X86, libc::SYS_clone as u32, ns), libc::SECCOMP_RET_USER_NOTIF);
assert_eq!(evaluate_args(&p, X86, libc::SYS_unshare as u32, ns), libc::SECCOMP_RET_USER_NOTIF);
let fork = with_arg(0, libc::SIGCHLD as u64);
assert_eq!(evaluate_args(&p, X86, libc::SYS_clone as u32, fork), SECCOMP_RET_ALLOW);
}

#[test]
fn trace_answers_as_a_zone() {
/* seccomp-trace hears of every call the zone filter refuses, and answers
* each as a zone hears it: a soft refusal's errno, or ENOSYS where a zone
* is killed. */
let zone = build_program(BASE_ALLOWLIST).unwrap();
let trace = build_program_with(BASE_ALLOWLIST, libc::SECCOMP_RET_USER_NOTIF).unwrap();
let ns = with_arg(0, (libc::CLONE_NEWUSER | libc::SIGCHLD) as u64);
let mut calls: Vec<(libc::c_long, [u64; 6])> = DENIED_RATIONALE
.iter()
.map(|&(nr, _)| nr)
.chain(REFUSED_SOFTLY.iter().map(|&(nr, _)| nr))
.map(|nr| (nr, [0; 6]))
.collect();
calls.extend([(libc::SYS_clone, ns), (libc::SYS_unshare, ns), (libc::SYS_ioctl, with_arg(1, TIOCSTI as u64))]);
for (nr, args) in calls {
assert_eq!(evaluate_args(&trace, X86, nr as u32, args), libc::SECCOMP_RET_USER_NOTIF, "syscall {nr} is not named");
let call = libc::seccomp_data { nr: nr as i32, arch: X86, instruction_pointer: 0, args };
let heard = soft_errno(&call).map_or(SECCOMP_RET_KILL_PROCESS, errno_action);
assert_eq!(evaluate_args(&zone, X86, nr as u32, args), heard, "syscall {nr}");
}
// Another architecture is killed in a zone whatever the number.
let i386 = libc::seccomp_data { nr: libc::SYS_unshare as i32, arch: 0x4000_0003, instruction_pointer: 0, args: ns };
assert_eq!(soft_errno(&i386), None);
}

#[test]
fn clone3_gets_enosys() {
let p = build_program(BASE_ALLOWLIST).unwrap();
Expand Down
24 changes: 22 additions & 2 deletions compartments/tests/adversarial.sh
Original file line number Diff line number Diff line change
Expand Up @@ -290,7 +290,7 @@ else

# setns alone would step into another zone's namespaces, defeating 1-4.
leaked=0
for sc in setns ptrace unshare mount bpf perf_event_open userfaultfd \
for sc in setns ptrace mount bpf perf_event_open userfaultfd \
keyctl init_module kexec_load process_vm_readv pivot_root chroot; do
"$KRYPTIKD" seccomp-test "$sc" >/dev/null 2>&1
rc=$?
Expand All @@ -301,10 +301,30 @@ else
done

if [[ "$leaked" -eq 0 ]]; then
pass "all 13 dangerous syscalls killed by SIGSYS (setns among them)"
pass "all 12 dangerous syscalls killed by SIGSYS (setns among them)"
else
fail "${leaked} dangerous syscall(s) reachable from inside a zone"
fi

# unshare(2) and a namespace clone fail with EPERM instead: programs probe
# for user namespaces and must hear no, as the kernel tells an unprivileged
# caller. This host lets the suite make one (the preconditions), so EPERM
# here is the filter's, and exit 7 means no namespace was made.
nested=0
for p in unshare-newuser clone-newuser; do
"$KRYPTIKD" seccomp-test "$p" >/dev/null 2>&1
rc=$?
if [[ "$rc" -ne 7 ]]; then
echo " ${p} was NOT refused with EPERM (rc=${rc})"
nested=$((nested + 1))
fi
done

if [[ "$nested" -eq 0 ]]; then
pass "unshare(2) and clone(CLONE_NEWUSER) fail with EPERM and make no namespace"
else
fail "${nested} namespace-creating call(s) not refused with EPERM"
fi
fi

# --- consistency with kryptikd ----------------------------------------------
Expand Down
27 changes: 22 additions & 5 deletions compartments/tests/launcher.sh
Original file line number Diff line number Diff line change
Expand Up @@ -856,8 +856,8 @@ probe "I1 a zone process reports seccomp mode 2 (filtered)" "2"
zrun alpha -- /bin/sh -c "$PRO echo PROBE=\$(grep '^Seccomp_filters:' /proc/self/status | awk '{print \$2}')"
probe "I2 exactly one filter is installed, not zero and not a stack" "1"

# Each is in seccomp.rs::DENIED_RATIONALE. A denied syscall kills the zone
# (SIGSYS, exit 159) rather than returning an error it could ignore.
# Each is in seccomp.rs::DENIED_RATIONALE and kills the zone (SIGSYS, exit
# 159) rather than returning an error it could ignore.
seccomp_kill() { # desc shell-command
local desc="$1" cmd="$2"
zrun alpha -- /bin/sh -c "$cmd"
Expand All @@ -874,8 +874,12 @@ seccomp_kill "I3 mount(2) kills the zone (re-mount under Landlock)" \
'/bin/mount -t tmpfs none /tmp'
seccomp_kill "I4 chroot(2) kills the zone (double-chroot escape)" \
'/usr/sbin/chroot / /bin/true'
seccomp_kill "I5 unshare(2) kills the zone (nested namespace LPE surface)" \
'/usr/bin/unshare -U /bin/true'
# unshare(2) is denied too, but fails with EPERM: Firefox, Chromium and
# bubblewrap probe for user namespaces and must hear no, as the kernel tells an
# unprivileged caller. unshare(1) lives to say so, and readlink never runs in a
# namespace of its own.
zrun alpha -- /bin/sh -c "$PRO o=\$(LC_ALL=C /usr/bin/unshare -U readlink /proc/self/ns/user 2>&1); rc=\$?; case \"\$rc:\$o\" in 159:*) echo PROBE=SIGSYS;; 0:user:*) echo PROBE=CREATED;; 1:*'Operation not permitted'*) echo PROBE=eperm;; *) echo \"PROBE=\$rc:\$o\";; esac"
probe "I5 unshare(2) fails with EPERM, not SIGSYS, and makes no namespace" "eperm"
# mknod(2) is allowed so mkfifo works; a device node is refused by Landlock
# (MAKE_CHAR/MAKE_BLOCK granted nowhere) and by nodev on every mount.
zrun alpha -- /bin/sh -c "$PRO /usr/bin/mknod /tmp/n c 1 3 2>/dev/null; if [ -e /tmp/n ]; then echo PROBE=CREATED; else echo PROBE=refused; fi"
Expand Down Expand Up @@ -1236,7 +1240,7 @@ filter_probe() { # desc probe expected
fi
}

filter_probe "L3 clone(CLONE_NEWUSER) is killed (nested user namespace)" clone-newuser 5
filter_probe "L3 clone(CLONE_NEWUSER) fails with EPERM rather than killing (no nested user namespace)" clone-newuser 7
filter_probe "L4 clone3 returns ENOSYS rather than killing (glibc falls back)" clone3 7
filter_probe "L5 socket(AF_VSOCK) is refused with an errno" socket-vsock 7
filter_probe "L6 socket(AF_NETLINK/NETFILTER) is refused with an errno" socket-netlink-nf 7
Expand All @@ -1256,6 +1260,19 @@ else
fail "L9 seccomp-trace did not report reboot(2) and inotify_init1(2) [$(tr '\n' ' ' <<<"$out")]"
fi

# clone(CLONE_NEWUSER) and unshare(2) are named too, marked soft, and fail with
# EPERM (1) as in a zone. A kernel may answer EPERM as well, so the names are
# what show that the filter refuses them.
out="$(timeout "$TIMEOUT" "$KRYPTIKD" seccomp-trace -- python3 -c \
'import ctypes; c = ctypes.CDLL(None, use_errno=True); [print(c.syscall(*a), ctypes.get_errno()) for a in ((56, 0x10000011, 0, 0, 0, 0), (272, 0x10000000))]' 2>&1)"
if grep -qx 'KRYPTIK_SECCOMP_DENIED 56 clone soft' <<<"$out" \
&& grep -qx 'KRYPTIK_SECCOMP_DENIED 272 unshare soft' <<<"$out" \
&& [[ "$(grep -cx -e '-1 1' <<<"$out")" == 2 ]]; then
pass "L9b seccomp-trace names a namespace clone and unshare(2) as soft refusals, and both fail with EPERM"
else
fail "L9b seccomp-trace did not refuse clone(CLONE_NEWUSER) and unshare(2) with EPERM [$(tr '\n' ' ' <<<"$out")]"
fi

# ncurses brackets each terminfo open with setfsuid and setfsgid. The filter
# answers them with EPERM; a kill would take every terminal program with it
# (tput would end with 159, SIGSYS).
Expand Down
11 changes: 8 additions & 3 deletions docs/design/zone-policy-files.md
Original file line number Diff line number Diff line change
Expand Up @@ -21,9 +21,9 @@ keep-capability CAP_NET_RAW # left in the bounding set
- `allow-syscall`: a name from `seccomp::ADDABLE`. A syscall on the base
denied list (`DENIED_RATIONALE`: `ptrace`, `mount`, `setns`, `bpf`, ...)
cannot be re-allowed, one on the base allowlist is reported as already
allowed, and any other name is an error. The id and capability calls on
the denied list fail with EPERM instead of killing the caller; the trace
paragraph below says why.
allowed, and any other name is an error. The id and capability calls and
`unshare` on the denied list fail with EPERM instead of killing the caller;
the trace paragraph below says why.
- `allow-socket`: `AF_PACKET`, `AF_KEY`, `AF_ALG`, `AF_VSOCK`, `AF_BLUETOOTH`,
`AF_CAN`, `AF_RDS`, `AF_TIPC` or `AF_XDP`; `AF_NETLINK` drops the netlink
protocol check. `socketpair(2)` stays `AF_UNIX` only whatever the file
Expand Down Expand Up @@ -55,6 +55,11 @@ after its name and gets the same errno here:
`setfsgid`, and `sudo`, `su` and daemons that drop privilege as root call
the rest, so a kill would take them down unexplained. They stay on the denied
list, and no id or capability changes either way.
- `unshare`, and `clone` with namespace flags, fail with EPERM: Firefox,
Chromium and bubblewrap probe for user namespaces at start and must hear
no, as the kernel tells an unprivileged caller. `unshare` stays on the
denied list, and `setns`, which would enter another zone's namespaces, is
killed.

A printed name can go on an `allow-syscall` line unless it is on the denied
list; a call refused for its arguments (namespace flags to `clone`,
Expand Down
16 changes: 9 additions & 7 deletions docs/hardening.md
Original file line number Diff line number Diff line change
Expand Up @@ -167,11 +167,13 @@ every sysctl reads back as `build/config/sysctl.d` says.

Every zoned process runs under a default-deny seccomp-bpf filter
(`compartments/kryptikd/src/seccomp.rs`) allowing about 200 syscalls; anything
else is `SECCOMP_RET_KILL_PROCESS`. `clone` with namespace flags is killed,
`clone3` fails with `ENOSYS` so libc falls back to `clone`, the `TIOCSTI` and
`TIOCLINUX` ioctls are killed, and `socket` is limited to `AF_UNIX`,
`AF_INET`, `AF_INET6` and `NETLINK_ROUTE`. A zone policy file can widen this
in named ways but never re-allow a denied syscall
else is `SECCOMP_RET_KILL_PROCESS`. `unshare` and `clone` with namespace flags
fail with `EPERM` instead: Firefox, Chromium and bubblewrap probe for user
namespaces at start and must hear no, as the kernel tells an unprivileged
caller. `clone3` fails with `ENOSYS` so libc falls back to `clone`, the
`TIOCSTI` and `TIOCLINUX` ioctls are killed, and `socket` is limited to
`AF_UNIX`, `AF_INET`, `AF_INET6` and `NETLINK_ROUTE`. A zone policy file can
widen this in named ways but never re-allow a denied syscall
([zone policy files](design/zone-policy-files.md)).

| Denied | Why |
Expand All @@ -186,8 +188,8 @@ in named ways but never re-allow a denied syscall
| `init_module`, `finit_module`, `kexec_load` | load kernel code |
| `io_uring_*` | does I/O without syscalls, past the filter |

`compartments/tests/adversarial.sh` makes 13 of these calls in a real process
and expects SIGSYS.
`compartments/tests/adversarial.sh` makes 12 of these calls in a real process
and expects SIGSYS; from `unshare` and a namespace `clone` it expects `EPERM`.

Other architectures are refused, and x32 calls (x86-64 numbers with bit 30
set) are killed before the allowlist. Each allowed syscall is a compare
Expand Down
Loading