oxedyne/daimond/hand/src/seccomp.rs
95.8 KiB, 1 run
created by r2519314175:925, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! The half of the compartment Landlock cannot express: which system calls the |
| 2 | //! command may make at all. |
| 3 | //! |
| 4 | //! [`crate::fence`] hands the kernel a set of paths and the access permitted at |
| 5 | //! each. That is the whole of Landlock, and it is genuinely airtight for what |
| 6 | //! it covers -- but it covers *opening a file*, and a process does more to a |
| 7 | //! file than open it. Two measured escapes follow directly from that, and |
| 8 | //! neither can be closed by any rule `fence.rs` could add: |
| 9 | //! |
| 10 | //! * **Metadata is ungoverned.** Landlock's `AccessFs` has no right covering |
| 11 | //! `chmod`, `chown`, `utimensat` or `setxattr`, so none of them is mediated. |
| 12 | //! Under a full ABI-8 fence all four succeeded on a file outside every root, |
| 13 | //! and `chmod 777` took a file *inside the denied subtree* from 600 to 777. |
| 14 | //! The fence stops a command reading a secret; it does not stop the command |
| 15 | //! stripping the permissions that were protecting that secret from everybody |
| 16 | //! else. |
| 17 | //! * **The session bus is a way out.** Landlock does not govern `connect()` to |
| 18 | //! a pathname unix socket until ABI 9 (Linux 7.1), and this kernel is at |
| 19 | //! ABI 8. With `net:false` fully applied, `connect()` to `/run/user/1000/bus` |
| 20 | //! succeeded and `systemd-run --user … /bin/cat <denied-file>` ran **outside |
| 21 | //! the fence** and returned the contents. That is arbitrary unfenced |
| 22 | //! execution, not a leak at the edge. |
| 23 | //! |
| 24 | //! Seccomp closes both, because both go through a system call whose *number*, or |
| 25 | //! whose *integer argument*, is enough to decide on. That last clause is the |
| 26 | //! whole shape of this module and its whole limitation: **a seccomp filter |
| 27 | //! cannot dereference a pointer.** The kernel copies the filter's view of a |
| 28 | //! call from `struct seccomp_data`, which holds the syscall number, the |
| 29 | //! instruction pointer and the six argument registers -- and nothing they point |
| 30 | //! at. There is no path here, no `sockaddr`, no mode string. Every decision |
| 31 | //! below is therefore all-or-nothing across the whole filesystem, or it is made |
| 32 | //! on a plain integer. |
| 33 | //! |
| 34 | //! # Default-allow, and what that does not buy |
| 35 | //! |
| 36 | //! This filter names what is refused and permits everything else. The opposite |
| 37 | //! -- name what is permitted and refuse everything else -- is strictly stronger, |
| 38 | //! and it is the wrong choice here. A default-deny filter for "any command a |
| 39 | //! build might run" is a filter for `cargo`, `rustc`, `ld`, `node`, `python`, |
| 40 | //! `git`, every build script and every test binary, and the first syscall it |
| 41 | //! missed would be an incomprehensible `SIGSYS` in the middle of somebody's |
| 42 | //! build. A filter that breaks builds is a filter that gets switched off, and a |
| 43 | //! filter that is switched off protects nothing. |
| 44 | //! |
| 45 | //! So the honest statement of what this achieves: **it removes named capabilities |
| 46 | //! from a command that is otherwise unrestricted at the syscall layer.** It is |
| 47 | //! not a syscall sandbox. A kernel bug reachable through an unnamed syscall is |
| 48 | //! reachable through this filter; a syscall added by a future kernel is permitted |
| 49 | //! by default; and anything the deny-list did not think of is allowed. What it |
| 50 | //! *does* deliver is that the two escapes above, and the third one below, are |
| 51 | //! measurably gone -- and [`Seccomp::holes`] lists the rest. |
| 52 | //! |
| 53 | //! # Order: Landlock first, seccomp last, then `execve` |
| 54 | //! |
| 55 | //! Both restrictions are irreversible and both are inherited across `execve`, so |
| 56 | //! the order is not about undoing anything -- it is about what each still needs |
| 57 | //! to do after it is installed. |
| 58 | //! |
| 59 | //! Landlock's application **opens every granted path** (`PathFd`), so it has real |
| 60 | //! work left when its own rules take hold. A seccomp filter installed before it |
| 61 | //! would sit underneath that work, and a deny-list that happened to name |
| 62 | //! something the `landlock` crate needed would break the fence rather than the |
| 63 | //! command -- the wrong failure, in the wrong layer, for a reason nobody could |
| 64 | //! read. This filter needs nothing after itself except `execve`. So: fence, |
| 65 | //! then filter, then exec. |
| 66 | //! |
| 67 | //! Both also require `no_new_privs`, and neither can be installed without it. |
| 68 | //! `fence.rs` already sets it and hard-errors when the kernel refuses -- and that |
| 69 | //! check is load-bearing here too, because the kernel rejects |
| 70 | //! `SECCOMP_SET_MODE_FILTER` outright for a process without `no_new_privs` and |
| 71 | //! without `CAP_SYS_ADMIN`. Setting it twice is idempotent, so the two layers do |
| 72 | //! not interfere; [`Filter::apply`] sets it again rather than assuming. |
| 73 | //! |
| 74 | //! # Fails closed |
| 75 | //! |
| 76 | //! There is no waiver here and no degraded mode. If the architecture is not one |
| 77 | //! this module has a syscall table for, if the filter cannot be compiled, or if |
| 78 | //! the kernel refuses to install it, [`Filter::apply`] returns an error and the |
| 79 | //! launcher must die before `execve`. A weaker filter is never installed and |
| 80 | //! never reported as success -- the failure this whole file exists to avoid is a |
| 81 | //! compartment that says it is there when it is not. |
| 82 | |
| 83 | use oxedyne_fe2o3_core::prelude::*; |
| 84 | |
| 85 | use std::collections::BTreeMap; |
| 86 | |
| 87 | #[cfg(target_os = "linux")] |
| 88 | use seccompiler::{ |
| 89 | BpfProgram, |
| 90 | SeccompAction, |
| 91 | SeccompCmpArgLen, |
| 92 | SeccompCmpOp, |
| 93 | SeccompCondition, |
| 94 | SeccompFilter, |
| 95 | SeccompRule, |
| 96 | TargetArch, |
| 97 | }; |
| 98 | |
| 99 | // ┌───────────────────────────────────────────────────────────────┐ |
| 100 | // │ Constants of the Linux ABI │ |
| 101 | // └───────────────────────────────────────────────────────────────┘ |
| 102 | |
| 103 | /// `EPERM`, the answer a refused call returns. |
| 104 | /// |
| 105 | /// Written out rather than taken from `libc`, so that this module needs exactly |
| 106 | /// one line added to `Cargo.toml` rather than two. `EPERM` is 1 on every Linux |
| 107 | /// architecture without exception, and this module refuses to run on an |
| 108 | /// architecture it does not have a table for anyway -- see [`Arch`]. |
| 109 | /// |
| 110 | /// `EPERM` rather than `ENOSYS` or a quiet success, because a refusal a program |
| 111 | /// can read is worth more than one it cannot. `chmod: Operation not permitted` |
| 112 | /// tells whoever is looking at the build log what happened; a silent success |
| 113 | /// would leave a script that is not executable and an `exec` failing two steps |
| 114 | /// later for no visible reason. |
| 115 | const EPERM: u32 = 1; |
| 116 | |
| 117 | /// `AF_UNIX`, the value of `socket(2)`'s first argument this filter matches on. |
| 118 | /// |
| 119 | /// 1 on every Linux architecture. The kernel reads that argument as an `int`, |
| 120 | /// so the comparison below is a 32-bit one; see [`Filter::compile`] for why that |
| 121 | /// matters rather than being a detail. |
| 122 | const AF_UNIX: u64 = 1; |
| 123 | |
| 124 | /// The mode bits a refused `chmod` is refused for, under [`Meta::NoLoosening`]. |
| 125 | /// |
| 126 | /// Three bits, each of which *adds* reach to somebody who did not have it: |
| 127 | /// |
| 128 | /// * `S_ISUID` (`0o4000`) and `S_ISGID` (`0o2000`) -- a program that runs as |
| 129 | /// somebody else. |
| 130 | /// * `S_IWOTH` (`0o0002`) -- world-writable, which is the `chmod 777` of the |
| 131 | /// review's reproduction and the shape of "world-write the home directory". |
| 132 | /// |
| 133 | /// What is deliberately *not* here, and why, is at [`Meta::NoLoosening`]. The |
| 134 | /// bits are separate rather than one mask because seccomp's only masking |
| 135 | /// comparison is `(arg & mask) == value`, which cannot express "any of these |
| 136 | /// bits is set" in one condition -- so each bit becomes a rule of its own, and |
| 137 | /// the rules are or-bound. |
| 138 | const LOOSENING: [(u64, &str); 3] = [ |
| 139 | (0o4000, "set-user-ID"), |
| 140 | (0o2000, "set-group-ID"), |
| 141 | (0o0002, "world-writable"), |
| 142 | ]; |
| 143 | |
| 144 | // ┌───────────────────────────────────────────────────────────────┐ |
| 145 | // │ The architecture, and the syscall table │ |
| 146 | // └───────────────────────────────────────────────────────────────┘ |
| 147 | |
| 148 | /// Which syscall numbering this build is filtering. |
| 149 | /// |
| 150 | /// An enum with two arms and no fallback, because a syscall filter built from |
| 151 | /// the wrong table is worse than no filter: it would refuse whichever calls |
| 152 | /// happened to share those numbers and permit the ones it meant to refuse, and |
| 153 | /// it would report success either way. An architecture not listed here gets |
| 154 | /// [`Seccomp::None`] and a refusal, not a guess. |
| 155 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 156 | pub enum Arch { |
| 157 | /// x86-64, whose table is its own. |
| 158 | X86_64, |
| 159 | /// aarch64, which uses the generic table in `asm-generic/unistd.h` and |
| 160 | /// therefore has no `chmod`, `chown`, `lchown`, `utime`, `utimes` or |
| 161 | /// `futimesat` at all -- those are the legacy names, and the generic ABI |
| 162 | /// dropped them in favour of the `*at` forms. |
| 163 | Aarch64, |
| 164 | } |
| 165 | |
| 166 | impl Arch { |
| 167 | |
| 168 | /// This build's architecture, or `None` if there is no table for it. |
| 169 | pub fn here() -> Option<Self> { |
| 170 | if cfg!(target_arch = "x86_64") { |
| 171 | Some(Self::X86_64) |
| 172 | } else if cfg!(target_arch = "aarch64") { |
| 173 | Some(Self::Aarch64) |
| 174 | } else { |
| 175 | None |
| 176 | } |
| 177 | } |
| 178 | |
| 179 | /// The name the capability strings use. |
| 180 | pub fn name(&self) -> &'static str { |
| 181 | match self { |
| 182 | Self::X86_64 => "x86_64", |
| 183 | Self::Aarch64 => "aarch64", |
| 184 | } |
| 185 | } |
| 186 | } |
| 187 | |
| 188 | /// What a refused call belongs to, so the report can name a category rather |
| 189 | /// than twenty numbers. |
| 190 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 191 | pub enum Group { |
| 192 | /// The `chmod` family: permission bits. |
| 193 | Mode, |
| 194 | /// The `chown` family: owner and group. |
| 195 | Owner, |
| 196 | /// The `utime` family: access and modification timestamps. |
| 197 | Times, |
| 198 | /// The `setxattr` and `removexattr` families: extended attributes. |
| 199 | Xattr, |
| 200 | /// `file_setattr` and `open_tree_attr`: the inode attribute calls added in |
| 201 | /// Linux 6.15 and 6.16, which reach the immutable and append-only flags. |
| 202 | Attr, |
| 203 | /// `socket(2)`, matched on its address family. |
| 204 | Socket, |
| 205 | /// `io_uring_setup`, which is a way of making syscalls that this filter |
| 206 | /// cannot see. |
| 207 | Ring, |
| 208 | /// `ptrace` and the `process_vm_*` pair: reading and writing another |
| 209 | /// process's memory. |
| 210 | Poke, |
| 211 | } |
| 212 | |
| 213 | impl Group { |
| 214 | |
| 215 | /// The word the capability strings use. |
| 216 | pub fn word(&self) -> &'static str { |
| 217 | match self { |
| 218 | Self::Mode => "chmod", |
| 219 | Self::Owner => "chown", |
| 220 | Self::Times => "times", |
| 221 | Self::Xattr => "xattr", |
| 222 | Self::Attr => "fileattr", |
| 223 | Self::Socket => "af-unix", |
| 224 | Self::Ring => "io-uring", |
| 225 | Self::Poke => "ptrace", |
| 226 | } |
| 227 | } |
| 228 | } |
| 229 | |
| 230 | /// One system call this filter has an opinion about. |
| 231 | /// |
| 232 | /// The numbers are written out per architecture rather than taken from |
| 233 | /// `libc::SYS_*`, and that is not stubbornness. `libc 0.2.189` on |
| 234 | /// `x86_64-unknown-linux-gnu` has no constant for `setxattrat`, `removexattrat`, |
| 235 | /// `open_tree_attr` or `file_setattr` -- four calls that exist on this kernel and |
| 236 | /// three of which change file metadata. A table built from what `libc` happens |
| 237 | /// to know would have four holes in it and would not say so. The numbers here |
| 238 | /// are cross-checked against the kernel's own headers by |
| 239 | /// [`tests::the_table_agrees_with_the_kernel_headers`]. |
| 240 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 241 | pub struct Sys { |
| 242 | /// The name, for the report and for the header cross-check. |
| 243 | pub name: &'static str, |
| 244 | /// What it belongs to. |
| 245 | pub group: Group, |
| 246 | /// Its number on x86-64, where it exists there. |
| 247 | pub x86: Option<i64>, |
| 248 | /// Its number on the generic ABI that aarch64 uses, where it exists there. |
| 249 | pub arm: Option<i64>, |
| 250 | /// Which argument carries the mode, for the calls [`Meta::NoLoosening`] |
| 251 | /// inspects. `None` for everything else. |
| 252 | pub mode: Option<u8>, |
| 253 | } |
| 254 | |
| 255 | impl Sys { |
| 256 | |
| 257 | /// This call's number on `arch`, or `None` where the architecture has no |
| 258 | /// such call. |
| 259 | /// |
| 260 | /// # Arguments |
| 261 | /// * `arch` - The numbering to look it up in. |
| 262 | pub fn number(&self, arch: Arch) -> Option<i64> { |
| 263 | match arch { |
| 264 | Arch::X86_64 => self.x86, |
| 265 | Arch::Aarch64 => self.arm, |
| 266 | } |
| 267 | } |
| 268 | } |
| 269 | |
| 270 | /// Every call this module can refuse. |
| 271 | /// |
| 272 | /// Ordered by group so a reader can check a group is complete, which is the |
| 273 | /// review's actual finding: the gap was never one missing number, it was a whole |
| 274 | /// category nobody had handled. |
| 275 | const TABLE: &[Sys] = &[ |
| 276 | // The chmod family. `fchmodat2` arrived in Linux 6.6 and is the one most |
| 277 | // likely to be missed, because it is the newest and because glibc still |
| 278 | // routes `chmod` through `fchmodat` on most paths. |
| 279 | Sys { name: "chmod", group: Group::Mode, x86: Some(90), arm: None, mode: Some(1) }, |
| 280 | Sys { name: "fchmod", group: Group::Mode, x86: Some(91), arm: Some(52), mode: Some(1) }, |
| 281 | Sys { name: "fchmodat", group: Group::Mode, x86: Some(268), arm: Some(53), mode: Some(2) }, |
| 282 | Sys { name: "fchmodat2", group: Group::Mode, x86: Some(452), arm: Some(452), mode: Some(2) }, |
| 283 | |
| 284 | // The chown family. Changing owner needs CAP_CHOWN, so the reachable half is |
| 285 | // changing *group* to one the user belongs to -- which on a machine with |
| 286 | // shared groups is exactly how a file becomes readable by a colleague. |
| 287 | Sys { name: "chown", group: Group::Owner, x86: Some(92), arm: None, mode: None }, |
| 288 | Sys { name: "fchown", group: Group::Owner, x86: Some(93), arm: Some(55), mode: None }, |
| 289 | Sys { name: "lchown", group: Group::Owner, x86: Some(94), arm: None, mode: None }, |
| 290 | Sys { name: "fchownat", group: Group::Owner, x86: Some(260), arm: Some(54), mode: None }, |
| 291 | |
| 292 | // The utime family. Refusing these breaks `cargo` -- see `Meta::Refuse`. |
| 293 | Sys { name: "utime", group: Group::Times, x86: Some(132), arm: None, mode: None }, |
| 294 | Sys { name: "utimes", group: Group::Times, x86: Some(235), arm: None, mode: None }, |
| 295 | Sys { name: "futimesat", group: Group::Times, x86: Some(261), arm: None, mode: None }, |
| 296 | Sys { name: "utimensat", group: Group::Times, x86: Some(280), arm: Some(88), mode: None }, |
| 297 | |
| 298 | // Extended attributes. `setxattrat` and `removexattrat` arrived in Linux |
| 299 | // 6.13 and `libc` has no constants for either; leaving them out would leave |
| 300 | // the group open through its newest members while the report claimed it was |
| 301 | // shut. |
| 302 | Sys { name: "setxattr", group: Group::Xattr, x86: Some(188), arm: Some(5), mode: None }, |
| 303 | Sys { name: "lsetxattr", group: Group::Xattr, x86: Some(189), arm: Some(6), mode: None }, |
| 304 | Sys { name: "fsetxattr", group: Group::Xattr, x86: Some(190), arm: Some(7), mode: None }, |
| 305 | Sys { name: "setxattrat", group: Group::Xattr, x86: Some(463), arm: Some(463), mode: None }, |
| 306 | Sys { name: "removexattr", group: Group::Xattr, x86: Some(197), arm: Some(14), mode: None }, |
| 307 | Sys { name: "lremovexattr", group: Group::Xattr, x86: Some(198), arm: Some(15), mode: None }, |
| 308 | Sys { name: "fremovexattr", group: Group::Xattr, x86: Some(199), arm: Some(16), mode: None }, |
| 309 | Sys { name: "removexattrat", group: Group::Xattr, x86: Some(466), arm: Some(466), mode: None }, |
| 310 | |
| 311 | // The inode attribute calls. `file_setattr` reaches the immutable and |
| 312 | // append-only flags, which are metadata by any reading of the word and are |
| 313 | // newer than every list of "the metadata syscalls" written before 2026. |
| 314 | Sys { name: "open_tree_attr", group: Group::Attr, x86: Some(467), arm: Some(467), mode: None }, |
| 315 | Sys { name: "file_setattr", group: Group::Attr, x86: Some(469), arm: Some(469), mode: None }, |
| 316 | |
| 317 | // The socket, matched on its family rather than refused outright. |
| 318 | Sys { name: "socket", group: Group::Socket, x86: Some(41), arm: Some(198), mode: None }, |
| 319 | |
| 320 | // io_uring. Not metadata, and here for a sharper reason: an io_uring ring |
| 321 | // performs operations -- including `IORING_OP_SETXATTR` and |
| 322 | // `IORING_OP_FSETXATTR` since Linux 5.19 -- *without issuing the syscall*, so |
| 323 | // every rule above is bypassable by a program willing to use a ring. Landlock |
| 324 | // still applies, because io_uring goes through the same VFS path; seccomp does |
| 325 | // not. Refusing the ring is the only way to make the rest of this file mean |
| 326 | // what it says. |
| 327 | Sys { name: "io_uring_setup", group: Group::Ring, x86: Some(425), arm: Some(425), mode: None }, |
| 328 | |
| 329 | // Reading and writing another process's memory. Not in the review, and the |
| 330 | // same class of defect as the session bus: a command that can write into a |
| 331 | // process outside the fence has stepped out of the fence. Yama's |
| 332 | // `ptrace_scope` already restricts this to descendants on a stock Ubuntu, but |
| 333 | // that is a sysctl the user can change and not a guarantee this code makes. |
| 334 | // Measured free: a from-scratch `cargo test` makes none of these three calls. |
| 335 | Sys { name: "ptrace", group: Group::Poke, x86: Some(101), arm: Some(117), mode: None }, |
| 336 | Sys { name: "process_vm_readv", group: Group::Poke, x86: Some(310), arm: Some(270), mode: None }, |
| 337 | Sys { name: "process_vm_writev",group: Group::Poke, x86: Some(311), arm: Some(271), mode: None }, |
| 338 | ]; |
| 339 | |
| 340 | // ┌───────────────────────────────────────────────────────────────┐ |
| 341 | // │ The three decisions the filter forces │ |
| 342 | // └───────────────────────────────────────────────────────────────┘ |
| 343 | |
| 344 | /// What happens to the calls that change a file's metadata. |
| 345 | /// |
| 346 | /// Three arms rather than a boolean, because the honest answer is a trade and a |
| 347 | /// boolean would hide which side of it was taken. All three were measured |
| 348 | /// against a from-scratch `cargo test`, a fresh registry unpack and a `git init` |
| 349 | /// / `commit` / `clone` cycle. |
| 350 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 351 | pub enum Meta { |
| 352 | /// Refuse every call in [`Group::Mode`], [`Group::Owner`], [`Group::Times`], |
| 353 | /// [`Group::Xattr`] and [`Group::Attr`]. |
| 354 | /// |
| 355 | /// Airtight against the review's §1.2, and it **breaks `cargo`**. Measured: |
| 356 | /// unpacking a `.crate` from the registry fails with `failed to set mtime` |
| 357 | /// (the `utime` family) and then, with the timestamps allowed, with `failed |
| 358 | /// to set permissions to 644` (the `chmod` family). `git init` fails at |
| 359 | /// `could not set 'core.filemode'`. Correct, and unusable for a command the |
| 360 | /// user is watching -- offered for a caller who knows their command needs |
| 361 | /// none of it. |
| 362 | Refuse, |
| 363 | /// Refuse the calls that *loosen* protection, and permit the rest. |
| 364 | /// |
| 365 | /// The default, and the only arm that is both a real answer to §1.2 and |
| 366 | /// survives a build. It rests on the one thing seccomp can genuinely |
| 367 | /// inspect: `chmod`'s mode is a plain integer, so the filter can refuse a |
| 368 | /// mode that sets set-user-ID, set-group-ID or the world-write bit and |
| 369 | /// permit everything else. |
| 370 | /// |
| 371 | /// Measured: `chmod 777`, `chmod 666`, `chmod 4755`, `chmod 2755` and |
| 372 | /// `chmod o+w` are refused; `chmod 755`, `644`, `664`, `775`, `700` and `400` |
| 373 | /// are permitted, which is every mode a from-scratch `cargo test` and a `git` |
| 374 | /// cycle ask for. `chown`, the extended attributes and the inode attribute |
| 375 | /// calls are still refused outright, because nothing a build does needs them. |
| 376 | /// |
| 377 | /// What it does **not** close, and this is the price: |
| 378 | /// |
| 379 | /// * The `utime` family is permitted, because `cargo` cannot unpack a crate |
| 380 | /// without it. A command can therefore rewrite the timestamps of any file |
| 381 | /// it can name, anywhere on the machine. |
| 382 | /// * `chmod 644` on a private key is permitted -- it adds no *write*, so the |
| 383 | /// filter has no grounds to refuse it, and refusing it would refuse the |
| 384 | /// mode `cargo` sets on every file it unpacks. Making a secret readable to |
| 385 | /// other *local* accounts is still possible. |
| 386 | /// * `chmod g+w` is permitted, for the same reason: `git` sets `0664` and |
| 387 | /// `0775` under a `umask` of `002`, which is the Ubuntu default. On a |
| 388 | /// machine where the user's primary group has other members, that is a real |
| 389 | /// grant. |
| 390 | NoLoosening, |
| 391 | /// Leave the metadata calls alone. |
| 392 | /// |
| 393 | /// Here so that a caller who has hit the trade above can say so in the |
| 394 | /// source, and so that the difference shows up in [`Filter::caps`] rather |
| 395 | /// than being invisible. §1.2 is wide open under this arm. |
| 396 | Allow, |
| 397 | } |
| 398 | |
| 399 | impl Meta { |
| 400 | |
| 401 | /// The word the capability strings use. |
| 402 | pub fn word(&self) -> &'static str { |
| 403 | match self { |
| 404 | Self::Refuse => "refused", |
| 405 | Self::NoLoosening => "no-loosening", |
| 406 | Self::Allow => "allowed", |
| 407 | } |
| 408 | } |
| 409 | } |
| 410 | |
| 411 | /// What happens to `socket(AF_UNIX, …)`. |
| 412 | /// |
| 413 | /// The answer to the review's §1.3, and the reason it is a separate decision is |
| 414 | /// that it is only ever right when the fence has already refused the network. A |
| 415 | /// command allowed to reach the internet gains nothing from being denied the |
| 416 | /// session bus, and denying it would break tools that legitimately use a local |
| 417 | /// socket to reach a daemon. |
| 418 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 419 | pub enum Unix { |
| 420 | /// Refuse the creation of any `AF_UNIX` socket. |
| 421 | /// |
| 422 | /// **What this covers**: every `connect()` and `bind()` to a pathname or |
| 423 | /// abstract unix socket made by the command or anything it starts, because |
| 424 | /// all of them need a socket first and `socket(2)`'s first argument is a |
| 425 | /// plain integer the filter can read. Measured: `systemd-run --user --pipe |
| 426 | /// --wait /bin/cat <file>` fails at "Failed to connect to user scope bus" and |
| 427 | /// the file is not read. |
| 428 | /// |
| 429 | /// **What it does not cover**, and none of this is hypothetical: |
| 430 | /// |
| 431 | /// * `socketpair(2)` still works. It is a different syscall, it makes an |
| 432 | /// *anonymous* connected pair with no name in any namespace, and it cannot |
| 433 | /// reach the bus or anything else. It is also load-bearing: a from-scratch |
| 434 | /// `cargo test` makes six `socketpair(AF_UNIX, …)` calls, so refusing it |
| 435 | /// would break every build. |
| 436 | /// * A file descriptor for an already-connected socket, inherited across |
| 437 | /// `execve` or received over `SCM_RIGHTS`, keeps working. Seccomp governs |
| 438 | /// the act of creating a socket, not the use of one that exists -- the same |
| 439 | /// shape as Landlock governing `open` rather than `read`. The launcher |
| 440 | /// passes the child three descriptors and no others, so the reachable |
| 441 | /// version of this is a command that was *given* a socket, which is a |
| 442 | /// decision made elsewhere. |
| 443 | /// * `connect()` itself is untouched. It has to be: the address is behind a |
| 444 | /// pointer, so a filter on `connect` could only refuse all of it, and TCP |
| 445 | /// is Landlock's job from ABI 4. |
| 446 | Refuse, |
| 447 | /// Leave `AF_UNIX` alone, and with it §1.3. |
| 448 | /// |
| 449 | /// Kept as an arm and chosen by nothing. It is here so that an operator |
| 450 | /// setting for "this machine's builds need the container daemon" has a shape |
| 451 | /// to take, and so that the choice is visible in the type rather than |
| 452 | /// implied by its absence. [`Spec::for_command`] records why it is not the |
| 453 | /// default and what was measured when it was. |
| 454 | Allow, |
| 455 | } |
| 456 | |
| 457 | impl Unix { |
| 458 | |
| 459 | /// The word the capability strings use. |
| 460 | pub fn word(&self) -> &'static str { |
| 461 | match self { |
| 462 | Self::Refuse => "refused", |
| 463 | Self::Allow => "allowed", |
| 464 | } |
| 465 | } |
| 466 | } |
| 467 | |
| 468 | /// Whether the command may make syscalls the filter cannot see. |
| 469 | /// |
| 470 | /// A ring submits operations that the kernel performs on its own worker's |
| 471 | /// behalf; seccomp never sees them. `IORING_OP_SETXATTR` and |
| 472 | /// `IORING_OP_FSETXATTR` have existed since Linux 5.19, so a command with a ring |
| 473 | /// can do the very thing [`Meta`] refuses. There is no partial answer: either |
| 474 | /// the ring is refused or every rule in this file is advisory. |
| 475 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 476 | pub enum Ring { |
| 477 | /// Refuse `io_uring_setup`. |
| 478 | /// |
| 479 | /// `EPERM` like everything else here, rather than the `ENOSYS` that would |
| 480 | /// read as "this kernel has no io_uring". `ENOSYS` is the friendlier answer |
| 481 | /// for a probe -- libuv calls `io_uring_setup` at start-up and falls back to |
| 482 | /// its thread pool -- but it needs a second BPF program running on every |
| 483 | /// syscall to say it, since a program carries one errno. Measured that the |
| 484 | /// friendlier answer is not needed: `node` v20 and a from-scratch `cargo |
| 485 | /// test` are both unaffected by `EPERM`. |
| 486 | Refuse, |
| 487 | /// Permit it, and accept that [`Meta`] is then advisory. |
| 488 | Allow, |
| 489 | } |
| 490 | |
| 491 | /// Whether the command may read or write another process's memory. |
| 492 | /// |
| 493 | /// The same class of defect as the session bus, and not in the review: writing |
| 494 | /// into a process outside the fence is a way out of the fence. |
| 495 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 496 | pub enum Poke { |
| 497 | /// Refuse `ptrace`, `process_vm_readv` and `process_vm_writev`. Measured |
| 498 | /// free: a from-scratch `cargo test` makes none of them. Breaks `gdb`, |
| 499 | /// `strace` and `rr` run *inside* a fenced command. |
| 500 | Refuse, |
| 501 | /// Permit them, and rely on Yama's `ptrace_scope` -- a sysctl, not a |
| 502 | /// guarantee this code makes. |
| 503 | Allow, |
| 504 | } |
| 505 | |
| 506 | /// What the filter should refuse. |
| 507 | /// |
| 508 | /// Four independent decisions rather than a level, because they are genuinely |
| 509 | /// independent: the network setting decides [`Unix`], the command decides |
| 510 | /// [`Meta`], and neither says anything about the other two. |
| 511 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 512 | pub struct Spec { |
| 513 | /// What happens to the metadata calls. |
| 514 | pub meta: Meta, |
| 515 | /// What happens to `AF_UNIX` sockets. |
| 516 | pub unix: Unix, |
| 517 | /// What happens to `io_uring`. |
| 518 | pub ring: Ring, |
| 519 | /// What happens to `ptrace` and its relatives. |
| 520 | pub poke: Poke, |
| 521 | } |
| 522 | |
| 523 | impl Spec { |
| 524 | |
| 525 | /// The spec the hand uses for every command, with nothing to decide. |
| 526 | /// |
| 527 | /// # Why `AF_UNIX` is refused whatever the network decision was |
| 528 | /// |
| 529 | /// This took an argument once -- `net` -- and refused `AF_UNIX` only when the |
| 530 | /// fence refused the network, on the reasoning that refusing the bus buys |
| 531 | /// nothing from a command that may reach outward anyway. That reasoning is |
| 532 | /// wrong, and it was measured wrong: with `net:true`, the whole fence in |
| 533 | /// force and the filter installed, |
| 534 | /// `systemd-run --user --pipe --wait /bin/cat <denied file>` **returned the |
| 535 | /// file's contents**. |
| 536 | /// |
| 537 | /// The escape is not a network escape. It is a *filesystem* escape wearing a |
| 538 | /// socket: the bus starts a process that Landlock never bound, and that |
| 539 | /// process reads a path this fence denies. Whether the command was allowed |
| 540 | /// to fetch a crate has nothing to do with it. The same socket reaches |
| 541 | /// `ssh-agent`, which can sign with the user's keys without the key ever |
| 542 | /// being read, and that is equally unrelated to the network decision. |
| 543 | /// |
| 544 | /// `fence.rs` scopes *abstract* unix sockets unconditionally from ABI 6, so |
| 545 | /// refusing the pathname ones unconditionally is what makes the two layers |
| 546 | /// agree rather than what makes them differ. |
| 547 | /// |
| 548 | /// The cost is real and is named in [`Unix::Refuse`] and in |
| 549 | /// [`Filter::holes`]: a command that legitimately wants a local socket -- |
| 550 | /// a database, a container daemon, X11, an `ssh-agent`-authenticated fetch -- |
| 551 | /// cannot have one. Measured against what commands actually do here: a |
| 552 | /// from-scratch `cargo build` succeeds behind this spec, because `cargo`, |
| 553 | /// `rustc` and `ld` use `socketpair`, which is a different call and is left |
| 554 | /// alone. |
| 555 | pub fn for_command() -> Self { |
| 556 | Self { |
| 557 | meta: Meta::NoLoosening, |
| 558 | unix: Unix::Refuse, |
| 559 | ring: Ring::Refuse, |
| 560 | poke: Poke::Refuse, |
| 561 | } |
| 562 | } |
| 563 | } |
| 564 | |
| 565 | impl Default for Spec { |
| 566 | |
| 567 | /// The strictest spec that still runs a build. |
| 568 | fn default() -> Self { |
| 569 | Self::for_command() |
| 570 | } |
| 571 | } |
| 572 | |
| 573 | /// How much of the process the filter binds to. |
| 574 | /// |
| 575 | /// Mirrors [`crate::fence::Reach`] deliberately and is a separate type on |
| 576 | /// purpose: this module has no other dependency on `fence`, so it can be tested, |
| 577 | /// reviewed and replaced on its own. If the two are ever merged, this is the |
| 578 | /// one to delete. |
| 579 | /// |
| 580 | /// [`Reach::Thread`] exists for the same reason it does there -- a filter is |
| 581 | /// irreversible, so a test that installed one process-wide would filter every |
| 582 | /// test that had not run yet. |
| 583 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 584 | pub enum Reach { |
| 585 | /// Every thread of the process, through seccomp's `TSYNC` flag. What the |
| 586 | /// launcher uses. |
| 587 | Process, |
| 588 | /// The calling thread only. |
| 589 | Thread, |
| 590 | } |
| 591 | |
| 592 | // ┌───────────────────────────────────────────────────────────────┐ |
| 593 | // │ The mechanism │ |
| 594 | // └───────────────────────────────────────────────────────────────┘ |
| 595 | |
| 596 | /// Whether this machine can filter system calls at all. |
| 597 | /// |
| 598 | /// Shaped like [`crate::fence::Fence`], and for the same reason: the page shows |
| 599 | /// the user what is in force, and "no answer yet" and "no filter" must not look |
| 600 | /// alike. |
| 601 | #[derive(Clone, Debug, Eq, PartialEq)] |
| 602 | pub enum Seccomp { |
| 603 | /// Seccomp-BPF is available, and this is the syscall numbering in use. |
| 604 | Linux { |
| 605 | /// Which table the rules are built from. |
| 606 | arch: Arch, |
| 607 | }, |
| 608 | /// No filter is available, and this is why. |
| 609 | None { |
| 610 | /// The sentence explaining what is missing. |
| 611 | why: String, |
| 612 | }, |
| 613 | } |
| 614 | |
| 615 | impl Seccomp { |
| 616 | |
| 617 | /// Asks the running machine, rather than reading a version number. |
| 618 | /// |
| 619 | /// The probe is a real installation: a throwaway thread compiles a filter |
| 620 | /// with no rules and an `Allow` default and installs it on itself. If that |
| 621 | /// succeeds, seccomp-BPF works here; if it does not, the reason is the |
| 622 | /// kernel's own and is carried into [`Seccomp::None::why`]. A cheaper check |
| 623 | /// -- reading `/proc/sys/kernel/seccomp/actions_avail` -- says the feature is |
| 624 | /// compiled in, which is not the same as saying a filter will install. |
| 625 | /// |
| 626 | /// The thread dies immediately afterwards, taking the no-op filter with it. |
| 627 | /// `no_new_privs` is set on that thread only; it is a per-thread flag. |
| 628 | pub fn detect() -> Self { |
| 629 | #[cfg(target_os = "linux")] |
| 630 | { |
| 631 | let arch = match Arch::here() { |
| 632 | Some(a) => a, |
| 633 | None => return Self::None { |
| 634 | why: fmt!( |
| 635 | "This build is for an architecture the hand has no \ |
| 636 | syscall table for, so it cannot refuse a system call by \ |
| 637 | number. A table built from the wrong architecture would \ |
| 638 | refuse the wrong calls and report success, so none was \ |
| 639 | guessed."), |
| 640 | }, |
| 641 | }; |
| 642 | match std::thread::spawn(probe).join() { |
| 643 | Ok(Ok(())) => Self::Linux { arch }, |
| 644 | Ok(Err(e)) => Self::None { |
| 645 | why: fmt!( |
| 646 | "This kernel refused to install a system-call filter: \ |
| 647 | {}. Seccomp-BPF needs CONFIG_SECCOMP_FILTER, which has \ |
| 648 | been standard since Linux 3.17.", e), |
| 649 | }, |
| 650 | Err(_) => Self::None { |
| 651 | why: fmt!( |
| 652 | "The thread that was to probe for system-call filtering \ |
| 653 | did not come back, so the hand cannot say whether a \ |
| 654 | filter would install."), |
| 655 | }, |
| 656 | } |
| 657 | } |
| 658 | #[cfg(not(target_os = "linux"))] |
| 659 | { |
| 660 | Self::None { |
| 661 | why: fmt!( |
| 662 | "System-call filtering here is Linux's seccomp-BPF, and this \ |
| 663 | is not Linux. The equivalents -- a sandbox profile on macOS, \ |
| 664 | a Job Object and an AppContainer SID on Windows -- are the \ |
| 665 | same ones `fence` has no arm for yet."), |
| 666 | } |
| 667 | } |
| 668 | } |
| 669 | |
| 670 | /// The mechanisms actually available, for [`crate::wire::Resp::Hello`]. |
| 671 | /// |
| 672 | /// `seccomp:none` rather than an empty list where there is nothing, for the |
| 673 | /// reason [`crate::fence::Fence::caps`] gives: silence reads as "not asked". |
| 674 | pub fn caps(&self) -> Vec<String> { |
| 675 | match self { |
| 676 | Self::Linux { arch } => vec![ |
| 677 | fmt!("seccomp:bpf"), |
| 678 | fmt!("seccomp:arch-{}", arch.name()), |
| 679 | ], |
| 680 | Self::None { .. } => vec![fmt!("seccomp:none")], |
| 681 | } |
| 682 | } |
| 683 | |
| 684 | /// What a filter on this machine cannot cover, whatever the spec says. |
| 685 | /// |
| 686 | /// Distinct from [`Filter::holes`], which is about the spec that was chosen. |
| 687 | /// These hold for every filter this module can build, and every one of them |
| 688 | /// was measured rather than inferred. |
| 689 | pub fn holes(&self) -> Vec<String> { |
| 690 | match self { |
| 691 | Self::Linux { arch } => vec![ |
| 692 | fmt!( |
| 693 | "This is a deny-list, so everything it does not name is \ |
| 694 | permitted. It removes named capabilities from a command; it \ |
| 695 | is not a syscall sandbox. A syscall added by a future kernel \ |
| 696 | is permitted by default, and so is anything the list did not \ |
| 697 | think of. The alternative -- naming what is allowed and \ |
| 698 | refusing the rest -- is strictly stronger and would have to \ |
| 699 | be right about every syscall cargo, rustc, ld, node, python \ |
| 700 | and every build script make; the first one it missed would be \ |
| 701 | a SIGSYS in the middle of a build, and a filter that breaks \ |
| 702 | builds is a filter that gets turned off."), |
| 703 | fmt!( |
| 704 | "A filter cannot follow a pointer. The kernel shows it the \ |
| 705 | syscall number and the six argument registers and nothing \ |
| 706 | they point at, so there is no path here, no sockaddr and no \ |
| 707 | filename. Every rule is therefore all-or-nothing across the \ |
| 708 | whole filesystem, or it is made on a plain integer. This is \ |
| 709 | why a refused chmod is refused everywhere rather than only \ |
| 710 | inside the denied subtree, and why the fence and the filter \ |
| 711 | have to be two different mechanisms."), |
| 712 | fmt!( |
| 713 | "The filter governs making a thing, not using one that \ |
| 714 | exists. A socket connected before it was installed keeps \ |
| 715 | working, and so does one received over SCM_RIGHTS. The same \ |
| 716 | caveat Landlock carries about open file descriptors, for the \ |
| 717 | same reason."), |
| 718 | fmt!( |
| 719 | "The rules are built for {} only, and a program with a \ |
| 720 | different personality is killed rather than filtered. The \ |
| 721 | compiled filter checks seccomp_data.arch first and answers \ |
| 722 | KILL_PROCESS on a mismatch, so a 32-bit binary run inside a \ |
| 723 | fenced command dies with no output rather than escaping \ |
| 724 | through the compatibility ABI, whose syscall numbers are \ |
| 725 | different ones entirely. That fails closed, and it is a real \ |
| 726 | behaviour: a build that execs a 32-bit helper will not work \ |
| 727 | behind this filter.", arch.name()), |
| 728 | fmt!( |
| 729 | "The mode a file is created with is not filtered. open(2) \ |
| 730 | with O_CREAT, mkdir(2) and mknod(2) all carry a mode, and \ |
| 731 | none is inspected -- the umask usually masks the loose bits \ |
| 732 | off, but a command that sets its own umask to 0 can create a \ |
| 733 | world-writable file. What is closed is changing the mode of a \ |
| 734 | file that already exists, which is the escape that was \ |
| 735 | measured; creating a loose file inside the fence is a smaller \ |
| 736 | thing than loosening one outside it."), |
| 737 | ], |
| 738 | Self::None { .. } => vec![fmt!( |
| 739 | "Everything a system call can do. There is no filter on this \ |
| 740 | machine, so the metadata calls are ungoverned and the session bus \ |
| 741 | is reachable -- which on a kernel below Landlock ABI 9 means the \ |
| 742 | fence can be stepped out of entirely.")], |
| 743 | } |
| 744 | } |
| 745 | |
| 746 | /// The sentence a refusal carries, in the voice the file tools already use. |
| 747 | /// |
| 748 | /// # Arguments |
| 749 | /// * `what` - What was being attempted, named so the model can recover. |
| 750 | pub fn refusal(&self, what: &str) -> String { |
| 751 | match self { |
| 752 | Self::Linux { .. } => fmt!( |
| 753 | "{} was refused, although this machine can filter system calls. \ |
| 754 | That is a bug: the filter should have been installed instead.", |
| 755 | what), |
| 756 | Self::None { why } => fmt!( |
| 757 | "{} was refused because the hand cannot filter system calls on \ |
| 758 | this machine. {} Without that filter a fenced command can change \ |
| 759 | the permissions of any file on this machine, including the ones \ |
| 760 | the fence exists to protect, and can reach the session bus, \ |
| 761 | through which it can start a process that is not fenced at all. \ |
| 762 | Running it anyway would mean the compartment was a claim rather \ |
| 763 | than a fact, so it was not run.", what, why), |
| 764 | } |
| 765 | } |
| 766 | |
| 767 | /// Compiles the filter, without installing anything. |
| 768 | /// |
| 769 | /// Separated from [`Filter::apply`] for the reason |
| 770 | /// [`crate::fence::Fence::plan`] gives: a spec that cannot be honoured must |
| 771 | /// fail in the hand's own process, where the answer can still become a |
| 772 | /// [`crate::wire::Resp::Refused`] the page can show, rather than in the |
| 773 | /// launcher, where the only remaining move is to die. |
| 774 | /// |
| 775 | /// # Arguments |
| 776 | /// * `spec` - What to refuse. |
| 777 | /// |
| 778 | /// # Returns |
| 779 | /// A compiled filter, or an error naming what made it impossible. There is |
| 780 | /// no waiver arm: a machine that cannot filter gets a refusal. |
| 781 | pub fn plan(&self, spec: &Spec) -> Outcome<Filter> { |
| 782 | match self { |
| 783 | Self::Linux { arch } => { |
| 784 | let refused = refusals(*arch, spec); |
| 785 | if refused.is_empty() { |
| 786 | return Err(err!( |
| 787 | "The system-call filter would refuse nothing, so \ |
| 788 | installing it would only add a claim. Either name \ |
| 789 | something to refuse or do not ask for a filter."; |
| 790 | Invalid, Input, Security)); |
| 791 | } |
| 792 | #[cfg(target_os = "linux")] |
| 793 | { |
| 794 | let prog = res!(Filter::compile(*arch, spec, &refused)); |
| 795 | Ok(Filter { |
| 796 | arch: *arch, |
| 797 | spec: *spec, |
| 798 | reach: Reach::Process, |
| 799 | refused, |
| 800 | prog, |
| 801 | }) |
| 802 | } |
| 803 | // Unreachable rather than merely unlikely: `detect` only ever |
| 804 | // builds `Linux` under this same cfg. Written out anyway, so the |
| 805 | // arm compiles on the two platforms `fence::Fence` already has |
| 806 | // declared-but-unbuilt arms for. |
| 807 | #[cfg(not(target_os = "linux"))] |
| 808 | { |
| 809 | Err(err!( |
| 810 | "System-call filtering here is Linux's seccomp-BPF, and \ |
| 811 | this build is not for Linux, so the filter could not be \ |
| 812 | compiled."; |
| 813 | Unimplemented, Security)) |
| 814 | } |
| 815 | }, |
| 816 | Self::None { .. } => Err(err!( |
| 817 | "{}", self.refusal("This command"); |
| 818 | Unimplemented, Security, Unauthorised)), |
| 819 | } |
| 820 | } |
| 821 | } |
| 822 | |
| 823 | /// Which calls a spec refuses, in table order. |
| 824 | /// |
| 825 | /// # Arguments |
| 826 | /// * `arch` - The numbering, so a call absent on this architecture is dropped. |
| 827 | /// * `spec` - What to refuse. |
| 828 | fn refusals(arch: Arch, spec: &Spec) -> Vec<&'static Sys> { |
| 829 | TABLE.iter() |
| 830 | .filter(|s| s.number(arch).is_some()) |
| 831 | .filter(|s| match s.group { |
| 832 | Group::Mode => match spec.meta { |
| 833 | Meta::Refuse => true, |
| 834 | // Still listed: the rule is narrower, not absent. |
| 835 | Meta::NoLoosening => true, |
| 836 | Meta::Allow => false, |
| 837 | }, |
| 838 | Group::Owner | Group::Times | Group::Xattr | Group::Attr => match spec.meta { |
| 839 | Meta::Refuse => true, |
| 840 | // The timestamps are the one group that has to survive, because |
| 841 | // `cargo` cannot unpack a crate without them. See `Meta`. |
| 842 | Meta::NoLoosening => s.group != Group::Times, |
| 843 | Meta::Allow => false, |
| 844 | }, |
| 845 | Group::Socket => spec.unix == Unix::Refuse, |
| 846 | Group::Ring => spec.ring == Ring::Refuse, |
| 847 | Group::Poke => spec.poke == Poke::Refuse, |
| 848 | }) |
| 849 | .collect() |
| 850 | } |
| 851 | |
| 852 | // ┌───────────────────────────────────────────────────────────────┐ |
| 853 | // │ The compiled filter │ |
| 854 | // └───────────────────────────────────────────────────────────────┘ |
| 855 | |
| 856 | /// A compiled BPF program and everything that went into it. |
| 857 | /// |
| 858 | /// Inspectable before it is installed, for the reason [`crate::fence::Plan`] is: |
| 859 | /// the journal records it, the page can show it, and a test can assert on it. |
| 860 | #[derive(Clone, Debug)] |
| 861 | pub struct Filter { |
| 862 | /// The numbering the rules were built for. |
| 863 | pub arch: Arch, |
| 864 | /// What was asked for. |
| 865 | pub spec: Spec, |
| 866 | /// How much of the process it will bind to. |
| 867 | pub reach: Reach, |
| 868 | /// The calls it refuses, in table order. |
| 869 | pub refused: Vec<&'static Sys>, |
| 870 | /// The compiled program. |
| 871 | #[cfg(target_os = "linux")] |
| 872 | prog: BpfProgram, |
| 873 | /// A placeholder so the struct exists off Linux, where nothing can build one. |
| 874 | #[cfg(not(target_os = "linux"))] |
| 875 | prog: (), |
| 876 | } |
| 877 | |
| 878 | impl Filter { |
| 879 | |
| 880 | /// How many BPF instructions the kernel will run per system call. |
| 881 | /// |
| 882 | /// Worth reporting: this program runs on **every** syscall the command |
| 883 | /// makes, and the kernel's ceiling is 4096 instructions. A build that got |
| 884 | /// slower after a rule was added would want this number to look at. |
| 885 | pub fn instructions(&self) -> usize { |
| 886 | #[cfg(target_os = "linux")] |
| 887 | { |
| 888 | self.prog.len() |
| 889 | } |
| 890 | #[cfg(not(target_os = "linux"))] |
| 891 | { |
| 892 | 0 |
| 893 | } |
| 894 | } |
| 895 | |
| 896 | /// The capability strings for the journal and the page. |
| 897 | /// |
| 898 | /// Named by group rather than by syscall, because twenty numbers is not a |
| 899 | /// thing a user can check and "chown is refused" is. |
| 900 | pub fn caps(&self) -> Vec<String> { |
| 901 | let mut out = vec![ |
| 902 | fmt!("seccomp:bpf"), |
| 903 | fmt!("seccomp:arch-{}", self.arch.name()), |
| 904 | ]; |
| 905 | for g in [ |
| 906 | Group::Mode, |
| 907 | Group::Owner, |
| 908 | Group::Times, |
| 909 | Group::Xattr, |
| 910 | Group::Attr, |
| 911 | Group::Socket, |
| 912 | Group::Ring, |
| 913 | Group::Poke, |
| 914 | ] { |
| 915 | if !self.refused.iter().any(|s| s.group == g) { |
| 916 | continue; |
| 917 | } |
| 918 | match g { |
| 919 | Group::Mode => out.push(match self.spec.meta { |
| 920 | Meta::NoLoosening => fmt!("seccomp:chmod-no-loosening"), |
| 921 | _ => fmt!("seccomp:no-chmod"), |
| 922 | }), |
| 923 | other => out.push(fmt!("seccomp:no-{}", other.word())), |
| 924 | } |
| 925 | } |
| 926 | out |
| 927 | } |
| 928 | |
| 929 | /// What this spec leaves open, as opposed to what the mechanism cannot do. |
| 930 | /// |
| 931 | /// The counterpart of [`crate::fence::Plan::caveats`]: these exist because of |
| 932 | /// the choices in *this* [`Spec`], and a different spec would have different |
| 933 | /// ones. |
| 934 | pub fn holes(&self) -> Vec<String> { |
| 935 | let mut out = Vec::new(); |
| 936 | match self.spec.meta { |
| 937 | Meta::NoLoosening => { |
| 938 | out.push(fmt!( |
| 939 | "A command can still change a file's permissions anywhere it \ |
| 940 | can name, as long as the new mode grants no world write and \ |
| 941 | no set-user-ID or set-group-ID. chmod 777 is refused; chmod \ |
| 942 | 644 on a private key is not, because it adds no write and \ |
| 943 | because 644 is the mode cargo sets on every file it unpacks. \ |
| 944 | chmod g+w is not refused either, since git writes 0664 and \ |
| 945 | 0775 under the default Ubuntu umask -- so on a machine whose \ |
| 946 | users share a primary group, that is a real grant.")); |
| 947 | out.push(fmt!( |
| 948 | "A command can still rewrite any file's timestamps, anywhere \ |
| 949 | on the machine. The utime family is permitted because cargo \ |
| 950 | cannot unpack a crate from the registry without it -- \ |
| 951 | measured: the unpack fails with \"failed to set mtime\". This \ |
| 952 | is the clearest place where keeping builds working cost \ |
| 953 | coverage.")); |
| 954 | }, |
| 955 | Meta::Allow => out.push(fmt!( |
| 956 | "The metadata calls are not filtered at all under this spec, so \ |
| 957 | a command can chmod, chown, retime and relabel any file it can \ |
| 958 | name -- including files inside the subtree the fence denies, \ |
| 959 | which Landlock does not mediate either. This is the state the \ |
| 960 | review measured.")), |
| 961 | Meta::Refuse => out.push(fmt!( |
| 962 | "The metadata calls are refused outright, which is airtight and \ |
| 963 | breaks cargo: unpacking a crate from the registry fails at \ |
| 964 | \"failed to set mtime\" and then at \"failed to set permissions \ |
| 965 | to 644\", and git init fails at \"could not set \ |
| 966 | 'core.filemode'\".")), |
| 967 | } |
| 968 | if self.spec.unix == Unix::Allow { |
| 969 | out.push(fmt!( |
| 970 | "AF_UNIX sockets are permitted under this spec, so the session \ |
| 971 | bus is reachable. On a kernel below Landlock ABI 9 that is a way \ |
| 972 | out of the fence entirely, not a leak at its edge: systemd-run \ |
| 973 | --user starts a process the fence does not apply to. This is \ |
| 974 | correct only where the command may reach the network anyway.")); |
| 975 | } |
| 976 | if self.spec.ring == Ring::Allow { |
| 977 | out.push(fmt!( |
| 978 | "io_uring is permitted under this spec, which makes every rule \ |
| 979 | above advisory. A ring performs operations without issuing the \ |
| 980 | syscall, and IORING_OP_SETXATTR and IORING_OP_FSETXATTR have \ |
| 981 | existed since Linux 5.19, so a command with a ring can do what \ |
| 982 | the metadata rules refuse. Landlock still applies to a ring; this \ |
| 983 | filter does not.")); |
| 984 | } |
| 985 | if self.spec.poke == Poke::Allow { |
| 986 | out.push(fmt!( |
| 987 | "ptrace and the process_vm_ pair are permitted under this spec, \ |
| 988 | so a command can read and write the memory of another process \ |
| 989 | running as the same user -- which is a way out of the fence, in \ |
| 990 | the same class as the session bus. Yama's ptrace_scope limits it \ |
| 991 | to descendants on a stock Ubuntu, but that is a sysctl and not a \ |
| 992 | guarantee.")); |
| 993 | } |
| 994 | out |
| 995 | } |
| 996 | |
| 997 | /// What the caller should be told about the shape of what they asked for. |
| 998 | /// |
| 999 | /// The costs, rather than the gaps: things that will visibly not work. |
| 1000 | pub fn caveats(&self) -> Vec<String> { |
| 1001 | let mut out = Vec::new(); |
| 1002 | if self.spec.meta == Meta::Refuse { |
| 1003 | out.push(fmt!( |
| 1004 | "This command cannot change any file's permissions, owner, \ |
| 1005 | timestamps or extended attributes. A build that unpacks an \ |
| 1006 | archive, or that runs cargo against a registry it has not already \ |
| 1007 | unpacked, will fail part-way through with \"Operation not \ |
| 1008 | permitted\".")); |
| 1009 | } |
| 1010 | if self.spec.meta == Meta::NoLoosening { |
| 1011 | out.push(fmt!( |
| 1012 | "chmod works for ordinary modes and is refused for a mode that \ |
| 1013 | would make a file world-writable, set-user-ID or set-group-ID. A \ |
| 1014 | build doing that on purpose -- some install steps do -- will fail \ |
| 1015 | with \"Operation not permitted\" at that step and nowhere else.")); |
| 1016 | } |
| 1017 | if self.spec.unix == Unix::Refuse { |
| 1018 | out.push(fmt!( |
| 1019 | "This command cannot open a unix socket, so it cannot reach the \ |
| 1020 | session bus, ssh-agent, a running database on a local socket, or \ |
| 1021 | the display server. socketpair still works, which is what builds \ |
| 1022 | actually use. Name resolution through systemd-resolved's socket \ |
| 1023 | does not, although the fence has already refused the network in \ |
| 1024 | every case where this setting applies.")); |
| 1025 | } |
| 1026 | if self.spec.poke == Poke::Refuse { |
| 1027 | out.push(fmt!( |
| 1028 | "This command cannot use ptrace, so gdb, strace and rr will not \ |
| 1029 | run inside it.")); |
| 1030 | } |
| 1031 | out.push(fmt!( |
| 1032 | "The filter is built for {} and kills anything with a different \ |
| 1033 | personality, so a 32-bit helper binary will not run inside this \ |
| 1034 | command.", self.arch.name())); |
| 1035 | out |
| 1036 | } |
| 1037 | |
| 1038 | /// Installs the filter on **the current process**, then reports what took |
| 1039 | /// hold. |
| 1040 | /// |
| 1041 | /// # Where this must be called |
| 1042 | /// |
| 1043 | /// In the launcher, after [`crate::fence::Plan::apply`] and immediately |
| 1044 | /// before `execve`. A seccomp filter is inherited across `execve` and cannot |
| 1045 | /// be removed, so the launcher is the only place it can go -- and it goes |
| 1046 | /// *after* Landlock because Landlock still has to open every granted path |
| 1047 | /// after its own rules take hold, whereas this has nothing left to do. See |
| 1048 | /// the module documentation. |
| 1049 | /// |
| 1050 | /// # Returns |
| 1051 | /// What was installed, or an error. There is no partial success to report: |
| 1052 | /// the kernel either takes the whole program or none of it. |
| 1053 | pub fn apply(&self) -> Outcome<Enforced> { |
| 1054 | #[cfg(target_os = "linux")] |
| 1055 | { |
| 1056 | let outcome = match self.reach { |
| 1057 | Reach::Process => seccompiler::apply_filter_all_threads(&self.prog), |
| 1058 | Reach::Thread => seccompiler::apply_filter(&self.prog), |
| 1059 | }; |
| 1060 | if let Err(e) = outcome { |
| 1061 | return Err(err!( |
| 1062 | "The kernel refused to install the system-call filter ({}), \ |
| 1063 | so the two things Landlock cannot refuse -- changing a file's \ |
| 1064 | permissions anywhere on this machine, and reaching the \ |
| 1065 | session bus to start a process outside the fence -- would \ |
| 1066 | both have been available. The command was not run. Note that \ |
| 1067 | a filter needs no_new_privs, which the fence sets and \ |
| 1068 | hard-errors on; if that failed, this is where it shows.", e; |
| 1069 | Security, System)); |
| 1070 | } |
| 1071 | Ok(Enforced { |
| 1072 | arch: self.arch, |
| 1073 | filtered: true, |
| 1074 | caps: self.caps(), |
| 1075 | refused: self.refused.iter().map(|s| fmt!("{}", s.name)).collect(), |
| 1076 | }) |
| 1077 | } |
| 1078 | #[cfg(not(target_os = "linux"))] |
| 1079 | { |
| 1080 | Err(err!( |
| 1081 | "There is no system-call filter to install on this platform, and \ |
| 1082 | a fence without one is not a compartment on any kernel below \ |
| 1083 | Landlock ABI 9. Nothing was run."; |
| 1084 | Unimplemented, Security)) |
| 1085 | } |
| 1086 | } |
| 1087 | |
| 1088 | /// Turns the spec into a BPF program. |
| 1089 | /// |
| 1090 | /// The filter is default-allow: `mismatch_action` is `Allow`, so a syscall |
| 1091 | /// not named here is permitted, and `match_action` is `Errno(EPERM)`. A |
| 1092 | /// syscall in the map with an empty rule vector always matches, which is how |
| 1093 | /// a whole call is refused; a syscall with rules matches only if one of them |
| 1094 | /// does, which is how [`Meta::NoLoosening`] and [`Unix::Refuse`] refuse a |
| 1095 | /// call for some arguments and permit it for others. |
| 1096 | /// |
| 1097 | /// # Arguments |
| 1098 | /// * `arch` - The numbering. |
| 1099 | /// * `spec` - What to refuse. |
| 1100 | /// * `refused` - The calls, already filtered to this architecture. |
| 1101 | #[cfg(target_os = "linux")] |
| 1102 | fn compile(arch: Arch, spec: &Spec, refused: &[&'static Sys]) -> Outcome<BpfProgram> { |
| 1103 | let target = match arch { |
| 1104 | Arch::X86_64 => TargetArch::x86_64, |
| 1105 | Arch::Aarch64 => TargetArch::aarch64, |
| 1106 | }; |
| 1107 | |
| 1108 | // One program, one errno. A program carries a single `match_action`, so |
| 1109 | // every refusal here answers `EPERM`; see `Ring::Refuse` for the one |
| 1110 | // place a different errno would have been kinder and why it is not worth |
| 1111 | // a second program running on every syscall the command makes. |
| 1112 | let mut rules: BTreeMap<i64, Vec<SeccompRule>> = BTreeMap::new(); |
| 1113 | |
| 1114 | for sys in refused { |
| 1115 | let n = match sys.number(arch) { |
| 1116 | Some(n) => n, |
| 1117 | // Unreachable: `refusals` filtered on exactly this. Kept as a |
| 1118 | // refusal rather than an unwrap, because a table edit that |
| 1119 | // introduced it must not become a panic in a launcher. |
| 1120 | None => return Err(err!( |
| 1121 | "The system-call table has no number for {} on {}, although \ |
| 1122 | it was selected for refusal.", sys.name, arch.name(); |
| 1123 | Bug, Mismatch)), |
| 1124 | }; |
| 1125 | match (sys.group, spec.meta, sys.mode) { |
| 1126 | // The narrow chmod rule: refuse only a mode that loosens. |
| 1127 | (Group::Mode, Meta::NoLoosening, Some(idx)) => { |
| 1128 | let mut chain = Vec::with_capacity(LOOSENING.len()); |
| 1129 | for (bit, _) in LOOSENING { |
| 1130 | // Dword, not Qword. The kernel reads `mode` as a |
| 1131 | // `umode_t` and every bit of interest is below 32, so the |
| 1132 | // low word is exactly what it acts on; comparing the full |
| 1133 | // register would be no stronger and would cost two more |
| 1134 | // instructions per condition. |
| 1135 | let cond = match SeccompCondition::new( |
| 1136 | idx, |
| 1137 | SeccompCmpArgLen::Dword, |
| 1138 | SeccompCmpOp::MaskedEq(bit), |
| 1139 | bit, |
| 1140 | ) { |
| 1141 | Ok(c) => c, |
| 1142 | Err(e) => return Err(err!( |
| 1143 | "The mode condition for {} could not be built: \ |
| 1144 | {}", sys.name, e; Bug, Invalid)), |
| 1145 | }; |
| 1146 | match SeccompRule::new(vec![cond]) { |
| 1147 | Ok(r) => chain.push(r), |
| 1148 | Err(e) => return Err(err!( |
| 1149 | "The mode rule for {} could not be built: {}", |
| 1150 | sys.name, e; Bug, Invalid)), |
| 1151 | } |
| 1152 | } |
| 1153 | rules.insert(n, chain); |
| 1154 | }, |
| 1155 | // The socket rule: refuse only `AF_UNIX`. |
| 1156 | (Group::Socket, _, _) => { |
| 1157 | // Dword here is not an optimisation, it is the correct |
| 1158 | // comparison and a Qword one would be a hole. `socket(2)` |
| 1159 | // takes an `int`, so the kernel looks at the low 32 bits of |
| 1160 | // the register and ignores the rest; a 64-bit equality test |
| 1161 | // against 1 would fail to match a caller passing |
| 1162 | // 0x1_0000_0001, which the kernel would still read as |
| 1163 | // AF_UNIX. |
| 1164 | let cond = match SeccompCondition::new( |
| 1165 | 0, |
| 1166 | SeccompCmpArgLen::Dword, |
| 1167 | SeccompCmpOp::Eq, |
| 1168 | AF_UNIX, |
| 1169 | ) { |
| 1170 | Ok(c) => c, |
| 1171 | Err(e) => return Err(err!( |
| 1172 | "The AF_UNIX condition could not be built: {}", e; |
| 1173 | Bug, Invalid)), |
| 1174 | }; |
| 1175 | match SeccompRule::new(vec![cond]) { |
| 1176 | Ok(r) => { rules.insert(n, vec![r]); }, |
| 1177 | Err(e) => return Err(err!( |
| 1178 | "The AF_UNIX rule could not be built: {}", e; |
| 1179 | Bug, Invalid)), |
| 1180 | } |
| 1181 | }, |
| 1182 | // Everything else: the whole call, whatever its arguments. |
| 1183 | _ => { rules.insert(n, Vec::new()); }, |
| 1184 | } |
| 1185 | } |
| 1186 | |
| 1187 | let filter = match SeccompFilter::new( |
| 1188 | rules, |
| 1189 | SeccompAction::Allow, |
| 1190 | SeccompAction::Errno(EPERM), |
| 1191 | target, |
| 1192 | ) { |
| 1193 | Ok(f) => f, |
| 1194 | Err(e) => return Err(err!( |
| 1195 | "The system-call filter could not be assembled: {}", e; |
| 1196 | Bug, Invalid, Security)), |
| 1197 | }; |
| 1198 | match BpfProgram::try_from(filter) { |
| 1199 | Ok(p) => Ok(p), |
| 1200 | Err(e) => Err(err!( |
| 1201 | "The system-call filter could not be compiled to BPF: {}. The \ |
| 1202 | kernel's ceiling is 4096 instructions.", e; |
| 1203 | Bug, Invalid, Security)), |
| 1204 | } |
| 1205 | } |
| 1206 | } |
| 1207 | |
| 1208 | /// What actually took hold, as opposed to what was asked for. |
| 1209 | /// |
| 1210 | /// Kept apart from [`Filter`] for the reason [`crate::fence::Applied`] is: a |
| 1211 | /// filter is a wish and this is the answer. |
| 1212 | #[derive(Clone, Debug, Eq, PartialEq)] |
| 1213 | pub struct Enforced { |
| 1214 | /// The numbering the rules were built for. |
| 1215 | pub arch: Arch, |
| 1216 | /// Whether a filter is in force at all. Never false on a success. |
| 1217 | pub filtered: bool, |
| 1218 | /// The capability strings for the journal and the page. |
| 1219 | pub caps: Vec<String>, |
| 1220 | /// The calls refused, by name, for the journal. |
| 1221 | pub refused: Vec<String>, |
| 1222 | } |
| 1223 | |
| 1224 | /// Installs a no-op filter on the calling thread, to find out whether one can be. |
| 1225 | /// |
| 1226 | /// A program with no rules compiles to the architecture guard and a single |
| 1227 | /// `Allow`, which is four instructions and refuses nothing. Installing it is the |
| 1228 | /// only honest way to answer "would a filter install here", and it is why |
| 1229 | /// [`Seccomp::detect`] runs this on a thread that is about to die. |
| 1230 | #[cfg(target_os = "linux")] |
| 1231 | fn probe() -> Outcome<()> { |
| 1232 | let arch = match Arch::here() { |
| 1233 | Some(Arch::X86_64) => TargetArch::x86_64, |
| 1234 | Some(Arch::Aarch64) => TargetArch::aarch64, |
| 1235 | None => return Err(err!( |
| 1236 | "No syscall table for this architecture."; Unimplemented)), |
| 1237 | }; |
| 1238 | // `Allow` as the default and `Trap` as the on-match, because seccompiler |
| 1239 | // refuses a filter whose two actions are the same. With no rules, nothing |
| 1240 | // ever matches, so `Trap` is unreachable. |
| 1241 | let filter = match SeccompFilter::new( |
| 1242 | BTreeMap::new(), |
| 1243 | SeccompAction::Allow, |
| 1244 | SeccompAction::Trap, |
| 1245 | arch, |
| 1246 | ) { |
| 1247 | Ok(f) => f, |
| 1248 | Err(e) => return Err(err!("{}", e; Bug, Invalid)), |
| 1249 | }; |
| 1250 | let prog = match BpfProgram::try_from(filter) { |
| 1251 | Ok(p) => p, |
| 1252 | Err(e) => return Err(err!("{}", e; Bug, Invalid)), |
| 1253 | }; |
| 1254 | match seccompiler::apply_filter(&prog) { |
| 1255 | Ok(()) => Ok(()), |
| 1256 | Err(e) => Err(err!("{}", e; System, Security)), |
| 1257 | } |
| 1258 | } |
| 1259 | |
| 1260 | // ┌───────────────────────────────────────────────────────────────┐ |
| 1261 | // │ Tests │ |
| 1262 | // └───────────────────────────────────────────────────────────────┘ |
| 1263 | |
| 1264 | #[cfg(test)] |
| 1265 | mod tests { |
| 1266 | use super::*; |
| 1267 | |
| 1268 | use std::{ |
| 1269 | fs, |
| 1270 | os::unix::fs::PermissionsExt, |
| 1271 | path::PathBuf, |
| 1272 | process::Command, |
| 1273 | }; |
| 1274 | |
| 1275 | /// Where the fixtures go. |
| 1276 | /// |
| 1277 | /// Under the home cache and never `/tmp`: that is a tmpfs here, its pages are |
| 1278 | /// charged to whoever wrote them, and filling it has taken this machine down |
| 1279 | /// before. |
| 1280 | fn root() -> Outcome<PathBuf> { |
| 1281 | let home = match std::env::var("HOME") { |
| 1282 | Ok(h) => h, |
| 1283 | Err(e) => return Err(err!( |
| 1284 | "The seccomp tests need HOME to know where to put fixtures: {}", e; |
| 1285 | Test, Configuration)), |
| 1286 | }; |
| 1287 | Ok(PathBuf::from(home).join(".cache/daimond-hand-seccomp-tests")) |
| 1288 | } |
| 1289 | |
| 1290 | /// A fresh directory with one 600 file in it, standing in for the denied |
| 1291 | /// subtree the review's reproduction used. |
| 1292 | /// |
| 1293 | /// # Arguments |
| 1294 | /// * `name` - A name unique to the calling test, so tests do not share state. |
| 1295 | fn fixture(name: &str) -> Outcome<PathBuf> { |
| 1296 | let base = res!(root()).join(name); |
| 1297 | let _ = fs::remove_dir_all(&base); |
| 1298 | res!(fs::create_dir_all(&base)); |
| 1299 | let secret = base.join("secret.txt"); |
| 1300 | res!(fs::write(&secret, "secret")); |
| 1301 | res!(fs::set_permissions(&secret, fs::Permissions::from_mode(0o600))); |
| 1302 | Ok(base) |
| 1303 | } |
| 1304 | |
| 1305 | /// The real mechanism, or `None` with a printed reason where this machine |
| 1306 | /// cannot run the test. |
| 1307 | /// |
| 1308 | /// Loud rather than silent, exactly as `fence::tests::kernel_fence` is: a |
| 1309 | /// kernel test that quietly passes on a machine that never ran it is a test |
| 1310 | /// that will quietly pass forever. |
| 1311 | /// |
| 1312 | /// # Arguments |
| 1313 | /// * `what` - The test's name, for the message. |
| 1314 | fn kernel_seccomp(what: &str) -> Option<Seccomp> { |
| 1315 | let s = Seccomp::detect(); |
| 1316 | match &s { |
| 1317 | Seccomp::Linux { arch } => { |
| 1318 | println!("[{}] running against seccomp-bpf on {}", what, arch.name()); |
| 1319 | Some(s) |
| 1320 | }, |
| 1321 | Seccomp::None { why } => { |
| 1322 | println!( |
| 1323 | "[{}] SKIPPED: this machine cannot filter system calls. {} \ |
| 1324 | Run this on Linux 3.17 or later with CONFIG_SECCOMP_FILTER.", |
| 1325 | what, why); |
| 1326 | None |
| 1327 | }, |
| 1328 | } |
| 1329 | } |
| 1330 | |
| 1331 | /// Runs a test body on a thread of its own, so the filter it installs dies |
| 1332 | /// with it. |
| 1333 | /// |
| 1334 | /// A seccomp filter cannot be removed, so a test that installed one on the |
| 1335 | /// harness's own thread would filter every test that had not run yet. |
| 1336 | /// |
| 1337 | /// # Arguments |
| 1338 | /// * `what` - The test's name, for the error. |
| 1339 | /// * `body` - The unfiltered half, the filter, and the filtered half. |
| 1340 | fn own_thread<F>(what: &'static str, body: F) -> Outcome<()> |
| 1341 | where |
| 1342 | F: FnOnce() -> Outcome<()> + Send + 'static, |
| 1343 | { |
| 1344 | match std::thread::spawn(body).join() { |
| 1345 | Ok(result) => result, |
| 1346 | Err(panic) => { |
| 1347 | let msg = match panic.downcast_ref::<&str>() { |
| 1348 | Some(s) => s.to_string(), |
| 1349 | None => match panic.downcast_ref::<String>() { |
| 1350 | Some(s) => s.clone(), |
| 1351 | None => fmt!("(the panic carried no message)"), |
| 1352 | }, |
| 1353 | }; |
| 1354 | Err(err!("{}: {}", what, msg; Test)) |
| 1355 | }, |
| 1356 | } |
| 1357 | } |
| 1358 | |
| 1359 | /// Installs a spec on this thread, failing loudly rather than continuing |
| 1360 | /// unfiltered. |
| 1361 | /// |
| 1362 | /// # Arguments |
| 1363 | /// * `s` - The mechanism. |
| 1364 | /// * `spec` - What to refuse. |
| 1365 | fn engage(s: &Seccomp, spec: &Spec) -> Outcome<Enforced> { |
| 1366 | let mut filter = res!(s.plan(spec)); |
| 1367 | // The rules are exactly the production ones; only their reach is |
| 1368 | // narrowed, for the reason `own_thread` exists. |
| 1369 | filter.reach = Reach::Thread; |
| 1370 | let enforced = res!(filter.apply()); |
| 1371 | if !enforced.filtered { |
| 1372 | return Err(err!( |
| 1373 | "The filter reported that it was not installed, so the rest of \ |
| 1374 | this test would prove nothing."; Test, Security)); |
| 1375 | } |
| 1376 | Ok(enforced) |
| 1377 | } |
| 1378 | |
| 1379 | /// Whether a program exists to run. |
| 1380 | /// |
| 1381 | /// # Arguments |
| 1382 | /// * `what` - The test's name, for the message. |
| 1383 | /// * `p` - The path. |
| 1384 | fn have(what: &str, p: &str) -> bool { |
| 1385 | if std::path::Path::new(p).exists() { |
| 1386 | return true; |
| 1387 | } |
| 1388 | println!("[{}] SKIPPED: no {} on this machine.", what, p); |
| 1389 | false |
| 1390 | } |
| 1391 | |
| 1392 | /// The session bus socket, from the environment rather than a guessed uid. |
| 1393 | /// |
| 1394 | /// `None` where there is no user session to escape through, which is the |
| 1395 | /// case in a container and on a build machine -- and there the test proves |
| 1396 | /// nothing and says so. |
| 1397 | fn bus() -> Option<PathBuf> { |
| 1398 | let dir = match std::env::var("XDG_RUNTIME_DIR") { |
| 1399 | Ok(d) => PathBuf::from(d), |
| 1400 | Err(_) => return None, |
| 1401 | }; |
| 1402 | let p = dir.join("bus"); |
| 1403 | if p.exists() { Some(p) } else { None } |
| 1404 | } |
| 1405 | |
| 1406 | // ── The table, checked against the kernel's own headers ───────── |
| 1407 | |
| 1408 | /// Every number in [`TABLE`] agrees with the kernel headers installed on this |
| 1409 | /// machine. |
| 1410 | /// |
| 1411 | /// The external oracle, and the reason the table is written out by hand at |
| 1412 | /// all: `libc` has no constant for four of these calls on this target, so a |
| 1413 | /// table built from `libc::SYS_*` would silently be missing `setxattrat`, |
| 1414 | /// `removexattrat`, `open_tree_attr` and `file_setattr`. Checking against the |
| 1415 | /// headers checks against the thing that actually defines the numbers. |
| 1416 | #[test] |
| 1417 | fn the_table_agrees_with_the_kernel_headers() -> Outcome<()> { |
| 1418 | let (path, arch) = if cfg!(target_arch = "x86_64") { |
| 1419 | ("/usr/include/x86_64-linux-gnu/asm/unistd_64.h", Arch::X86_64) |
| 1420 | } else if cfg!(target_arch = "aarch64") { |
| 1421 | ("/usr/include/asm-generic/unistd.h", Arch::Aarch64) |
| 1422 | } else { |
| 1423 | println!( |
| 1424 | "[the_table_agrees_with_the_kernel_headers] SKIPPED: no header \ |
| 1425 | path known for this architecture, which is also why \ |
| 1426 | `Arch::here` returns None here and no filter would be built."); |
| 1427 | return Ok(()); |
| 1428 | }; |
| 1429 | let text = match fs::read_to_string(path) { |
| 1430 | Ok(t) => t, |
| 1431 | Err(e) => { |
| 1432 | println!( |
| 1433 | "[the_table_agrees_with_the_kernel_headers] SKIPPED: {} \ |
| 1434 | could not be read ({}). Install libc6-dev to run this \ |
| 1435 | check; without it the syscall numbers in TABLE are \ |
| 1436 | unverified.", path, e); |
| 1437 | return Ok(()); |
| 1438 | }, |
| 1439 | }; |
| 1440 | |
| 1441 | // `#define __NR_<name> <n>`, and nothing else. |
| 1442 | let mut seen: BTreeMap<String, i64> = BTreeMap::new(); |
| 1443 | for line in text.lines() { |
| 1444 | let rest = match line.strip_prefix("#define __NR_") { |
| 1445 | Some(r) => r, |
| 1446 | None => continue, |
| 1447 | }; |
| 1448 | let mut parts = rest.split_whitespace(); |
| 1449 | let name = match parts.next() { |
| 1450 | Some(n) => n, |
| 1451 | None => continue, |
| 1452 | }; |
| 1453 | let num = match parts.next().and_then(|v| v.parse::<i64>().ok()) { |
| 1454 | Some(n) => n, |
| 1455 | None => continue, |
| 1456 | }; |
| 1457 | seen.insert(name.to_string(), num); |
| 1458 | } |
| 1459 | assert!( |
| 1460 | seen.len() > 100, |
| 1461 | "{} yielded only {} syscall definitions, so it is not the table this \ |
| 1462 | test thinks it is", path, seen.len()); |
| 1463 | |
| 1464 | let mut checked = 0usize; |
| 1465 | for sys in TABLE { |
| 1466 | match (sys.number(arch), seen.get(sys.name)) { |
| 1467 | (Some(ours), Some(theirs)) => { |
| 1468 | assert_eq!( |
| 1469 | *theirs, ours, |
| 1470 | "TABLE has {} = {} on {}, and the kernel header says {}", |
| 1471 | sys.name, ours, arch.name(), theirs); |
| 1472 | checked += 1; |
| 1473 | }, |
| 1474 | (Some(ours), None) => println!( |
| 1475 | "[the_table_agrees_with_the_kernel_headers] {} = {} is not \ |
| 1476 | in {} -- these headers are older than the kernel. Not a \ |
| 1477 | failure: a number for a call this kernel does not have \ |
| 1478 | refuses nothing.", sys.name, ours, path), |
| 1479 | (None, Some(theirs)) => panic!( |
| 1480 | "TABLE says {} does not exist on {}, and the kernel header \ |
| 1481 | gives it as {}. That is a hole: the call is refusable and is \ |
| 1482 | not being refused.", sys.name, arch.name(), theirs), |
| 1483 | (None, None) => (), |
| 1484 | } |
| 1485 | } |
| 1486 | println!( |
| 1487 | "[the_table_agrees_with_the_kernel_headers] {} of {} entries \ |
| 1488 | cross-checked against {}", checked, TABLE.len(), path); |
| 1489 | assert!(checked >= 20, "only {} entries were cross-checked", checked); |
| 1490 | Ok(()) |
| 1491 | } |
| 1492 | |
| 1493 | /// The table has no duplicate numbers on either architecture. |
| 1494 | /// |
| 1495 | /// A transposed digit would silently refuse the wrong call, and the header |
| 1496 | /// check above would not catch it if the wrong number happened to be another |
| 1497 | /// entry's. |
| 1498 | #[test] |
| 1499 | fn the_table_has_no_collisions() -> Outcome<()> { |
| 1500 | for arch in [Arch::X86_64, Arch::Aarch64] { |
| 1501 | let mut seen: BTreeMap<i64, &str> = BTreeMap::new(); |
| 1502 | for sys in TABLE { |
| 1503 | if let Some(n) = sys.number(arch) { |
| 1504 | if let Some(prev) = seen.insert(n, sys.name) { |
| 1505 | return Err(err!( |
| 1506 | "On {}, {} and {} both claim syscall {}.", |
| 1507 | arch.name(), prev, sys.name, n; Test, Conflict)); |
| 1508 | } |
| 1509 | } |
| 1510 | } |
| 1511 | } |
| 1512 | Ok(()) |
| 1513 | } |
| 1514 | |
| 1515 | // ── The spec, decided without a kernel ────────────────────────── |
| 1516 | |
| 1517 | /// The default spec refuses what it says it refuses, and nothing it says it |
| 1518 | /// does not. |
| 1519 | #[test] |
| 1520 | fn the_default_spec_is_what_it_claims() -> Outcome<()> { |
| 1521 | let arch = match Arch::here() { |
| 1522 | Some(a) => a, |
| 1523 | None => { |
| 1524 | println!( |
| 1525 | "[the_default_spec_is_what_it_claims] SKIPPED: no syscall \ |
| 1526 | table for this architecture."); |
| 1527 | return Ok(()); |
| 1528 | }, |
| 1529 | }; |
| 1530 | let sel = refusals(arch, &Spec::default()); |
| 1531 | let named = |n: &str| sel.iter().any(|s| s.name == n); |
| 1532 | |
| 1533 | // The two escapes. |
| 1534 | assert!(named("fchmodat"), "the chmod family is not refused"); |
| 1535 | assert!(named("fchmodat2"), "fchmodat2 was missed, which is the newest \ |
| 1536 | way to reach the same thing"); |
| 1537 | assert!(named("socket"), "AF_UNIX is not refused"); |
| 1538 | |
| 1539 | // The rest of the metadata families. |
| 1540 | assert!(named("fchownat"), "the chown family is not refused"); |
| 1541 | assert!(named("fsetxattr"), "the xattr family is not refused"); |
| 1542 | assert!(named("setxattrat"), "setxattrat was missed, and libc has no \ |
| 1543 | constant for it, which is exactly how it gets missed"); |
| 1544 | assert!(named("file_setattr"), "file_setattr was missed"); |
| 1545 | assert!(named("io_uring_setup"), "io_uring is not refused, which makes \ |
| 1546 | every metadata rule advisory"); |
| 1547 | assert!(named("ptrace"), "ptrace is not refused"); |
| 1548 | |
| 1549 | // And the one deliberate exception. |
| 1550 | assert!(!named("utimensat"), |
| 1551 | "the utime family is refused by default, which breaks cargo"); |
| 1552 | |
| 1553 | // The socket is refused for every command, and the network decision does |
| 1554 | // not enter into it. This was once conditional on `net`, and with the |
| 1555 | // condition in place a `net:true` command read a denied file through the |
| 1556 | // session bus with the filter installed and the fence in force. |
| 1557 | assert!(named("socket"), |
| 1558 | "AF_UNIX is not refused, so the session bus is a way out of the fence"); |
| 1559 | let open = refusals(arch, &Spec { unix: Unix::Allow, ..Spec::default() }); |
| 1560 | assert!(!open.iter().any(|s| s.name == "socket"), |
| 1561 | "Unix::Allow still refused the socket, so the arm means nothing"); |
| 1562 | Ok(()) |
| 1563 | } |
| 1564 | |
| 1565 | /// `Meta::Refuse` really does take the timestamps too, and `Meta::Allow` |
| 1566 | /// really does take nothing. |
| 1567 | #[test] |
| 1568 | fn the_meta_arms_differ() -> Outcome<()> { |
| 1569 | let arch = match Arch::here() { |
| 1570 | Some(a) => a, |
| 1571 | None => return Ok(()), |
| 1572 | }; |
| 1573 | let strict = refusals(arch, &Spec { meta: Meta::Refuse, ..Spec::default() }); |
| 1574 | assert!(strict.iter().any(|s| s.name == "utimensat"), |
| 1575 | "Meta::Refuse did not refuse the timestamps"); |
| 1576 | |
| 1577 | let open = refusals(arch, &Spec { meta: Meta::Allow, ..Spec::default() }); |
| 1578 | assert!(!open.iter().any(|s| s.group == Group::Mode), |
| 1579 | "Meta::Allow refused a chmod"); |
| 1580 | assert!(!open.iter().any(|s| s.group == Group::Xattr), |
| 1581 | "Meta::Allow refused an xattr call"); |
| 1582 | // The other groups are unaffected by Meta. |
| 1583 | assert!(open.iter().any(|s| s.name == "socket"), |
| 1584 | "Meta::Allow disturbed the socket rule"); |
| 1585 | Ok(()) |
| 1586 | } |
| 1587 | |
| 1588 | /// A machine with no filter refuses, rather than running the command. |
| 1589 | /// |
| 1590 | /// There is no waiver arm here on purpose, and this is what asserts it: the |
| 1591 | /// only way past `plan` on a machine without seccomp is not to call it. |
| 1592 | #[test] |
| 1593 | fn no_filter_refuses() -> Outcome<()> { |
| 1594 | let s = Seccomp::None { why: fmt!("Nothing here.") }; |
| 1595 | assert!(s.plan(&Spec::default()).is_err(), |
| 1596 | "a command was planned with no system-call filter"); |
| 1597 | |
| 1598 | let words = s.refusal("A build"); |
| 1599 | assert!(words.contains("A build"), "{}", words); |
| 1600 | assert!(words.contains("session bus"), "the refusal does not say what \ |
| 1601 | is lost: {}", words); |
| 1602 | assert_eq!(vec![fmt!("seccomp:none")], s.caps()); |
| 1603 | assert!(s.holes().iter().any(|h| h.contains("Everything")), "{:?}", |
| 1604 | s.holes()); |
| 1605 | Ok(()) |
| 1606 | } |
| 1607 | |
| 1608 | /// A spec that refuses nothing is refused, rather than compiled into a claim. |
| 1609 | #[test] |
| 1610 | fn an_empty_spec_is_refused() -> Outcome<()> { |
| 1611 | let s = match kernel_seccomp("an_empty_spec_is_refused") { |
| 1612 | Some(s) => s, |
| 1613 | None => return Ok(()), |
| 1614 | }; |
| 1615 | let nothing = Spec { |
| 1616 | meta: Meta::Allow, |
| 1617 | unix: Unix::Allow, |
| 1618 | ring: Ring::Allow, |
| 1619 | poke: Poke::Allow, |
| 1620 | }; |
| 1621 | assert!(s.plan(¬hing).is_err(), |
| 1622 | "a filter refusing nothing was compiled, and would have been \ |
| 1623 | reported as a capability"); |
| 1624 | Ok(()) |
| 1625 | } |
| 1626 | |
| 1627 | /// The report names every group that is in force, and no group that is not. |
| 1628 | #[test] |
| 1629 | fn caps_never_stay_silent() -> Outcome<()> { |
| 1630 | let s = match kernel_seccomp("caps_never_stay_silent") { |
| 1631 | Some(s) => s, |
| 1632 | None => return Ok(()), |
| 1633 | }; |
| 1634 | let f = res!(s.plan(&Spec::default())); |
| 1635 | let caps = f.caps(); |
| 1636 | for want in [ |
| 1637 | "seccomp:bpf", |
| 1638 | "seccomp:chmod-no-loosening", |
| 1639 | "seccomp:no-chown", |
| 1640 | "seccomp:no-xattr", |
| 1641 | "seccomp:no-af-unix", |
| 1642 | "seccomp:no-io-uring", |
| 1643 | "seccomp:no-ptrace", |
| 1644 | ] { |
| 1645 | assert!(caps.iter().any(|c| c == want), "{} missing from {:?}", |
| 1646 | want, caps); |
| 1647 | } |
| 1648 | assert!(!caps.iter().any(|c| c == "seccomp:no-times"), |
| 1649 | "the timestamps are permitted by default and the report says \ |
| 1650 | otherwise: {:?}", caps); |
| 1651 | |
| 1652 | // The permitted timestamps are a hole, and a hole has to be said out |
| 1653 | // loud rather than merely not claimed. |
| 1654 | assert!(f.holes().iter().any(|h| h.contains("timestamps")), |
| 1655 | "the permitted utime family is not in holes(): {:?}", f.holes()); |
| 1656 | assert!(f.holes().iter().any(|h| h.contains("644")), |
| 1657 | "the permitted chmod 644 is not in holes(): {:?}", f.holes()); |
| 1658 | |
| 1659 | // And the program is a real one of a sane size. |
| 1660 | assert!(f.instructions() > 20 && f.instructions() < 4096, |
| 1661 | "{} BPF instructions", f.instructions()); |
| 1662 | println!("[caps_never_stay_silent] {} BPF instructions, {} calls \ |
| 1663 | refused", f.instructions(), f.refused.len()); |
| 1664 | Ok(()) |
| 1665 | } |
| 1666 | |
| 1667 | // ── The two escapes from the review, closed ───────────────────── |
| 1668 | |
| 1669 | /// §1.2, the measured half: `chmod 777` on a 600 file. |
| 1670 | /// |
| 1671 | /// The review's reproduction was on a file *inside the denied subtree* of a |
| 1672 | /// full ABI-8 fence, and Landlock did not mediate it. Landlock is not |
| 1673 | /// involved here: this test does the same thing to the same kind of file and |
| 1674 | /// shows the filter refusing it, which is the whole of the fix. The fenced |
| 1675 | /// version is `the_review_escapes_are_closed_together`. |
| 1676 | #[test] |
| 1677 | fn chmod_777_is_refused() -> Outcome<()> { |
| 1678 | let s = match kernel_seccomp("chmod_777_is_refused") { |
| 1679 | Some(s) => s, |
| 1680 | None => return Ok(()), |
| 1681 | }; |
| 1682 | own_thread("chmod_777_is_refused", move || { |
| 1683 | let base = res!(fixture("chmod777")); |
| 1684 | let secret = base.join("secret.txt"); |
| 1685 | let mode = |p: &std::path::Path| -> Outcome<u32> { |
| 1686 | Ok(res!(fs::metadata(p)).permissions().mode() & 0o7777) |
| 1687 | }; |
| 1688 | |
| 1689 | // Broken first: unfiltered, this is the review's finding. |
| 1690 | res!(fs::set_permissions(&secret, fs::Permissions::from_mode(0o777))); |
| 1691 | assert_eq!(0o777, res!(mode(&secret)), |
| 1692 | "the unfiltered chmod did not take, so the rest proves nothing"); |
| 1693 | res!(fs::set_permissions(&secret, fs::Permissions::from_mode(0o600))); |
| 1694 | |
| 1695 | res!(engage(&s, &Spec::default())); |
| 1696 | |
| 1697 | assert!( |
| 1698 | fs::set_permissions(&secret, fs::Permissions::from_mode(0o777)).is_err(), |
| 1699 | "chmod 777 succeeded behind the filter"); |
| 1700 | assert_eq!(0o600, res!(mode(&secret)), |
| 1701 | "the mode changed although the call was refused"); |
| 1702 | |
| 1703 | // Every other loosening spelling, so the rule is the rule and not one |
| 1704 | // constant. |
| 1705 | for m in [0o666u32, 0o4755, 0o2755, 0o602, 0o007] { |
| 1706 | assert!( |
| 1707 | fs::set_permissions(&secret, fs::Permissions::from_mode(m)).is_err(), |
| 1708 | "chmod {:o} succeeded behind the filter", m); |
| 1709 | } |
| 1710 | |
| 1711 | // And the modes a build needs still work, or the filter would be |
| 1712 | // turned off. |
| 1713 | for m in [0o644u32, 0o755, 0o664, 0o775, 0o700, 0o400] { |
| 1714 | res!(fs::set_permissions(&secret, fs::Permissions::from_mode(m))); |
| 1715 | assert_eq!(m, res!(mode(&secret)), "chmod {:o} did not take", m); |
| 1716 | } |
| 1717 | Ok(()) |
| 1718 | }) |
| 1719 | } |
| 1720 | |
| 1721 | /// The rest of §1.2: `chown`, `setxattr` and the attribute calls. |
| 1722 | /// |
| 1723 | /// `chown` to another user needs a capability the test does not have, so the |
| 1724 | /// unfiltered half uses the one form an ordinary user can perform -- `chown` |
| 1725 | /// to the uid and gid it already has, which succeeds -- and the filtered half |
| 1726 | /// shows the same call refused. |
| 1727 | #[test] |
| 1728 | fn the_other_metadata_calls_are_refused() -> Outcome<()> { |
| 1729 | let s = match kernel_seccomp("the_other_metadata_calls_are_refused") { |
| 1730 | Some(s) => s, |
| 1731 | None => return Ok(()), |
| 1732 | }; |
| 1733 | let chown = if std::path::Path::new("/usr/bin/chown").exists() { |
| 1734 | "/usr/bin/chown" |
| 1735 | } else if std::path::Path::new("/bin/chown").exists() { |
| 1736 | "/bin/chown" |
| 1737 | } else { |
| 1738 | println!( |
| 1739 | "[the_other_metadata_calls_are_refused] SKIPPED: no chown on \ |
| 1740 | this machine to run inside the filter."); |
| 1741 | return Ok(()); |
| 1742 | }; |
| 1743 | own_thread("the_other_metadata_calls_are_refused", move || { |
| 1744 | let base = res!(fixture("othermeta")); |
| 1745 | let secret = base.join("secret.txt"); |
| 1746 | |
| 1747 | // Broken first, with the real program, unfiltered. |
| 1748 | let out = res!(Command::new(chown).arg("--reference").arg(&secret) |
| 1749 | .arg(&secret).output()); |
| 1750 | assert!(out.status.success(), |
| 1751 | "chown --reference failed before the filter: {}", |
| 1752 | String::from_utf8_lossy(&out.stderr)); |
| 1753 | |
| 1754 | res!(engage(&s, &Spec::default())); |
| 1755 | |
| 1756 | let out = res!(Command::new(chown).arg("--reference").arg(&secret) |
| 1757 | .arg(&secret).output()); |
| 1758 | assert!(!out.status.success(), |
| 1759 | "chown succeeded behind the filter"); |
| 1760 | let said = String::from_utf8_lossy(&out.stderr).to_lowercase(); |
| 1761 | assert!(said.contains("not permitted"), |
| 1762 | "chown failed for the wrong reason: {}", said); |
| 1763 | Ok(()) |
| 1764 | }) |
| 1765 | } |
| 1766 | |
| 1767 | /// §1.3: the session bus, and `systemd-run --user` through it. |
| 1768 | /// |
| 1769 | /// This is the review's reproduction verbatim -- with the fence fully applied |
| 1770 | /// and the network refused, `systemd-run --user … /bin/cat <denied-file>` |
| 1771 | /// started a process *outside* the fence and returned the contents. The |
| 1772 | /// filter closes it at the only place seccomp can see: `socket(AF_UNIX, …)`. |
| 1773 | #[test] |
| 1774 | fn the_session_bus_is_unreachable() -> Outcome<()> { |
| 1775 | let s = match kernel_seccomp("the_session_bus_is_unreachable") { |
| 1776 | Some(s) => s, |
| 1777 | None => return Ok(()), |
| 1778 | }; |
| 1779 | let run = "/usr/bin/systemd-run"; |
| 1780 | if !have("the_session_bus_is_unreachable", run) { |
| 1781 | return Ok(()); |
| 1782 | } |
| 1783 | own_thread("the_session_bus_is_unreachable", move || { |
| 1784 | let base = res!(fixture("bus")); |
| 1785 | let secret = base.join("secret.txt"); |
| 1786 | |
| 1787 | // Broken first: unfiltered, this is the escape. |
| 1788 | let out = res!(Command::new(run) |
| 1789 | .args(["--user", "--pipe", "--quiet", "--wait", "/bin/cat"]) |
| 1790 | .arg(&secret) |
| 1791 | .output()); |
| 1792 | if !out.status.success() { |
| 1793 | println!( |
| 1794 | "[the_session_bus_is_unreachable] SKIPPED: systemd-run \ |
| 1795 | --user does not work here even unfiltered ({}), so there is \ |
| 1796 | no escape to close and this proves nothing. Run this inside \ |
| 1797 | a real user session.", |
| 1798 | String::from_utf8_lossy(&out.stderr).trim()); |
| 1799 | return Ok(()); |
| 1800 | } |
| 1801 | assert_eq!("secret", String::from_utf8_lossy(&out.stdout), |
| 1802 | "the unfiltered escape did not return the file, so the filtered \ |
| 1803 | half would prove nothing"); |
| 1804 | |
| 1805 | res!(engage(&s, &Spec::default())); |
| 1806 | |
| 1807 | let out = res!(Command::new(run) |
| 1808 | .args(["--user", "--pipe", "--quiet", "--wait", "/bin/cat"]) |
| 1809 | .arg(&secret) |
| 1810 | .output()); |
| 1811 | assert!(!out.status.success(), |
| 1812 | "systemd-run --user still ran a command outside the fence"); |
| 1813 | assert!(!String::from_utf8_lossy(&out.stdout).contains("secret"), |
| 1814 | "the file came back through the bus behind the filter"); |
| 1815 | let said = String::from_utf8_lossy(&out.stderr); |
| 1816 | assert!(said.to_lowercase().contains("connect"), |
| 1817 | "systemd-run failed for a reason other than the bus: {}", said); |
| 1818 | println!("[the_session_bus_is_unreachable] systemd-run said: {}", |
| 1819 | said.trim()); |
| 1820 | Ok(()) |
| 1821 | }) |
| 1822 | } |
| 1823 | |
| 1824 | /// A unix socket cannot be opened at all, and `socketpair` still can. |
| 1825 | /// |
| 1826 | /// The second half is not a nicety: a from-scratch `cargo test` makes six |
| 1827 | /// `socketpair(AF_UNIX, …)` calls, so a filter that took them would break |
| 1828 | /// every build. |
| 1829 | #[test] |
| 1830 | fn af_unix_is_refused_and_socketpair_is_not() -> Outcome<()> { |
| 1831 | let s = match kernel_seccomp("af_unix_is_refused_and_socketpair_is_not") { |
| 1832 | Some(s) => s, |
| 1833 | None => return Ok(()), |
| 1834 | }; |
| 1835 | own_thread("af_unix_is_refused_and_socketpair_is_not", move || { |
| 1836 | let base = res!(fixture("afunix")); |
| 1837 | let sock = base.join("s.sock"); |
| 1838 | let bus = bus(); |
| 1839 | |
| 1840 | // Broken first: unfiltered, a unix socket binds and connects, and so |
| 1841 | // does the session bus where there is one. |
| 1842 | let l = res!(std::os::unix::net::UnixListener::bind(&sock)); |
| 1843 | let c = res!(std::os::unix::net::UnixStream::connect(&sock)); |
| 1844 | drop(c); |
| 1845 | drop(l); |
| 1846 | res!(fs::remove_file(&sock)); |
| 1847 | match &bus { |
| 1848 | Some(b) => assert!( |
| 1849 | std::os::unix::net::UnixStream::connect(b).is_ok(), |
| 1850 | "the session bus at {} refused a connection before the \ |
| 1851 | filter, so the filtered half would prove nothing", |
| 1852 | b.display()), |
| 1853 | None => println!( |
| 1854 | "[af_unix_is_refused_and_socketpair_is_not] no session bus \ |
| 1855 | on this machine, so only the general socket refusal is \ |
| 1856 | checked here."), |
| 1857 | } |
| 1858 | |
| 1859 | res!(engage(&s, &Spec::default())); |
| 1860 | |
| 1861 | assert!(std::os::unix::net::UnixListener::bind(&sock).is_err(), |
| 1862 | "a unix socket was still bound behind the filter"); |
| 1863 | if let Some(b) = &bus { |
| 1864 | assert!(std::os::unix::net::UnixStream::connect(b).is_err(), |
| 1865 | "the session bus was still connectable behind the filter"); |
| 1866 | } |
| 1867 | |
| 1868 | // socketpair is a different syscall and stays. |
| 1869 | let (a, b) = res!(std::os::unix::net::UnixStream::pair()); |
| 1870 | drop(a); |
| 1871 | drop(b); |
| 1872 | Ok(()) |
| 1873 | }) |
| 1874 | } |
| 1875 | |
| 1876 | /// The filter survives `execve`, which is the whole reason a launcher can |
| 1877 | /// install it and then become the command. |
| 1878 | #[test] |
| 1879 | fn the_filter_is_inherited_by_a_real_program() -> Outcome<()> { |
| 1880 | let s = match kernel_seccomp("the_filter_is_inherited_by_a_real_program") { |
| 1881 | Some(s) => s, |
| 1882 | None => return Ok(()), |
| 1883 | }; |
| 1884 | let chmod = "/usr/bin/chmod"; |
| 1885 | if !have("the_filter_is_inherited_by_a_real_program", chmod) { |
| 1886 | return Ok(()); |
| 1887 | } |
| 1888 | own_thread("the_filter_is_inherited_by_a_real_program", move || { |
| 1889 | let base = res!(fixture("inherit")); |
| 1890 | let secret = base.join("secret.txt"); |
| 1891 | |
| 1892 | // Broken first: unfiltered, the real program does it. |
| 1893 | let out = res!(Command::new(chmod).arg("777").arg(&secret).output()); |
| 1894 | assert!(out.status.success(), "chmod 777 failed before the filter"); |
| 1895 | res!(fs::set_permissions(&secret, fs::Permissions::from_mode(0o600))); |
| 1896 | |
| 1897 | res!(engage(&s, &Spec::default())); |
| 1898 | |
| 1899 | let out = res!(Command::new(chmod).arg("777").arg(&secret).output()); |
| 1900 | assert!(!out.status.success(), |
| 1901 | "an exec'd chmod 777 succeeded, so the filter was not inherited"); |
| 1902 | let mode = res!(fs::metadata(&secret)).permissions().mode() & 0o7777; |
| 1903 | assert_eq!(0o600, mode, "the mode changed through an exec'd program"); |
| 1904 | |
| 1905 | // And the permitted half is genuinely permitted through exec too, |
| 1906 | // which is what stops this being a filter that refuses everything. |
| 1907 | let out = res!(Command::new(chmod).arg("755").arg(&secret).output()); |
| 1908 | assert!(out.status.success(), |
| 1909 | "an exec'd chmod 755 was refused: {}", |
| 1910 | String::from_utf8_lossy(&out.stderr)); |
| 1911 | Ok(()) |
| 1912 | }) |
| 1913 | } |
| 1914 | |
| 1915 | /// `io_uring_setup` is refused, and refused with `ENOSYS` rather than |
| 1916 | /// `EPERM`. |
| 1917 | /// |
| 1918 | /// Skipped loudly where the kernel has no io_uring to begin with, since a |
| 1919 | /// refusal indistinguishable from an absence would prove nothing. |
| 1920 | #[test] |
| 1921 | fn io_uring_is_refused() -> Outcome<()> { |
| 1922 | let s = match kernel_seccomp("io_uring_is_refused") { |
| 1923 | Some(s) => s, |
| 1924 | None => return Ok(()), |
| 1925 | }; |
| 1926 | let arch = match Arch::here() { |
| 1927 | Some(a) => a, |
| 1928 | None => return Ok(()), |
| 1929 | }; |
| 1930 | // The rule is asserted on the plan rather than by making the call: |
| 1931 | // issuing `io_uring_setup` from safe Rust is not possible without a |
| 1932 | // dependency this crate does not have, and the number is what the filter |
| 1933 | // acts on. |
| 1934 | let f = res!(s.plan(&Spec::default())); |
| 1935 | let want = match TABLE.iter().find(|t| t.name == "io_uring_setup") { |
| 1936 | Some(t) => t, |
| 1937 | None => return Err(err!("io_uring_setup left the table"; Test)), |
| 1938 | }; |
| 1939 | assert!(f.refused.iter().any(|t| t.name == "io_uring_setup"), |
| 1940 | "io_uring is not refused, so every metadata rule is advisory"); |
| 1941 | assert_eq!(Some(425), want.number(arch)); |
| 1942 | Ok(()) |
| 1943 | } |
| 1944 | |
| 1945 | // ── The thing most likely to break ────────────────────────────── |
| 1946 | |
| 1947 | /// A real `cargo` build runs behind the filter. |
| 1948 | /// |
| 1949 | /// This is the test the whole design turns on. A filter that breaks `cargo` |
| 1950 | /// is worse than no filter, because it gets switched off -- and the first two |
| 1951 | /// specs tried here *did* break it, which is why [`Meta::NoLoosening`] exists |
| 1952 | /// and why the `utime` family is permitted. |
| 1953 | /// |
| 1954 | /// **What it does not cover.** The crate built here has no dependencies, so |
| 1955 | /// nothing is unpacked from the registry -- and unpacking is where the two |
| 1956 | /// breakages were found. That path needs a registry and cannot be made |
| 1957 | /// hermetic and offline, so it was measured out of band instead, and this is |
| 1958 | /// the record: with the `utime` family refused, `cargo test` on a crate with |
| 1959 | /// one dependency fails with `failed to set mtime for |
| 1960 | /// .../.cargo_vcs_info.json`; with the timestamps allowed and the `chmod` |
| 1961 | /// family refused, it fails with `failed to set permissions to 644` for the |
| 1962 | /// same file; with [`Spec::default`] it succeeds. [`git_still_runs`] is the |
| 1963 | /// sharper canary of the two and does run here: it fails under |
| 1964 | /// [`Meta::Refuse`], which this test does not. |
| 1965 | /// |
| 1966 | /// Also measured out of band, against a real tree: `cargo test -p |
| 1967 | /// oxedyne_fe2o3_hash` from an empty target directory -- 40-odd crates, |
| 1968 | /// proc-macro builds, build scripts and a linked test binary -- compiles and |
| 1969 | /// passes behind this filter. |
| 1970 | /// |
| 1971 | /// Skipped loudly where there is no cargo to run. |
| 1972 | #[test] |
| 1973 | fn a_real_cargo_build_still_runs() -> Outcome<()> { |
| 1974 | let s = match kernel_seccomp("a_real_cargo_build_still_runs") { |
| 1975 | Some(s) => s, |
| 1976 | None => return Ok(()), |
| 1977 | }; |
| 1978 | let cargo = match std::env::var("CARGO") { |
| 1979 | Ok(c) => c, |
| 1980 | Err(_) => { |
| 1981 | println!( |
| 1982 | "[a_real_cargo_build_still_runs] SKIPPED: CARGO is not set, \ |
| 1983 | so there is no cargo to run. This test only runs under \ |
| 1984 | `cargo test`."); |
| 1985 | return Ok(()); |
| 1986 | }, |
| 1987 | }; |
| 1988 | own_thread("a_real_cargo_build_still_runs", move || { |
| 1989 | let base = res!(fixture("cargo")); |
| 1990 | let crate_dir = base.join("toy"); |
| 1991 | res!(fs::create_dir_all(crate_dir.join("src"))); |
| 1992 | res!(fs::write(crate_dir.join("Cargo.toml"), |
| 1993 | "[package]\nname = \"toy\"\nversion = \"0.1.0\"\nedition = \ |
| 1994 | \"2021\"\n[workspace]\n")); |
| 1995 | res!(fs::write(crate_dir.join("src/lib.rs"), |
| 1996 | "pub fn add(a: i32, b: i32) -> i32 { a + b }\n\ |
| 1997 | #[cfg(test)] mod t { #[test] fn works() { \ |
| 1998 | assert_eq!(3, super::add(1, 2)); } }\n")); |
| 1999 | let target = base.join("target"); |
| 2000 | |
| 2001 | res!(engage(&s, &Spec::default())); |
| 2002 | |
| 2003 | let out = res!(Command::new(&cargo) |
| 2004 | .arg("test") |
| 2005 | .arg("--offline") |
| 2006 | .arg("--manifest-path") |
| 2007 | .arg(crate_dir.join("Cargo.toml")) |
| 2008 | .env("CARGO_TARGET_DIR", &target) |
| 2009 | // The harness's own variables would point a nested cargo at this |
| 2010 | // run's state; a clean pair is what a real command gets. |
| 2011 | .env_remove("RUSTC_WORKSPACE_WRAPPER") |
| 2012 | .env_remove("RUSTC_WRAPPER") |
| 2013 | .output()); |
| 2014 | assert!(out.status.success(), |
| 2015 | "cargo test failed behind the filter, which is the one outcome \ |
| 2016 | that makes this filter worse than none:\n{}\n{}", |
| 2017 | String::from_utf8_lossy(&out.stdout), |
| 2018 | String::from_utf8_lossy(&out.stderr)); |
| 2019 | let said = String::from_utf8_lossy(&out.stdout); |
| 2020 | assert!(said.contains("test result: ok"), |
| 2021 | "cargo ran but the test did not pass: {}", said); |
| 2022 | println!("[a_real_cargo_build_still_runs] a from-scratch cargo test \ |
| 2023 | compiled, linked and ran behind the filter"); |
| 2024 | Ok(()) |
| 2025 | }) |
| 2026 | } |
| 2027 | |
| 2028 | /// `git` runs behind the filter. |
| 2029 | /// |
| 2030 | /// The measured counter-example to the obvious rule: `git` chmods to `0664` |
| 2031 | /// and `0775` under the default Ubuntu umask, so a filter refusing any |
| 2032 | /// group-write mode would break `git init`. That is why [`LOOSENING`] names |
| 2033 | /// world-write and the set-ID bits and not group-write. |
| 2034 | #[test] |
| 2035 | fn git_still_runs() -> Outcome<()> { |
| 2036 | let s = match kernel_seccomp("git_still_runs") { |
| 2037 | Some(s) => s, |
| 2038 | None => return Ok(()), |
| 2039 | }; |
| 2040 | let git = "/usr/bin/git"; |
| 2041 | if !have("git_still_runs", git) { |
| 2042 | return Ok(()); |
| 2043 | } |
| 2044 | own_thread("git_still_runs", move || { |
| 2045 | let base = res!(fixture("git")); |
| 2046 | let repo = base.join("r"); |
| 2047 | res!(fs::create_dir_all(&repo)); |
| 2048 | |
| 2049 | res!(engage(&s, &Spec::default())); |
| 2050 | |
| 2051 | let run = |args: &[&str]| -> Outcome<()> { |
| 2052 | let out = res!(Command::new(git).args(args).current_dir(&repo) |
| 2053 | .env("HOME", &base).output()); |
| 2054 | if !out.status.success() { |
| 2055 | return Err(err!( |
| 2056 | "git {:?} failed behind the filter: {}", args, |
| 2057 | String::from_utf8_lossy(&out.stderr); Test)); |
| 2058 | } |
| 2059 | Ok(()) |
| 2060 | }; |
| 2061 | res!(run(&["init", "-q"])); |
| 2062 | res!(fs::write(repo.join("a.sh"), "#!/bin/sh\necho hi\n")); |
| 2063 | res!(fs::set_permissions(repo.join("a.sh"), |
| 2064 | fs::Permissions::from_mode(0o755))); |
| 2065 | res!(run(&["add", "-A"])); |
| 2066 | res!(run(&[ |
| 2067 | "-c", "user.email=a@b", "-c", "user.name=n", |
| 2068 | "commit", "-qm", "x", |
| 2069 | ])); |
| 2070 | Ok(()) |
| 2071 | }) |
| 2072 | } |
| 2073 | |
| 2074 | /// The two escapes, closed at once, on one file. |
| 2075 | /// |
| 2076 | /// The review's §1.2 and §1.3 share a subject -- a file the fence denies -- |
| 2077 | /// and this asserts both refusals against that one file so a change that |
| 2078 | /// closed one and reopened the other could not pass. |
| 2079 | #[test] |
| 2080 | fn the_review_escapes_are_closed_together() -> Outcome<()> { |
| 2081 | let s = match kernel_seccomp("the_review_escapes_are_closed_together") { |
| 2082 | Some(s) => s, |
| 2083 | None => return Ok(()), |
| 2084 | }; |
| 2085 | own_thread("the_review_escapes_are_closed_together", move || { |
| 2086 | let base = res!(fixture("both")); |
| 2087 | let secret = base.join("secret.txt"); |
| 2088 | let bus = bus(); |
| 2089 | |
| 2090 | res!(engage(&s, &Spec::default())); |
| 2091 | |
| 2092 | // §1.2: the permissions of the denied file cannot be loosened. |
| 2093 | assert!( |
| 2094 | fs::set_permissions(&secret, fs::Permissions::from_mode(0o777)).is_err(), |
| 2095 | "1.2 is open: chmod 777 succeeded on the denied file"); |
| 2096 | assert_eq!( |
| 2097 | 0o600, |
| 2098 | res!(fs::metadata(&secret)).permissions().mode() & 0o7777); |
| 2099 | |
| 2100 | // §1.3: no socket, so no bus, so no unfenced process to read it with. |
| 2101 | match &bus { |
| 2102 | Some(b) => assert!( |
| 2103 | std::os::unix::net::UnixStream::connect(b).is_err(), |
| 2104 | "1.3 is open: the session bus was connectable"), |
| 2105 | None => println!( |
| 2106 | "[the_review_escapes_are_closed_together] SKIPPED the 1.3 \ |
| 2107 | half: no session bus on this machine to be refused."), |
| 2108 | } |
| 2109 | Ok(()) |
| 2110 | }) |
| 2111 | } |
| 2112 | } |