diff --git a/docs/wayland-port.md b/docs/wayland-port.md index 6146a19c3..396edb1a7 100644 --- a/docs/wayland-port.md +++ b/docs/wayland-port.md @@ -14024,3 +14024,34 @@ that in place the whole round trip closes: type `2`, commit, the model takes `14 Two halves of one class, found one after the other: 0028 parses and 0029 displays. A formatter that can do neither looks like a text field that ignores the keyboard and then like a field that shows nonsense, and neither symptom names the formatter. + +### Two cider shell calls in a row, and the check that tested the wrong syscall + +`scripts/run-tests.nu` could not run a single suite here: after the first `cider shell` every later +one died with `Cannot join mnt namespace of pid NNNNN`, and since the runner spends its first call on +a preflight probe, the compile always failed and it blamed the toolchain. + +**A rootless container lives in a user namespace of its own**, and `setns(CLONE_NEWNS)` demands +CAP_SYS_ADMIN in the user namespace that OWNS that mount namespace. The invocation that created the +container has it; a later invocation does not. `nsenter` says the same thing in three lines: + + nsenter -t PID -m true reassociate to namespaces failed: EPERM + nsenter -t PID -U -m true gets past the join, fails later in setgroups + nsenter -t PID -U --preserve-credentials -m ls / lists the guest root + +So the launcher joins the user namespace first when the container has one of its own, exactly as +`nsenter -U -m` does. Six sequential `cider shell echo ok` calls now answer `ok` in about a fifth of a +second each, where the second used to fail outright. + +**And the guard that was supposed to catch this tested the wrong thing.** `container_joinable` OPENED +`/proc//ns/mnt` and answered yes if that worked. Opening succeeds for any live process this user +can see; `setns` is what fails, so a container that could not be entered was declared joinable, the +reap-and-restart path above it was skipped, and the join a few lines later died. **The only honest +test of a syscall is the syscall** - it now forks a child, tries the real join, and reads the exit +status, because `setns` cannot be undone in the calling process. + +One more macOS divergence fell out of it: **`/tmp` is a symlink to `private/tmp` on macOS and was a +real directory or missing here**, while the runtime sets `TMPDIR=/private/tmp`. Anything writing to +the literal `/tmp` in the guest went somewhere else than everything that asks the system for a +temporary directory. The launcher now makes that symlink when `/tmp` is absent or empty, and leaves a +populated one alone. diff --git a/src/linux/launcher/src/main.rs b/src/linux/launcher/src/main.rs index 39e8d3157..9b833f633 100644 --- a/src/linux/launcher/src/main.rs +++ b/src/linux/launcher/src/main.rs @@ -164,7 +164,11 @@ fn main() { } // Join the container mnt ns so we can connect to the overlay-resident socket - // (cider.c:285-287, the Linux 4.11 / overlayfs socket hack). + // (cider.c:285-287, the Linux 4.11 / overlayfs socket hack), through its user ns when it has + // one of its own: see try_enter_container for why the order matters. + if !same_namespace(pid_init, "user") { + join_namespace(pid_init, libc::CLONE_NEWUSER, "user"); + } join_namespace(pid_init, libc::CLONE_NEWNS, "mnt"); // Drop euid (cider.c:289; no-op rootless). @@ -338,6 +342,41 @@ fn ensure_prefix_dirs(ctx: &Ctx) { ] { create_dir(&format!("{}{}", ctx.prefix, d)); } + + /* + * /tmp IS A SYMLINK ON macOS, and here it was a directory or nothing at all. + * + * The runtime sets TMPDIR to /private/tmp and creates it, so anything that asks the system for a + * temporary directory lands there, while anything that writes to the literal /tmp lands in a + * different place or fails. scripts/run-tests.nu copies its sources to /private/tmp and + * compiles them at /tmp, which is the same path on macOS and two places here: cd said No such + * file or directory and the harness reported that the toolchain was broken. + * + * Only an ABSENT or EMPTY /tmp is replaced. A prefix whose /tmp already holds files keeps it, + * because moving a running container's temporary files is not this function's business. /var and + * /etc are symlinks on macOS too and are left as real directories here deliberately: they hold + * runtime state already and nothing has asked for that yet. + */ + ensure_private_symlink(&ctx.prefix, "/tmp", "private/tmp"); +} + +fn ensure_private_symlink(prefix: &str, at: &str, target: &str) { + let path = format!("{prefix}{at}"); + + match std::fs::symlink_metadata(&path) { + Ok(md) if md.file_type().is_symlink() => return, + Ok(md) if md.is_dir() => { + let empty = std::fs::read_dir(&path).map(|mut d| d.next().is_none()).unwrap_or(false); + + if !empty { + return; + } + let _ = std::fs::remove_dir(&path); + } + Ok(_) => return, + Err(_) => {} + } + let _ = std::os::unix::fs::symlink(target, &path); } fn setup_prefix(ctx: &Ctx) { @@ -443,15 +482,64 @@ fn status_matches_ids(pid: i32, uid: u32, gid: u32) -> bool { uok && gok } +/// Whether a surviving container can actually be ENTERED, which is not the same question as +/// whether its namespace files can be opened. +/// +/// This used to open /proc//ns/mnt and answer yes if that succeeded. Opening succeeds for any +/// live process this user can see; setns is what fails. So a container left by a previous +/// invocation was declared joinable, the reap-and-restart path above was skipped, and the join a +/// few lines later died with "Cannot join mnt namespace of pid N". THE ONLY HONEST TEST OF A +/// SYSCALL IS THE SYSCALL, and setns cannot be undone in this process, so a forked child does it +/// and reports by exit status. fn container_joinable(pid: i32) -> bool { - let p = cstr(&format!("/proc/{pid}/ns/mnt")); + let child = unsafe { libc::fork() }; + + if child < 0 { + return false; + } + if child == 0 { + let ok = try_enter_container(pid); + unsafe { libc::_exit(if ok { 0 } else { 1 }) }; + } + + let mut status: c_int = 0; + if unsafe { libc::waitpid(child, &mut status, 0) } != child { + return false; + } + libc::WIFEXITED(status) && libc::WEXITSTATUS(status) == 0 +} + +/// The two setns calls a caller needs, in the order the kernel requires. +/// +/// A rootless container lives in a user namespace of its own, and setns(CLONE_NEWNS) demands +/// CAP_SYS_ADMIN in the user namespace that OWNS the mount namespace. The invocation that created +/// the container has that; a later one does not, which is why a second cider shell in the same +/// prefix used to fail while the first succeeded. Joining the user namespace first grants those +/// capabilities, exactly as nsenter -U -m does. +fn try_enter_container(pid: i32) -> bool { + if !same_namespace(pid, "user") && !setns_path(pid, "user", libc::CLONE_NEWUSER) { + return false; + } + setns_path(pid, "mnt", libc::CLONE_NEWNS) +} + +fn same_namespace(pid: i32, name: &str) -> bool { + let mine = std::fs::read_link(format!("/proc/self/ns/{name}")).ok(); + let theirs = std::fs::read_link(format!("/proc/{pid}/ns/{name}")).ok(); + + mine.is_some() && mine == theirs +} + +fn setns_path(pid: i32, name: &str, nstype: c_int) -> bool { + let p = cstr(&format!("/proc/{pid}/ns/{name}")); let fd = unsafe { libc::open(p.as_ptr(), libc::O_RDONLY) }; - if fd >= 0 { - unsafe { libc::close(fd) }; - true - } else { - false + + if fd < 0 { + return false; } + let ok = unsafe { libc::setns(fd, nstype) } == 0; + unsafe { libc::close(fd) }; + ok } fn join_namespace(pid: i32, nstype: c_int, name: &str) {