//! The unix socket this daemon listens on, and who is allowed to reach it. //! //! Four of the five guarantees below are lifted from //! `crates/didbot-serve/src/estop_admin.rs`, which shipped them after an //! exposure review found its own socket landing world-writable at a //! predictable path. Repeating that mistake here would be worse: what is //! behind e-stop's socket stops a server, and what is behind this one issues //! credentials. //! //! The fifth is new. E-stop's commands are uniform and the socket's //! permissions decide everything; this socket hands a different answer to //! each caller, so it reads the connecting process's credentials rather than //! believing what the connection says about itself. //! //! # What reaching this socket means, and what it does not //! //! Reaching it **is** the authorization. `plan/cred-delivery.md` publishes //! this path through `CLAUDE_ENV_FILE`, which any shell command the model //! runs can print — so the path is not a secret and nothing here treats it //! as one. E-stop deleted its own operator secret for exactly this reason: //! the same caller could read it off the same filesystem, and the socket's //! permissions were doing the real work. Anything later carried in that //! environment file selects which context is calling. It does not decide //! whether the caller is allowed to. //! //! The ceiling is therefore the user account, as `docs/trust-model.md` //! already states: anything running as this user can connect, and this //! design defends against a credential leaving the machine rather than //! against one being misused on it. use std::io; use std::os::unix::fs::{MetadataExt, PermissionsExt}; use std::path::{Path, PathBuf}; use tokio::net::{UnixListener, UnixStream}; use tracing::warn; /// The directory this daemon creates and owns, under the runtime directory. const DIR_NAME: &str = "didbot-agent"; /// The socket file inside it. const SOCKET_NAME: &str = "agent.sock"; /// The credentials of a connected process, as the kernel reports them. /// /// Not as the connection claims them: these come from the socket itself, so /// a caller cannot choose what they say. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct Peer { /// The connecting process's effective user. pub uid: u32, /// Its effective group. pub gid: u32, /// Its process id, where the platform reports one. Useful for finding /// which session a caller belongs to, and useless as a check on its own: /// a process id is reused once the process is gone. pub pid: Option, } /// Where the socket lives when a deployment has not said otherwise. /// /// `XDG_RUNTIME_DIR` when it is set — a per-user directory the system clears /// on logout — and the temporary directory otherwise. Either way the socket /// sits inside a subdirectory this process creates and owns, never directly /// at the base: a predictable path stops being a weakness once the directory /// holding it cannot be pre-empted, and that is what [`Listener::bind`] /// enforces. pub fn default_socket_path() -> PathBuf { std::env::var_os("XDG_RUNTIME_DIR") .map(PathBuf::from) .unwrap_or_else(std::env::temp_dir) .join(DIR_NAME) .join(SOCKET_NAME) } /// A listening socket that only yields connections from this user. #[derive(Debug)] pub struct Listener { inner: UnixListener, owner: u32, path: PathBuf, } impl Listener { /// Binds `path`, hardening the directory and the socket file. /// /// The directory is created at `0700`, or tightened to `0700` if it is /// already there — which fails rather than proceeding when this process /// does not own it, and that refusal is the point. An attacker who /// pre-creates the directory is caught here, and one who tries to bind /// first inside a directory they do not own never gets the chance. /// /// The socket file is then set to `0600` explicitly, rather than left to /// whatever `umask` this process happens to be running under. pub fn bind(path: impl Into) -> io::Result { let path = path.into(); if let Some(dir) = path.parent().filter(|dir| !dir.as_os_str().is_empty()) { secure_dir(dir)?; } let inner = UnixListener::bind(&path)?; std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o600))?; // The socket this process just created is owned by this process, so // its own metadata answers "who am I" without a second way to ask. let owner = std::fs::metadata(&path)?.uid(); Ok(Self { inner, owner, path }) } /// The user this socket belongs to. Every accepted peer matches it. pub fn owner(&self) -> u32 { self.owner } /// The path bound. pub fn path(&self) -> &Path { &self.path } /// Accepts the next connection from this user, dropping any other. /// /// A peer from another user should not be able to reach a `0600` socket /// in a `0700` directory at all. It is checked anyway, because the cost /// is one syscall and the alternative is that a deployment which relaxes /// either permission by accident relaxes credential issuance with it. pub async fn accept(&self) -> io::Result<(UnixStream, Peer)> { loop { let (stream, _) = self.inner.accept().await?; let peer = match peer_of(&stream) { Ok(peer) => peer, Err(error) => { // A connection whose credentials cannot be read is not // one to serve on the assumption that it is probably // fine. warn!(%error, "dropped a connection whose peer credentials could not be read"); continue; } }; if peer.uid != self.owner { warn!( peer_uid = peer.uid, owner_uid = self.owner, "dropped a connection from another user" ); continue; } return Ok((stream, peer)); } } } /// Reads the connected process's credentials off the socket. pub fn peer_of(stream: &UnixStream) -> io::Result { let cred = stream.peer_cred()?; Ok(Peer { uid: cred.uid(), gid: cred.gid(), pid: cred.pid(), }) } /// Removes `path` when, and only when, nothing is listening on it. /// /// Best effort and silent on any doubt. Skipping a socket that should have /// been removed costs a clear "address in use" from [`Listener::bind`]; /// removing one a live process still holds takes the socket away from a /// running daemon, which then serves a path nothing can reach while a second /// process binds a fresh one and quietly takes over credential issuance. pub fn remove_if_stale(path: &Path) { use std::os::unix::fs::FileTypeExt; // `symlink_metadata`, not `exists`: a dangling symlink here is not a // socket this daemon left behind, and `Path::exists` answers `false` for // any stat error at all, which is the conflation this function exists to // avoid. let Ok(metadata) = std::fs::symlink_metadata(path) else { return; }; if !metadata.file_type().is_socket() { return; } // Only `ECONNREFUSED` means nothing is listening. Every other failure — // `EACCES` for a socket belonging to somebody else, `EAGAIN` or // `ENOBUFS` when a live backlog is full, `EMFILE` when this process is // out of descriptors — means a listener may well be there. match std::os::unix::net::UnixStream::connect(path) { Err(error) if error.kind() == io::ErrorKind::ConnectionRefused => { let _ = std::fs::remove_file(path); } _ => {} } } /// Creates `dir` at `0700`, or tightens an existing one to `0700`. /// /// Tightening fails when this process does not own the directory, which is /// the check that closes the pre-creation race. The state directory holding /// the node key gets the same treatment, from the same function. pub(crate) fn secure_dir(dir: &Path) -> io::Result<()> { if dir.is_dir() { return std::fs::set_permissions(dir, std::fs::Permissions::from_mode(0o700)); } use std::os::unix::fs::DirBuilderExt; std::fs::DirBuilder::new() .recursive(true) .mode(0o700) .create(dir) } #[cfg(test)] mod tests { use super::*; use crate::scratch::Scratch; impl Scratch { fn socket(&self) -> PathBuf { self.0.join(SOCKET_NAME) } } fn mode_of(path: &Path) -> u32 { std::fs::metadata(path).unwrap().permissions().mode() & 0o777 } #[tokio::test] async fn binding_leaves_the_directory_and_the_socket_shut() { let scratch = Scratch::new("shut"); let listener = Listener::bind(scratch.socket()).unwrap(); assert_eq!(mode_of(&scratch.0), 0o700, "directory"); assert_eq!(mode_of(listener.path()), 0o600, "socket"); } #[tokio::test] async fn binding_tightens_a_directory_that_is_already_open() { let scratch = Scratch::new("tighten"); std::fs::create_dir_all(&scratch.0).unwrap(); std::fs::set_permissions(&scratch.0, std::fs::Permissions::from_mode(0o777)).unwrap(); let _listener = Listener::bind(scratch.socket()).unwrap(); assert_eq!(mode_of(&scratch.0), 0o700); } #[tokio::test] async fn an_accepted_peer_is_this_process() { let scratch = Scratch::new("peer"); let listener = Listener::bind(scratch.socket()).unwrap(); let path = listener.path().to_path_buf(); let connect = tokio::spawn(async move { UnixStream::connect(&path).await.unwrap() }); let (_stream, peer) = listener.accept().await.unwrap(); let _client = connect.await.unwrap(); assert_eq!(peer.uid, listener.owner()); assert_eq!( peer.pid, Some(std::process::id() as i32), "the connecting process is this one" ); } #[tokio::test] async fn a_socket_nothing_listens_on_is_removed() { let scratch = Scratch::new("stale"); let socket = scratch.socket(); drop(Listener::bind(&socket).unwrap()); remove_if_stale(&socket); assert!(!socket.exists(), "a dead socket is cleared out of the way"); } #[tokio::test] async fn a_socket_something_listens_on_is_left_alone() { let scratch = Scratch::new("live"); let listener = Listener::bind(scratch.socket()).unwrap(); remove_if_stale(listener.path()); assert!( listener.path().exists(), "a live socket is never taken from under its listener" ); } #[tokio::test] async fn something_that_is_not_a_socket_is_left_alone() { let scratch = Scratch::new("notasocket"); std::fs::create_dir_all(&scratch.0).unwrap(); let file = scratch.0.join("ordinary"); std::fs::write(&file, b"").unwrap(); remove_if_stale(&file); assert!(file.exists()); } }