diff --git a/.github/workflows/e2e-cache.yml b/.github/workflows/e2e-cache.yml new file mode 100644 index 0000000..b77766c --- /dev/null +++ b/.github/workflows/e2e-cache.yml @@ -0,0 +1,201 @@ +name: "e2e (cache: disk + S3/MinIO)" + +# Boots a real OCI microVM under KVM and validates `bsdkrun cache` end to end: +# save a guest directory and restore it byte for byte in every archive format +# (gzip / zstd / estargz / none), against both backends. +# +# The S3 half runs against a real MinIO rather than a mock. +# SigV4 either produces a request a server accepts or it does not, and when it +# does not the answer is a flat 403 that names none of the five derivation +# steps — so the only signing test worth having is one that talks to something +# that verifies signatures. It also pins the behaviour that broke on first +# contact with a real bucket: the 404 from the "is this key already cached?" +# probe is an answer, not an error. +# +# Only GitHub's x86_64 runners expose /dev/kvm (the arm64 runners don't), so CI +# runs amd64 here; arm64 (macOS Hypervisor.framework) is validated on a dev +# machine. + +on: + workflow_dispatch: {} + pull_request: + paths: + - "core/**" + - "agent/**" + - "src/**" + - "build.rs" + - "tests/e2e_cache.sh" + - "tests/estargz_interop.sh" + - ".github/workflows/e2e-cache.yml" + push: + branches: + - "main" + paths: + - "core/**" + - "agent/**" + - "src/**" + - "build.rs" + - "tests/e2e_cache.sh" + - "tests/estargz_interop.sh" + - ".github/workflows/e2e-cache.yml" + +permissions: + contents: read + +jobs: + cache: + name: x86_64 + runs-on: ubuntu-latest # x86_64; has /dev/kvm + timeout-minutes: 30 # never hang the runner if a guest doesn't power off + + steps: + # `submodules` for library/solo5: core/build.rs builds the Solo5 tender + # from it, and without it the build merely warns — so the binary under + # test would quietly differ from the one that ships. + - uses: actions/checkout@v4 + with: + submodules: recursive + + - name: Enable KVM + run: | + set -eux + test -e /dev/kvm || { echo "no /dev/kvm on this runner"; exit 1; } + echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \ + | sudo tee /etc/udev/rules.d/99-kvm4all.rules + sudo udevadm control --reload-rules + sudo udevadm trigger --name-match=kvm + sudo chmod 0666 /dev/kvm + ls -l /dev/kvm + + - name: Install build deps (incl. libkrun/libkrunfw build requirements) + run: | + set -eux + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + build-essential git curl ca-certificates xz-utils python3 \ + python3-pyelftools pkg-config patchelf patch cpio kmod rsync zstd \ + flex bison bc libelf-dev libssl-dev \ + clang libclang-dev llvm-dev musl-tools libseccomp-dev + + - name: Install gvproxy (user-mode networking) + run: | + set -eux + sudo curl -fL -o /usr/local/bin/gvproxy \ + https://github.com/containers/gvisor-tap-vsock/releases/latest/download/gvproxy-linux-amd64 + sudo chmod +x /usr/local/bin/gvproxy + gvproxy --help >/dev/null 2>&1 || which gvproxy + + - uses: dtolnay/rust-toolchain@stable + with: + targets: x86_64-unknown-linux-musl + - uses: Swatinem/rust-cache@v2 + + # Same cache key as e2e-linux.yml / e2e-linux-disk.yml so the three share + # one libkrunfw build (a Linux kernel from source, ~15-20 min). + - name: Restore libkrun build + id: libkrun-cache + uses: actions/cache/restore@v4 + with: + path: | + libkrunfw + libkrun + key: libkrun-build-x86_64-pvh-v16 + + - name: Build libkrun (cache miss) + if: steps.libkrun-cache.outputs.cache-hit != 'true' + run: | + set -eux + git clone --depth 1 -b v5.5.0 https://github.com/containers/libkrunfw + ( cd libkrunfw && make -j"$(nproc)" && sudo make install ) + git clone --depth 1 -b feat/pvh-boot https://github.com/tsirysndr/libkrun + ( cd libkrun && make -j"$(nproc)" BLK=1 NET=1 ) + + - name: Save libkrun build + if: steps.libkrun-cache.outputs.cache-hit != 'true' && success() + uses: actions/cache/save@v4 + with: + path: | + libkrunfw + libkrun + key: libkrun-build-x86_64-pvh-v16 + + - name: Install libkrun + run: | + set -eux + ( cd libkrunfw && sudo make install ) + ( cd libkrun && sudo make install BLK=1 NET=1 ) + echo "PKG_CONFIG_PATH=/usr/local/lib64/pkgconfig:/usr/local/lib/pkgconfig${PKG_CONFIG_PATH:+:$PKG_CONFIG_PATH}" >> "$GITHUB_ENV" + echo /usr/local/lib64 | sudo tee /etc/ld.so.conf.d/libkrun.conf + sudo ldconfig + ls -l /usr/local/lib64/libkrun.so* + + - name: Build bsdkrun + run: cargo build --release + + - name: Build guest agent (static musl) + run: | + set -eux + ( cd agent && cargo build --release --target x86_64-unknown-linux-musl ) + echo "BSDKRUN_AGENT_LINUX=$PWD/agent/target/x86_64-unknown-linux-musl/release/bsdkrun-agent" >> "$GITHUB_ENV" + + - name: bsdkrun doctor + # Not a gate on the cache itself — it is the fastest way to see, in the + # log of a failed run, whether the host was ever able to run machines. + run: target/release/bsdkrun doctor || true + + # Started with `docker run`, not a `services:` container: MinIO's + # entrypoint needs the `server /data` argument, and `services:` has no way + # to pass one — the workaround people reach for is a different image whose + # default command happens to be right, which is a worse dependency than + # one explicit line. + - name: Start MinIO + run: | + set -eux + docker run -d --name minio -p 9000:9000 \ + -e MINIO_ROOT_USER=bsdkruntest \ + -e MINIO_ROOT_PASSWORD=bsdkruntest123 \ + minio/minio:latest server /data + # Wait for it rather than sleeping: a fixed sleep is either too short + # on a slow runner or wasted on a fast one. + for _ in $(seq 60); do + if curl -fsS http://127.0.0.1:9000/minio/health/live >/dev/null 2>&1; then + echo "minio is live"; break + fi + sleep 1 + done + curl -fsS http://127.0.0.1:9000/minio/health/live >/dev/null + + # `mc` rather than curl: creating a bucket is itself a signed request, and + # using the vendor's client keeps this step from becoming a second, + # untested implementation of SigV4. + - name: Create the MinIO bucket + env: + MC_HOST_local: http://bsdkruntest:bsdkruntest123@127.0.0.1:9000 + run: | + set -eux + curl -fsSL https://dl.min.io/client/mc/release/linux-amd64/mc -o /tmp/mc + chmod +x /tmp/mc + /tmp/mc mb --ignore-existing local/bsdkrun-cache + /tmp/mc ls local/ + + - name: e2e — cache save/restore across formats, on disk and on S3 + env: + BSDKRUN_BIN: target/release/bsdkrun + S3_ENDPOINT: http://127.0.0.1:9000 + S3_BUCKET: bsdkrun-cache + AWS_ACCESS_KEY_ID: bsdkruntest + AWS_SECRET_ACCESS_KEY: bsdkruntest123 + AWS_REGION: us-east-1 + run: ./tests/e2e_cache.sh + + # The archive is meant to be the artifact containerd consumes, so check it + # with containerd's own reader rather than ours. A format that only our + # code can read would pass every test above and still be wrong. + - name: Verify the estargz archive with containerd's reader + env: + BSDKRUN_BIN: target/release/bsdkrun + run: ./tests/estargz_interop.sh + + - name: MinIO logs (on failure) + if: failure() + run: docker logs minio 2>&1 | tail -50 diff --git a/Cargo.lock b/Cargo.lock index c7cab3e..c53cad5 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -626,6 +626,8 @@ dependencies = [ "anyhow", "clap", "crossterm", + "flate2", + "hmac", "libc", "mime_guess", "nucleo-matcher", @@ -633,11 +635,14 @@ dependencies = [ "rust-embed", "serde", "serde_json", + "sha2 0.10.9", "sqlx", + "tar", "tempfile", "tokio", "toml", "tracing", + "zstd", ] [[package]] @@ -1194,6 +1199,7 @@ checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" dependencies = [ "block-buffer 0.10.4", "crypto-common 0.1.7", + "subtle", ] [[package]] @@ -1328,6 +1334,16 @@ dependencies = [ "winapi", ] +[[package]] +name = "filetime" +version = "0.2.29" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5c287a33c7f0a620c38e641e7f60827713987b3c0f26e8ddc9462cc69cf75759" +dependencies = [ + "cfg-if", + "libc", +] + [[package]] name = "find-msvc-tools" version = "0.1.9" @@ -1652,6 +1668,15 @@ version = "0.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" +[[package]] +name = "hmac" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6c49c37c09c17a53d937dfbb742eb3a961d65a994e6bcdcf37e7399d0cc8ab5e" +dependencies = [ + "digest 0.10.7", +] + [[package]] name = "http" version = "0.2.12" @@ -3670,6 +3695,17 @@ dependencies = [ "windows", ] +[[package]] +name = "tar" +version = "0.4.46" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f6221d9a6003c78398e3b239969f352578258df48c8eb051caadae0015bc840" +dependencies = [ + "filetime", + "libc", + "xattr", +] + [[package]] name = "tempfile" version = "3.27.0" @@ -4718,6 +4754,16 @@ version = "0.6.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" +[[package]] +name = "xattr" +version = "1.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32e45ad4206f6d2479085147f02bc2ef834ac85886624a23575ae137c8aa8156" +dependencies = [ + "libc", + "rustix", +] + [[package]] name = "yoke" version = "0.8.3" diff --git a/README.md b/README.md index 6f199bb..3f96045 100644 --- a/README.md +++ b/README.md @@ -879,7 +879,10 @@ bsdkrun logs -f $id # follow it live bsdkrun exec $id uname -a # run a command inside the guest (-t for a PTY, -e K=V for env) bsdkrun cp ./app.py $id:/app/app.py # copy a file in (-r for a directory, - for stdin/stdout) bsdkrun cp $id:/var/log/app.log ./ # ...and back out +bsdkrun cache save $id:/root/.cargo --key deps-v1 # archive a guest dir under a key +bsdkrun cache restore $id --key deps-v1 # ...and put it back later bsdkrun shell $id # open an interactive shell in the guest +bsdkrun doctor # check this host can run machines, and what to fix if not bsdkrun stop $id # stop a running machine (BSD guests clean-poweroff first) bsdkrun start $id # re-boot a stopped machine in place — resumes its own disk/rootfs bsdkrun update $id --cpus 4 --mem 2048 # change recorded vCPU / RAM (applies on next start) diff --git a/core/Cargo.toml b/core/Cargo.toml index c1284c1..857fb06 100644 --- a/core/Cargo.toml +++ b/core/Cargo.toml @@ -54,6 +54,19 @@ serde_json = "1" toml = "0.8" sqlx = { version = "0.8", default-features = false, features = ["runtime-tokio", "sqlite"] } tokio = { version = "1", features = ["rt", "time"] } +# `bsdkrun cache` archives. Linked rather than shelled out to: `zstd` is not on +# a stock macOS or a minimal Linux, and estargz cannot be produced by any CLI — +# it needs one gzip member per tar entry, which no gzip(1) exposes. +flate2 = "1" +zstd = "0.13" +tar = "0.4" +# SigV4 for the S3 cache backend. Pure Rust and no async runtime, so the +# "all HTTP goes through curl" rule in oci.rs still holds — we sign, curl sends. +sha2 = "0.10" +hmac = "0.12" +# estargz stages the incoming tar so it can find entry boundaries; a temp file +# rather than memory keeps a multi-gigabyte cache off the heap. +tempfile = "3" actix-web = { version = "4", optional = true } rust-embed = { version = "8", optional = true } mime_guess = { version = "2", optional = true } diff --git a/core/src/cache/archive.rs b/core/src/cache/archive.rs new file mode 100644 index 0000000..35ff73c --- /dev/null +++ b/core/src/cache/archive.rs @@ -0,0 +1,297 @@ +//! Archive formats for `bsdkrun cache`. +//! +//! A cache entry is always a tar of the guest directory's *contents*; the +//! format only decides how that tar is wrapped. Compression happens on the +//! **host**, never in the guest: the guest already has to provide `tar` for the +//! copy, and requiring `zstd` in every image on top of that would rule out +//! most of them. + +use std::io::{Read, Write}; +use std::path::Path; +use std::str::FromStr; + +use anyhow::{bail, Result}; + +pub mod estargz; + +/// How a cache archive is wrapped. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, serde::Serialize, serde::Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum Compression { + /// gzip — the default. Universally readable, including by plain `tar -xzf`. + #[default] + Gzip, + /// zstd — markedly faster to compress and decompress at similar ratios. + Zstd, + /// eStargz — a gzip archive with one member per file plus a table of + /// contents, so an individual entry can be fetched without reading the + /// whole thing. See [`estargz`] for what that does and does not buy here. + Estargz, + /// A bare tar. For a store that compresses on its own, or for content that + /// does not compress (an already-packed cache). + None, +} + +impl Compression { + /// File extension for an archive in this format, including the `.tar`. + pub fn extension(self) -> &'static str { + match self { + Compression::Gzip => "tar.gz", + Compression::Zstd => "tar.zst", + Compression::Estargz => "tar.estargz", + Compression::None => "tar", + } + } + + /// Every accepted spelling, for CLI help and error messages. + pub const ALL: [&'static str; 4] = ["gzip", "zstd", "estargz", "none"]; +} + +impl FromStr for Compression { + type Err = anyhow::Error; + + fn from_str(s: &str) -> Result { + match s.to_ascii_lowercase().as_str() { + "gzip" | "gz" => Ok(Compression::Gzip), + "zstd" | "zst" => Ok(Compression::Zstd), + "estargz" | "stargz" => Ok(Compression::Estargz), + "none" | "uncompressed" | "tar" => Ok(Compression::None), + other => bail!( + "unknown compression {other:?} — expected one of: {}", + Compression::ALL.join(", ") + ), + } + } +} + +impl std::fmt::Display for Compression { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + // `pad`, not `write_str`: the latter ignores the formatter's width, so + // `{:<9}` in the `cache ls` table would silently do nothing. + f.pad(match self { + Compression::Gzip => "gzip", + Compression::Zstd => "zstd", + Compression::Estargz => "estargz", + Compression::None => "none", + }) + } +} + +/// zstd level 3 — its default, and the point on the curve where it already +/// beats gzip on both ratio and speed. Higher levels cost far more time than +/// they save in bytes for a cache that is written as often as it is read. +const ZSTD_LEVEL: i32 = 3; + +/// gzip level 6 (the default). Level 1 is what `oci.rs` uses for a throwaway +/// initramfs; a cache is written once and restored many times, so the ratio +/// matters more here. +const GZIP_LEVEL: u32 = 6; + +/// Wrap `tar` bytes written to `out` in `format`, returning a writer to feed +/// the tar stream into. Call [`Sink::finish`] to flush the trailer. +pub enum Sink { + Gzip(flate2::write::GzEncoder), + Zstd(zstd::stream::write::Encoder<'static, W>), + Estargz(estargz::Writer), + Plain(W), +} + +impl Sink { + pub fn new(out: W, format: Compression) -> Result { + Ok(match format { + Compression::Gzip => Sink::Gzip(flate2::write::GzEncoder::new( + out, + flate2::Compression::new(GZIP_LEVEL), + )), + Compression::Zstd => Sink::Zstd(zstd::stream::write::Encoder::new(out, ZSTD_LEVEL)?), + Compression::Estargz => Sink::Estargz(estargz::Writer::new(out)), + Compression::None => Sink::Plain(out), + }) + } + + /// Finish the archive and return the underlying writer. + pub fn finish(self) -> Result { + Ok(match self { + Sink::Gzip(e) => e.finish()?, + Sink::Zstd(e) => e.finish()?, + Sink::Estargz(e) => e.finish()?, + Sink::Plain(w) => w, + }) + } +} + +impl Write for Sink { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + match self { + Sink::Gzip(e) => e.write(buf), + Sink::Zstd(e) => e.write(buf), + Sink::Estargz(e) => e.write(buf), + Sink::Plain(w) => w.write(buf), + } + } + + fn flush(&mut self) -> std::io::Result<()> { + match self { + Sink::Gzip(e) => e.flush(), + Sink::Zstd(e) => e.flush(), + Sink::Estargz(e) => e.flush(), + Sink::Plain(w) => w.flush(), + } + } +} + +/// Unwrap an archive in `format` back into a plain tar stream. +/// +/// estargz decodes through the gzip reader: its members concatenate into one +/// valid gzip stream, which is exactly why the format stays readable by tools +/// that know nothing about its table of contents. +pub fn reader<'a, R: Read + Send + 'a>( + input: R, + format: Compression, +) -> Result> { + Ok(match format { + Compression::Gzip | Compression::Estargz => { + Box::new(flate2::read::MultiGzDecoder::new(input)) + } + Compression::Zstd => Box::new(zstd::stream::read::Decoder::new(input)?), + Compression::None => Box::new(input), + }) +} + +/// Copy a tar stream, dropping the entries eStargz adds for its own use. +/// +/// The TOC and the landmark are real tar members — that is what keeps the +/// archive readable by plain `tar` — so a restore that just piped the stream +/// into the guest would leave a `stargz.index.json` and a +/// `.no.prefetch.landmark` sitting in the restored directory. Nothing else +/// strips them: the guest's `tar` does not know the format, and `--exclude` is +/// not something busybox tar can be relied on for. +pub fn strip_estargz(src: R, mut out: W) -> Result { + let mut builder = tar::Builder::new(&mut out); + let mut archive = tar::Archive::new(src); + for entry in archive.entries()? { + let mut entry = entry?; + let path = entry.path()?.display().to_string(); + let name = path.trim_start_matches("./"); + if name == estargz::TOC_TAR_NAME || name == estargz::NO_PREFETCH_LANDMARK { + continue; + } + let header = entry.header().clone(); + let mut body = Vec::new(); + entry.read_to_end(&mut body)?; + builder.append(&header, &body[..])?; + } + builder.into_inner()?; + Ok(out) +} + +/// Infer the format from an archive's file name, for restoring an entry whose +/// metadata was lost. +pub fn from_path(path: &Path) -> Option { + let name = path.file_name()?.to_str()?; + if name.ends_with(".tar.estargz") { + Some(Compression::Estargz) + } else if name.ends_with(".tar.gz") || name.ends_with(".tgz") { + Some(Compression::Gzip) + } else if name.ends_with(".tar.zst") { + Some(Compression::Zstd) + } else if name.ends_with(".tar") { + Some(Compression::None) + } else { + None + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn tar_bytes() -> Vec { + let mut builder = tar::Builder::new(Vec::new()); + let body = b"hello cache\n"; + let mut header = tar::Header::new_gnu(); + header.set_path("a.txt").unwrap(); + header.set_size(body.len() as u64); + header.set_mode(0o644); + header.set_cksum(); + builder.append(&header, &body[..]).unwrap(); + builder.into_inner().unwrap() + } + + fn round_trip(format: Compression) -> Vec { + let tar = tar_bytes(); + let mut sink = Sink::new(Vec::new(), format).unwrap(); + sink.write_all(&tar).unwrap(); + let archive = sink.finish().unwrap(); + + let mut out = Vec::new(); + reader(&archive[..], format) + .unwrap() + .read_to_end(&mut out) + .unwrap(); + out + } + + fn entries_of(tar_bytes: &[u8]) -> Vec<(String, String)> { + tar::Archive::new(tar_bytes) + .entries() + .unwrap() + .map(|e| { + let mut e = e.unwrap(); + let mut body = String::new(); + e.read_to_string(&mut body).unwrap(); + (e.path().unwrap().display().to_string(), body) + }) + .collect() + } + + /// Every format has to hand back the exact tar it was given — a cache that + /// restores *almost* the right bytes is worse than one that fails. + #[test] + fn every_format_round_trips_the_tar_byte_for_byte() { + for format in [Compression::Gzip, Compression::Zstd, Compression::None] { + assert_eq!( + entries_of(&round_trip(format)), + vec![("a.txt".to_string(), "hello cache\n".to_string())], + "{format} did not round-trip" + ); + } + } + + /// estargz is the exception: its TOC and landmark are entries *in* the tar, + /// by design. They must never reach the guest, so [`strip_estargz`] takes + /// them back out on the way in — this pins both halves of that. + #[test] + fn estargz_carries_bookkeeping_entries_that_restore_strips() { + let raw = entries_of(&round_trip(Compression::Estargz)); + let names: Vec<_> = raw.iter().map(|(n, _)| n.as_str()).collect(); + assert_eq!( + names, + vec!["a.txt", ".no.prefetch.landmark", "stargz.index.json"] + ); + + let mut stripped = Vec::new(); + strip_estargz(&round_trip(Compression::Estargz)[..], &mut stripped).unwrap(); + assert_eq!( + entries_of(&stripped), + vec![("a.txt".to_string(), "hello cache\n".to_string())], + ); + } + + #[test] + fn compression_names_and_extensions_agree() { + for name in Compression::ALL { + let parsed: Compression = name.parse().unwrap(); + assert_eq!(parsed.to_string(), name); + let path = std::path::PathBuf::from(format!("c.{}", parsed.extension())); + assert_eq!(from_path(&path), Some(parsed), "{name} extension"); + } + } + + #[test] + fn unknown_compression_lists_the_valid_ones() { + let err = "brotli".parse::().unwrap_err().to_string(); + assert!(err.contains("gzip"), "{err}"); + assert!(err.contains("estargz"), "{err}"); + } +} diff --git a/core/src/cache/archive/estargz.rs b/core/src/cache/archive/estargz.rs new file mode 100644 index 0000000..21d8b59 --- /dev/null +++ b/core/src/cache/archive/estargz.rs @@ -0,0 +1,573 @@ +//! An eStargz writer. +//! +//! eStargz is a tar.gz laid out so an individual file can be fetched without +//! reading the archive: every entry's payload begins its own gzip member, a +//! `stargz.index.json` table of contents records each member's byte offset, and +//! a 51-byte footer at the very end says where that TOC begins. Because gzip +//! members concatenate into one valid gzip stream, the result is still an +//! ordinary `.tar.gz` to anything that does not care — `tar -xzf` reads it. +//! +//! **What it buys here, today: interoperability, not speed.** `bsdkrun cache +//! restore` unpacks the whole tree, so it never seeks, and the per-member +//! framing makes the archive slightly *larger* than plain gzip. The reason to +//! write it is that the artifact is the same one containerd, stargz-snapshotter +//! and `ctr-remote` consume — a cache saved here can be served to a lazy-pulling +//! runtime, and a future partial restore has the offsets it would need. +//! +//! Spec: + +use std::io::{Read, Seek, SeekFrom, Write}; + +use anyhow::{Context, Result}; +use sha2::{Digest, Sha256}; + +/// Tar entry name of the table of contents. +pub(super) const TOC_TAR_NAME: &str = "stargz.index.json"; + +/// Landmark marking "this archive has no prefetch range". The spec requires one +/// of the two landmarks to be present; without it a verifying reader rejects the +/// archive, which is exactly the interoperability we are here for. +pub(super) const NO_PREFETCH_LANDMARK: &str = ".no.prefetch.landmark"; + +/// The footer is a gzip member carrying no data, whose Extra field points at the +/// TOC. Its length is fixed by the spec, and readers seek to `len - 51`. +pub const FOOTER_SIZE: usize = 51; + +/// TOC schema version. +const TOC_VERSION: u32 = 1; + +/// One entry in `stargz.index.json`. +/// +/// Field names are the wire format's, so they are camelCase rather than Rust's +/// convention. Absent fields are omitted rather than sent as null — a reader +/// distinguishes "no link" from `"linkName": ""`. +#[derive(Debug, serde::Serialize, serde::Deserialize)] +struct TocEntry { + name: String, + #[serde(rename = "type")] + kind: String, + #[serde(skip_serializing_if = "is_zero_u64", default)] + size: u64, + #[serde(rename = "modtime", skip_serializing_if = "String::is_empty", default)] + mod_time: String, + #[serde(rename = "linkName", skip_serializing_if = "String::is_empty", default)] + link_name: String, + mode: i64, + #[serde(skip_serializing_if = "is_zero_u64", default)] + uid: u64, + #[serde(skip_serializing_if = "is_zero_u64", default)] + gid: u64, + #[serde(rename = "userName", skip_serializing_if = "String::is_empty", default)] + user_name: String, + #[serde( + rename = "groupName", + skip_serializing_if = "String::is_empty", + default + )] + group_name: String, + #[serde(rename = "devMajor", skip_serializing_if = "is_zero_u64", default)] + dev_major: u64, + #[serde(rename = "devMinor", skip_serializing_if = "is_zero_u64", default)] + dev_minor: u64, + /// Offset of the gzip member holding this entry's payload. + #[serde(skip_serializing_if = "is_zero_u64", default)] + offset: u64, + /// `sha256:…` over the file's contents. + #[serde(skip_serializing_if = "String::is_empty", default)] + digest: String, +} + +fn is_zero_u64(n: &u64) -> bool { + *n == 0 +} + +#[derive(Debug, serde::Serialize, serde::Deserialize)] +struct Toc { + version: u32, + entries: Vec, +} + +/// Buffers the incoming tar, then converts it on [`finish`](Writer::finish). +/// +/// The conversion has to read the tar back to find entry boundaries, so it +/// cannot be done in a single forward pass over a `Write`. Staging goes to a +/// temp *file*, not memory, so a multi-gigabyte cache costs disk rather than +/// RAM — and that file is the only extra cost of choosing this format. +pub struct Writer { + staging: Option, + out: Option, +} + +impl Writer { + pub fn new(out: W) -> Self { + Writer { + staging: None, + out: Some(out), + } + } + + fn staging(&mut self) -> std::io::Result<&mut tempfile::NamedTempFile> { + if self.staging.is_none() { + self.staging = Some(tempfile::NamedTempFile::new()?); + } + Ok(self.staging.as_mut().expect("just created")) + } + + /// Convert the staged tar into eStargz and return the underlying writer. + pub fn finish(mut self) -> Result { + let out = self.out.take().expect("finish called once"); + let Some(mut staging) = self.staging.take() else { + // Nothing was ever written. Emit a well-formed empty archive so a + // reader gets an empty tar rather than a truncated file. + return build(std::io::empty(), out); + }; + staging.flush()?; + staging.as_file_mut().seek(SeekFrom::Start(0))?; + let file = staging.reopen().context("reopening the staged tar")?; + build(file, out) + } +} + +impl Write for Writer { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.staging()?.write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + match self.staging.as_mut() { + Some(f) => f.flush(), + None => Ok(()), + } + } +} + +/// Counts bytes on their way out, so a TOC offset is just `counter.n`. +struct Counting { + inner: W, + n: u64, +} + +impl Write for Counting { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + let n = self.inner.write(buf)?; + self.n += n as u64; + Ok(n) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.inner.flush() + } +} + +/// Write one gzip member containing exactly `body`, and return the output +/// offset it started at. +fn member(out: &mut Counting, body: &[u8]) -> Result { + let at = out.n; + let mut gz = flate2::write::GzEncoder::new(&mut *out, flate2::Compression::default()); + gz.write_all(body)?; + gz.finish()?; + Ok(at) +} + +/// Read a tar from `src` and write the eStargz form to `out`. +/// +/// Iteration is over the *raw* entries, so GNU long-name and long-link pseudo +/// entries arrive as themselves and are re-emitted byte for byte. That keeps the +/// decompressed stream identical to the input — the alternative, rebuilding +/// headers through `tar::Builder`, both loses the original bytes and terminates +/// the archive early, because its `into_inner` writes the end-of-archive blocks. +fn build(src: R, out: W) -> Result { + let mut out = Counting { inner: out, n: 0 }; + let mut toc = Toc { + version: TOC_VERSION, + entries: Vec::new(), + }; + + // A GNU long name/link arrives as a pseudo entry *before* the entry it + // describes; hold it until that entry shows up so the TOC records the real + // path rather than the 100-byte truncation in the header. + let mut pending_name: Option = None; + let mut pending_link: Option = None; + + let mut archive = tar::Archive::new(src); + let entries = archive + .entries() + .context("reading the staged tar")? + .raw(true); + for entry in entries { + let mut entry = entry.context("reading a tar entry")?; + let header = entry.header().clone(); + let size = header.size().unwrap_or(0); + let entry_type = header.entry_type(); + + // The header block, verbatim, in its own member. + member(&mut out, header.as_bytes())?; + + let mut payload = Vec::with_capacity(size as usize); + entry + .read_to_end(&mut payload) + .context("reading a tar entry body")?; + + let offset = if size > 0 { + let mut padded = payload.clone(); + pad_to_block(&mut padded); + member(&mut out, &padded)? + } else { + out.n + }; + + // Pseudo entries carry a name, not a file: stash and move on. + if entry_type == tar::EntryType::GNULongName { + pending_name = Some(cstr(&payload)); + continue; + } + if entry_type == tar::EntryType::GNULongLink { + pending_link = Some(cstr(&payload)); + continue; + } + + let name = match pending_name.take() { + Some(n) => n, + None => header + .path() + .map(|p| p.display().to_string()) + .unwrap_or_default(), + }; + let link = match pending_link.take() { + Some(l) => l, + None => header + .link_name() + .ok() + .flatten() + .map(|p| p.display().to_string()) + .unwrap_or_default(), + }; + + toc.entries.push(TocEntry { + name: normalize(&name), + kind: kind_of(&header).to_string(), + size, + mod_time: rfc3339(header.mtime().unwrap_or(0)), + link_name: normalize(&link), + mode: header.mode().unwrap_or(0) as i64, + uid: header.uid().unwrap_or(0), + gid: header.gid().unwrap_or(0), + user_name: header.username().ok().flatten().unwrap_or("").to_string(), + group_name: header.groupname().ok().flatten().unwrap_or("").to_string(), + dev_major: header.device_major().ok().flatten().unwrap_or(0) as u64, + dev_minor: header.device_minor().ok().flatten().unwrap_or(0) as u64, + offset, + digest: if payload.is_empty() { + String::new() + } else { + format!("sha256:{:x}", Sha256::digest(&payload)) + }, + }); + } + + append_generated(&mut out, &mut toc, NO_PREFETCH_LANDMARK, &[0u8])?; + + // The TOC, last, in a member of its own — the footer points at where it + // starts. Its header and its JSON go in the *same* member, unlike a file's: + // a reader takes one gzip member at the footer's offset, calls tar.Next() + // for the header and then decodes the JSON from the same stream. Splitting + // them the way regular entries are split ends that stream after the header, + // and the TOC decode fails with "unexpected EOF". + let json = serde_json::to_vec(&toc).context("serializing the estargz TOC")?; + let mut toc_blocks = plain_header(TOC_TAR_NAME, json.len() as u64)?.to_vec(); + toc_blocks.extend_from_slice(&json); + pad_to_block(&mut toc_blocks); + let toc_offset = member(&mut out, &toc_blocks)?; + + // A tar ends with two zero blocks; without them `tar -xzf` warns. + member(&mut out, &[0u8; 1024])?; + + out.write_all(&footer(toc_offset))?; + out.flush()?; + Ok(out.inner) +} + +/// A 512-byte ustar header for a short-named regular file we generate. +fn plain_header(name: &str, size: u64) -> Result<[u8; 512]> { + let mut h = tar::Header::new_gnu(); + h.set_path(name) + .with_context(|| format!("naming the generated entry {name}"))?; + h.set_size(size); + h.set_mode(0o644); + h.set_entry_type(tar::EntryType::Regular); + h.set_cksum(); + Ok(*h.as_bytes()) +} + +/// Append a small file bsdkrun generates (the landmark) as a real tar entry, so +/// the archive stays a valid tar, and record it in the TOC. +fn append_generated( + out: &mut Counting, + toc: &mut Toc, + name: &str, + body: &[u8], +) -> Result<()> { + member(out, &plain_header(name, body.len() as u64)?)?; + let mut padded = body.to_vec(); + pad_to_block(&mut padded); + let offset = member(out, &padded)?; + + toc.entries.push(TocEntry { + name: name.to_string(), + kind: "reg".to_string(), + size: body.len() as u64, + mod_time: String::new(), + link_name: String::new(), + mode: 0o644, + uid: 0, + gid: 0, + user_name: String::new(), + group_name: String::new(), + dev_major: 0, + dev_minor: 0, + offset, + digest: format!("sha256:{:x}", Sha256::digest(body)), + }); + Ok(()) +} + +/// A GNU pseudo entry's body is the name, NUL-terminated. +fn cstr(bytes: &[u8]) -> String { + let end = bytes.iter().position(|&b| b == 0).unwrap_or(bytes.len()); + String::from_utf8_lossy(&bytes[..end]).into_owned() +} + +/// The 51-byte footer: an empty gzip member whose Extra field is the subfield +/// `S G` carrying `%016xSTARGZ`, the TOC's offset in hex. +fn footer(toc_offset: u64) -> Vec { + let payload = format!("{toc_offset:016x}STARGZ"); + debug_assert_eq!(payload.len(), 22); + let mut extra = Vec::with_capacity(4 + payload.len()); + extra.push(b'S'); + extra.push(b'G'); + extra.extend_from_slice(&(payload.len() as u16).to_le_bytes()); + extra.extend_from_slice(payload.as_bytes()); + + let mut buf = Vec::new(); + let gz = flate2::GzBuilder::new() + .extra(extra) + .write(&mut buf, flate2::Compression::none()); + gz.finish().expect("writing to a Vec cannot fail"); + buf +} + +/// Read a footer back, returning the TOC offset it points at. The inverse of +/// [`footer`], and what a reader does first. +pub fn parse_footer(bytes: &[u8]) -> Option { + if bytes.len() != FOOTER_SIZE { + return None; + } + // Locate the subfield payload rather than assuming a fixed position: the + // gzip header before it is fixed-width today, but reading it out by pattern + // keeps this honest if flate2 ever emits an OS byte or MTIME differently. + let marker = bytes.windows(6).position(|w| w == b"STARGZ")?; + let hex = bytes.get(marker.checked_sub(16)?..marker)?; + u64::from_str_radix(std::str::from_utf8(hex).ok()?, 16).ok() +} + +fn pad_to_block(buf: &mut Vec) { + let rem = buf.len() % 512; + if rem != 0 { + buf.resize(buf.len() + (512 - rem), 0); + } +} + +/// eStargz spells tar's type flags out in words. +fn kind_of(header: &tar::Header) -> &'static str { + use tar::EntryType::*; + match header.entry_type() { + Directory => "dir", + Symlink => "symlink", + Link => "hardlink", + Char => "char", + Block => "block", + Fifo => "fifo", + _ => "reg", + } +} + +/// TOC paths are relative and unprefixed; tar writes them as `./foo`. +fn normalize(path: &str) -> String { + path.trim_start_matches("./").to_string() +} + +/// Format unix seconds as RFC 3339 UTC, which is what the TOC's `modtime` is. +/// +/// Hand-rolled rather than pulling in a date library for one field: this is the +/// civil-from-days algorithm, exact for any date after 1970. +fn rfc3339(secs: u64) -> String { + let days = (secs / 86_400) as i64; + let tod = secs % 86_400; + // Days since 1970-01-01 -> y/m/d (Howard Hinnant's civil_from_days). + let z = days + 719_468; + let era = z.div_euclid(146_097); + let doe = z.rem_euclid(146_097); + let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365; + let y = yoe + era * 400; + let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); + let mp = (5 * doy + 2) / 153; + let d = doy - (153 * mp + 2) / 5 + 1; + let m = if mp < 10 { mp + 3 } else { mp - 9 }; + let y = if m <= 2 { y + 1 } else { y }; + format!( + "{y:04}-{m:02}-{d:02}T{:02}:{:02}:{:02}Z", + tod / 3600, + (tod % 3600) / 60, + tod % 60 + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn sample_tar() -> Vec { + let mut b = tar::Builder::new(Vec::new()); + for (name, body) in [ + ("small.txt", "hello\n".as_bytes()), + ("nested/deep/file.bin", &[0u8, 1, 2, 255, 254][..]), + ] { + let mut h = tar::Header::new_gnu(); + h.set_size(body.len() as u64); + h.set_mode(0o644); + h.set_mtime(1_760_000_000); + h.set_entry_type(tar::EntryType::Regular); + b.append_data(&mut h, name, body).unwrap(); + } + b.into_inner().unwrap() + } + + fn build_sample() -> Vec { + let mut w = Writer::new(Vec::new()); + w.write_all(&sample_tar()).unwrap(); + w.finish().unwrap() + } + + /// The whole point of concatenated gzip members: anything that can read a + /// .tar.gz can read this, TOC or no TOC. + #[test] + fn a_plain_gzip_reader_sees_the_original_files() { + let archive = build_sample(); + let mut tar = Vec::new(); + flate2::read::MultiGzDecoder::new(&archive[..]) + .read_to_end(&mut tar) + .unwrap(); + + let mut found = Vec::new(); + for e in tar::Archive::new(&tar[..]).entries().unwrap() { + let mut e = e.unwrap(); + let name = e.path().unwrap().display().to_string(); + let mut body = Vec::new(); + e.read_to_end(&mut body).unwrap(); + found.push((name, body)); + } + assert_eq!(found[0].0, "small.txt"); + assert_eq!(found[0].1, b"hello\n"); + assert_eq!(found[1].0, "nested/deep/file.bin"); + assert_eq!(found[1].1, vec![0u8, 1, 2, 255, 254]); + } + + #[test] + fn the_footer_is_the_size_the_spec_fixes() { + assert_eq!(footer(0).len(), FOOTER_SIZE); + assert_eq!(footer(u32::MAX as u64).len(), FOOTER_SIZE); + } + + #[test] + fn the_footer_round_trips_the_toc_offset() { + for off in [0u64, 1, 512, 123_456, 0xdead_beef] { + assert_eq!(parse_footer(&footer(off)), Some(off), "offset {off}"); + } + } + + /// A reader seeks to `len - 51`, reads the offset, and gunzips there. If the + /// offset is off by even one byte it lands mid-member and gets nothing — + /// so follow it and check the TOC really is where it says. + #[test] + fn the_toc_is_where_the_footer_says_it_is() { + let archive = build_sample(); + let toc_offset = + parse_footer(&archive[archive.len() - FOOTER_SIZE..]).expect("footer parses") as usize; + + let mut tar_bytes = Vec::new(); + flate2::read::MultiGzDecoder::new(&archive[toc_offset..]) + .read_to_end(&mut tar_bytes) + .unwrap(); + let mut entries = tar::Archive::new(&tar_bytes[..]); + let mut e = entries.entries().unwrap().next().unwrap().unwrap(); + assert_eq!(e.path().unwrap().display().to_string(), TOC_TAR_NAME); + + let mut json = String::new(); + e.read_to_string(&mut json).unwrap(); + let toc: Toc = serde_json::from_str(&json).unwrap(); + assert_eq!(toc.version, TOC_VERSION); + let names: Vec<_> = toc.entries.iter().map(|e| e.name.as_str()).collect(); + assert!(names.contains(&"small.txt"), "{names:?}"); + assert!(names.contains(&NO_PREFETCH_LANDMARK), "{names:?}"); + } + + /// Each entry's `offset` must point at the gzip member holding its *payload* + /// — not its tar header. Getting this wrong yields 512 bytes of header where + /// the file contents should be, which is the mistake the format invites. + #[test] + fn each_entry_offset_lands_on_its_own_payload() { + let archive = build_sample(); + let toc_offset = + parse_footer(&archive[archive.len() - FOOTER_SIZE..]).expect("footer parses") as usize; + let mut json_tar = Vec::new(); + flate2::read::MultiGzDecoder::new(&archive[toc_offset..]) + .read_to_end(&mut json_tar) + .unwrap(); + let mut e = tar::Archive::new(&json_tar[..]); + let mut first = e.entries().unwrap().next().unwrap().unwrap(); + let mut json = String::new(); + first.read_to_string(&mut json).unwrap(); + let toc: Toc = serde_json::from_str(&json).unwrap(); + + for want in [ + ("small.txt", &b"hello\n"[..]), + ("nested/deep/file.bin", &[0u8, 1, 2, 255, 254][..]), + ] { + let entry = toc + .entries + .iter() + .find(|e| e.name == want.0) + .unwrap_or_else(|| panic!("{} missing from the TOC", want.0)); + let mut got = vec![0u8; entry.size as usize]; + let mut gz = flate2::read::GzDecoder::new(&archive[entry.offset as usize..]); + gz.read_exact(&mut got).unwrap(); + assert_eq!(got, want.1, "{} payload at its TOC offset", want.0); + assert_eq!(entry.digest, format!("sha256:{:x}", Sha256::digest(want.1))); + } + } + + #[test] + fn modtimes_are_rfc3339_utc() { + assert_eq!(rfc3339(0), "1970-01-01T00:00:00Z"); + assert_eq!(rfc3339(1_760_000_000), "2025-10-09T08:53:20Z"); + } + + /// An empty input still has to produce something a reader accepts, rather + /// than a zero-byte file that looks like a failed upload. + #[test] + fn an_empty_archive_is_still_well_formed() { + let out = Writer::new(Vec::new()).finish().unwrap(); + assert!(parse_footer(&out[out.len() - FOOTER_SIZE..]).is_some()); + let mut tar = Vec::new(); + flate2::read::MultiGzDecoder::new(&out[..]) + .read_to_end(&mut tar) + .unwrap(); + // The landmark and the TOC, and nothing else. + let names: Vec<_> = tar::Archive::new(&tar[..]) + .entries() + .unwrap() + .map(|e| e.unwrap().path().unwrap().display().to_string()) + .collect(); + assert_eq!(names, vec![NO_PREFETCH_LANDMARK, TOC_TAR_NAME]); + } +} diff --git a/core/src/cache/config.rs b/core/src/cache/config.rs new file mode 100644 index 0000000..c7ee987 --- /dev/null +++ b/core/src/cache/config.rs @@ -0,0 +1,235 @@ +//! Where cache entries are stored, and how to reach it. +//! +//! Resolution order is environment, then `cache.toml`, then the default — so CI +//! can point a run at a bucket with `BSDKRUN_CACHE_*` without writing a file, +//! and a workstation can set it once and forget. +//! +//! ```toml +//! # ~/.config/bsdkrun/cache.toml +//! backend = "s3" # or "disk" (the default) +//! +//! [s3] +//! bucket = "my-ci-cache" +//! region = "us-east-1" +//! prefix = "bsdkrun" # optional key prefix +//! endpoint = "https://.r2.cloudflarestorage.com" # optional: R2, MinIO, … +//! ``` +//! +//! Credentials are **never** read from that file — they come from +//! `AWS_ACCESS_KEY_ID` / `AWS_SECRET_ACCESS_KEY` (+ `AWS_SESSION_TOKEN`), so a +//! config you can commit never holds a secret. + +use anyhow::{bail, Context, Result}; +use std::path::PathBuf; + +/// Which store cache entries live in. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Backend { + /// A directory on this host — the default, and the whole feature for a + /// single machine. + Disk, + /// An S3-compatible bucket, for sharing a cache between hosts or CI runs. + S3(S3Config), +} + +#[derive(Debug, Clone, PartialEq, Eq, Default, serde::Deserialize)] +pub struct S3Config { + pub bucket: String, + #[serde(default = "default_region")] + pub region: String, + /// Key prefix inside the bucket, so one bucket can hold several projects. + #[serde(default)] + pub prefix: String, + /// Override the endpoint for a non-AWS implementation (R2, MinIO, Ceph). + #[serde(default)] + pub endpoint: Option, +} + +fn default_region() -> String { + "us-east-1".to_string() +} + +#[derive(Debug, Default, serde::Deserialize)] +struct File { + #[serde(default)] + backend: Option, + #[serde(default)] + s3: Option, +} + +impl S3Config { + /// Base URL for the bucket, honouring a custom endpoint. + /// + /// AWS is addressed virtual-host style (`bucket.s3.region.amazonaws.com`); + /// a custom endpoint is addressed path style (`endpoint/bucket`), which is + /// what MinIO defaults to and what R2 requires. + pub fn base_url(&self) -> String { + match &self.endpoint { + Some(ep) => format!("{}/{}", ep.trim_end_matches('/'), self.bucket), + None => format!("https://{}.s3.{}.amazonaws.com", self.bucket, self.region), + } + } + + /// Host header for signing — the URL's authority. + pub fn host(&self) -> String { + let url = self.base_url(); + let after_scheme = url.split_once("://").map(|(_, r)| r).unwrap_or(&url); + after_scheme + .split('/') + .next() + .unwrap_or(after_scheme) + .to_string() + } + + /// Full object key for a cache entry, including any configured prefix. + pub fn object_key(&self, name: &str) -> String { + match self.prefix.trim_matches('/') { + "" => name.to_string(), + p => format!("{p}/{name}"), + } + } +} + +/// Resolve the backend from the environment, then `cache.toml`, then the +/// default. +pub fn resolve() -> Result { + let file = load_file(); + + let chosen = env("BSDKRUN_CACHE_BACKEND") + .or_else(|| file.backend.clone()) + .unwrap_or_else(|| "disk".to_string()); + + match chosen.to_ascii_lowercase().as_str() { + "disk" | "local" | "host" => Ok(Backend::Disk), + "s3" => Ok(Backend::S3(s3_config(file.s3)?)), + other => bail!("unknown cache backend {other:?} — expected `disk` or `s3`"), + } +} + +fn s3_config(from_file: Option) -> Result { + let mut cfg = from_file.unwrap_or_default(); + if let Some(v) = env("BSDKRUN_CACHE_S3_BUCKET") { + cfg.bucket = v; + } + if let Some(v) = env("BSDKRUN_CACHE_S3_REGION").or_else(|| env("AWS_REGION")) { + cfg.region = v; + } + if let Some(v) = env("BSDKRUN_CACHE_S3_PREFIX") { + cfg.prefix = v; + } + if let Some(v) = env("BSDKRUN_CACHE_S3_ENDPOINT") { + cfg.endpoint = Some(v); + } + if cfg.region.is_empty() { + cfg.region = default_region(); + } + if cfg.bucket.is_empty() { + bail!( + "the S3 cache backend needs a bucket. Set BSDKRUN_CACHE_S3_BUCKET, or add one to {}:\n\ + \n backend = \"s3\"\n \n [s3]\n bucket = \"my-ci-cache\"\n region = \"us-east-1\"", + config_path() + .map(|p| p.display().to_string()) + .unwrap_or_else(|_| "~/.config/bsdkrun/cache.toml".to_string()) + ); + } + Ok(cfg) +} + +/// S3 credentials, from the environment only. +pub struct Credentials { + pub access_key: String, + pub secret_key: String, + pub session_token: Option, +} + +pub fn credentials() -> Result { + let access_key = env("AWS_ACCESS_KEY_ID").context( + "the S3 cache backend needs AWS_ACCESS_KEY_ID and AWS_SECRET_ACCESS_KEY in the \ + environment (they are deliberately not read from cache.toml, so the file stays \ + safe to commit)", + )?; + let secret_key = env("AWS_SECRET_ACCESS_KEY") + .context("AWS_ACCESS_KEY_ID is set but AWS_SECRET_ACCESS_KEY is not")?; + Ok(Credentials { + access_key, + secret_key, + session_token: env("AWS_SESSION_TOKEN"), + }) +} + +fn env(key: &str) -> Option { + std::env::var(key).ok().filter(|v| !v.is_empty()) +} + +/// `$XDG_CONFIG_HOME/bsdkrun/cache.toml`, else `~/.config/bsdkrun/cache.toml`. +pub fn config_path() -> Result { + let base = match env("XDG_CONFIG_HOME") { + Some(dir) => PathBuf::from(dir), + None => PathBuf::from(env("HOME").context("neither XDG_CONFIG_HOME nor HOME is set")?) + .join(".config"), + }; + Ok(base.join("bsdkrun").join("cache.toml")) +} + +fn load_file() -> File { + let Ok(path) = config_path() else { + return File::default(); + }; + let Ok(text) = std::fs::read_to_string(&path) else { + return File::default(); + }; + match toml::from_str(&text) { + Ok(f) => f, + Err(e) => { + tracing::warn!("ignoring {}: {e}", path.display()); + File::default() + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn aws_is_virtual_host_addressed_and_custom_endpoints_are_path_addressed() { + let aws = S3Config { + bucket: "b".into(), + region: "eu-west-1".into(), + ..Default::default() + }; + assert_eq!(aws.base_url(), "https://b.s3.eu-west-1.amazonaws.com"); + assert_eq!(aws.host(), "b.s3.eu-west-1.amazonaws.com"); + + let r2 = S3Config { + bucket: "b".into(), + region: "auto".into(), + endpoint: Some("https://acct.r2.cloudflarestorage.com/".into()), + ..Default::default() + }; + assert_eq!(r2.base_url(), "https://acct.r2.cloudflarestorage.com/b"); + assert_eq!(r2.host(), "acct.r2.cloudflarestorage.com"); + } + + #[test] + fn a_prefix_is_applied_and_slashes_are_not_doubled() { + let mut cfg = S3Config { + bucket: "b".into(), + ..Default::default() + }; + assert_eq!(cfg.object_key("x.tar.gz"), "x.tar.gz"); + cfg.prefix = "/team/ci/".into(); + assert_eq!(cfg.object_key("x.tar.gz"), "team/ci/x.tar.gz"); + } + + /// The error has to say what to set; a bare "missing bucket" sends people + /// looking for a flag that does not exist. + #[test] + fn a_bucketless_s3_config_explains_how_to_set_one() { + let err = s3_config(Some(S3Config::default())) + .unwrap_err() + .to_string(); + assert!(err.contains("BSDKRUN_CACHE_S3_BUCKET"), "{err}"); + assert!(err.contains("cache.toml"), "{err}"); + } +} diff --git a/core/src/cache/mod.rs b/core/src/cache/mod.rs new file mode 100644 index 0000000..b4c6904 --- /dev/null +++ b/core/src/cache/mod.rs @@ -0,0 +1,467 @@ +//! `bsdkrun cache` — save a guest directory to a backing store under a key, and +//! restore it into any machine later. +//! +//! The shape is GitHub Actions': you save under an exact `--key`, and restore +//! by that key with `--restore-keys` prefixes as fallbacks, so a lockfile-hashed +//! key that misses still lands on the most recent compatible cache instead of +//! nothing. +//! +//! The archive is produced the same way `bsdkrun cp -r` moves a directory — +//! `tar` in the guest, streamed out over the exec agent — and compressed on the +//! host, so an image needs no compressor of its own. See [`archive`] for the +//! formats and [`config`] for where entries are stored. + +use std::io::{Read, Write}; +use std::path::{Path, PathBuf}; + +use anyhow::{bail, Context, Result}; +use sha2::{Digest, Sha256}; + +pub mod archive; +pub mod config; +pub mod s3; + +use archive::Compression; +use config::Backend; + +/// What a saved cache entry records about itself. Written beside the archive so +/// a restore knows how to unwrap it, and `ls` can describe it without reading +/// gigabytes. +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] +pub struct Entry { + /// The exact key it was saved under. + pub key: String, + /// Guest path the tree came from — informational, and the default target + /// when restoring without one. + pub path: String, + pub compression: Compression, + /// Archive size in bytes. + pub size: u64, + /// Unix seconds when it was saved. + pub created: u64, + /// `sha256:…` over the archive, so a restore can tell a truncated download + /// from a corrupt cache. + pub digest: String, +} + +impl Entry { + /// Storage name shared by the archive and its metadata. + /// + /// A readable slug plus a hash of the full key: the slug keeps a store + /// browsable, and the hash makes collisions impossible even though the slug + /// throws away characters and length. + pub fn storage_name(key: &str) -> String { + let slug: String = key + .chars() + .map(|c| { + if c.is_ascii_alphanumeric() || c == '-' || c == '_' || c == '.' { + c + } else { + '-' + } + }) + .collect(); + let slug: String = slug.trim_matches('-').chars().take(40).collect(); + let hash = format!("{:x}", Sha256::digest(key.as_bytes())); + if slug.is_empty() { + hash[..16].to_string() + } else { + format!("{slug}-{}", &hash[..8]) + } + } + + fn archive_name(&self) -> String { + format!( + "{}.{}", + Entry::storage_name(&self.key), + self.compression.extension() + ) + } + + fn meta_name(&self) -> String { + format!("{}.json", Entry::storage_name(&self.key)) + } +} + +/// A place cache entries live. Disk and S3 differ only in how bytes move. +pub enum Store { + Disk(PathBuf), + S3(config::S3Config), +} + +impl Store { + /// Open the store the environment and config select. + pub fn open() -> Result { + match config::resolve()? { + Backend::Disk => Ok(Store::Disk(disk_dir()?)), + Backend::S3(cfg) => Ok(Store::S3(cfg)), + } + } + + /// One line naming where entries go, for `cache ls` and `doctor`. + pub fn describe(&self) -> String { + match self { + Store::Disk(dir) => format!("host disk at {}", dir.display()), + Store::S3(cfg) => format!("s3://{}/{}", cfg.bucket, cfg.prefix.trim_matches('/')), + } + } + + /// Publish an archive and its metadata. Metadata goes **last**, so a store + /// never advertises an entry whose archive did not finish uploading — the + /// same reason `oci.rs` writes its completion marker last. + pub fn put_entry(&self, entry: &Entry, archive: &Path) -> Result<()> { + let meta = serde_json::to_vec_pretty(entry)?; + match self { + Store::Disk(dir) => { + std::fs::create_dir_all(dir) + .with_context(|| format!("creating {}", dir.display()))?; + let dest = dir.join(entry.archive_name()); + std::fs::copy(archive, &dest) + .with_context(|| format!("writing {}", dest.display()))?; + std::fs::write(dir.join(entry.meta_name()), meta)?; + } + Store::S3(cfg) => { + s3::put_file(cfg, &entry.archive_name(), archive)?; + s3::put_bytes(cfg, &entry.meta_name(), &meta)?; + } + } + Ok(()) + } + + /// Fetch an entry's archive into `dest`. `Ok(false)` means it is not there. + pub fn fetch_entry(&self, entry: &Entry, dest: &Path) -> Result { + match self { + Store::Disk(dir) => { + let src = dir.join(entry.archive_name()); + if !src.exists() { + return Ok(false); + } + std::fs::copy(&src, dest).with_context(|| format!("reading {}", src.display()))?; + Ok(true) + } + Store::S3(cfg) => s3::get_file(cfg, &entry.archive_name(), dest), + } + } + + /// Every entry in the store, newest first. + pub fn list(&self) -> Result> { + let mut out = Vec::new(); + match self { + Store::Disk(dir) => { + let Ok(rd) = std::fs::read_dir(dir) else { + return Ok(out); + }; + for e in rd.filter_map(|e| e.ok()) { + let p = e.path(); + if p.extension().and_then(|s| s.to_str()) != Some("json") { + continue; + } + if let Ok(bytes) = std::fs::read(&p) { + if let Ok(entry) = serde_json::from_slice::(&bytes) { + out.push(entry); + } + } + } + } + Store::S3(cfg) => { + for name in s3::list(cfg, "")? { + if !name.ends_with(".json") { + continue; + } + if let Some(bytes) = s3::get_bytes(cfg, &name)? { + if let Ok(entry) = serde_json::from_slice::(&bytes) { + out.push(entry); + } + } + } + } + } + out.sort_by_key(|e| std::cmp::Reverse(e.created)); + Ok(out) + } + + /// Metadata for one exact key. + fn get(&self, key: &str) -> Result> { + let name = format!("{}.json", Entry::storage_name(key)); + let bytes = match self { + Store::Disk(dir) => std::fs::read(dir.join(&name)).ok(), + Store::S3(cfg) => s3::get_bytes(cfg, &name)?, + }; + Ok(bytes.and_then(|b| serde_json::from_slice(&b).ok())) + } + + /// Remove an entry. Returns whether there was one. + pub fn remove(&self, key: &str) -> Result { + let Some(entry) = self.get(key)? else { + return Ok(false); + }; + match self { + Store::Disk(dir) => { + let _ = std::fs::remove_file(dir.join(entry.archive_name())); + let _ = std::fs::remove_file(dir.join(entry.meta_name())); + } + Store::S3(cfg) => { + s3::delete(cfg, &entry.archive_name())?; + s3::delete(cfg, &entry.meta_name())?; + } + } + Ok(true) + } +} + +/// Resolve `key`, then each `restore_keys` prefix in turn. +/// +/// Exact match wins. Otherwise the prefixes are tried **in the order given** — +/// they are a preference list, not a set — and within one prefix the newest +/// matching entry wins. That is GitHub Actions' rule, and the reason a +/// lockfile-hashed key can fall back to "any cache for this OS" instead of +/// rebuilding from nothing. +pub fn resolve_key(store: &Store, key: &str, restore_keys: &[String]) -> Result> { + if let Some(hit) = store.get(key)? { + return Ok(Some(hit)); + } + if restore_keys.is_empty() { + return Ok(None); + } + let all = store.list()?; // already newest-first + for prefix in restore_keys { + if let Some(hit) = all.iter().find(|e| e.key.starts_with(prefix.as_str())) { + return Ok(Some(hit.clone())); + } + } + Ok(None) +} + +/// Default disk store: `/caches`, beside the image cache. +fn disk_dir() -> Result { + Ok(crate::fetch::cache_dir()?.join("caches")) +} + +/// Compress a tar stream from `tar_source` into a new archive file. +/// +/// Returns the temp file holding it, its size and its digest — the caller +/// uploads it and then throws it away. +pub fn write_archive( + compression: Compression, + tar_source: F, +) -> Result<(tempfile::NamedTempFile, u64, String)> +where + F: FnOnce(&mut dyn Write) -> Result<()>, +{ + let tmp = tempfile::NamedTempFile::new().context("creating a temporary archive")?; + let file = tmp.reopen()?; + let mut sink = archive::Sink::new(std::io::BufWriter::new(file), compression)?; + tar_source(&mut sink)?; + sink.finish()?.flush()?; + + let mut hasher = Sha256::new(); + let mut f = std::fs::File::open(tmp.path())?; + let mut buf = vec![0u8; 64 * 1024]; + let mut size = 0u64; + loop { + let n = f.read(&mut buf)?; + if n == 0 { + break; + } + size += n as u64; + hasher.update(&buf[..n]); + } + Ok((tmp, size, format!("sha256:{:x}", hasher.finalize()))) +} + +/// Decompress an archive into a plain tar, ready to stream into the guest. +/// +/// eStargz's TOC and landmark are stripped here rather than in the guest: they +/// are real tar members, so a guest `tar -xf` would otherwise drop a +/// `stargz.index.json` into the restored directory. +pub fn open_archive(path: &Path, compression: Compression) -> Result> { + let file = std::fs::File::open(path) + .with_context(|| format!("opening the cache archive {}", path.display()))?; + if compression == Compression::Estargz { + let mut stripped = tempfile::tempfile()?; + archive::strip_estargz(archive::reader(file, compression)?, &mut stripped)?; + use std::io::Seek; + stripped.seek(std::io::SeekFrom::Start(0))?; + return Ok(Box::new(stripped)); + } + archive::reader(file, compression) +} + +/// Verify a fetched archive against the digest its metadata recorded. +pub fn verify(path: &Path, expected: &str) -> Result<()> { + if expected.is_empty() { + return Ok(()); + } + let mut hasher = Sha256::new(); + let mut f = std::fs::File::open(path)?; + let mut buf = vec![0u8; 64 * 1024]; + loop { + let n = f.read(&mut buf)?; + if n == 0 { + break; + } + hasher.update(&buf[..n]); + } + let got = format!("sha256:{:x}", hasher.finalize()); + if got != expected { + bail!("the cache archive is corrupt: expected {expected}, got {got}"); + } + Ok(()) +} + +/// Unix seconds, for an entry's `created`. +pub fn now() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_secs()) + .unwrap_or(0) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn entry(key: &str, created: u64) -> Entry { + Entry { + key: key.to_string(), + path: "/root/.cargo".into(), + compression: Compression::Gzip, + size: 1, + created, + digest: String::new(), + } + } + + fn disk_store() -> (tempfile::TempDir, Store) { + let dir = tempfile::tempdir().unwrap(); + let store = Store::Disk(dir.path().to_path_buf()); + (dir, store) + } + + /// A key is arbitrary user text — a lockfile hash, a branch name, a path. + /// The name it maps to has to be a safe filename *and* an S3 key, and two + /// different keys must never collide onto one. + #[test] + fn storage_names_are_safe_and_collision_free() { + let name = Entry::storage_name("deps/linux amd64:v2"); + assert!( + name.chars() + .all(|c| c.is_ascii_alphanumeric() || c == '-' || c == '_' || c == '.'), + "{name}" + ); + assert_ne!( + Entry::storage_name("deps/linux"), + Entry::storage_name("deps-linux"), + "keys that slugify the same must still differ" + ); + // Long keys stay bounded rather than producing an unusable filename. + assert!(Entry::storage_name(&"x".repeat(500)).len() <= 49); + } + + #[test] + fn an_empty_or_unslugifiable_key_still_gets_a_name() { + assert!(!Entry::storage_name("").is_empty()); + assert!(!Entry::storage_name("///").is_empty()); + } + + #[test] + fn saving_then_listing_round_trips_the_metadata() { + let (_dir, store) = disk_store(); + let archive = tempfile::NamedTempFile::new().unwrap(); + std::fs::write(archive.path(), b"not really a tar").unwrap(); + + let e = entry("deps-v1", 100); + store.put_entry(&e, archive.path()).unwrap(); + + let listed = store.list().unwrap(); + assert_eq!(listed.len(), 1); + assert_eq!(listed[0].key, "deps-v1"); + assert_eq!(listed[0].path, "/root/.cargo"); + + let dest = tempfile::NamedTempFile::new().unwrap(); + assert!(store.fetch_entry(&e, dest.path()).unwrap()); + assert_eq!(std::fs::read(dest.path()).unwrap(), b"not really a tar"); + } + + #[test] + fn removing_takes_both_the_archive_and_its_metadata() { + let (dir, store) = disk_store(); + let archive = tempfile::NamedTempFile::new().unwrap(); + std::fs::write(archive.path(), b"x").unwrap(); + store.put_entry(&entry("k", 1), archive.path()).unwrap(); + + assert!(store.remove("k").unwrap()); + assert!( + !store.remove("k").unwrap(), + "removing twice is not an error" + ); + assert_eq!(std::fs::read_dir(dir.path()).unwrap().count(), 0); + } + + /// The fallback rule is the whole reason keys are worth having: an exact + /// miss should still find the newest compatible entry. + #[test] + fn restore_keys_fall_back_in_order_and_prefer_the_newest() { + let (_dir, store) = disk_store(); + let archive = tempfile::NamedTempFile::new().unwrap(); + std::fs::write(archive.path(), b"x").unwrap(); + for (key, created) in [ + ("deps-linux-aaa", 100), + ("deps-linux-bbb", 300), + ("deps-macos-ccc", 200), + ] { + store + .put_entry(&entry(key, created), archive.path()) + .unwrap(); + } + + // Exact hit wins outright. + let hit = resolve_key(&store, "deps-linux-aaa", &["deps-".into()]) + .unwrap() + .unwrap(); + assert_eq!(hit.key, "deps-linux-aaa"); + + // A miss falls back to the newest entry under the prefix. + let hit = resolve_key(&store, "deps-linux-zzz", &["deps-linux-".into()]) + .unwrap() + .unwrap(); + assert_eq!(hit.key, "deps-linux-bbb"); + + // Prefixes are a preference list, tried in order. + let hit = resolve_key( + &store, + "nope", + &["deps-macos-".into(), "deps-linux-".into()], + ) + .unwrap() + .unwrap(); + assert_eq!(hit.key, "deps-macos-ccc"); + + // No prefix matches, and no restore keys at all. + assert!(resolve_key(&store, "nope", &["other-".into()]) + .unwrap() + .is_none()); + assert!(resolve_key(&store, "nope", &[]).unwrap().is_none()); + } + + #[test] + fn a_corrupt_archive_is_caught_by_its_digest() { + let f = tempfile::NamedTempFile::new().unwrap(); + std::fs::write(f.path(), b"hello").unwrap(); + let good = format!("sha256:{:x}", Sha256::digest(b"hello")); + verify(f.path(), &good).unwrap(); + verify(f.path(), "").unwrap(); // no digest recorded: nothing to check + + let err = verify(f.path(), "sha256:deadbeef").unwrap_err().to_string(); + assert!(err.contains("corrupt"), "{err}"); + } + + /// Metadata is written after the archive so an interrupted save leaves an + /// invisible entry rather than one that resolves to a partial file. + #[test] + fn an_archive_without_metadata_is_not_listed() { + let (dir, store) = disk_store(); + std::fs::write(dir.path().join("orphan-12345678.tar.gz"), b"x").unwrap(); + assert!(store.list().unwrap().is_empty()); + } +} diff --git a/core/src/cache/s3.rs b/core/src/cache/s3.rs new file mode 100644 index 0000000..8b2fe76 --- /dev/null +++ b/core/src/cache/s3.rs @@ -0,0 +1,563 @@ +//! The S3 cache backend: AWS Signature Version 4, sent with `curl`. +//! +//! Signing is a pure function of the request, so it needs no HTTP client — we +//! compute the `Authorization` header and hand the request to `curl`, which +//! this crate already depends on for every image pull. That keeps the "no HTTP +//! crate, no async runtime" rule in `oci.rs` intact and works against anything +//! that speaks S3: AWS, Cloudflare R2, MinIO, Backblaze B2. +//! +//! Reference: + +use std::path::Path; +use std::process::Command; + +use anyhow::{bail, Context, Result}; +use hmac::{Hmac, Mac}; +use sha2::{Digest, Sha256}; + +use super::config::{credentials, Credentials, S3Config}; + +type HmacSha256 = Hmac; + +/// `UNSIGNED-PAYLOAD` is allowed over HTTPS, but signing the real hash is what +/// MinIO and older gateways expect, and it costs one pass over a file we have +/// on disk anyway. +fn sha256_hex(bytes: &[u8]) -> String { + format!("{:x}", Sha256::digest(bytes)) +} + +fn sha256_file(path: &Path) -> Result { + use std::io::Read; + let mut f = std::fs::File::open(path) + .with_context(|| format!("opening {} to sign it", path.display()))?; + let mut hasher = Sha256::new(); + let mut buf = vec![0u8; 64 * 1024]; + loop { + let n = f.read(&mut buf)?; + if n == 0 { + break; + } + hasher.update(&buf[..n]); + } + Ok(format!("{:x}", hasher.finalize())) +} + +fn hmac(key: &[u8], data: &str) -> Vec { + let mut mac = HmacSha256::new_from_slice(key).expect("hmac takes a key of any length"); + mac.update(data.as_bytes()); + mac.finalize().into_bytes().to_vec() +} + +/// Percent-encode a path segment per RFC 3986, leaving `/` alone. +/// +/// S3 signs the *encoded* path, so this has to agree byte for byte with what +/// curl puts on the wire — which is why keys are restricted to a safe alphabet +/// upstream rather than relying on this to normalise anything exotic. +fn uri_encode(s: &str, encode_slash: bool) -> String { + let mut out = String::with_capacity(s.len()); + for b in s.bytes() { + match b { + b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => { + out.push(b as char) + } + b'/' if !encode_slash => out.push('/'), + _ => out.push_str(&format!("%{b:02X}")), + } + } + out +} + +/// A signed request, ready to hand to curl. +struct Signed { + url: String, + headers: Vec<(String, String)>, +} + +/// Build the SigV4 headers for one request. +/// +/// `query` must already be in canonical form: sorted by key, URI-encoded. The +/// only query this module sends is ListObjectsV2, which builds it that way. +fn sign( + cfg: &S3Config, + creds: &Credentials, + method: &str, + key: &str, + query: &str, + payload_hash: &str, + now: (String, String), +) -> Signed { + let (date_stamp, amz_date) = now; // (YYYYMMDD, YYYYMMDDTHHMMSSZ) + let host = cfg.host(); + + // A path-style endpoint puts the bucket in the path, so the canonical URI + // has to include it — take whatever follows the host in the base URL. + let base = cfg.base_url(); + let path_prefix = base + .split_once("://") + .map(|(_, rest)| rest) + .and_then(|rest| rest.split_once('/')) + .map(|(_, p)| format!("/{p}")) + .unwrap_or_default(); + let canonical_uri = format!("{}/{}", path_prefix, uri_encode(key, false)); + + let mut headers: Vec<(String, String)> = vec![ + ("host".into(), host.clone()), + ("x-amz-content-sha256".into(), payload_hash.to_string()), + ("x-amz-date".into(), amz_date.clone()), + ]; + if let Some(token) = &creds.session_token { + headers.push(("x-amz-security-token".into(), token.clone())); + } + headers.sort_by(|a, b| a.0.cmp(&b.0)); + + let signed_headers = headers + .iter() + .map(|(k, _)| k.as_str()) + .collect::>() + .join(";"); + let canonical_headers = headers + .iter() + .map(|(k, v)| format!("{k}:{}\n", v.trim())) + .collect::(); + + let canonical_request = format!( + "{method}\n{canonical_uri}\n{query}\n{canonical_headers}\n{signed_headers}\n{payload_hash}" + ); + + let scope = format!("{date_stamp}/{}/s3/aws4_request", cfg.region); + let string_to_sign = format!( + "AWS4-HMAC-SHA256\n{amz_date}\n{scope}\n{}", + sha256_hex(canonical_request.as_bytes()) + ); + + let k_date = hmac(format!("AWS4{}", creds.secret_key).as_bytes(), &date_stamp); + let k_region = hmac(&k_date, &cfg.region); + let k_service = hmac(&k_region, "s3"); + let k_signing = hmac(&k_service, "aws4_request"); + let signature = hex(&hmac(&k_signing, &string_to_sign)); + + let authorization = format!( + "AWS4-HMAC-SHA256 Credential={}/{scope}, SignedHeaders={signed_headers}, Signature={signature}", + creds.access_key + ); + + let mut out: Vec<(String, String)> = headers + .into_iter() + .filter(|(k, _)| k != "host") // curl derives Host from the URL + .collect(); + out.push(("authorization".into(), authorization)); + + let url = if query.is_empty() { + format!("{base}/{key}") + } else { + format!("{base}/{key}?{query}") + }; + Signed { url, headers: out } +} + +fn hex(bytes: &[u8]) -> String { + bytes.iter().map(|b| format!("{b:02x}")).collect() +} + +/// `(YYYYMMDD, YYYYMMDDTHHMMSSZ)` for right now, which is what SigV4 wants. +fn timestamps() -> (String, String) { + let secs = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_secs()) + .unwrap_or(0); + let (date, time) = civil(secs); + (date.clone(), format!("{date}T{time}Z")) +} + +/// Split unix seconds into `(YYYYMMDD, HHMMSS)` — the civil-from-days algorithm, +/// same as the estargz TOC's timestamps. +fn civil(secs: u64) -> (String, String) { + let days = (secs / 86_400) as i64; + let tod = secs % 86_400; + let z = days + 719_468; + let era = z.div_euclid(146_097); + let doe = z.rem_euclid(146_097); + let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365; + let y = yoe + era * 400; + let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); + let mp = (5 * doy + 2) / 153; + let d = doy - (153 * mp + 2) / 5 + 1; + let m = if mp < 10 { mp + 3 } else { mp - 9 }; + let y = if m <= 2 { y + 1 } else { y }; + ( + format!("{y:04}{m:02}{d:02}"), + format!("{:02}{:02}{:02}", tod / 3600, (tod % 3600) / 60, tod % 60), + ) +} + +/// Send a signed request, writing the response body to `body_to` and returning +/// the HTTP status. +/// +/// The body always goes to a file and the status always comes back on stdout, +/// because the alternative — `--fail-with-body` plus curl's exit code — cannot +/// tell a 404 from a network error once `-o` has taken the body away. Reading +/// the status directly is what lets "not cached yet" be an ordinary answer +/// rather than a failed command. +fn curl(signed: &Signed, extra: &[&str], body_to: &Path, what: &str) -> Result { + let mut cmd = Command::new("curl"); + cmd.args(["-sS", "--max-time", "900", "-w", "%{http_code}", "-o"]) + .arg(body_to); + for (k, v) in &signed.headers { + cmd.arg("-H").arg(format!("{k}: {v}")); + } + cmd.args(extra).arg(&signed.url); + + let out = cmd + .output() + .map_err(|e| crate::fetch::spawn_error(&cmd, what, e))?; + if !out.status.success() { + bail!( + "{what} failed: {}", + String::from_utf8_lossy(&out.stderr).trim() + ); + } + String::from_utf8_lossy(&out.stdout) + .trim() + .parse() + .with_context(|| format!("{what}: curl reported no HTTP status")) +} + +/// Fail unless the status is a success, quoting S3's own explanation. +fn expect_ok(status: u16, body_to: &Path, what: &str) -> Result<()> { + if (200..300).contains(&status) { + return Ok(()); + } + let body = std::fs::read_to_string(body_to).unwrap_or_default(); + // S3 explains itself in an XML body; surfacing that beats "HTTP 403". + let detail = extract(&body, "Message") + .or_else(|| extract(&body, "Code")) + .unwrap_or_else(|| body.trim().to_string()); + if detail.is_empty() { + bail!("{what} failed with HTTP {status}"); + } + bail!("{what} failed (HTTP {status}): {detail}"); +} + +/// Pull the text out of the first `…`. S3's error and listing bodies +/// are small, flat XML; a parser crate would be the only dependency here that +/// earned nothing. +fn extract(xml: &str, tag: &str) -> Option { + let open = format!("<{tag}>"); + let close = format!(""); + let start = xml.find(&open)? + open.len(); + let end = xml[start..].find(&close)? + start; + Some(xml[start..end].to_string()) +} + +/// Every `` value in the document, in order. +fn extract_all(xml: &str, tag: &str) -> Vec { + let open = format!("<{tag}>"); + let close = format!(""); + let mut out = Vec::new(); + let mut rest = xml; + while let Some(start) = rest.find(&open) { + let from = start + open.len(); + let Some(end) = rest[from..].find(&close) else { + break; + }; + out.push(rest[from..from + end].to_string()); + rest = &rest[from + end + close.len()..]; + } + out +} + +/// Upload `file` to `key`. +pub fn put_file(cfg: &S3Config, key: &str, file: &Path) -> Result<()> { + let creds = credentials()?; + let hash = sha256_file(file)?; + let signed = sign( + cfg, + &creds, + "PUT", + &cfg.object_key(key), + "", + &hash, + timestamps(), + ); + let path = file.to_string_lossy().into_owned(); + let what = format!("uploading the cache entry to s3://{}/{key}", cfg.bucket); + let sink = tempfile::NamedTempFile::new()?; + let status = curl( + &signed, + &["-X", "PUT", "--upload-file", &path], + sink.path(), + &what, + )?; + expect_ok(status, sink.path(), &what) +} + +/// Upload `body` to `key`. Used for the small JSON metadata objects. +pub fn put_bytes(cfg: &S3Config, key: &str, body: &[u8]) -> Result<()> { + let mut tmp = tempfile::NamedTempFile::new()?; + std::io::Write::write_all(&mut tmp, body)?; + std::io::Write::flush(&mut tmp)?; + put_file(cfg, key, tmp.path()) +} + +/// Download `key` to `dest`. `Ok(false)` means it is simply not there. +pub fn get_file(cfg: &S3Config, key: &str, dest: &Path) -> Result { + let creds = credentials()?; + let signed = sign( + cfg, + &creds, + "GET", + &cfg.object_key(key), + "", + &sha256_hex(b""), + timestamps(), + ); + let what = format!("downloading s3://{}/{key}", cfg.bucket); + let status = curl(&signed, &[], dest, &what)?; + // 404 is the answer to "is this cached?", not a failure — and 403 is what a + // bucket that hides missing keys returns instead, so treat both as a miss. + if status == 404 || status == 403 { + let _ = std::fs::remove_file(dest); + return Ok(false); + } + expect_ok(status, dest, &what)?; + Ok(true) +} + +/// Fetch `key` into memory. Used for metadata. +pub fn get_bytes(cfg: &S3Config, key: &str) -> Result>> { + let tmp = tempfile::NamedTempFile::new()?; + if !get_file(cfg, key, tmp.path())? { + return Ok(None); + } + Ok(Some(std::fs::read(tmp.path())?)) +} + +pub fn delete(cfg: &S3Config, key: &str) -> Result<()> { + let creds = credentials()?; + let signed = sign( + cfg, + &creds, + "DELETE", + &cfg.object_key(key), + "", + &sha256_hex(b""), + timestamps(), + ); + let what = format!("deleting s3://{}/{key}", cfg.bucket); + let sink = tempfile::NamedTempFile::new()?; + let status = curl(&signed, &["-X", "DELETE"], sink.path(), &what)?; + // S3 answers 204 for a delete whether or not the key was there. + if status == 404 { + return Ok(()); + } + expect_ok(status, sink.path(), &what) +} + +/// List object keys under `prefix`, relative to any configured bucket prefix. +pub fn list(cfg: &S3Config, prefix: &str) -> Result> { + let creds = credentials()?; + let full = cfg.object_key(prefix); + // Canonical query: sorted by key, values URI-encoded. + let query = format!("list-type=2&prefix={}", uri_encode(&full, true)); + let signed = sign( + cfg, + &creds, + "GET", + "", + &query, + &sha256_hex(b""), + timestamps(), + ); + let what = format!("listing s3://{}/{full}", cfg.bucket); + let sink = tempfile::NamedTempFile::new()?; + let status = curl(&signed, &[], sink.path(), &what)?; + expect_ok(status, sink.path(), &what)?; + let xml = std::fs::read_to_string(sink.path()).unwrap_or_default(); + + let strip = cfg.object_key(""); + Ok(extract_all(&xml, "Key") + .into_iter() + .map(|k| k.trim_start_matches(&strip).to_string()) + .collect()) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn creds() -> Credentials { + Credentials { + access_key: "AKIDEXAMPLE".into(), + secret_key: "wJalrXUtnFEMI/K7MDENG+bPxRfiCYEXAMPLEKEY".into(), + session_token: None, + } + } + + fn cfg() -> S3Config { + S3Config { + bucket: "examplebucket".into(), + region: "us-east-1".into(), + prefix: String::new(), + endpoint: None, + } + } + + fn authorization(signed: &Signed) -> String { + signed + .headers + .iter() + .find(|(k, _)| k == "authorization") + .map(|(_, v)| v.clone()) + .expect("every signed request carries an authorization header") + } + + /// The signature is pinned against a value computed by an independent + /// implementation of SigV4 (python hmac/hashlib) over the same inputs. A + /// wrong signature is a flat 403 with no hint as to which of the five + /// derivation steps drifted, so this is the test that has to be exact + /// rather than approximate. + #[test] + fn signing_matches_an_independent_sigv4_implementation() { + let signed = sign( + &cfg(), + &creds(), + "GET", + "test.txt", + "", + &sha256_hex(b""), + ("20130524".into(), "20130524T000000Z".into()), + ); + let auth = authorization(&signed); + assert!( + auth.contains("Credential=AKIDEXAMPLE/20130524/us-east-1/s3/aws4_request"), + "{auth}" + ); + assert!( + auth.contains("SignedHeaders=host;x-amz-content-sha256;x-amz-date"), + "{auth}" + ); + assert!( + auth.contains( + "Signature=0cee62862edd8e0aec93e9fbb49b3463c45da6ba7363da395910103de3775840" + ), + "{auth}" + ); + } + + /// The empty-payload hash is a constant AWS documents; getting it wrong + /// signs a body we never send. + #[test] + fn the_empty_payload_hash_is_the_documented_constant() { + assert_eq!( + sha256_hex(b""), + "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + ); + } + + /// A session token has to be both sent and signed; signing without it is a + /// 403 that reads like bad credentials. + #[test] + fn a_session_token_is_signed_not_just_sent() { + let mut c = creds(); + c.session_token = Some("tok".into()); + let signed = sign( + &cfg(), + &c, + "GET", + "k", + "", + &sha256_hex(b""), + ("20130524".into(), "20130524T000000Z".into()), + ); + let auth = authorization(&signed); + assert!(auth.contains("x-amz-security-token"), "{auth}"); + assert!(signed + .headers + .iter() + .any(|(k, v)| k == "x-amz-security-token" && v == "tok")); + } + + /// A path-style endpoint puts the bucket in the URI, and the *signature* + /// covers that path — sign the AWS-style path against MinIO and every + /// request fails authentication. + #[test] + fn a_path_style_endpoint_signs_the_bucket_in_the_uri() { + let cfg = S3Config { + bucket: "b".into(), + region: "auto".into(), + prefix: String::new(), + endpoint: Some("https://minio.local:9000".into()), + }; + let signed = sign( + &cfg, + &creds(), + "GET", + "obj", + "", + &sha256_hex(b""), + ("20130524".into(), "20130524T000000Z".into()), + ); + assert_eq!(signed.url, "https://minio.local:9000/b/obj"); + } + + #[test] + fn uri_encoding_leaves_safe_characters_alone() { + assert_eq!(uri_encode("a/b-c_d.e~f", false), "a/b-c_d.e~f"); + assert_eq!(uri_encode("a/b", true), "a%2Fb"); + assert_eq!(uri_encode("a b+c", false), "a%20b%2Bc"); + } + + #[test] + fn error_bodies_are_read_for_their_message() { + let xml = "NoSuchKeyThe specified key does not exist."; + assert_eq!(extract(xml, "Code").as_deref(), Some("NoSuchKey")); + assert_eq!( + extract(xml, "Message").as_deref(), + Some("The specified key does not exist.") + ); + } + + #[test] + fn listings_yield_every_key() { + let xml = "a.json\ + b.tar.gz"; + assert_eq!(extract_all(xml, "Key"), vec!["a.json", "b.tar.gz"]); + } + + /// A 404 has to stay distinguishable from a failure: it is the answer to + /// "is this key cached yet?", which every `cache save` asks first. Reading + /// it out of curl's exit code instead of the status made the very first + /// save against a real bucket fail. + #[test] + fn a_successful_status_passes_and_a_failure_quotes_s3s_own_message() { + let body = tempfile::NamedTempFile::new().unwrap(); + std::fs::write(body.path(), "").unwrap(); + expect_ok(200, body.path(), "x").unwrap(); + expect_ok(204, body.path(), "x").unwrap(); + + std::fs::write( + body.path(), + "AccessDeniedAccess Denied.", + ) + .unwrap(); + let err = expect_ok(403, body.path(), "uploading") + .unwrap_err() + .to_string(); + assert!(err.contains("403"), "{err}"); + assert!(err.contains("Access Denied."), "{err}"); + + // An empty body still has to name the status rather than say nothing. + std::fs::write(body.path(), "").unwrap(); + let err = expect_ok(500, body.path(), "uploading") + .unwrap_err() + .to_string(); + assert!(err.contains("500"), "{err}"); + } + + #[test] + fn timestamps_are_the_shape_sigv4_requires() { + let (date, time) = civil(1_369_353_600); // 2013-05-24T00:00:00Z + assert_eq!(date, "20130524"); + assert_eq!(time, "000000"); + } +} diff --git a/core/src/cli.rs b/core/src/cli.rs index d0605da..58f9b38 100644 --- a/core/src/cli.rs +++ b/core/src/cli.rs @@ -162,6 +162,12 @@ pub enum Command { /// Copy files between the host and a running machine (like `docker cp`). Cp(CpArgs), + /// Save and restore a guest directory under a key (host disk or S3). + Cache(CacheArgs), + + /// Check that this host can run machines, and say what to fix if not. + Doctor(DoctorArgs), + /// Manage tailscale inside a running machine (install/start/status/setup). Tailscale(TailscaleArgs), @@ -705,6 +711,106 @@ pub struct CpArgs { pub dst: String, } +/// Save and restore guest directories under a key, so a rebuild can pick up +/// where the last one left off. +#[derive(Parser, Serialize, Deserialize)] +#[command(after_help = "\ +EXAMPLES: + bsdkrun cache save web:/root/.cargo --key cargo-$(shasum Cargo.lock | cut -c1-12) + bsdkrun cache restore web --key cargo-abc123 --restore-keys cargo- + bsdkrun cache ls + bsdkrun cache rm cargo-abc123 + +Entries go to the host disk by default. Point them at S3 with +BSDKRUN_CACHE_BACKEND=s3 and BSDKRUN_CACHE_S3_BUCKET, or ~/.config/bsdkrun/cache.toml.")] +pub struct CacheArgs { + #[command(subcommand)] + pub cmd: CacheCmd, +} + +#[derive(clap::Subcommand, Serialize, Deserialize)] +pub enum CacheCmd { + /// Archive a guest directory and store it under a key. + Save(CacheSaveArgs), + /// Restore a stored tree into a machine. + Restore(CacheRestoreArgs), + /// List stored entries. + Ls(CacheLsArgs), + /// Remove stored entries. + Rm(CacheRmArgs), +} + +#[derive(Parser, Serialize, Deserialize)] +pub struct CacheSaveArgs { + /// What to archive, as `ID:PATH` (the machine must be running). + #[arg(value_name = "ID:PATH")] + pub target: String, + + /// Key to store it under. Make it name the content — a lockfile hash is the + /// usual choice — so a changed dependency set gets a different entry. + #[arg(short = 'k', long)] + pub key: String, + + /// Archive format: gzip (default), zstd, estargz, or none. + #[arg(short = 'c', long, default_value = "gzip", value_name = "FORMAT")] + pub compression: String, + + /// Replace an entry that already has this key. + #[arg(short = 'f', long)] + pub force: bool, + + /// Print the result as JSON instead of a human summary. + #[arg(long)] + pub json: bool, +} + +#[derive(Parser, Serialize, Deserialize)] +pub struct CacheRestoreArgs { + /// Where to restore, as `ID` or `ID:PATH`. Without a path, the entry goes + /// back to the directory it was saved from. + #[arg(value_name = "ID[:PATH]")] + pub target: String, + + /// Key to look for. + #[arg(short = 'k', long)] + pub key: String, + + /// Prefixes to fall back on when the key misses, most preferred first. + /// Within a prefix the newest matching entry wins. + #[arg(long, value_name = "PREFIX", num_args = 1..)] + pub restore_keys: Vec, + + /// Print the result as JSON. `restored` says whether anything was found, + /// and `key` which entry a `--restore-keys` fallback landed on. + #[arg(long)] + pub json: bool, +} + +#[derive(Parser, Default, Serialize, Deserialize)] +pub struct CacheLsArgs { + /// Print the entries as JSON. + #[arg(long)] + pub json: bool, +} + +#[derive(Parser, Serialize, Deserialize)] +pub struct CacheRmArgs { + /// Remove every entry in the store. + #[arg(long, conflicts_with = "keys")] + pub all: bool, + + /// Key(s) to remove. + #[arg(value_name = "KEY", required_unless_present = "all")] + pub keys: Vec, +} + +#[derive(Parser, Default, Serialize, Deserialize)] +pub struct DoctorArgs { + /// Print the report as JSON. + #[arg(long)] + pub json: bool, +} + #[derive(Parser, Serialize, Deserialize)] pub struct TailscaleArgs { /// machine id (a unique prefix is enough). diff --git a/core/src/commands/cache.rs b/core/src/commands/cache.rs new file mode 100644 index 0000000..50eeb79 --- /dev/null +++ b/core/src/commands/cache.rs @@ -0,0 +1,268 @@ +//! `bsdkrun cache` — the CLI over [`crate::cache`]. + +use anyhow::{bail, Context, Result}; + +use crate::cache::{self, archive::Compression, Entry, Store}; + +use super::guest::{agent_error, agent_target}; +use super::truncate; + +/// What `--json` prints for a save or a restore. +/// +/// A restore's headline fact is whether it *hit*, and on a fallback which key +/// it actually landed on — neither of which a caller should have to read out of +/// a human sentence on stderr. The SDKs are built on this. +#[derive(serde::Serialize)] +struct Outcome<'a> { + /// Restore only: whether anything was found. Omitted for a save, where it + /// would be a field with no meaning. + #[serde(skip_serializing_if = "Option::is_none")] + restored: Option, + /// The key asked for. + requested_key: &'a str, + /// The entry actually used, when there was one. Differs from + /// `requested_key` when a `--restore-keys` prefix matched. + #[serde(skip_serializing_if = "Option::is_none")] + key: Option<&'a str>, + #[serde(skip_serializing_if = "Option::is_none")] + path: Option<&'a str>, + #[serde(skip_serializing_if = "Option::is_none")] + size: Option, + #[serde(skip_serializing_if = "Option::is_none")] + compression: Option, + #[serde(skip_serializing_if = "Option::is_none")] + created: Option, +} + +/// Save the guest directory at `path` under `key`. +pub(crate) fn cmd_save( + id: &str, + path: &str, + key: &str, + compression: Compression, + force: bool, + json: bool, +) -> Result<()> { + if key.trim().is_empty() { + bail!("a cache entry needs a --key"); + } + let store = Store::open()?; + if !force { + if let Some(existing) = cache::resolve_key(&store, key, &[])? { + bail!( + "{key:?} is already cached ({} from {}, saved {}). Pass --force to replace it, \ + or use a key that names this exact content — a lockfile hash, say.", + human_size(existing.size), + existing.path, + ago(existing.created) + ); + } + } + + let (vm, port) = agent_target(id)?; + if !json { + eprintln!("saving {}:{path} as {key:?} ({compression})", vm.id); + } + + let (tmp, size, digest) = cache::write_archive(compression, |sink| { + super::cp::stream_dir_out(&vm.id, port, path, sink).map_err(|e| agent_error(&vm.kind, e)) + })?; + + let entry = Entry { + key: key.to_string(), + path: path.to_string(), + compression, + size, + created: cache::now(), + digest, + }; + store.put_entry(&entry, tmp.path())?; + if json { + println!( + "{}", + serde_json::to_string_pretty(&Outcome { + restored: None, + requested_key: key, + key: Some(key), + path: Some(path), + size: Some(size), + compression: Some(compression.to_string()), + created: Some(entry.created), + })? + ); + } else { + eprintln!( + "cached {key:?} — {} to {}", + human_size(size), + store.describe() + ); + } + Ok(()) +} + +/// Restore a cached tree into a machine. +pub(crate) fn cmd_restore( + id: &str, + path: Option<&str>, + key: &str, + restore_keys: &[String], + json: bool, +) -> Result<()> { + let store = Store::open()?; + let Some(entry) = cache::resolve_key(&store, key, restore_keys)? else { + // A cache miss is the normal case on a first run, not a failure — the + // caller builds from scratch and saves afterwards. Exit 0 and say so, + // the way actions/cache does, so `cache restore || true` is unnecessary. + if json { + println!( + "{}", + serde_json::to_string_pretty(&Outcome { + restored: Some(false), + requested_key: key, + key: None, + path: None, + size: None, + compression: None, + created: None, + })? + ); + } else { + eprintln!( + "cache miss for {key:?}{} — nothing restored", + if restore_keys.is_empty() { + String::new() + } else { + format!(" (and {} restore-key prefix(es))", restore_keys.len()) + } + ); + } + return Ok(()); + }; + if entry.key != key && !json { + eprintln!("cache miss for {key:?}; restoring {:?} instead", entry.key); + } + + let target = path.unwrap_or(&entry.path); + let (vm, port) = agent_target(id)?; + + let tmp = tempfile::NamedTempFile::new().context("creating a temporary archive")?; + if !store.fetch_entry(&entry, tmp.path())? { + bail!( + "the metadata for {:?} is in {} but its archive is gone — remove it with \ + `bsdkrun cache rm {}`", + entry.key, + store.describe(), + entry.key + ); + } + cache::verify(tmp.path(), &entry.digest)?; + + let stream = cache::open_archive(tmp.path(), entry.compression)?; + super::cp::stream_dir_in(&vm.id, port, target, stream).map_err(|e| agent_error(&vm.kind, e))?; + + if json { + println!( + "{}", + serde_json::to_string_pretty(&Outcome { + restored: Some(true), + requested_key: key, + key: Some(&entry.key), + path: Some(target), + size: Some(entry.size), + compression: Some(entry.compression.to_string()), + created: Some(entry.created), + })? + ); + } else { + eprintln!( + "restored {:?} ({}, saved {}) into {}:{target}", + entry.key, + human_size(entry.size), + ago(entry.created), + vm.id + ); + } + Ok(()) +} + +pub(crate) fn cmd_ls(json: bool) -> Result<()> { + let store = Store::open()?; + let entries = store.list()?; + + if json { + println!("{}", serde_json::to_string_pretty(&entries)?); + return Ok(()); + } + if entries.is_empty() { + println!("no cache entries in {}", store.describe()); + return Ok(()); + } + println!( + "{:<32} {:<24} {:>10} {:<9} SAVED", + "KEY", "PATH", "SIZE", "FORMAT" + ); + for e in entries { + println!( + "{:<32} {:<24} {:>10} {:<9} {}", + truncate(&e.key, 32), + truncate(&e.path, 24), + human_size(e.size), + e.compression, + ago(e.created) + ); + } + Ok(()) +} + +pub(crate) fn cmd_rm(keys: &[String], all: bool) -> Result<()> { + let store = Store::open()?; + let owned: Vec; + let keys = if all { + owned = store.list()?.into_iter().map(|e| e.key).collect(); + if owned.is_empty() { + println!("no cache entries in {}", store.describe()); + return Ok(()); + } + &owned + } else { + keys + }; + for key in keys { + if store.remove(key)? { + println!("{key}"); + } else { + eprintln!("no cache entry for {key:?}"); + } + } + Ok(()) +} + +fn human_size(bytes: u64) -> String { + crate::oci::human_size(bytes) +} + +/// A rough "3 hours ago", matching how `ps` renders ages. +fn ago(unix: u64) -> String { + let now = cache::now(); + let secs = now.saturating_sub(unix); + match secs { + 0..=59 => "just now".to_string(), + 60..=3599 => format!("{} minutes ago", secs / 60), + 3600..=86_399 => format!("{} hours ago", secs / 3600), + _ => format!("{} days ago", secs / 86_400), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn ages_read_as_prose() { + let now = cache::now(); + assert_eq!(ago(now), "just now"); + assert_eq!(ago(now - 120), "2 minutes ago"); + assert_eq!(ago(now - 7200), "2 hours ago"); + assert_eq!(ago(now - 3 * 86_400), "3 days ago"); + } +} diff --git a/core/src/commands/cp.rs b/core/src/commands/cp.rs index 0a05d4b..9753e4c 100644 --- a/core/src/commands/cp.rs +++ b/core/src/commands/cp.rs @@ -155,7 +155,8 @@ fn upload(id: &str, from: &Endpoint, dst: &str, recursive: bool) -> Result<()> { /// directory's *contents* (`-C dir .`), which is what makes `upload` land them /// in the destination rather than one level below it. fn tar_from(dir: &Path) -> Result> { - let mut child = Command::new("tar") + let mut cmd = Command::new("tar"); + cmd // macOS `tar` is bsdtar, which stores each file's extended attributes // in a sidecar AppleDouble member — so an uploaded directory arrives in // the guest with a `._main.py` next to every `main.py`, and a `._.` at @@ -168,9 +169,10 @@ fn tar_from(dir: &Path) -> Result> { .arg(dir) .arg(".") .stdout(Stdio::piped()) - .stderr(Stdio::null()) + .stderr(Stdio::null()); + let mut child = cmd .spawn() - .with_context(|| format!("running tar to pack {}", dir.display()))?; + .map_err(|e| crate::fetch::spawn_error(&cmd, "packing a directory to copy", e))?; Ok(Box::new(child.stdout.take().expect("tar stdout is piped"))) } @@ -205,14 +207,15 @@ fn download(id: &str, src: &str, to: &Endpoint, recursive: bool) -> Result<()> { } Endpoint::Host(p) if recursive => { std::fs::create_dir_all(p).with_context(|| format!("creating {}", p.display()))?; - let mut child = Command::new("tar") - .arg("-xf") + let mut cmd = Command::new("tar"); + cmd.arg("-xf") .arg("-") .arg("-C") .arg(p) - .stdin(Stdio::piped()) + .stdin(Stdio::piped()); + let mut child = cmd .spawn() - .with_context(|| format!("running tar to unpack into {}", p.display()))?; + .map_err(|e| crate::fetch::spawn_error(&cmd, "unpacking a copied directory", e))?; let mut sink = child.stdin.take().expect("tar stdin is piped"); let res = agent::exec_stream(port, &argv, None, &mut sink); drop(sink); // EOF, or tar waits forever for more of the archive @@ -257,6 +260,46 @@ fn download(id: &str, src: &str, to: &Endpoint, recursive: bool) -> Result<()> { Ok(()) } +// --- reused by `bsdkrun cache` ----------------------------------------------- +// +// Saving a cache is `cp -r` out of the guest with the stream compressed instead +// of untarred, and restoring is `cp -r` in. Sharing the scripts keeps one +// definition of what a directory transfer means, so a fix to the guest side +// lands in both. + +/// Stream a tar of the guest directory at `path` into `out`. +pub(crate) fn stream_dir_out( + id: &str, + port: u16, + path: &str, + out: &mut dyn std::io::Write, +) -> Result<()> { + let (code, err) = agent::exec_stream(port, &sh(GET_DIR, &[path.to_string()]), None, out)?; + if code != 0 { + bail!("{}", guest_failure(id, path, code, &err, true)); + } + Ok(()) +} + +/// Unpack a tar stream into the guest at `path`, creating it if needed. +pub(crate) fn stream_dir_in( + id: &str, + port: u16, + path: &str, + input: Box, +) -> Result<()> { + let (code, err) = agent::exec_stream( + port, + &sh(PUT_DIR, &[path.to_string()]), + Some(input), + &mut std::io::sink(), + )?; + if code != 0 { + bail!("{}", guest_failure(id, path, code, &err, true)); + } + Ok(()) +} + // --- shared ------------------------------------------------------------------ /// Wrap a script as `sh -c SCRIPT sh ARGS…`. The paths travel as positional diff --git a/core/src/commands/doctor.rs b/core/src/commands/doctor.rs new file mode 100644 index 0000000..b8ac494 --- /dev/null +++ b/core/src/commands/doctor.rs @@ -0,0 +1,369 @@ +//! `bsdkrun doctor` — check that this host can actually run machines, and say +//! what to do about it when it can't. +//! +//! Everything here is a thing that has, at some point, failed in a way that did +//! not name itself: a missing `curl` surfaces as `No such file or directory` +//! attached to an image pull, an unsigned binary as a bare `EINVAL` from +//! `krun_start_enter`, a case-insensitive store as a nix build that cannot +//! create a directory that already exists. One command, one place to look. + +use std::path::PathBuf; +use std::process::Command; + +use anyhow::Result; +use serde::Serialize; + +/// How a single check came out. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +enum State { + /// Working. + Ok, + /// Works, but something will be missing or slower. + Warn, + /// Machines will not run until this is fixed. + Fail, +} + +impl State { + fn marker(self) -> &'static str { + match self { + State::Ok => "ok ", + State::Warn => "warn", + State::Fail => "FAIL", + } + } +} + +#[derive(Debug, Serialize)] +struct Check { + name: &'static str, + state: State, + detail: String, + /// What to do about it. Only set when there is something to do. + #[serde(skip_serializing_if = "Option::is_none")] + fix: Option, +} + +impl Check { + fn ok(name: &'static str, detail: impl Into) -> Check { + Check { + name, + state: State::Ok, + detail: detail.into(), + fix: None, + } + } + + fn warn(name: &'static str, detail: impl Into, fix: impl Into) -> Check { + Check { + name, + state: State::Warn, + detail: detail.into(), + fix: Some(fix.into()), + } + } + + fn fail(name: &'static str, detail: impl Into, fix: impl Into) -> Check { + Check { + name, + state: State::Fail, + detail: detail.into(), + fix: Some(fix.into()), + } + } +} + +/// Run every check and report. Exits 1 if anything failed, so CI can gate on it. +pub(crate) fn cmd_doctor(json: bool) -> Result<()> { + let checks = gather(); + + if json { + println!( + "{}", + serde_json::to_string_pretty(&serde_json::json!({ + "ok": !checks.iter().any(|c| c.state == State::Fail), + "version": crate::VERSION, + "checks": checks, + }))? + ); + } else { + println!("bsdkrun {} on {}", crate::VERSION, platform()); + println!(); + let width = checks.iter().map(|c| c.name.len()).max().unwrap_or(0); + for c in &checks { + println!("[{}] {: Vec { + let mut checks = vec![ + // The host tools bsdkrun shells out to instead of linking. Missing curl + // is the one that reads worst: it fails inside an image pull, so it + // looks like the registry is unreachable. + tool( + "curl", + true, + "every image pull and download goes through it", + ), + tool("tar", true, "unpacking image layers, `cp -r`, and `cache`"), + tool( + "gzip", + false, + "not required — archives are compressed in-process", + ), + ]; + checks.push(hypervisor()); + #[cfg(target_os = "macos")] + checks.push(signature()); + checks.push(networking()); + checks.push(writable("state directory", crate::db::state_dir())); + checks.push(writable("cache directory", crate::fetch::cache_dir())); + #[cfg(target_os = "macos")] + checks.push(store()); + checks.push(cache_backend()); + checks +} + +fn platform() -> String { + let arch = crate::host::Arch::current() + .map(|a| a.slug().to_string()) + .unwrap_or_else(|_| "unknown".into()); + format!("{}/{arch}", std::env::consts::OS) +} + +/// Is `program` on PATH? `required` decides whether its absence is fatal. +fn tool(program: &'static str, required: bool, why: &str) -> Check { + match which(program) { + Some(path) => Check::ok(program, path.display().to_string()), + None if required => Check::fail( + program, + format!("not on PATH — {why}"), + format!("brew install {program} (macOS), or your distribution's package"), + ), + None => Check::warn(program, format!("not on PATH — {why}"), "optional"), + } +} + +fn which(program: &str) -> Option { + let out = Command::new("/usr/bin/which").arg(program).output().ok()?; + if !out.status.success() { + return None; + } + let path = String::from_utf8_lossy(&out.stdout).trim().to_string(); + (!path.is_empty()).then(|| PathBuf::from(path)) +} + +/// Can we actually create a VM? This is `bsdkrun probe`, folded in — it is the +/// check that subsumes libkrun linking, the hypervisor, and (on macOS) whether +/// the signature took. +fn hypervisor() -> Check { + #[cfg(feature = "boot")] + { + if let Err(e) = crate::host::check_kvm() { + return Check::fail( + "hypervisor", + format!("{e:#}"), + "on Linux, load kvm and add yourself to the kvm group", + ); + } + match crate::krun::Ctx::new().and_then(|ctx| ctx.set_vm_config(1, 256).map(|_| ())) { + Ok(()) => Check::ok("hypervisor", "libkrun linked; a VM context was created"), + Err(e) => Check::fail( + "hypervisor", + format!("{e:#}"), + #[cfg(target_os = "macos")] + "an EINVAL here is almost always a stripped code signature — see the next line", + #[cfg(not(target_os = "macos"))] + "check that /dev/kvm exists and is accessible", + ), + } + } + #[cfg(not(feature = "boot"))] + Check::warn( + "hypervisor", + "this build cannot start machines (compiled without `boot`)", + "use the bsdkrun CLI binary rather than a daemon-only build", + ) +} + +/// macOS refuses `hv_vm_create` to a process without +/// `com.apple.security.hypervisor`, and an entitlement only counts inside a +/// signature — which `cargo build` strips on every rebuild. +#[cfg(target_os = "macos")] +fn signature() -> Check { + let exe = match std::env::current_exe() { + Ok(p) => p, + Err(e) => { + return Check::warn( + "code signature", + format!("cannot locate this binary: {e}"), + "", + ) + } + }; + let out = Command::new("codesign") + .args(["-d", "--entitlements", "-"]) + .arg(&exe) + .output(); + match out { + Ok(o) + if String::from_utf8_lossy(&o.stdout).contains("com.apple.security.hypervisor") + || String::from_utf8_lossy(&o.stderr).contains("com.apple.security.hypervisor") => + { + Check::ok("code signature", "signed, with the hypervisor entitlement") + } + Ok(_) => Check::fail( + "code signature", + "signed, but without com.apple.security.hypervisor — krun_start_enter \ + will fail with a bare EINVAL that says nothing about signing", + "every `cargo build` relinks, and the linker's ad-hoc signature carries no \n\ + entitlements. Re-sign: make sign-release", + ), + Err(e) => Check::warn("code signature", format!("could not run codesign: {e}"), ""), + } +} + +/// gvproxy is what gives a guest a network, port forwards and the exec agent — +/// without it a machine still boots, but `exec`, `cp` and `cache` cannot reach +/// it. +fn networking() -> Check { + match crate::net::locate() { + Ok(p) => Check::ok("networking", format!("gvproxy at {}", p.display())), + Err(e) => Check::warn( + "networking", + format!("{e:#}"), + "brew install gvproxy — without it machines boot with no network, \ + so exec/cp/cache cannot reach them", + ), + } +} + +fn writable(name: &'static str, dir: Result) -> Check { + let dir = match dir { + Ok(d) => d, + Err(e) => return Check::fail(name, format!("{e:#}"), "set HOME"), + }; + if let Err(e) = std::fs::create_dir_all(&dir) { + return Check::fail( + name, + format!("{} is not usable: {e}", dir.display()), + "check the permissions on it", + ); + } + // Creating the directory can succeed on a read-only mount; writing proves it. + let probe = dir.join(".bsdkrun-doctor"); + match std::fs::write(&probe, b"") { + Ok(()) => { + let _ = std::fs::remove_file(&probe); + Check::ok(name, dir.display().to_string()) + } + Err(e) => Check::fail( + name, + format!("{} is not writable: {e}", dir.display()), + "check the permissions on it", + ), + } +} + +#[cfg(target_os = "macos")] +fn store() -> Check { + match crate::store::describe() { + Ok(s) if s.contains("case-INSENSITIVE") => Check::warn( + "store", + s, + "nix guests and Linux kernel sources need case-sensitive storage; \ + bsdkrun creates it automatically on the next Linux run", + ), + Ok(s) => Check::ok("store", s), + Err(e) => Check::warn("store", format!("{e:#}"), ""), + } +} + +/// Where `bsdkrun cache` would put things, and — for S3 — whether it has what +/// it needs to get there. Deliberately does not make a network call: doctor +/// should be fast and safe to run anywhere. +fn cache_backend() -> Check { + match crate::cache::Store::open() { + Ok(store @ crate::cache::Store::Disk(_)) => Check::ok("cache", store.describe()), + Ok(store) => match crate::cache::config::credentials() { + Ok(_) => Check::ok( + "cache", + format!("{} (credentials present)", store.describe()), + ), + Err(e) => Check::warn("cache", format!("{}: {e}", store.describe()), ""), + }, + Err(e) => Check::warn("cache", format!("{e:#}"), ""), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Doctor exists to be read under stress; a check that fails has to carry a + /// fix, or it is just another error message. + #[test] + fn every_failing_or_warning_check_offers_something_to_do() { + for c in gather() { + if c.state == State::Ok { + continue; + } + assert!( + c.fix.is_some(), + "check {:?} is {:?} but suggests nothing", + c.name, + c.state + ); + } + } + + /// The tools bsdkrun cannot work without have to be reported as failures, + /// not warnings — this machine has them, so the shape is checked directly. + #[test] + fn a_missing_required_tool_is_fatal_and_an_optional_one_is_not() { + let missing = tool("definitely-not-a-real-tool", true, "why"); + assert_eq!(missing.state, State::Fail); + assert!(missing.fix.unwrap().contains("brew install")); + + let optional = tool("definitely-not-a-real-tool", false, "why"); + assert_eq!(optional.state, State::Warn); + } + + #[test] + fn a_writable_directory_passes_and_a_bad_one_fails() { + let dir = tempfile::tempdir().unwrap(); + assert_eq!(writable("t", Ok(dir.path().to_path_buf())).state, State::Ok); + assert_eq!( + writable("t", Ok(PathBuf::from("/dev/null/nope"))).state, + State::Fail + ); + } + + /// The probe file must not survive the check that wrote it. + #[test] + fn the_writability_probe_cleans_up_after_itself() { + let dir = tempfile::tempdir().unwrap(); + writable("t", Ok(dir.path().to_path_buf())); + assert_eq!(std::fs::read_dir(dir.path()).unwrap().count(), 0); + } +} diff --git a/core/src/commands/mod.rs b/core/src/commands/mod.rs index 409440e..ae6cdb9 100644 --- a/core/src/commands/mod.rs +++ b/core/src/commands/mod.rs @@ -12,7 +12,9 @@ #[cfg(feature = "boot")] pub mod boot; +pub mod cache; pub mod cp; +pub mod doctor; pub mod domains; pub mod flavor; pub mod guest; diff --git a/core/src/fetch.rs b/core/src/fetch.rs index f0dc167..e35197f 100644 --- a/core/src/fetch.rs +++ b/core/src/fetch.rs @@ -813,9 +813,42 @@ fn write_freebsd_loader_env(raw: &Path, env: &str) -> Result<()> { /// Run a command, streaming its stdout/stderr, and error if it fails. pub(crate) fn run(cmd: &mut Command, what: &str) -> Result<()> { - let status = cmd.status().with_context(|| format!("spawning {what}"))?; + let status = cmd.status().map_err(|e| spawn_error(cmd, what, e))?; if !status.success() { bail!("{what} exited with {status}"); } Ok(()) } + +/// Explain a failed spawn, naming the tool when it simply isn't installed. +/// +/// bsdkrun shells out to a handful of host tools rather than linking their +/// libraries — `curl` for every download, `tar` for every archive. When one is +/// absent the raw error is `No such file or directory (os error 2)` attached to +/// whatever we were *doing*, which reads like the URL or the path was wrong and +/// sends you looking in the wrong place entirely. +pub(crate) fn spawn_error(cmd: &Command, what: &str, e: std::io::Error) -> anyhow::Error { + let program = cmd.get_program().to_string_lossy().into_owned(); + if e.kind() == std::io::ErrorKind::NotFound { + anyhow::anyhow!( + "{what} needs `{program}`, which is not on PATH.{}", + install_hint(&program) + ) + } else { + anyhow::Error::new(e).context(format!("spawning {program} for {what}")) + } +} + +/// How to get one of the host tools bsdkrun shells out to. +fn install_hint(program: &str) -> String { + let pkg = match program { + "curl" => "curl", + "tar" => "tar (bsdtar or GNU tar)", + "gzip" => "gzip", + other => return format!(" Install {other} and try again."), + }; + format!( + " Install {pkg}: `brew install {program}` on macOS, \ + or your distribution's package of the same name." + ) +} diff --git a/core/src/lib.rs b/core/src/lib.rs index 6628caa..ff9e926 100644 --- a/core/src/lib.rs +++ b/core/src/lib.rs @@ -21,6 +21,7 @@ pub mod agent; pub mod api; +pub mod cache; pub mod cli; pub mod commands; pub mod console; @@ -137,6 +138,22 @@ pub fn dispatch(cmd: Command) -> Result<()> { commands::guest::cmd_exec(&args.id, &args.command, &args.env, args.tty) } Command::Cp(args) => commands::cp::cmd_cp(&args.src, &args.dst, args.recursive), + Command::Cache(args) => match args.cmd { + CacheCmd::Save(a) => { + let (id, path) = split_target(&a.target)?; + let path = path.ok_or_else(|| { + anyhow::anyhow!("`cache save` needs the directory to archive: ID:PATH") + })?; + commands::cache::cmd_save(id, path, &a.key, a.compression.parse()?, a.force, a.json) + } + CacheCmd::Restore(a) => { + let (id, path) = split_target(&a.target)?; + commands::cache::cmd_restore(id, path, &a.key, &a.restore_keys, a.json) + } + CacheCmd::Ls(a) => commands::cache::cmd_ls(a.json), + CacheCmd::Rm(a) => commands::cache::cmd_rm(&a.keys, a.all), + }, + Command::Doctor(args) => commands::doctor::cmd_doctor(args.json), Command::Tailscale(args) => commands::guest::cmd_tailscale(&args.id, &args.args), Command::Ssh(args) => commands::guest::cmd_ssh(&args.id, &args.args), Command::Systemd(args) => { @@ -221,3 +238,14 @@ fn serve_ui(args: cli::UiArgs) -> Result<()> { fn serve_ui(_args: cli::UiArgs) -> Result<()> { anyhow::bail!("this build has no web UI compiled in") } + +/// Split a `cache` target into `(id, path)`. The path is optional so +/// `cache restore web` can mean "back where it came from". +#[cfg(feature = "boot")] +fn split_target(target: &str) -> Result<(&str, Option<&str>)> { + match target.split_once(':') { + Some((id, path)) if !id.is_empty() && !path.is_empty() => Ok((id, Some(path))), + Some(_) => anyhow::bail!("expected ID:PATH, got {target:?}"), + None => Ok((target, None)), + } +} diff --git a/sdk/clojure/README.md b/sdk/clojure/README.md index e01468c..55041cf 100644 --- a/sdk/clojure/README.md +++ b/sdk/clojure/README.md @@ -17,14 +17,14 @@ and every namespace is a set of functions over plain maps. ```clojure (require '[bsdkrun.sandbox :as sandbox]) -(def box (sandbox/create! {:os :linux :image "alpine"})) +(def sbx (sandbox/create! {:os :linux :image "alpine"})) ;; exec argv directly, with env / stdin / a PTY / a working dir: -(println (:stdout (sandbox/exec! box ["uname" "-a"]))) -(sandbox/exec! box ["apk" "add" "curl"] {:throw-on-error true}) -(sandbox/run-command! box "curl" ["-fsSL" "https://example.com"]) +(println (:stdout (sandbox/exec! sbx ["uname" "-a"]))) +(sandbox/exec! sbx ["apk" "add" "curl"] {:throw-on-error true}) +(sandbox/run-command! sbx "curl" ["-fsSL" "https://example.com"]) -(sandbox/stop! box) +(sandbox/stop! sbx) ``` ## Install @@ -99,9 +99,9 @@ pass an argv vector (or a bare program name plus `:args`). (require '[bsdkrun.types :as types] '[clojure.java.io :as io]) -(sandbox/exec! box ["ls" "-la" "/etc"]) +(sandbox/exec! sbx ["ls" "-la" "/etc"]) -(sandbox/exec! box "ruby" +(sandbox/exec! sbx "ruby" {:args ["-e" "puts ENV['X']"] :env {"X" "hi"} :cwd "/app" @@ -112,7 +112,7 @@ pass an argv vector (or a bare program name plus `:args`). :throw-on-error true}) ; throw on non-zero exit (default: false) ;; Vercel-Sandbox-style alias: -(def result (sandbox/run-command! box "uname" ["-a"])) +(def result (sandbox/run-command! sbx "uname" ["-a"])) (:stdout result) ; raw stdout (types/text result) ; stdout, trailing newlines trimmed (:exit-code result) @@ -128,6 +128,33 @@ The callbacks receive byte arrays as chunks arrive, while `:stdout` and `:stderr` remain fully buffered in the result. Streaming is independent of `:tty`; a PTY changes command semantics and may merge stderr into stdout. +## Caching + +`bsdkrun.cache` saves a guest directory under a key and restores it later, so a +rebuild can pick up where the last one left off. **A miss is not an error** — +check `:restored` rather than catching. + +```clojure +(require '[bsdkrun.cache :as cache]) + +(let [k (str "deps-" lock-hash) + hit (cache/restore "web" {:key k :restore-keys ["deps-"]})] + (when-not (:restored hit) + (sandbox/exec "web" ["npm" "ci"]) + (cache/save "web" "/app/node_modules" {:key k :compression "zstd"}))) + +(cache/ls) ; every stored entry, newest first +(cache/rm [k]) ; or (cache/rm [] {:all true}) +``` + +`:restore-keys` are prefixes tried in order when the exact key misses; within a +prefix the newest matching entry wins, and `:key` on the result says which one +was used. Formats are `"gzip"` (default), `"zstd"`, `"estargz"` and `"none"`. + +Where entries live is host configuration, not an SDK concern: the default is +this host's disk, and `BSDKRUN_CACHE_BACKEND=s3` + `BSDKRUN_CACHE_S3_*` (or +`~/.config/bsdkrun/cache.toml`) points them at a bucket instead. + ## Files `bsdkrun.filesystem` reads and writes files in the guest. Parent directories are @@ -163,22 +190,22 @@ Failures throw an `ex-info` whose `ex-data` is ```clojure (require '[bsdkrun.sandbox :as sandbox]) -(def box (sandbox/create! {:os :linux :image "alpine" :name "web-1" :command ["sleep" "300"]})) -(def same (sandbox/get (:id box))) ; reconnect (id prefix ok) +(def sbx (sandbox/create! {:os :linux :image "alpine" :name "web-1" :command ["sleep" "300"]})) +(def same (sandbox/get (:id sbx))) ; reconnect (id prefix ok) (def same (sandbox/get "web-1")) ; ...or by its exact --name (def all (sandbox/list {:all true})) ; vector of sandbox-info maps -(sandbox/status box) ; sandbox-info map, or nil -(sandbox/running? box) ; true / false -(sandbox/logs box) ; console log (string) -(sandbox/shell! box) ; interactive shell (inherits the terminal) -(sandbox/stop! box) ; BSD guests clean-poweroff; Linux SIGTERM -(sandbox/start! box) ; restart in place — resumes its own disk/rootfs -(sandbox/update! box {:cpus 4 :mem 2048}) ; applies on next start -(sandbox/remove! box {:force true}) +(sandbox/status sbx) ; sandbox-info map, or nil +(sandbox/running? sbx) ; true / false +(sandbox/logs sbx) ; console log (string) +(sandbox/shell! sbx) ; interactive shell (inherits the terminal) +(sandbox/stop! sbx) ; BSD guests clean-poweroff; Linux SIGTERM +(sandbox/start! sbx) ; restart in place — resumes its own disk/rootfs +(sandbox/update! sbx {:cpus 4 :mem 2048}) ; applies on next start +(sandbox/remove! sbx {:force true}) ``` -Every function above takes a `ref` — a sandbox map (`box`) **or** a bare +Every function above takes a `ref` — a sandbox map (`sbx`) **or** a bare id/name string — so you never have to reconnect first just to act on a machine you already know the name of: @@ -198,7 +225,7 @@ you want the sandbox back at the end instead of the last call's result: (sandbox/exec! ["uname" "-a"]) :stdout) -;; doto : same box driven through every step; you get the box back +;; doto : same sbx driven through every step; you get the sbx back (doto (sandbox/get "web-1") sandbox/start! (sandbox/exec! ["setup.sh"] {:throw-on-error true}) @@ -229,11 +256,11 @@ Host-level namespaces: (sandbox/create! {:os :linux :image "alpine" :net {:ports ["2222:22"]}}) ;; agent-managed key-based SSH -(sandbox/ssh-setup! box) ; install local ~/.ssh/*.pub keys -(sandbox/ssh-setup! box {:user "tsiry" :key "~/.ssh/work.pub"}) +(sandbox/ssh-setup! sbx) ; install local ~/.ssh/*.pub keys +(sandbox/ssh-setup! sbx {:user "tsiry" :key "~/.ssh/work.pub"}) ;; put a guest on your tailnet -(sandbox/tailscale-up! box {:authkey "tskey-auth-..." :hostname "web"}) +(sandbox/tailscale-up! sbx {:authkey "tskey-auth-..." :hostname "web"}) ``` ### Global networks — reach machines by name @@ -394,7 +421,7 @@ Pattern-match on `(:bsdkrun/error (ex-data e))`: (require '[bsdkrun.sandbox :as sandbox]) (try - (sandbox/exec! box ["false"] {:throw-on-error true}) + (sandbox/exec! sbx ["false"] {:throw-on-error true}) (catch clojure.lang.ExceptionInfo e (case (:bsdkrun/error (ex-data e)) :command-failed (println "exit" (:exit-code (ex-data e)) (:stderr (ex-data e))) diff --git a/sdk/clojure/src/bsdkrun/cache.clj b/sdk/clojure/src/bsdkrun/cache.clj new file mode 100644 index 0000000..c57172b --- /dev/null +++ b/sdk/clojure/src/bsdkrun/cache.clj @@ -0,0 +1,66 @@ +(ns bsdkrun.cache + "Cached guest directories. + + Entries are keyed, so a rebuild can pick up where the last one left off: + + (require '[bsdkrun.cache :as cache]) + + (let [hit (cache/restore \"web\" {:key k :restore-keys [\"deps-\"]})] + (when-not (:restored hit) + (sandbox/exec \"web\" [\"npm\" \"ci\"]) + (cache/save \"web\" \"/app/node_modules\" {:key k}))) + + Where entries live — host disk or S3 — is host configuration, not an SDK + concern: set `BSDKRUN_CACHE_BACKEND` / `BSDKRUN_CACHE_S3_*`, or write + `~/.config/bsdkrun/cache.toml`." + (:require [clojure.data.json :as json] + [clojure.string :as str] + [bsdkrun.process :as process])) + +(defn- decode + "Parse the CLI's JSON, keywordising keys. An empty body means `empty`." + [text empty] + (let [body (str/trim (or text ""))] + (json/read-str (if (str/blank? body) empty body) :key-fn keyword))) + +(defn- run-json [args label empty] + (-> (process/run! args {:label label}) :stdout (decode empty))) + +(defn save + "Archive the guest directory at `path` under `:key`. + + Options: `:key` (required), `:compression` (\"gzip\" by default, or \"zstd\", + \"estargz\", \"none\"), `:force` to replace an existing entry. Returns the + stored entry." + [id path {:keys [key compression force] :or {compression "gzip"}}] + (cond-> ["cache" "save" (str id ":" path) "--key" key "--json"] + (not= compression "gzip") (into ["--compression" compression]) + force (conj "--force") + :always (run-json "bsdkrun cache save" "{}"))) + +(defn restore + "Restore a stored tree. + + Options: `:key` (required), `:path` (defaults to where it was saved from), + `:restore-keys` — prefixes tried in order when the key misses. A miss is not + an error: check `:restored` on the result." + [id {:keys [key path restore-keys]}] + (let [target (if path (str id ":" path) id)] + (cond-> ["cache" "restore" target "--key" key "--json"] + (seq restore-keys) (into (cons "--restore-keys" restore-keys)) + :always (run-json "bsdkrun cache restore" "{}")))) + +(defn ls + "Every stored cache entry, newest first." + [] + (run-json ["cache" "ls" "--json"] "bsdkrun cache ls" "[]")) + +(defn rm + "Remove entries by key, or every one of them with `{:all true}`. Returns nil." + ([keys] (rm keys {})) + ([keys {:keys [all]}] + (process/run! (if all + ["cache" "rm" "--all"] + (into ["cache" "rm"] (if (string? keys) [keys] keys))) + {:label "bsdkrun cache rm"}) + nil)) diff --git a/sdk/elixir/README.md b/sdk/elixir/README.md index 73b7359..d4b4d87 100644 --- a/sdk/elixir/README.md +++ b/sdk/elixir/README.md @@ -7,14 +7,14 @@ The SDK shells out to the `bsdkrun` binary via `System.cmd/3`, so its only runtime dependency is [`jason`](https://hex.pm/packages/jason) for JSON parsing. ```elixir -{:ok, box} = Bsdkrun.create(os: :linux, image: "alpine") +{:ok, sbx} = Bsdkrun.create(os: :linux, image: "alpine") # argv exec — no shell parsing; env / stdin / a PTY / a working dir: -{:ok, res} = Bsdkrun.exec(box, ["uname", "-a"]) +{:ok, res} = Bsdkrun.exec(sbx, ["uname", "-a"]) IO.puts(Bsdkrun.Types.Result.text(res)) -{:ok, _} = Bsdkrun.exec(box, ["apk", "add", "curl"]) -:ok = Bsdkrun.stop(box) +{:ok, _} = Bsdkrun.exec(sbx, ["apk", "add", "curl"]) +:ok = Bsdkrun.stop(sbx) ``` Or, with the bang variants, as one `|>` chain: @@ -92,10 +92,10 @@ Bsdkrun.create(os: :kernel, kernel: "netbsd", format: "elf", disk: "root.raw") parsing) or a bare program name with `:args`, plus options: ```elixir -Bsdkrun.exec(box, ["ls", "-la", "/etc"]) +Bsdkrun.exec(sbx, ["ls", "-la", "/etc"]) {:ok, res} = - Bsdkrun.exec(box, "node", + Bsdkrun.exec(sbx, "node", args: ["-e", "IO.puts System.get_env(\"X\")"], env: %{"X" => "hi"}, cwd: "/app", @@ -116,6 +116,35 @@ The callbacks receive binary chunks in real time while the complete streams remain buffered in the returned result. They are independent of `:tty`; a PTY changes command semantics and may merge stderr into stdout. +## Caching + +`Bsdkrun.Cache` saves a guest directory under a key and restores it later, so a +rebuild can pick up where the last one left off. **A miss is not an error** — it +comes back as `{:ok, %{restored: false}}`. + +```elixir +alias Bsdkrun.Cache + +key = "deps-" <> lock_hash +{:ok, hit} = Cache.restore("web", key: key, restore_keys: ["deps-"]) + +unless hit.restored do + Bsdkrun.Sandbox.exec("web", ["npm", "ci"]) + Cache.save("web", "/app/node_modules", key: key, compression: "zstd") +end + +Cache.list() # every stored entry, newest first +Cache.remove([key]) # or Cache.remove([], all: true) +``` + +`:restore_keys` are prefixes tried in order when the exact key misses; within a +prefix the newest matching entry wins, and `hit.key` says which one was used. +Formats are `"gzip"` (default), `"zstd"`, `"estargz"` and `"none"`. + +Where entries live is host configuration, not an SDK concern: the default is +this host's disk, and `BSDKRUN_CACHE_BACKEND=s3` + `BSDKRUN_CACHE_S3_*` (or +`~/.config/bsdkrun/cache.toml`) points them at a bucket instead. + ## Files `Bsdkrun.FileSystem` reads and writes files in the guest. Parent directories are @@ -146,17 +175,17 @@ carries the offending path. ## Lifecycle & inventory ```elixir -{:ok, box} = Bsdkrun.create(os: :linux, image: "alpine", command: ["sleep", "300"]) -{:ok, same} = Bsdkrun.get(box.id) # reconnect (prefix ok) +{:ok, sbx} = Bsdkrun.create(os: :linux, image: "alpine", command: ["sleep", "300"]) +{:ok, same} = Bsdkrun.get(sbx.id) # reconnect (prefix ok) {:ok, list} = Bsdkrun.list(all: true) # [%Bsdkrun.Types.SandboxInfo{}] -Bsdkrun.Sandbox.status(box) # {:ok, %SandboxInfo{} | nil} -Bsdkrun.Sandbox.running?(box) # boolean -Bsdkrun.logs(box) # {:ok, console_log} -Bsdkrun.stop(box) # BSD guests clean-poweroff; Linux SIGTERM -Bsdkrun.start(box) # restart in place — resumes disk/rootfs -Bsdkrun.Sandbox.update(box, cpus: 4, mem: 2048) # applies on next start -Bsdkrun.remove(box, force: true) +Bsdkrun.Sandbox.status(sbx) # {:ok, %SandboxInfo{} | nil} +Bsdkrun.Sandbox.running?(sbx) # boolean +Bsdkrun.logs(sbx) # {:ok, console_log} +Bsdkrun.stop(sbx) # BSD guests clean-poweroff; Linux SIGTERM +Bsdkrun.start(sbx) # restart in place — resumes disk/rootfs +Bsdkrun.Sandbox.update(sbx, cpus: 4, mem: 2048) # applies on next start +Bsdkrun.remove(sbx, force: true) ``` `SandboxInfo.kind` is an atom — `:linux`, `:freebsd`, `:netbsd`, `:firmware`, diff --git a/sdk/elixir/lib/bsdkrun/cache.ex b/sdk/elixir/lib/bsdkrun/cache.ex new file mode 100644 index 0000000..0b86f9f --- /dev/null +++ b/sdk/elixir/lib/bsdkrun/cache.ex @@ -0,0 +1,144 @@ +defmodule Bsdkrun.Cache do + @moduledoc """ + Cached guest directories. + + Entries are keyed, so a rebuild can pick up where the last one left off: + + case Bsdkrun.Cache.restore("web", key: key, restore_keys: ["deps-"]) do + {:ok, %{restored: false}} -> + Bsdkrun.Sandbox.exec("web", ["npm", "ci"]) + Bsdkrun.Cache.save("web", "/app/node_modules", key: key) + + {:ok, _hit} -> + :ok + end + + Where entries live — host disk or S3 — is host configuration, not an SDK + concern: set `BSDKRUN_CACHE_BACKEND` / `BSDKRUN_CACHE_S3_*`, or write + `~/.config/bsdkrun/cache.toml`. + """ + + alias Bsdkrun.{Cli, Error} + + @typedoc "A stored cache entry, as `cache ls` reports it." + @type entry :: %{ + key: String.t(), + path: String.t(), + compression: String.t(), + size: non_neg_integer(), + created: non_neg_integer(), + digest: String.t() + } + + @typedoc "What a restore did. A miss is not an error — check `:restored`." + @type result :: %{ + restored: boolean(), + requested_key: String.t(), + key: String.t() | nil, + path: String.t() | nil, + size: non_neg_integer() | nil, + compression: String.t() | nil, + created: non_neg_integer() | nil + } + + @doc """ + Archive the guest directory at `path` under `:key`. + + Options: `:key` (required), `:compression` (`"gzip"` by default, or `"zstd"`, + `"estargz"`, `"none"`), `:force` to replace an existing entry. + """ + @spec save(String.t(), String.t(), keyword()) :: {:ok, entry()} | {:error, Error.t()} + def save(id, path, opts) do + key = Keyword.fetch!(opts, :key) + compression = Keyword.get(opts, :compression, "gzip") + + args = ["cache", "save", "#{id}:#{path}", "--key", key, "--json"] + args = if compression == "gzip", do: args, else: args ++ ["--compression", compression] + args = if Keyword.get(opts, :force, false), do: args ++ ["--force"], else: args + + with {:ok, map} <- json(args, "bsdkrun cache save") do + {:ok, to_entry(map)} + end + end + + @doc """ + Restore a stored tree. + + Options: `:key` (required), `:path` (defaults to where it was saved from), + `:restore_keys` — prefixes tried in order when the key misses. + """ + @spec restore(String.t(), keyword()) :: {:ok, result()} | {:error, Error.t()} + def restore(id, opts) do + key = Keyword.fetch!(opts, :key) + target = if path = Keyword.get(opts, :path), do: "#{id}:#{path}", else: id + restore_keys = Keyword.get(opts, :restore_keys, []) + + args = ["cache", "restore", target, "--key", key, "--json"] + args = if restore_keys == [], do: args, else: args ++ ["--restore-keys" | restore_keys] + + with {:ok, map} <- json(args, "bsdkrun cache restore") do + {:ok, + %{ + restored: Map.get(map, "restored", false), + requested_key: Map.get(map, "requested_key", key), + key: Map.get(map, "key"), + path: Map.get(map, "path"), + size: Map.get(map, "size"), + compression: Map.get(map, "compression"), + created: Map.get(map, "created") + }} + end + end + + @doc "Every stored cache entry, newest first." + @spec list() :: {:ok, [entry()]} | {:error, Error.t()} + def list do + res = Cli.run(["cache", "ls", "--json"]) + + if res.exit_code == 0 do + {:ok, res.stdout |> decode("[]") |> Enum.map(&to_entry/1)} + else + {:error, Error.command_failed(res.exit_code, res.stdout, res.stderr, "bsdkrun cache ls")} + end + end + + @doc "Remove entries by key, or every one of them with `all: true`." + @spec remove([String.t()], keyword()) :: :ok | {:error, Error.t()} + def remove(keys \\ [], opts \\ []) do + args = ["cache", "rm"] + args = if Keyword.get(opts, :all, false), do: args ++ ["--all"], else: args ++ keys + + case Cli.checked(args, "bsdkrun cache rm") do + {:ok, _} -> :ok + error -> error + end + end + + defp json(args, label) do + res = Cli.run(args) + + if res.exit_code == 0 do + {:ok, decode(res.stdout, "{}")} + else + {:error, Error.command_failed(res.exit_code, res.stdout, res.stderr, label)} + end + end + + defp decode(text, empty) do + case String.trim(text) do + "" -> Jason.decode!(empty) + body -> Jason.decode!(body) + end + end + + defp to_entry(map) do + %{ + key: Map.get(map, "key", ""), + path: Map.get(map, "path", ""), + compression: Map.get(map, "compression", ""), + size: Map.get(map, "size", 0), + created: Map.get(map, "created", 0), + digest: Map.get(map, "digest", "") + } + end +end diff --git a/sdk/gleam/README.md b/sdk/gleam/README.md index 617d7e0..6b846b5 100644 --- a/sdk/gleam/README.md +++ b/sdk/gleam/README.md @@ -47,10 +47,10 @@ import bsdkrun/args import bsdkrun/types pub fn main() { - let assert Ok(box) = bsdkrun.create(args.linux("alpine")) - let assert Ok(res) = bsdkrun.exec(box, ["uname", "-a"]) + let assert Ok(sbx) = bsdkrun.create(args.linux("alpine")) + let assert Ok(res) = bsdkrun.exec(sbx, ["uname", "-a"]) echo types.text(res) - let assert Ok(box) = bsdkrun.stop(box) + let assert Ok(sbx) = bsdkrun.stop(sbx) } ``` @@ -116,7 +116,7 @@ import bsdkrun/types let assert Ok(res) = sandbox.exec( - box, + sbx, ["sh", "-c", "cat > out.txt && wc -l < out.txt"], sandbox.exec_options() |> sandbox.with_env([#("RUST_LOG", "debug")]) @@ -142,6 +142,34 @@ The callbacks receive chunks as they arrive, while the completed of `with_tty`; a PTY changes command semantics and may merge stderr into stdout. +## Caching + +`bsdkrun/cache` saves a guest directory under a key and restores it later, so a +rebuild can pick up where the last one left off. **A miss is not an error** — it +comes back as `Ok` with `restored: False`. + +```gleam +import bsdkrun/cache +import gleam/option.{None, Some} + +let assert Ok(hit) = cache.restore("web", key, None, ["deps-"]) +case hit.restored { + False -> cache.save("web", "/app/node_modules", key, cache.Zstd, False) + True -> Ok(cache.CacheEntry("", "", "", 0, 0, "")) +} + +cache.list() // every stored entry, newest first +cache.remove([key], False) // or ([], True) for all +``` + +The restore keys are prefixes tried in order when the exact key misses; within a +prefix the newest matching entry wins, and `hit.key` says which one was used. +Formats are `Gzip` (default), `Zstd`, `Estargz` and `Uncompressed`. + +Where entries live is host configuration, not an SDK concern: the default is +this host's disk, and `BSDKRUN_CACHE_BACKEND=s3` + `BSDKRUN_CACHE_S3_*` (or +`~/.config/bsdkrun/cache.toml`) points them at a bucket instead. + ## Files `bsdkrun/filesystem` reads and writes files in the guest. Parent directories are @@ -173,23 +201,23 @@ Failures are `error.FileTransferFailed(path, message)`. ## Lifecycle ```gleam -bsdkrun.stop(box) // Ok(box) back — not Ok(Nil) -bsdkrun.start(box) // restart in place: same id, disk, resources -bsdkrun.remove(box, True) // force: stop first if running -bsdkrun.status(box) // Ok(Some(SandboxInfo)) or Ok(None) if gone -bsdkrun.is_running(box) -bsdkrun.logs(box) // console log -sandbox.boot_logs(box) // bsdkrun's own boot log -sandbox.update(box, Some(4), Some(4096)) // cpus, mem — applies on next start -sandbox.connect_network(box, "devnet") // join/switch — applies on next start -sandbox.disconnect_network(box) -sandbox.shell(box) // interactive shell, inherits stdio +bsdkrun.stop(sbx) // Ok(sbx) back — not Ok(Nil) +bsdkrun.start(sbx) // restart in place: same id, disk, resources +bsdkrun.remove(sbx, True) // force: stop first if running +bsdkrun.status(sbx) // Ok(Some(SandboxInfo)) or Ok(None) if gone +bsdkrun.is_running(sbx) +bsdkrun.logs(sbx) // console log +sandbox.boot_logs(sbx) // bsdkrun's own boot log +sandbox.update(sbx, Some(4), Some(4096)) // cpus, mem — applies on next start +sandbox.connect_network(sbx, "devnet") // join/switch — applies on next start +sandbox.disconnect_network(sbx) +sandbox.shell(sbx) // interactive shell, inherits stdio ``` `stop`, `start`, `remove`, `update`, `connect_network` and `disconnect_network` -all return `Result(Sandbox, Error)` — the same `box`, not `Nil` — so a +all return `Result(Sandbox, Error)` — the same `sbx`, not `Nil` — so a sequence of them chains with `|>` through `gleam/result.try` instead of -re-threading `box` by hand: +re-threading `sbx` by hand: ```gleam import gleam/result @@ -237,11 +265,11 @@ resolve via the network's DNS, NetBSD via a synced `/etc/hosts` block. ```gleam // install key-based SSH via the guest agent -sandbox.ssh_setup(box, None, []) // your local ~/.ssh/*.pub -sandbox.ssh_setup(box, Some("tsiry"), ["~/.ssh/work.pub"]) +sandbox.ssh_setup(sbx, None, []) // your local ~/.ssh/*.pub +sandbox.ssh_setup(sbx, Some("tsiry"), ["~/.ssh/work.pub"]) // put the guest on your tailnet -sandbox.tailscale_up(box, Some("tskey-auth-…"), Some("web"), []) +sandbox.tailscale_up(sbx, Some("tskey-auth-…"), Some("web"), []) ``` The Tailscale auth key travels in the environment as `TS_AUTHKEY`, so it never diff --git a/sdk/gleam/src/bsdkrun/cache.gleam b/sdk/gleam/src/bsdkrun/cache.gleam new file mode 100644 index 0000000..229e4b7 --- /dev/null +++ b/sdk/gleam/src/bsdkrun/cache.gleam @@ -0,0 +1,124 @@ +//// Cached guest directories. +//// +//// Entries are keyed, so a rebuild can pick up where the last one left off: +//// +//// ```gleam +//// import bsdkrun/cache +//// +//// let assert Ok(hit) = cache.restore("web", "deps-abc", None, ["deps-"]) +//// case hit.restored { +//// False -> { +//// let assert Ok(_) = sandbox.exec("web", ["npm", "ci"]) +//// cache.save("web", "/app/node_modules", "deps-abc", cache.Gzip, False) +//// } +//// True -> Ok(Nil) +//// } +//// ``` +//// +//// Where entries live — host disk or S3 — is host configuration, not an SDK +//// concern: set `BSDKRUN_CACHE_BACKEND` / `BSDKRUN_CACHE_S3_*`, or write +//// `~/.config/bsdkrun/cache.toml`. + +import bsdkrun/cli +import bsdkrun/error.{type Error} +import bsdkrun/types.{type CacheEntry, type RestoreResult} +import gleam/list +import gleam/option.{type Option, None, Some} +import gleam/result + +/// An archive format a cache entry can be stored in. +pub type Compression { + Gzip + Zstd + Estargz + Uncompressed +} + +fn compression_name(c: Compression) -> String { + case c { + Gzip -> "gzip" + Zstd -> "zstd" + Estargz -> "estargz" + Uncompressed -> "none" + } +} + +/// Archive the guest directory at `path` under `key`. +pub fn save( + id: String, + path: String, + key: String, + compression: Compression, + force: Bool, +) -> Result(CacheEntry, Error) { + let flags = case compression { + Gzip -> [] + other -> ["--compression", compression_name(other)] + } + let flags = case force { + True -> list.append(flags, ["--force"]) + False -> flags + } + let args = + list.append( + ["cache", "save", id <> ":" <> path, "--key", key, "--json"], + flags, + ) + + use out <- result.try(cli.checked(args, "bsdkrun cache save", cli.options())) + types.decode_one( + out.stdout, + "bsdkrun cache save", + types.cache_entry_decoder(), + ) +} + +/// Restore a stored tree. `path` defaults to where the entry was saved from; +/// `restore_keys` are prefixes tried in order when `key` misses. +pub fn restore( + id: String, + key: String, + path: Option(String), + restore_keys: List(String), +) -> Result(RestoreResult, Error) { + let target = case path { + Some(p) -> id <> ":" <> p + None -> id + } + let args = ["cache", "restore", target, "--key", key, "--json"] + let args = case restore_keys { + [] -> args + keys -> list.append(list.append(args, ["--restore-keys"]), keys) + } + + use out <- result.try(cli.checked( + args, + "bsdkrun cache restore", + cli.options(), + )) + types.decode_one( + out.stdout, + "bsdkrun cache restore", + types.restore_result_decoder(), + ) +} + +/// Every stored cache entry, newest first. +pub fn list() -> Result(List(CacheEntry), Error) { + use out <- result.try(cli.checked( + ["cache", "ls", "--json"], + "bsdkrun cache ls", + cli.options(), + )) + + types.decode_rows(out.stdout, "bsdkrun cache ls", types.cache_entry_decoder()) +} + +/// Remove entries by key, or every one of them with `all`. +pub fn remove(keys: List(String), all: Bool) -> Result(Nil, Error) { + let args = case all { + True -> ["cache", "rm", "--all"] + False -> list.append(["cache", "rm"], keys) + } + cli.checked_unit(args, "bsdkrun cache rm", cli.options()) +} diff --git a/sdk/gleam/src/bsdkrun/types.gleam b/sdk/gleam/src/bsdkrun/types.gleam index bacd1c0..4059824 100644 --- a/sdk/gleam/src/bsdkrun/types.gleam +++ b/sdk/gleam/src/bsdkrun/types.gleam @@ -82,6 +82,32 @@ pub type VolumeInfo { ) } +/// A stored cache entry, as reported by `bsdkrun cache ls --json`. +pub type CacheEntry { + CacheEntry( + key: String, + path: String, + compression: String, + size: Int, + created: Int, + digest: String, + ) +} + +/// What a `bsdkrun cache restore --json` did. A miss is not an error — check +/// `restored`. +pub type RestoreResult { + RestoreResult( + restored: Bool, + requested_key: String, + key: Option(String), + path: Option(String), + size: Option(Int), + compression: Option(String), + created: Option(Int), + ) +} + /// A global network, as reported by `bsdkrun network ls --json`. pub type NetworkInfo { NetworkInfo( @@ -211,6 +237,39 @@ pub fn port_forward_decoder() -> Decoder(PortForward) { decode.success(PortForward(bind:, host:, guest:)) } +/// Decoder for one `cache ls --json` row, and for `cache save --json`. +pub fn cache_entry_decoder() -> Decoder(CacheEntry) { + use key <- field_or("key", "", decode.string) + use path <- field_or("path", "", decode.string) + use compression <- field_or("compression", "", decode.string) + use size <- field_or("size", 0, lenient_int()) + use created <- field_or("created", 0, lenient_int()) + use digest <- field_or("digest", "", decode.string) + + decode.success(CacheEntry(key:, path:, compression:, size:, created:, digest:)) +} + +/// Decoder for `cache restore --json`. +pub fn restore_result_decoder() -> Decoder(RestoreResult) { + use restored <- field_or("restored", False, decode.bool) + use requested_key <- field_or("requested_key", "", decode.string) + use key <- optional_field("key", decode.string) + use path <- optional_field("path", decode.string) + use size <- optional_field("size", lenient_int()) + use compression <- optional_field("compression", decode.string) + use created <- optional_field("created", lenient_int()) + + decode.success(RestoreResult( + restored:, + requested_key:, + key:, + path:, + size:, + compression:, + created:, + )) +} + /// Decoder for one `ps --json` row. pub fn sandbox_info_decoder() -> Decoder(SandboxInfo) { use id <- field_or("id", "", decode.string) @@ -473,6 +532,23 @@ pub fn string_field(dyn: Dynamic, name: String, default: String) -> String { /// Decode a `--json` list payload. Blank output — which the CLI emits when /// there is nothing to list — decodes as the empty list. +/// Decode a single JSON object, as `decode_rows` does for a list. +pub fn decode_one( + raw: String, + label: String, + row: Decoder(a), +) -> Result(a, Error) { + let payload = case string.trim(raw) { + "" -> "{}" + trimmed -> trimmed + } + + case json.parse(payload, row) { + Ok(value) -> Ok(value) + Error(_) -> Error(DecodeFailed(label, raw)) + } +} + pub fn decode_rows( raw: String, label: String, diff --git a/sdk/go/README.md b/sdk/go/README.md index 0c37c21..35d9fe3 100644 --- a/sdk/go/README.md +++ b/sdk/go/README.md @@ -11,19 +11,19 @@ fluent: builders chain and end in a terminal call returning `(T, error)`. ```go import bsdkrun "github.com/tsirysndr/bsdkrun/sdk/go" -box, err := bsdkrun.Linux("alpine").Create() +sbx, err := bsdkrun.Linux("alpine").Create() if err != nil { log.Fatal(err) } // exec argv directly, or chain env / stdin / a PTY / a working dir: -res, _ := box.Exec("uname", "-a") +res, _ := sbx.Exec("uname", "-a") fmt.Println(res.Text()) -box.Exec("apk", "add", "curl") -box.Command("curl").Args("-fsSL", "https://example.com").Run() +sbx.Exec("apk", "add", "curl") +sbx.Command("curl").Args("-fsSL", "https://example.com").Run() -box.Stop() +sbx.Stop() ``` ## Install @@ -87,9 +87,9 @@ Every `Create` runs the machine **detached** and returns a `*Sandbox` handle Pass an argv directly to `Exec`, or chain options on `Command`: ```go -box.Exec("ls", "-la", "/etc") +sbx.Exec("ls", "-la", "/etc") -res, err := box.Command("node"). +res, err := sbx.Command("node"). Args("-e", "print(1)"). Env("X", "hi"). Cwd("/app"). @@ -112,13 +112,39 @@ rendering of Python's `throw_if_failed`). and also retained in the returned `Result`. Streaming is independent of `TTY`; a PTY changes command semantics and may merge stderr into stdout. +## Caching + +`Sandbox.Cache()` saves a guest directory under a key and restores it later, so +a rebuild can pick up where the last one left off. **A miss is not an error** — +check `Restored` rather than the error. + +```go +key := "deps-" + lockHash +hit, _ := sbx.Cache().Restore(bsdkrun.RestoreOptions{Key: key, RestoreKeys: []string{"deps-"}}) +if !hit.Restored { + sbx.Exec("npm", "ci") + sbx.Cache().Save("/app/node_modules", bsdkrun.SaveOptions{Key: key, Compression: bsdkrun.Zstd}) +} + +bsdkrun.ListCaches() // every stored entry, newest first +bsdkrun.RemoveCache([]string{key}, false) // or (nil, true) for all +``` + +`RestoreKeys` are prefixes tried in order when the exact key misses; within a +prefix the newest matching entry wins, and `hit.Key` says which one was used. +Formats are `Gzip` (default), `Zstd`, `Estargz` and `NoCompression`. + +Where entries live is host configuration, not an SDK concern: the default is +this host's disk, and `BSDKRUN_CACHE_BACKEND=s3` + `BSDKRUN_CACHE_S3_*` (or +`~/.config/bsdkrun/cache.toml`) points them at a bucket instead. + ## Files `Sandbox.FS()` reads and writes files in the guest. Parent directories are created for you, and everything is byte-exact. ```go -fs := box.FS() +fs := sbx.FS() fs.WriteTextFile("/app/main.py", "print('hi')") fs.WriteFile("/app/logo.png", pngBytes) @@ -143,18 +169,18 @@ Failures are a `*FileTransferError`, which carries the offending `Path`. ## Lifecycle & inventory ```go -box, _ := bsdkrun.Linux("alpine").Command("sleep", "300").Create() -same, _ := bsdkrun.GetSandbox(box.ID) // reconnect (prefix ok) +sbx, _ := bsdkrun.Linux("alpine").Command("sleep", "300").Create() +same, _ := bsdkrun.GetSandbox(sbx.ID) // reconnect (prefix ok) rows, _ := bsdkrun.ListSandboxes(true) // []SandboxInfo, incl. exited -box.Status() // *SandboxInfo (nil if gone) -box.IsRunning() // bool -box.Logs() // console log; box.BootLogs() for the boot log -box.Shell() // interactive shell (inherits the terminal) -box.Stop() // BSD guests clean-poweroff; Linux SIGTERM -box.Start() // restart in place — resumes its own disk/rootfs (data persists) -box.Update().Cpus(4).Mem(2048).Apply() // applies on next start -box.Remove(true) // force: stop first if running +sbx.Status() // *SandboxInfo (nil if gone) +sbx.IsRunning() // bool +sbx.Logs() // console log; sbx.BootLogs() for the boot log +sbx.Shell() // interactive shell (inherits the terminal) +sbx.Stop() // BSD guests clean-poweroff; Linux SIGTERM +sbx.Start() // restart in place — resumes its own disk/rootfs (data persists) +sbx.Update().Cpus(4).Mem(2048).Apply() // applies on next start +sbx.Remove(true) // force: stop first if running ``` Host-level namespaces: @@ -206,11 +232,11 @@ refreshes an existing network without restarting members. ```go // agent-managed key-based SSH -box.SSHSetup().Run() // install local ~/.ssh/*.pub keys -box.SSHSetup().User("tsiry").Key("~/.ssh/work.pub").Run() +sbx.SSHSetup().Run() // install local ~/.ssh/*.pub keys +sbx.SSHSetup().User("tsiry").Key("~/.ssh/work.pub").Run() // put a guest on your tailnet -box.TailscaleUp().AuthKey("tskey-auth-...").Hostname("web").Run() +sbx.TailscaleUp().AuthKey("tskey-auth-...").Hostname("web").Run() ``` ## Connecting to a remote daemon diff --git a/sdk/go/cache.go b/sdk/go/cache.go new file mode 100644 index 0000000..7908cca --- /dev/null +++ b/sdk/go/cache.go @@ -0,0 +1,160 @@ +package bsdkrun + +import "encoding/json" + +// Compression is an archive format a cache entry can be stored in. +type Compression string + +const ( + Gzip Compression = "gzip" + Zstd Compression = "zstd" + Estargz Compression = "estargz" + NoCompression Compression = "none" +) + +// CacheEntry is a stored cache entry, as `cache ls` reports it. +type CacheEntry struct { + Key string `json:"key"` + // Path is the guest directory the tree came from. + Path string `json:"path"` + Compression Compression `json:"compression"` + // Size of the archive in bytes. + Size int64 `json:"size"` + // Created is unix seconds. + Created int64 `json:"created"` + // Digest is `sha256:…` over the archive. + Digest string `json:"digest"` +} + +// RestoreResult is what a restore did. A miss is not an error — check Restored. +type RestoreResult struct { + Restored bool `json:"restored"` + // RequestedKey is the key that was asked for. + RequestedKey string `json:"requested_key"` + // Key is the entry actually used. It differs from RequestedKey when a + // RestoreKeys prefix matched, and is empty on a miss. + Key string `json:"key"` + Path string `json:"path"` + Size int64 `json:"size"` + Compression Compression `json:"compression"` + Created int64 `json:"created"` +} + +// SaveOptions tunes Cache.Save. +type SaveOptions struct { + // Key to store under. Make it name the content — a lockfile hash. + Key string + // Compression defaults to gzip. + Compression Compression + // Force replaces an entry that already has this key. + Force bool +} + +// RestoreOptions tunes Cache.Restore. +type RestoreOptions struct { + Key string + // Path defaults to the directory the entry was saved from. + Path string + // RestoreKeys are prefixes tried in order when Key misses; within a prefix + // the newest matching entry wins. + RestoreKeys []string +} + +// Cache saves and restores guest directories under a key, so a rebuild can pick +// up where the last one left off. Reach it through Sandbox.Cache. +// +// hit, _ := box.Cache().Restore(bsdkrun.RestoreOptions{Key: key, RestoreKeys: []string{"deps-"}}) +// if !hit.Restored { +// box.Exec("npm", "ci") +// box.Cache().Save("/app/node_modules", bsdkrun.SaveOptions{Key: key}) +// } +// +// Where entries live — host disk or S3 — is host configuration, not an SDK +// concern: set BSDKRUN_CACHE_BACKEND / BSDKRUN_CACHE_S3_*, or write +// ~/.config/bsdkrun/cache.toml. +type Cache struct { + id string +} + +// Cache returns a handle to this machine's keyed directory cache. +func (s *Sandbox) Cache() *Cache { return &Cache{id: s.ID} } + +// Save archives the guest directory at path under opts.Key. +func (c *Cache) Save(path string, opts SaveOptions) (*CacheEntry, error) { + args := []string{"cache", "save", c.id + ":" + path, "--key", opts.Key, "--json"} + if opts.Compression != "" && opts.Compression != Gzip { + args = append(args, "--compression", string(opts.Compression)) + } + if opts.Force { + args = append(args, "--force") + } + out, err := cacheJSON(args, "bsdkrun cache save") + if err != nil { + return nil, err + } + var entry CacheEntry + if err := json.Unmarshal(out, &entry); err != nil { + return nil, err + } + return &entry, nil +} + +// Restore puts a stored tree back. A miss is reported through +// RestoreResult.Restored, not as an error. +func (c *Cache) Restore(opts RestoreOptions) (*RestoreResult, error) { + target := c.id + if opts.Path != "" { + target = c.id + ":" + opts.Path + } + args := []string{"cache", "restore", target, "--key", opts.Key, "--json"} + if len(opts.RestoreKeys) > 0 { + args = append(args, "--restore-keys") + args = append(args, opts.RestoreKeys...) + } + out, err := cacheJSON(args, "bsdkrun cache restore") + if err != nil { + return nil, err + } + var result RestoreResult + if err := json.Unmarshal(out, &result); err != nil { + return nil, err + } + return &result, nil +} + +// ListCaches returns every stored cache entry, newest first. +func ListCaches() ([]CacheEntry, error) { + out, err := cacheJSON([]string{"cache", "ls", "--json"}, "bsdkrun cache ls") + if err != nil { + return nil, err + } + var entries []CacheEntry + if err := json.Unmarshal(out, &entries); err != nil { + return nil, err + } + return entries, nil +} + +// RemoveCache removes entries by key. With all set, it removes every entry and +// keys is ignored. +func RemoveCache(keys []string, all bool) error { + args := []string{"cache", "rm"} + if all { + args = append(args, "--all") + } else { + args = append(args, keys...) + } + _, err := RunChecked(args, "bsdkrun cache rm", nil) + return err +} + +func cacheJSON(args []string, label string) ([]byte, error) { + res, err := RunChecked(args, label, nil) + if err != nil { + return nil, err + } + if res.Stdout == "" { + return []byte("{}"), nil + } + return []byte(res.Stdout), nil +} diff --git a/sdk/python/README.md b/sdk/python/README.md index 15a374e..a218661 100644 --- a/sdk/python/README.md +++ b/sdk/python/README.md @@ -10,14 +10,14 @@ runtime dependencies** — stdlib only, Python 3.10+. ```python from bsdkrun import Sandbox -box = Sandbox.create(os="linux", image="alpine") +sbx = Sandbox.create(os="linux", image="alpine") # exec argv directly, with env / stdin / a PTY / a working dir: -print(box.exec(["uname", "-a"]).text()) -box.exec(["apk", "add", "curl"]) -box.run_command("curl", ["-fsSL", "https://example.com"]) +print(sbx.exec(["uname", "-a"]).text()) +sbx.exec(["apk", "add", "curl"]) +sbx.run_command("curl", ["-fsSL", "https://example.com"]) -box.stop() +sbx.stop() ``` ## Install @@ -84,9 +84,9 @@ Pass an argv list (no shell parsing), or a program name plus `args`: ```python import sys -box.exec(["ls", "-la", "/etc"]) +sbx.exec(["ls", "-la", "/etc"]) -box.exec( +sbx.exec( "node", args=["-e", "print(1)"], env={"X": "hi"}, @@ -99,7 +99,7 @@ box.exec( ) # Vercel-Sandbox-style alias: -result = box.run_command("uname", ["-a"]) +result = sbx.run_command("uname", ["-a"]) print(result.stdout, result.exit_code) ``` @@ -110,21 +110,48 @@ The stream callbacks receive `bytes` as they arrive while the complete output is still captured in the returned `Result`. They are independent of `tty`; allocating a TTY changes command behavior and may merge stderr into stdout. +## Caching + +``sbx.cache`` saves a guest directory under a key and restores it later, so a +rebuild can pick up where the last one left off. **A miss is not an error** — +check ``restored`` rather than catching. + +```python +from bsdkrun import caches + +key = f"deps-{lock_hash}" +hit = sbx.cache.restore(key=key, restore_keys=["deps-"]) +if not hit.restored: + sbx.exec(["npm", "ci"]) + sbx.cache.save("/app/node_modules", key=key, compression="zstd") + +caches.ls() # every stored entry, newest first +caches.rm([key]) # or caches.rm(all=True) +``` + +``restore_keys`` are prefixes tried in order when the exact key misses; within a +prefix the newest matching entry wins, and ``hit.key`` says which one was used. +Formats are ``gzip`` (default), ``zstd``, ``estargz`` and ``none``. + +Where entries live is host configuration, not an SDK concern: the default is +this host's disk, and `BSDKRUN_CACHE_BACKEND=s3` + `BSDKRUN_CACHE_S3_*` (or +`~/.config/bsdkrun/cache.toml`) points them at a bucket instead. + ## Files -``box.fs`` reads and writes files in the guest. Parent directories are created +``sbx.fs`` reads and writes files in the guest. Parent directories are created for you, and everything is byte-exact — ``read_file`` returns ``bytes``, so a PNG survives the round trip. ```python -box.fs.write_file("/app/main.py", "print('hi')") -box.fs.write_file("/app/logo.png", png_bytes) +sbx.fs.write_file("/app/main.py", "print('hi')") +sbx.fs.write_file("/app/logo.png", png_bytes) -text = box.fs.read_text("/app/out.json") -data = box.fs.read_file("/app/logo.png") +text = sbx.fs.read_text("/app/out.json") +data = sbx.fs.read_file("/app/logo.png") -box.fs.upload("./src", "/app/src") # file or directory -box.fs.download("/app/dist", "./dist", recursive=True) +sbx.fs.upload("./src", "/app/src") # file or directory +sbx.fs.download("/app/dist", "./dist", recursive=True) ``` ``upload`` looks at the local path to decide whether to recurse; ``download`` @@ -141,18 +168,18 @@ Failures raise ``FileTransferError``, which carries the offending ``path``. ## Lifecycle & inventory ```python -box = Sandbox.create(os="linux", image="alpine", command=["sleep", "300"]) -same = Sandbox.get(box.id) # reconnect (prefix ok) +sbx = Sandbox.create(os="linux", image="alpine", command=["sleep", "300"]) +same = Sandbox.get(sbx.id) # reconnect (prefix ok) rows = Sandbox.list(all=True) # list[SandboxInfo] -box.status() # SandboxInfo | None -box.is_running() # bool -box.logs() # console log (str) -box.shell() # interactive shell (inherits the terminal) -box.stop() # BSD guests clean-poweroff; Linux SIGTERM -box.start() # restart in place — resumes its own disk/rootfs (data persists) -box.update(cpus=4, mem=2048) # applies on next start -box.remove(force=True) +sbx.status() # SandboxInfo | None +sbx.is_running() # bool +sbx.logs() # console log (str) +sbx.shell() # interactive shell (inherits the terminal) +sbx.stop() # BSD guests clean-poweroff; Linux SIGTERM +sbx.start() # restart in place — resumes its own disk/rootfs (data persists) +sbx.update(cpus=4, mem=2048) # applies on next start +sbx.remove(force=True) ``` Host-level namespaces: @@ -208,11 +235,11 @@ an existing network without restarting members. ```python # agent-managed key-based SSH -box.ssh_setup() # install local ~/.ssh/*.pub keys -box.ssh_setup(user="tsiry", key="~/.ssh/work.pub") +sbx.ssh_setup() # install local ~/.ssh/*.pub keys +sbx.ssh_setup(user="tsiry", key="~/.ssh/work.pub") # put a guest on your tailnet -box.tailscale_up(authkey="tskey-auth-...", hostname="web") +sbx.tailscale_up(authkey="tskey-auth-...", hostname="web") ``` ## Connecting to a remote daemon diff --git a/sdk/python/src/bsdkrun/__init__.py b/sdk/python/src/bsdkrun/__init__.py index b0ddccb..81d0046 100644 --- a/sdk/python/src/bsdkrun/__init__.py +++ b/sdk/python/src/bsdkrun/__init__.py @@ -6,9 +6,9 @@ shells out, and parses the JSON output. from bsdkrun import Sandbox, networks - box = Sandbox.create(os="linux", image="alpine") - print(box.exec(["uname", "-a"]).text()) - box.stop() + sbx = Sandbox.create(os="linux", image="alpine") + print(sbx.exec(["uname", "-a"]).text()) + sbx.stop() Host-level operations live in the :mod:`bsdkrun.images`, :mod:`bsdkrun.volumes`, :mod:`bsdkrun.networks`, and :mod:`bsdkrun.system` namespaces. @@ -16,9 +16,10 @@ Host-level operations live in the :mod:`bsdkrun.images`, :mod:`bsdkrun.volumes`, from __future__ import annotations -from . import images, networks, system, volumes +from . import caches, images, networks, system, volumes from .args import build_create_args from .binary import reset_binary_cache, resolve_binary, set_binary_path +from .cache import Cache, CacheEntry, RestoreResult from .client import Client, ShellSession from .errors import ( AuthError, @@ -56,6 +57,7 @@ __all__ = [ "volumes", "networks", "system", + "caches", # binary resolution "set_binary_path", "resolve_binary", @@ -69,6 +71,10 @@ __all__ = [ "BinaryResult", # guest filesystem "FileSystem", + # guest directory cache + "Cache", + "CacheEntry", + "RestoreResult", # argv builder "build_create_args", # data types diff --git a/sdk/python/src/bsdkrun/cache.py b/sdk/python/src/bsdkrun/cache.py new file mode 100644 index 0000000..24a2eb7 --- /dev/null +++ b/sdk/python/src/bsdkrun/cache.py @@ -0,0 +1,152 @@ +"""Cached guest directories — ``sandbox.cache``, plus host-level listing. + +Entries are keyed, so a rebuild can pick up where the last one left off:: + + hit = box.cache.restore(key=key, restore_keys=["deps-"]) + if not hit.restored: + box.exec(["npm", "ci"]) + box.cache.save("/app/node_modules", key=key) + +Where entries live — host disk or S3 — is host configuration, not an SDK +concern: set ``BSDKRUN_CACHE_BACKEND`` / ``BSDKRUN_CACHE_S3_*``, or write +``~/.config/bsdkrun/cache.toml``. +""" + +from __future__ import annotations + +import json +from dataclasses import dataclass +from typing import Literal + +from .errors import CommandFailed +from .process import run + +__all__ = ["Cache", "CacheEntry", "RestoreResult", "Compression", "list_caches", "remove_cache"] + +#: An archive format a cache entry can be stored in. +Compression = Literal["gzip", "zstd", "estargz", "none"] + + +@dataclass(frozen=True) +class CacheEntry: + """A stored cache entry, as ``cache ls`` reports it.""" + + key: str + #: Guest path the tree came from. + path: str + compression: str + #: Archive size in bytes. + size: int + #: Unix seconds when it was saved. + created: int + #: ``sha256:…`` over the archive. + digest: str = "" + + @staticmethod + def _from(row: dict) -> CacheEntry: + return CacheEntry( + key=str(row.get("key", "")), + path=str(row.get("path", "")), + compression=str(row.get("compression", "")), + size=int(row.get("size", 0)), + created=int(row.get("created", 0)), + digest=str(row.get("digest", "")), + ) + + +@dataclass(frozen=True) +class RestoreResult: + """What a restore did. A miss is not an error — check :attr:`restored`.""" + + restored: bool + #: The key asked for. + requested_key: str + #: The entry actually used; differs from :attr:`requested_key` when a + #: ``restore_keys`` prefix matched, and is None on a miss. + key: str | None = None + #: Guest path it was restored into. + path: str | None = None + size: int | None = None + compression: str | None = None + created: int | None = None + + +def _json(args: list[str], label: str) -> dict: + result = run(args) + if result.exit_code != 0: + raise CommandFailed(result.exit_code, result.stdout, result.stderr, label) + return json.loads(result.stdout or "{}") + + +class Cache: + """Save and restore guest directories under a key.""" + + def __init__(self, sandbox_id: str) -> None: + self.id = sandbox_id + + def __repr__(self) -> str: + return f"Cache(id={self.id!r})" + + def save( + self, + path: str, + *, + key: str, + compression: Compression = "gzip", + force: bool = False, + ) -> CacheEntry: + """Archive the guest directory at ``path`` under ``key``.""" + args = ["cache", "save", f"{self.id}:{path}", "--key", key, "--json"] + if compression != "gzip": + args += ["--compression", compression] + if force: + args.append("--force") + return CacheEntry._from(_json(args, "bsdkrun cache save")) + + def restore( + self, + *, + key: str, + path: str | None = None, + restore_keys: list[str] | None = None, + ) -> RestoreResult: + """Restore a stored tree. + + ``path`` defaults to the directory the entry was saved from. + ``restore_keys`` are prefixes tried in order when ``key`` misses; within + a prefix the newest matching entry wins. + """ + target = f"{self.id}:{path}" if path else self.id + args = ["cache", "restore", target, "--key", key, "--json"] + if restore_keys: + args += ["--restore-keys", *restore_keys] + row = _json(args, "bsdkrun cache restore") + return RestoreResult( + restored=bool(row.get("restored")), + requested_key=str(row.get("requested_key", key)), + key=row.get("key"), + path=row.get("path"), + size=row.get("size"), + compression=row.get("compression"), + created=row.get("created"), + ) + + +def list_caches() -> list[CacheEntry]: + """Every stored cache entry, newest first.""" + result = run(["cache", "ls", "--json"]) + if result.exit_code != 0: + raise CommandFailed(result.exit_code, result.stdout, result.stderr, "bsdkrun cache ls") + return [CacheEntry._from(r) for r in json.loads(result.stdout or "[]")] + + +def remove_cache(keys: str | list[str] | None = None, *, all: bool = False) -> None: + """Remove entries by key, or every one of them with ``all=True``.""" + args = ["cache", "rm"] + if all: + args.append("--all") + else: + args += [keys] if isinstance(keys, str) else list(keys or []) + result = run(args) + if result.exit_code != 0: + raise CommandFailed(result.exit_code, result.stdout, result.stderr, "bsdkrun cache rm") diff --git a/sdk/python/src/bsdkrun/caches.py b/sdk/python/src/bsdkrun/caches.py new file mode 100644 index 0000000..497a695 --- /dev/null +++ b/sdk/python/src/bsdkrun/caches.py @@ -0,0 +1,16 @@ +"""Host-level cache operations — listing and removing stored entries. + +Mirrors :mod:`bsdkrun.images` / :mod:`bsdkrun.volumes`: the per-sandbox half +lives on :attr:`bsdkrun.Sandbox.cache`. +""" + +from __future__ import annotations + +from .cache import CacheEntry, list_caches, remove_cache + +__all__ = ["CacheEntry", "ls", "rm"] + +#: Every stored cache entry, newest first. +ls = list_caches +#: Remove entries by key, or every one of them with ``all=True``. +rm = remove_cache diff --git a/sdk/python/src/bsdkrun/sandbox.py b/sdk/python/src/bsdkrun/sandbox.py index 5d81e0b..817e645 100644 --- a/sdk/python/src/bsdkrun/sandbox.py +++ b/sdk/python/src/bsdkrun/sandbox.py @@ -8,6 +8,7 @@ from collections.abc import Callable, Mapping, Sequence from typing import Any from .args import build_create_args +from .cache import Cache from .errors import CommandFailed, SandboxNotFound from .filesystem import FileSystem from .process import run, run_checked, spawn @@ -25,9 +26,9 @@ class Sandbox: Create one with :meth:`create`, reconnect with :meth:`get`, or enumerate with :meth:`list`:: - box = Sandbox.create(os="linux", image="alpine") - box.exec(["uname", "-a"]) - box.stop() + sbx = Sandbox.create(os="linux", image="alpine") + sbx.exec(["uname", "-a"]) + sbx.stop() """ def __init__(self, sandbox_id: str, ssh_port: int | None = None) -> None: @@ -37,6 +38,8 @@ class Sandbox: self.ssh_port = ssh_port #: Read and write files in the guest. self.fs = FileSystem(sandbox_id) + #: Save and restore guest directories under a key. + self.cache = Cache(sandbox_id) def __repr__(self) -> str: return f"Sandbox(id={self.id!r})" @@ -120,8 +123,8 @@ class Sandbox: chunks in real time while the same output remains available in the returned result. - box.exec(["ls", "-la", "/etc"]) - box.exec("node", args=["-e", "print(1)"], env={"X": "1"}) + sbx.exec(["ls", "-la", "/etc"]) + sbx.exec("node", args=["-e", "print(1)"], env={"X": "1"}) """ if isinstance(command, str): argv = [command, *(args or [])] diff --git a/sdk/ruby/README.md b/sdk/ruby/README.md index d5f37dd..176f25c 100644 --- a/sdk/ruby/README.md +++ b/sdk/ruby/README.md @@ -9,14 +9,14 @@ dependencies** — just the Ruby standard library (`open3`, `json`, `pathname`). ```ruby require "bsdkrun" -box = Bsdkrun::Sandbox.create(os: "linux", image: "alpine") +sbx = Bsdkrun::Sandbox.create(os: "linux", image: "alpine") # exec argv directly, with env / stdin / a PTY / a working dir: -puts box.exec(["uname", "-a"]).text -box.exec(["apk", "add", "curl"], throw_on_error: true) -box.run_command("curl", ["-fsSL", "https://example.com"]) +puts sbx.exec(["uname", "-a"]).text +sbx.exec(["apk", "add", "curl"], throw_on_error: true) +sbx.run_command("curl", ["-fsSL", "https://example.com"]) -box.stop +sbx.stop ``` ## Install @@ -83,9 +83,9 @@ Every `create` runs the machine **detached** and returns a `Sandbox` handle. array (or a program name plus `args:`). ```ruby -box.exec(["ls", "-la", "/etc"]) +sbx.exec(["ls", "-la", "/etc"]) -box.exec("ruby", +sbx.exec("ruby", args: ["-e", "puts ENV['X']"], env: { "X" => "hi" }, cwd: "/app", @@ -96,7 +96,7 @@ box.exec("ruby", throw_on_error: true) # raise on non-zero exit (default: false) # Vercel-Sandbox-style alias: -result = box.run_command("uname", ["-a"]) +result = sbx.run_command("uname", ["-a"]) result.stdout # raw stdout result.text # stdout, trailing newlines trimmed result.exit_code @@ -111,20 +111,46 @@ The callbacks run as chunks arrive, and the same bytes remain buffered in the returned result. They do not require `tty`; a PTY changes command semantics and may merge stderr into stdout. +## Caching + +`sbx.cache` saves a guest directory under a key and restores it later, so a +rebuild can pick up where the last one left off. **A miss is not an error** — +check `restored` rather than rescuing. + +```ruby +key = "deps-#{lock_hash}" +hit = sbx.cache.restore(key: key, restore_keys: ["deps-"]) +unless hit.restored + sbx.exec(["npm", "ci"]) + sbx.cache.save("/app/node_modules", key: key, compression: "zstd") +end + +Bsdkrun::Caches.ls # every stored entry, newest first +Bsdkrun::Caches.rm([key]) # or Bsdkrun::Caches.rm(all: true) +``` + +`restore_keys` are prefixes tried in order when the exact key misses; within a +prefix the newest matching entry wins, and `hit.key` says which one was used. +Formats are `gzip` (default), `zstd`, `estargz` and `none`. + +Where entries live is host configuration, not an SDK concern: the default is +this host's disk, and `BSDKRUN_CACHE_BACKEND=s3` + `BSDKRUN_CACHE_S3_*` (or +`~/.config/bsdkrun/cache.toml`) points them at a bucket instead. + ## Files -`box.fs` reads and writes files in the guest. Parent directories are created +`sbx.fs` reads and writes files in the guest. Parent directories are created for you, and everything is byte-exact — `read_file` returns a binary string. ```ruby -box.fs.write_file("/app/main.py", "print('hi')") -box.fs.write_file("/app/logo.png", png_bytes) +sbx.fs.write_file("/app/main.py", "print('hi')") +sbx.fs.write_file("/app/logo.png", png_bytes) -text = box.fs.read_text("/app/out.json") -bytes = box.fs.read_file("/app/logo.png") +text = sbx.fs.read_text("/app/out.json") +bytes = sbx.fs.read_file("/app/logo.png") -box.fs.upload("./src", "/app/src") # file or directory -box.fs.download("/app/dist", "./dist", recursive: true) +sbx.fs.upload("./src", "/app/src") # file or directory +sbx.fs.download("/app/dist", "./dist", recursive: true) ``` `upload` looks at the local path to decide whether to recurse; `download` cannot @@ -141,18 +167,18 @@ Failures raise `Bsdkrun::FileTransferFailed`, which carries the offending `path` ## Lifecycle & inventory ```ruby -box = Bsdkrun::Sandbox.create(os: "linux", image: "alpine", command: ["sleep", "300"]) -same = Bsdkrun::Sandbox.get(box.id) # reconnect (prefix ok) +sbx = Bsdkrun::Sandbox.create(os: "linux", image: "alpine", command: ["sleep", "300"]) +same = Bsdkrun::Sandbox.get(sbx.id) # reconnect (prefix ok) all = Bsdkrun::Sandbox.list(all: true) # Array -box.status # SandboxInfo | nil -box.running? # true / false -box.logs # console log (String) -box.shell # interactive shell (inherits the terminal) -box.stop # BSD guests clean-poweroff; Linux SIGTERM -box.start # restart in place — resumes its own disk/rootfs (data persists) -box.update(cpus: 4, mem: 2048) # applies on next start -box.remove(force: true) +sbx.status # SandboxInfo | nil +sbx.running? # true / false +sbx.logs # console log (String) +sbx.shell # interactive shell (inherits the terminal) +sbx.stop # BSD guests clean-poweroff; Linux SIGTERM +sbx.start # restart in place — resumes its own disk/rootfs (data persists) +sbx.update(cpus: 4, mem: 2048) # applies on next start +sbx.remove(force: true) ``` Host-level namespaces: @@ -174,11 +200,11 @@ Bsdkrun::System.versions("netbsd") # Array Bsdkrun::Sandbox.create(os: "linux", image: "alpine", net: { ports: ["2222:22"] }) # agent-managed key-based SSH -box.ssh_setup # install local ~/.ssh/*.pub keys -box.ssh_setup(user: "tsiry", key: "~/.ssh/work.pub") +sbx.ssh_setup # install local ~/.ssh/*.pub keys +sbx.ssh_setup(user: "tsiry", key: "~/.ssh/work.pub") # put a guest on your tailnet -box.tailscale_up(authkey: "tskey-auth-...", hostname: "web") +sbx.tailscale_up(authkey: "tskey-auth-...", hostname: "web") ``` ### Global networks — reach machines by name diff --git a/sdk/ruby/lib/bsdkrun.rb b/sdk/ruby/lib/bsdkrun.rb index 35ecffd..3a6f7d0 100644 --- a/sdk/ruby/lib/bsdkrun.rb +++ b/sdk/ruby/lib/bsdkrun.rb @@ -5,6 +5,7 @@ require_relative "bsdkrun/errors" require_relative "bsdkrun/binary" require_relative "bsdkrun/process" require_relative "bsdkrun/args" +require_relative "bsdkrun/cache" require_relative "bsdkrun/filesystem" require_relative "bsdkrun/types" require_relative "bsdkrun/sandbox" @@ -24,9 +25,9 @@ require_relative "bsdkrun/client" # @example # require "bsdkrun" # -# box = Bsdkrun::Sandbox.create(os: "linux", image: "alpine") -# puts box.exec(["uname", "-a"]).text -# box.stop +# sbx = Bsdkrun::Sandbox.create(os: "linux", image: "alpine") +# puts sbx.exec(["uname", "-a"]).text +# sbx.stop module Bsdkrun class << self # Force the SDK to use a specific +bsdkrun+ binary, bypassing discovery. diff --git a/sdk/ruby/lib/bsdkrun/cache.rb b/sdk/ruby/lib/bsdkrun/cache.rb new file mode 100644 index 0000000..60a56f9 --- /dev/null +++ b/sdk/ruby/lib/bsdkrun/cache.rb @@ -0,0 +1,112 @@ +# frozen_string_literal: true + +require "json" + +module Bsdkrun + # A stored cache entry, as +cache ls+ reports it. + CacheEntry = Struct.new(:key, :path, :compression, :size, :created, :digest, keyword_init: true) do + def self.from(row) + new( + key: row["key"].to_s, + path: row["path"].to_s, + compression: row["compression"].to_s, + size: row["size"].to_i, + created: row["created"].to_i, + digest: row["digest"].to_s + ) + end + end + + # What a restore did. A miss is not an error — check +restored+. + RestoreResult = Struct.new( + :restored, :requested_key, :key, :path, :size, :compression, :created, + keyword_init: true + ) + + # Save and restore guest directories under a key, reached as {Sandbox#cache}. + # + # Entries are keyed, so a rebuild can pick up where the last one left off: + # + # hit = sbx.cache.restore(key: key, restore_keys: ["deps-"]) + # unless hit.restored + # sbx.exec(["npm", "ci"]) + # sbx.cache.save("/app/node_modules", key: key) + # end + # + # Where entries live — host disk or S3 — is host configuration, not an SDK + # concern: set +BSDKRUN_CACHE_BACKEND+ / +BSDKRUN_CACHE_S3_*+, or write + # +~/.config/bsdkrun/cache.toml+. + class Cache + # @param id [String] the machine's id. + def initialize(id) + @id = id + end + + # Archive the guest directory at +path+ under +key+. + # + # @param path [String] absolute path in the guest. + # @param key [String] key to store under. + # @param compression [String] gzip (default), zstd, estargz or none. + # @param force [Boolean] replace an entry that already has this key. + # @return [CacheEntry] + def save(path, key:, compression: "gzip", force: false) + args = ["cache", "save", "#{@id}:#{path}", "--key", key, "--json"] + args += ["--compression", compression] unless compression == "gzip" + args << "--force" if force + CacheEntry.from(json(args, "bsdkrun cache save")) + end + + # Restore a stored tree. + # + # @param key [String] + # @param path [String, nil] defaults to where the entry was saved from. + # @param restore_keys [Array] prefixes tried in order on a miss. + # @return [RestoreResult] + def restore(key:, path: nil, restore_keys: []) + target = path ? "#{@id}:#{path}" : @id + args = ["cache", "restore", target, "--key", key, "--json"] + args += ["--restore-keys", *restore_keys] unless restore_keys.empty? + row = json(args, "bsdkrun cache restore") + RestoreResult.new( + restored: !!row["restored"], + requested_key: row["requested_key"].to_s, + key: row["key"], + path: row["path"], + size: row["size"], + compression: row["compression"], + created: row["created"] + ) + end + + private + + def json(args, label) + out = Process.run!(args, label: label).stdout + JSON.parse(out.strip.empty? ? "{}" : out) + end + end + + # Host-level cache operations, mirroring {Bsdkrun.volumes}. + module Caches + module_function + + # @return [Array] every stored entry, newest first. + def ls + out = Process.run!(["cache", "ls", "--json"], label: "bsdkrun cache ls").stdout + JSON.parse(out.strip.empty? ? "[]" : out).map { |row| CacheEntry.from(row) } + end + + # Remove entries by key, or every one with all: true. + # @return [void] + def rm(keys = [], all: false) + args = ["cache", "rm"] + if all + args << "--all" + else + args += Array(keys) + end + Process.run!(args, label: "bsdkrun cache rm") + nil + end + end +end diff --git a/sdk/ruby/lib/bsdkrun/sandbox.rb b/sdk/ruby/lib/bsdkrun/sandbox.rb index b7501b9..9691658 100644 --- a/sdk/ruby/lib/bsdkrun/sandbox.rb +++ b/sdk/ruby/lib/bsdkrun/sandbox.rb @@ -9,9 +9,9 @@ module Bsdkrun # with {Sandbox.list}. # # @example - # box = Bsdkrun::Sandbox.create(os: "linux", image: "alpine") - # box.exec(["uname", "-a"]).text - # box.stop + # sbx = Bsdkrun::Sandbox.create(os: "linux", image: "alpine") + # sbx.exec(["uname", "-a"]).text + # sbx.stop class Sandbox ID_RE = /\A[0-9a-f]{6,}\z/ SSH_PORT_RE = /ssh -p (\d+)/ @@ -38,6 +38,12 @@ module Bsdkrun @fs ||= FileSystem.new(@id) end + # Save and restore guest directories under a key. + # @return [Cache] + def cache + @cache ||= Cache.new(@id) + end + class << self # Boot a new microVM and return a handle to it. # diff --git a/sdk/rust/README.md b/sdk/rust/README.md index 0aceca7..bdeb5e9 100644 --- a/sdk/rust/README.md +++ b/sdk/rust/README.md @@ -132,6 +132,34 @@ struct Info { hostname: String } let info: Info = sandbox.exec(["cat", "/etc/info.json"])?.json()?; ``` +## Caching + +`Sandbox::cache()` saves a guest directory under a key and restores it later, so +a rebuild can pick up where the last one left off. **A miss is not an error** — +check `restored` rather than the `Result`. + +```rust +use bsdkrun_sdk::cache::{self, Compression}; + +let key = format!("deps-{lock_hash}"); +let hit = sbx.cache().restore(&key, None, &["deps-".to_string()])?; +if !hit.restored { + sbx.exec(["npm", "ci"])?; + sbx.cache().save("/app/node_modules", &key, Compression::Zstd, false)?; +} + +cache::list()?; // every stored entry, newest first +cache::remove(&[key.clone()], false)?; // or (&[], true) for all +``` + +The restore keys are prefixes tried in order when the exact key misses; within a +prefix the newest matching entry wins, and `hit.key` says which one was used. +Formats are `Gzip` (default), `Zstd`, `Estargz` and `None`. + +Where entries live is host configuration, not an SDK concern: the default is +this host's disk, and `BSDKRUN_CACHE_BACKEND=s3` + `BSDKRUN_CACHE_S3_*` (or +`~/.config/bsdkrun/cache.toml`) points them at a bucket instead. + ## Files `Sandbox::fs()` reads and writes files in the guest. Parent directories are diff --git a/sdk/rust/src/cache.rs b/sdk/rust/src/cache.rs new file mode 100644 index 0000000..7d28ad6 --- /dev/null +++ b/sdk/rust/src/cache.rs @@ -0,0 +1,242 @@ +//! Cached guest directories — [`Sandbox::cache`](crate::Sandbox::cache), plus +//! host-level listing. +//! +//! Entries are keyed, so a rebuild can pick up where the last one left off. +//! Where they live — host disk or S3 — is host configuration, not an SDK +//! concern: set `BSDKRUN_CACHE_BACKEND` / `BSDKRUN_CACHE_S3_*`, or write +//! `~/.config/bsdkrun/cache.toml`. + +use serde_json::Value; + +use crate::error::{Error, Result}; +use crate::process::run; + +/// An archive format a cache entry can be stored in. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum Compression { + #[default] + Gzip, + Zstd, + Estargz, + None, +} + +impl Compression { + fn as_str(self) -> &'static str { + match self { + Compression::Gzip => "gzip", + Compression::Zstd => "zstd", + Compression::Estargz => "estargz", + Compression::None => "none", + } + } +} + +/// A stored cache entry, as `cache ls` reports it. +#[derive(Debug, Clone, Default)] +pub struct CacheEntry { + /// The exact key it was saved under. + pub key: String, + /// Guest path the tree came from. + pub path: String, + pub compression: String, + /// Archive size in bytes. + pub size: u64, + /// Unix seconds when it was saved. + pub created: u64, + /// `sha256:…` over the archive. + pub digest: String, +} + +impl CacheEntry { + fn from_value(v: &Value) -> CacheEntry { + CacheEntry { + key: str_at(v, "key"), + path: str_at(v, "path"), + compression: str_at(v, "compression"), + size: num_at(v, "size").unwrap_or(0), + created: num_at(v, "created").unwrap_or(0), + digest: str_at(v, "digest"), + } + } +} + +/// What a restore did. A miss is not an error — check [`restored`](Self::restored). +#[derive(Debug, Clone, Default)] +pub struct RestoreResult { + pub restored: bool, + /// The key asked for. + pub requested_key: String, + /// The entry actually used. Differs from [`requested_key`](Self::requested_key) + /// when a restore-key prefix matched, and is `None` on a miss. + pub key: Option, + /// Guest path it was restored into. + pub path: Option, + pub size: Option, + pub compression: Option, + pub created: Option, +} + +impl RestoreResult { + fn from_value(v: &Value) -> RestoreResult { + RestoreResult { + restored: v.get("restored").and_then(Value::as_bool).unwrap_or(false), + requested_key: str_at(v, "requested_key"), + key: opt_str_at(v, "key"), + path: opt_str_at(v, "path"), + size: num_at(v, "size"), + compression: opt_str_at(v, "compression"), + created: num_at(v, "created"), + } + } +} + +// Lenient accessors, matching `types.rs`: the SDK reads the CLI's JSON through +// `serde_json::Value` rather than deriving, so a field the CLI adds later never +// turns into a decode error. +fn str_at(v: &Value, key: &str) -> String { + v.get(key) + .and_then(Value::as_str) + .unwrap_or_default() + .to_string() +} + +fn opt_str_at(v: &Value, key: &str) -> Option { + v.get(key).and_then(Value::as_str).map(str::to_string) +} + +fn num_at(v: &Value, key: &str) -> Option { + v.get(key).and_then(Value::as_u64) +} + +/// Save and restore guest directories under a key. +/// +/// ```no_run +/// # use bsdkrun_sdk::{Sandbox, cache::Compression}; +/// # fn main() -> bsdkrun_sdk::Result<()> { +/// let sbx = Sandbox::get("web")?; +/// let hit = sbx.cache().restore("deps-abc123", None, &["deps-".to_string()])?; +/// if !hit.restored { +/// sbx.exec(["npm", "ci"])?; +/// sbx.cache().save("/app/node_modules", "deps-abc123", Compression::Zstd, false)?; +/// } +/// # Ok(()) +/// # } +/// ``` +pub struct Cache { + id: String, +} + +impl Cache { + pub(crate) fn new(id: impl Into) -> Self { + Cache { id: id.into() } + } + + /// Archive the guest directory at `path` under `key`. + pub fn save( + &self, + path: &str, + key: &str, + compression: Compression, + force: bool, + ) -> Result { + let mut args = vec![ + "cache".to_string(), + "save".to_string(), + format!("{}:{}", self.id, path), + "--key".to_string(), + key.to_string(), + "--json".to_string(), + ]; + if compression != Compression::Gzip { + args.push("--compression".to_string()); + args.push(compression.as_str().to_string()); + } + if force { + args.push("--force".to_string()); + } + Ok(CacheEntry::from_value(&json(&args, "bsdkrun cache save")?)) + } + + /// Restore a stored tree. + /// + /// `path` defaults to the directory the entry was saved from. + /// `restore_keys` are prefixes tried in order when `key` misses; within a + /// prefix the newest matching entry wins. + pub fn restore( + &self, + key: &str, + path: Option<&str>, + restore_keys: &[String], + ) -> Result { + let target = match path { + Some(p) => format!("{}:{}", self.id, p), + None => self.id.clone(), + }; + let mut args = vec![ + "cache".to_string(), + "restore".to_string(), + target, + "--key".to_string(), + key.to_string(), + "--json".to_string(), + ]; + if !restore_keys.is_empty() { + args.push("--restore-keys".to_string()); + args.extend(restore_keys.iter().cloned()); + } + Ok(RestoreResult::from_value(&json( + &args, + "bsdkrun cache restore", + )?)) + } +} + +/// Every stored cache entry, newest first. +pub fn list() -> Result> { + let v = json( + &["cache".to_string(), "ls".to_string(), "--json".to_string()], + "bsdkrun cache ls", + )?; + Ok(v.as_array() + .map(|rows| rows.iter().map(CacheEntry::from_value).collect()) + .unwrap_or_default()) +} + +/// Remove entries by key, or every one of them with `all`. +pub fn remove(keys: &[String], all: bool) -> Result<()> { + let mut args = vec!["cache".to_string(), "rm".to_string()]; + if all { + args.push("--all".to_string()); + } else { + args.extend(keys.iter().cloned()); + } + let res = run(args)?; + if res.exit_code != 0 { + return Err(Error::CommandFailed { + exit_code: res.exit_code, + stdout: res.stdout, + stderr: res.stderr, + command: "bsdkrun cache rm".to_string(), + }); + } + Ok(()) +} + +fn json(args: &[String], label: &str) -> Result { + let res = run(args.to_vec())?; + if res.exit_code != 0 { + return Err(Error::CommandFailed { + exit_code: res.exit_code, + stdout: res.stdout, + stderr: res.stderr, + command: label.to_string(), + }); + } + serde_json::from_str(res.stdout.trim()).map_err(|e| Error::CommandFailed { + exit_code: 0, + stdout: res.stdout.clone(), + stderr: format!("could not decode {label} output: {e}"), + command: label.to_string(), + }) +} diff --git a/sdk/rust/src/lib.rs b/sdk/rust/src/lib.rs index f176916..52b6c70 100644 --- a/sdk/rust/src/lib.rs +++ b/sdk/rust/src/lib.rs @@ -27,6 +27,7 @@ mod args; mod binary; mod client; +pub mod cache; mod error; mod filesystem; mod process; @@ -46,6 +47,7 @@ pub use client::{ ShellSession, Subscription, }; pub use error::{Error, Result}; +pub use cache::{Cache, CacheEntry, RestoreResult}; pub use filesystem::FileSystem; pub use process::{run, run_binary, run_checked, spawn, BinaryResult, RawResult}; pub use sandbox::{ diff --git a/sdk/rust/src/sandbox.rs b/sdk/rust/src/sandbox.rs index 1ba598b..23f7c67 100644 --- a/sdk/rust/src/sandbox.rs +++ b/sdk/rust/src/sandbox.rs @@ -4,14 +4,14 @@ //! ```no_run //! use bsdkrun_sdk::Sandbox; //! -//! let boxx = Sandbox::linux("alpine") +//! let sbx = Sandbox::linux("alpine") //! .cpus(2) //! .mem(1024) //! .port("8080:80") //! .command(["sleep", "300"]) //! .create()?; -//! println!("{}", boxx.exec(["uname", "-a"])?.text()); -//! boxx.stop()?; +//! println!("{}", sbx.exec(["uname", "-a"])?.text()); +//! sbx.stop()?; //! # Ok::<(), bsdkrun_sdk::Error>(()) //! ``` @@ -449,6 +449,11 @@ impl Sandbox { crate::filesystem::FileSystem::new(&self.id) } + /// Save and restore guest directories under a key. + pub fn cache(&self) -> crate::Cache { + crate::cache::Cache::new(&self.id) + } + /// Boot an OCI image as a Linux microVM. pub fn linux(image: impl Into) -> LinuxBuilder { LinuxBuilder { diff --git a/sdk/typescript/README.md b/sdk/typescript/README.md index 487c42c..d82e94b 100644 --- a/sdk/typescript/README.md +++ b/sdk/typescript/README.md @@ -11,16 +11,16 @@ binary, so it has zero npm dependencies. ```ts import { Sandbox } from "@bsdkrun/sdk"; -const box = await Sandbox.create({ os: "linux", image: "alpine" }); +const sbx = await Sandbox.create({ os: "linux", image: "alpine" }); // Tagged-template shell — interpolations are shell-quoted for you: -const kernel = await box.sh`uname -a`.text(); +const kernel = await sbx.sh`uname -a`.text(); // ...or exec argv directly, with env / stdin / a PTY / a working dir: -await box.exec(["apk", "add", "curl"]); -await box.runCommand("curl", ["-fsSL", "https://example.com"]); +await sbx.exec(["apk", "add", "curl"]); +await sbx.runCommand("curl", ["-fsSL", "https://example.com"]); -await box.stop(); +await sbx.stop(); ``` ## Install @@ -110,14 +110,14 @@ Best for quick shell one-liners. Interpolated values are **single-quoted** ```ts const dir = "/etc"; -await box.sh`ls -la ${dir}`; // quoted -await box.sh`grep ${pattern} /var/log/*`; // quoted +await sbx.sh`ls -la ${dir}`; // quoted +await sbx.sh`grep ${pattern} /var/log/*`; // quoted -await box.sh`cat /nope`.nothrow(); // don't throw on non-zero exit -await box.sh`echo $X`.env({ X: "1" }).text(); +await sbx.sh`cat /nope`.nothrow(); // don't throw on non-zero exit +await sbx.sh`echo $X`.env({ X: "1" }).text(); import { raw } from "@bsdkrun/sdk"; -await box.sh`ls ${raw("-la /var")}`; // spliced verbatim (trusted only) +await sbx.sh`ls ${raw("-la /var")}`; // spliced verbatim (trusted only) ``` An `sh` call is lazy and awaitable; `.text()`, `.json()`, `.lines()` are @@ -130,9 +130,9 @@ The primary programmatic entrypoint. No shell parsing; pass an argv array (or a program name plus `args`). Richer options than `sh`: ```ts -await box.exec(["ls", "-la", "/etc"]); +await sbx.exec(["ls", "-la", "/etc"]); -await box.exec("node", { +await sbx.exec("node", { args: ["-e", "console.log(process.env.X)"], env: { X: "hi" }, cwd: "/app", @@ -144,7 +144,7 @@ await box.exec("node", { }); // Vercel-Sandbox-style alias: -const { stdout, exitCode } = await box.runCommand("uname", ["-a"]); +const { stdout, exitCode } = await sbx.runCommand("uname", ["-a"]); ``` `exec` returns a `CommandResult` with `.stdout`, `.stderr`, `.exitCode`, `.ok`, @@ -159,6 +159,34 @@ semantics and commonly merges stderr into stdout. > Linux guests get it injected automatically; on BSD you install it once — see > the [bsdkrun README](../../README.md#the-exec-agent). +## Caching + +`sandbox.cache` saves a guest directory under a key and restores it later, so a +rebuild can pick up where the last one left off. **A miss is not an error** — +check `restored` rather than catching. + +```ts +import { caches } from "@bsdkrun/sdk"; + +const key = `deps-${lockHash}`; +const hit = await sbx.cache.restore({ key, restoreKeys: ["deps-"] }); +if (!hit.restored) { + await sbx.exec(["npm", "ci"]); + await sbx.cache.save("/app/node_modules", { key, compression: "zstd" }); +} + +await caches.list(); // every stored entry, newest first +await caches.remove([key]); // or removeCache([], { all: true }) +``` + +`restoreKeys` are prefixes tried in order when the exact key misses; within a +prefix the newest matching entry wins, and `hit.key` says which one was used. +Formats are `gzip` (default), `zstd`, `estargz` and `none`. + +Where entries live is host configuration, not an SDK concern: the default is +this host's disk, and `BSDKRUN_CACHE_BACKEND=s3` + `BSDKRUN_CACHE_S3_*` (or +`~/.config/bsdkrun/cache.toml`) points them at a bucket instead. + ## Files `sandbox.fs` reads and writes files in the guest. Parent directories are created @@ -166,14 +194,14 @@ for you, and everything is byte-exact — `readFile` hands back a `Buffer`, so a PNG survives the round trip. ```ts -await box.fs.writeFile("/app/main.py", "print('hi')"); -await box.fs.writeFile("/app/logo.png", pngBytes); +await sbx.fs.writeFile("/app/main.py", "print('hi')"); +await sbx.fs.writeFile("/app/logo.png", pngBytes); -const text = await box.fs.readTextFile("/app/out.json"); -const bytes = await box.fs.readFile("/app/logo.png"); +const text = await sbx.fs.readTextFile("/app/out.json"); +const bytes = await sbx.fs.readFile("/app/logo.png"); -await box.fs.upload("./src", "/app/src"); // file or directory -await box.fs.download("/app/dist", "./dist", { recursive: true }); +await sbx.fs.upload("./src", "/app/src"); // file or directory +await sbx.fs.download("/app/dist", "./dist", { recursive: true }); ``` `upload` looks at the local path to decide whether to recurse; `download` cannot @@ -190,19 +218,19 @@ Failures throw `FileTransferError`, which carries the offending `path`. ## Lifecycle & inventory ```ts -const box = await Sandbox.create({ os: "linux", image: "alpine", command: ["sleep","300"] }); -const same = await Sandbox.get(box.id); // reconnect (prefix ok) +const sbx = await Sandbox.create({ os: "linux", image: "alpine", command: ["sleep","300"] }); +const same = await Sandbox.get(sbx.id); // reconnect (prefix ok) const list = await Sandbox.list({ all: true }); // SandboxInfo[] -await box.status(); // SandboxInfo | null -await box.isRunning(); // boolean -await box.logs(); // console log (string) -box.followLogs(); // live stream (child process) -box.shell(); // interactive shell (inherits the terminal) -await box.stop(); // BSD guests clean-poweroff; Linux SIGTERM -await box.start(); // restart in place — resumes its own disk/rootfs (data persists) -await box.update({ cpus: 4, mem: 2048 }); // applies on next start -await box.remove({ force: true }); +await sbx.status(); // SandboxInfo | null +await sbx.isRunning(); // boolean +await sbx.logs(); // console log (string) +sbx.followLogs(); // live stream (child process) +sbx.shell(); // interactive shell (inherits the terminal) +await sbx.stop(); // BSD guests clean-poweroff; Linux SIGTERM +await sbx.start(); // restart in place — resumes its own disk/rootfs (data persists) +await sbx.update({ cpus: 4, mem: 2048 }); // applies on next start +await sbx.remove({ force: true }); ``` `stop`/`start` **persist your data**: `start` resumes the machine's own @@ -224,13 +252,13 @@ await versions("netbsd"); ## Interactive terminal (xterm.js in the browser) -`box.terminal()` opens a PTY session in the guest, streamed over the agent's TCP +`sbx.terminal()` opens a PTY session in the guest, streamed over the agent's TCP protocol — with **live window-resize**. It's shaped to drop straight into [xterm.js](https://xtermjs.org): pipe output in, forward keystrokes out, resize on demand. ```ts -const term = await box.terminal({ command: ["/bin/sh"], cols: 120, rows: 30 }); +const term = await sbx.terminal({ command: ["/bin/sh"], cols: 120, rows: 30 }); term.onData((chunk) => xterm.write(chunk)); // guest → xterm xterm.onData((input) => term.write(input)); // xterm → guest @@ -243,7 +271,7 @@ Server-side, bridge it to a browser over a WebSocket in one call: ```ts wss.on("connection", async (ws) => { - const term = await box.terminal(); + const term = await sbx.terminal(); term.bindWebSocket(ws); // wires output, input, and {"resize":[c,r]} frames }); ``` @@ -258,18 +286,18 @@ complete Bun server + xterm.js page. await Sandbox.create({ os: "linux", image: "alpine", net: { ports: ["2222:22"] } }); // agent-managed key-based SSH (typed helpers) -await box.ssh.setup(); // install local ~/.ssh/*.pub keys -await box.ssh.setup({ user: "tsiry", key: "~/.ssh/work.pub" }); -await box.ssh.addKey("ssh-ed25519 AAAA..."); -await box.ssh.status(); +await sbx.ssh.setup(); // install local ~/.ssh/*.pub keys +await sbx.ssh.setup({ user: "tsiry", key: "~/.ssh/work.pub" }); +await sbx.ssh.addKey("ssh-ed25519 AAAA..."); +await sbx.ssh.status(); // put a guest on your tailnet -await box.tailscale.up({ authkey: "tskey-auth-...", hostname: "web" }); -await box.tailscale.status(); +await sbx.tailscale.up({ authkey: "tskey-auth-...", hostname: "web" }); +await sbx.tailscale.status(); // turn a Linux guest into a systemd system (debian/ubuntu/fedora only — // not Alpine, not the BSD guests) -await box.systemd.setup(); +await sbx.systemd.setup(); ``` ### Global networks — reach machines by name diff --git a/sdk/typescript/src/cache.ts b/sdk/typescript/src/cache.ts new file mode 100644 index 0000000..a06f360 --- /dev/null +++ b/sdk/typescript/src/cache.ts @@ -0,0 +1,174 @@ +import { CommandFailedError } from "./errors.js"; +import { runCli } from "./process.js"; +import { CommandResult } from "./shell.js"; + +/** An archive format a cache entry can be stored in. */ +export type Compression = "gzip" | "zstd" | "estargz" | "none"; + +/** A stored cache entry, as `cache ls` reports it. */ +export interface CacheEntry { + /** The exact key it was saved under. */ + key: string; + /** Guest path the tree came from. */ + path: string; + compression: Compression; + /** Archive size in bytes. */ + size: number; + /** Unix seconds when it was saved. */ + created: number; + /** `sha256:…` over the archive. */ + digest: string; +} + +/** What a restore did. */ +export interface RestoreResult { + /** Whether anything was found. A miss is not an error. */ + restored: boolean; + /** The key asked for. */ + requestedKey: string; + /** + * The entry actually used. Differs from {@link requestedKey} when a + * `restoreKeys` prefix matched, and is undefined on a miss. + */ + key?: string; + /** Guest path it was restored into. */ + path?: string; + size?: number; + compression?: Compression; + created?: number; +} + +export interface SaveOptions { + /** Key to store under. Make it name the content — a lockfile hash. */ + key: string; + /** Archive format. Defaults to gzip. */ + compression?: Compression; + /** Replace an entry that already has this key. */ + force?: boolean; +} + +export interface RestoreOptions { + key: string; + /** + * Where to restore to. Defaults to the directory the entry was saved from. + */ + path?: string; + /** + * Prefixes to fall back on when the key misses, most preferred first. Within + * a prefix the newest matching entry wins. + */ + restoreKeys?: string[]; +} + +function mapResult(row: Record): RestoreResult { + return { + restored: Boolean(row.restored), + requestedKey: String(row.requested_key), + key: row.key == null ? undefined : String(row.key), + path: row.path == null ? undefined : String(row.path), + size: row.size == null ? undefined : Number(row.size), + compression: row.compression as Compression | undefined, + created: row.created == null ? undefined : Number(row.created), + }; +} + +async function json(args: string[], label: string): Promise> { + const res = await runCli(args); + if (res.exitCode !== 0) { + throw new CommandFailedError( + new CommandResult(res.stdout, res.stderr, res.exitCode, label), + ); + } + return JSON.parse(res.stdout || "{}") as Record; +} + +/** + * Cached guest directories for one sandbox. + * + * Reached as `sandbox.cache`. Entries are keyed, so a rebuild can pick up where + * the last one left off: + * + * ```ts + * const hit = await box.cache.restore({ key, restoreKeys: ["deps-"] }); + * if (!hit.restored) { + * await box.exec(["npm", "ci"]); + * await box.cache.save("/app/node_modules", { key }); + * } + * ``` + * + * Where entries live — host disk or S3 — is host configuration, not an SDK + * concern: set `BSDKRUN_CACHE_BACKEND` / `BSDKRUN_CACHE_S3_*`, or + * `~/.config/bsdkrun/cache.toml`. + */ +export class Cache { + constructor(private readonly id: string) {} + + /** Archive a guest directory under a key. */ + async save(path: string, opts: SaveOptions): Promise { + const args = ["cache", "save", `${this.id}:${path}`, "--key", opts.key, "--json"]; + if (opts.compression) args.push("--compression", opts.compression); + if (opts.force) args.push("--force"); + const row = await json(args, "bsdkrun cache save"); + return { + key: String(row.key), + path: String(row.path), + compression: row.compression as Compression, + size: Number(row.size), + created: Number(row.created), + digest: String(row.digest ?? ""), + }; + } + + /** + * Restore a stored tree. **A miss is not an error** — check `restored` on the + * result rather than catching. + */ + async restore(opts: RestoreOptions): Promise { + const target = opts.path ? `${this.id}:${opts.path}` : this.id; + const args = ["cache", "restore", target, "--key", opts.key, "--json"]; + if (opts.restoreKeys?.length) args.push("--restore-keys", ...opts.restoreKeys); + return mapResult(await json(args, "bsdkrun cache restore")); + } +} + +/** List every stored cache entry, newest first. */ +export async function listCaches(): Promise { + const res = await runCli(["cache", "ls", "--json"]); + if (res.exitCode !== 0) { + throw new CommandFailedError( + new CommandResult(res.stdout, res.stderr, res.exitCode, "bsdkrun cache ls"), + ); + } + const rows = JSON.parse(res.stdout || "[]") as Record[]; + return rows.map((r) => ({ + key: String(r.key), + path: String(r.path), + compression: r.compression as Compression, + size: Number(r.size), + created: Number(r.created), + digest: String(r.digest ?? ""), + })); +} + +/** Remove stored entries by key, or every one of them with `{ all: true }`. */ +export async function removeCache( + keys: string | string[] = [], + opts: { all?: boolean } = {}, +): Promise { + const list = Array.isArray(keys) ? keys : [keys]; + const args = ["cache", "rm"]; + if (opts.all) args.push("--all"); + else args.push(...list); + const res = await runCli(args); + if (res.exitCode !== 0) { + throw new CommandFailedError( + new CommandResult(res.stdout, res.stderr, res.exitCode, "bsdkrun cache rm"), + ); + } +} + +/** Namespace grouping host-level cache operations. */ +export const caches = { + list: listCaches, + remove: removeCache, +}; diff --git a/sdk/typescript/src/index.ts b/sdk/typescript/src/index.ts index 33856bc..20e5d43 100644 --- a/sdk/typescript/src/index.ts +++ b/sdk/typescript/src/index.ts @@ -6,10 +6,10 @@ * ```ts * import { Sandbox } from "@bsdkrun/sdk"; * - * const box = await Sandbox.create({ os: "linux", image: "alpine" }); - * const out = await box.sh`uname -a`.text(); - * await box.exec(["apk", "add", "curl"]); - * await box.stop(); + * const sbx = await Sandbox.create({ os: "linux", image: "alpine" }); + * const out = await sbx.sh`uname -a`.text(); + * await sbx.exec(["apk", "add", "curl"]); + * await sbx.stop(); * ``` */ @@ -73,6 +73,15 @@ export type { export { FileSystem, FileTransferError } from "./filesystem.js"; export type { FsOptions, DownloadOptions } from "./filesystem.js"; +export { Cache, caches, listCaches, removeCache } from "./cache.js"; +export type { + CacheEntry, + Compression, + RestoreResult, + SaveOptions, + RestoreOptions, +} from "./cache.js"; + export { buildCreateArgs } from "./args.js"; export { diff --git a/sdk/typescript/src/sandbox.ts b/sdk/typescript/src/sandbox.ts index 5866f18..c3ae948 100644 --- a/sdk/typescript/src/sandbox.ts +++ b/sdk/typescript/src/sandbox.ts @@ -1,6 +1,7 @@ import { buildCreateArgs } from "./args.js"; import { readAgentPort } from "./agent-protocol.js"; import { CommandFailedError, SandboxNotFoundError } from "./errors.js"; +import { Cache } from "./cache.js"; import { FileSystem } from "./filesystem.js"; import { runCli, spawnCli } from "./process.js"; import { @@ -113,10 +114,10 @@ export function fromGraphQLMachine(m: Record): SandboxInfo { * {@link Sandbox.list}. * * ```ts - * const box = await Sandbox.create({ os: "linux", image: "alpine" }); - * await box.sh`echo hello`; - * await box.exec(["uname", "-a"]); - * await box.stop(); + * const sbx = await Sandbox.create({ os: "linux", image: "alpine" }); + * await sbx.sh`echo hello`; + * await sbx.exec(["uname", "-a"]); + * await sbx.stop(); * ``` */ export class Sandbox { @@ -131,6 +132,9 @@ export class Sandbox { /** Read and write files in the guest. */ readonly fs: FileSystem; + /** Save and restore guest directories under a key. */ + readonly cache: Cache; + #stateDirCache?: string; private constructor(id: string, sshPort?: number) { @@ -138,6 +142,7 @@ export class Sandbox { this.sshPort = sshPort; this.sh = createSh((script, opts) => this.#shRunner(script, opts)); this.fs = new FileSystem(id); + this.cache = new Cache(id); } /** Boot a new microVM and return a handle to it. */ @@ -207,8 +212,8 @@ export class Sandbox { * env, a PTY, stdin, or a working directory. * * ```ts - * await box.exec(["ls", "-la", "/etc"]); - * await box.exec("node", { args: ["-e", "console.log(1)"], env: { X: "1" } }); + * await sbx.exec(["ls", "-la", "/etc"]); + * await sbx.exec("node", { args: ["-e", "console.log(1)"], env: { X: "1" } }); * ``` */ async exec( @@ -252,7 +257,7 @@ export class Sandbox { * Vercel-Sandbox-style alias for {@link exec}: a program plus its args. * * ```ts - * const { stdout } = await box.runCommand("uname", ["-a"]); + * const { stdout } = await sbx.runCommand("uname", ["-a"]); * ``` */ runCommand( @@ -386,10 +391,10 @@ export class Sandbox { * arbitrary args. * * ```ts - * await box.ssh.setup(); // install local ~/.ssh/*.pub - * await box.ssh.setup({ user: "tsiry", key: "~/.ssh/work.pub" }); - * await box.ssh.addKey("ssh-ed25519 AAAA..."); - * await box.ssh.status(); + * await sbx.ssh.setup(); // install local ~/.ssh/*.pub + * await sbx.ssh.setup({ user: "tsiry", key: "~/.ssh/work.pub" }); + * await sbx.ssh.addKey("ssh-ed25519 AAAA..."); + * await sbx.ssh.status(); * ``` */ get ssh() { @@ -420,8 +425,8 @@ export class Sandbox { * `tailscaled` (userspace networking by default), and joins your tailnet. * * ```ts - * await box.tailscale.up({ authkey: "tskey-auth-...", hostname: "web" }); - * await box.tailscale.status(); + * await sbx.tailscale.up({ authkey: "tskey-auth-...", hostname: "web" }); + * await sbx.tailscale.status(); * ``` */ get tailscale() { @@ -458,8 +463,8 @@ export class Sandbox { * `setup` errors clearly there. Manage BSD services with `rc.d` instead. * * ```ts - * await box.systemd.setup(); // install + mark for next boot (debian/ubuntu/fedora) - * await box.systemd.status(); + * await sbx.systemd.setup(); // install + mark for next boot (debian/ubuntu/fedora) + * await sbx.systemd.status(); * ``` */ get systemd() { @@ -490,7 +495,7 @@ export class Sandbox { * `onResize` to {@link Terminal.resize}. See `examples/08-browser-terminal`. * * ```ts - * const term = await box.terminal({ cols: 120, rows: 30 }); + * const term = await sbx.terminal({ cols: 120, rows: 30 }); * term.onData((chunk) => process.stdout.write(chunk)); * term.write("uname -a\n"); * ``` diff --git a/skills/bsdkrun-cli/SKILL.md b/skills/bsdkrun-cli/SKILL.md index 3f97c27..6744dda 100644 --- a/skills/bsdkrun-cli/SKILL.md +++ b/skills/bsdkrun-cli/SKILL.md @@ -39,6 +39,10 @@ Lifecycle: Interact: - `bsdkrun exec [-t] [-e K=V]... ...` — run a command in a guest (via its agent). - `bsdkrun cp [-r] ` — copy files host<->guest, `docker cp`-style (`ID:PATH`, `-` = stdio). +- `bsdkrun cache save : --key K [-c gzip|zstd|estargz|none]` — archive a guest dir. +- `bsdkrun cache restore [:] --key K [--restore-keys PREFIX...]` — put it back (a miss exits 0). +- `bsdkrun cache ls` / `bsdkrun cache rm ... | --all` — list and remove entries. +- `bsdkrun doctor [--json]` — check the host can run machines; exits 1 on any failure. - `bsdkrun shell ` — attach an interactive console to a detached machine. - `bsdkrun logs [-f] [--boot] ` — show the console log (`--boot` = bsdkrun's own boot log). diff --git a/skills/bsdkrun-cli/references/cli-reference.md b/skills/bsdkrun-cli/references/cli-reference.md index 2fc4f7e..3419ac7 100644 --- a/skills/bsdkrun-cli/references/cli-reference.md +++ b/skills/bsdkrun-cli/references/cli-reference.md @@ -101,6 +101,42 @@ The transfer rides the guest's exec agent (`cat`, plus `tar` for `-r`), so the m running and its image needs a shell — which every image that boots under bsdkrun already has. An image without `tar` can still copy files one at a time; `-r` reports that specifically. +### `cache save --key [-c FORMAT] [--force]` +Archive a guest directory and store it under a key. `--compression` is `gzip` (default), `zstd`, +`estargz` or `none`. Saving over an existing key needs `--force`. + +### `cache restore --key [--restore-keys PREFIX...]` +Restore a stored tree. Without a path it goes back where it was saved from. `--restore-keys` are +prefixes tried in order when the exact key misses; within a prefix the newest entry wins. **A miss +is not an error** — it prints `cache miss` and exits 0, so a first run needs no `|| true`. + +### `cache ls [--json]` / `cache rm ... | --all` +List and remove entries. + +```sh +bsdkrun cache save web:/root/.cargo --key cargo-$(shasum Cargo.lock | cut -c1-12) +bsdkrun cache restore web --key cargo-abc123 --restore-keys cargo- +bsdkrun cache ls +bsdkrun cache rm --all +``` + +Entries go to the host disk (`/caches`) by default. For a shared store, set +`BSDKRUN_CACHE_BACKEND=s3` and `BSDKRUN_CACHE_S3_BUCKET`, or write `~/.config/bsdkrun/cache.toml`: + +```toml +backend = "s3" + +[s3] +bucket = "my-ci-cache" +region = "us-east-1" +prefix = "bsdkrun" # optional +endpoint = "https://.r2.cloudflarestorage.com" # optional: R2, MinIO, … +``` + +Credentials come from `AWS_ACCESS_KEY_ID` / `AWS_SECRET_ACCESS_KEY` (+ `AWS_SESSION_TOKEN`) only, +never from the file, so the config stays safe to commit. Saving needs `tar` in the guest; +compression happens on the host, so the image needs no compressor of its own. + ### `shell ` Attach an interactive console to a running (detached) machine. @@ -218,6 +254,14 @@ Check that libkrun links and a context/HVF can be initialized (connectivity/heal --- +### `doctor [--json]` +Check that this host can run machines and print what to fix. Covers the host tools bsdkrun shells +out to (`curl`, `tar`), the hypervisor, the macOS code signature and its hypervisor entitlement, +gvproxy, the state/cache directories, the case-sensitive store, and the cache backend. Exits 1 if +anything failed, so CI can gate on it. + +--- + ## Environment variables - `BSDKRUN_CACHE` — override the cache dir (images/kernels/agent/flavor builds). diff --git a/tests/e2e_cache.sh b/tests/e2e_cache.sh new file mode 100755 index 0000000..c821d30 --- /dev/null +++ b/tests/e2e_cache.sh @@ -0,0 +1,198 @@ +#!/usr/bin/env bash +# +# End-to-end test for `bsdkrun cache`: save a guest directory under a key and +# restore it into a different path, byte for byte, in every archive format — +# against both backends. +# +# The disk backend needs nothing. The S3 backend runs against a real MinIO, +# because that is the half that unit tests cannot reach: SigV4 either signs a +# request a server accepts or it does not, and the failure is a flat 403 with +# no hint as to which of the five derivation steps drifted. It also pins the +# behaviour that broke on first contact with a real bucket — a 404 from the +# "is this key already cached?" probe is an *answer*, not an error. +# +# MinIO is optional: without S3_ENDPOINT the S3 half is skipped and the disk +# half still runs, so the script is useful on a laptop. +# +# Exit 0 on success, 1 on failure, 2 on missing prerequisites. Overridable via +# environment: +# BSDKRUN_BIN (default target/debug/bsdkrun — build + sign with `make build`) +# IMAGE (default "alpine") +# TIMEOUT (seconds per boot, default 180) +# S3_ENDPOINT (e.g. http://127.0.0.1:19000; unset skips the S3 half) +# S3_BUCKET (default bsdkrun-cache) +set -uo pipefail + +BIN="${BSDKRUN_BIN:-target/debug/bsdkrun}" +IMAGE="${IMAGE:-alpine}" +TIMEOUT="${TIMEOUT:-180}" +NAME="e2e-cache-$$" +S3_BUCKET="${S3_BUCKET:-bsdkrun-cache}" + +if [ ! -x "$BIN" ]; then + echo "e2e: missing bsdkrun binary: $BIN (run 'make build' first)" >&2 + exit 2 +fi + +ID="" +# Every key this run creates, so a failure part-way still cleans the store. +KEYS=() + +cleanup() { + for key in "${KEYS[@]:-}"; do + [ -n "$key" ] && "$BIN" cache rm "$key" >/dev/null 2>&1 + done + [ -n "$ID" ] && "$BIN" rm -f "$ID" >/dev/null 2>&1 +} +trap cleanup EXIT INT TERM + +fail() { + echo "e2e: FAIL — $*" >&2 + [ -n "$ID" ] && "$BIN" logs "$ID" 2>/dev/null | tail -20 >&2 + exit 1 +} + +wait_for_agent() { + local deadline=$(( SECONDS + TIMEOUT )) + while [ "$SECONDS" -lt "$deadline" ]; do + if "$BIN" exec "$ID" /bin/true >/dev/null 2>&1; then + return 0 + fi + sleep 2 + done + fail "guest agent not reachable within ${TIMEOUT}s" +} + +# The manifest of a directory in the guest: every file's path and sha256. This +# is what "restored correctly" means — a tree that merely has the right *names* +# has silently lost content, and a tree with an extra file (estargz's TOC and +# landmark are real tar members) has silently gained one. +manifest() { + "$BIN" exec "$ID" sh -c "cd $1 && find . -type f | sort | xargs sha256sum" 2>/dev/null +} + +echo "e2e: booting $IMAGE" +ID="$("$BIN" linux -d --name "$NAME" "$IMAGE" 2>/dev/null /work/deps/a.txt + echo lib-b > /work/deps/nested/deep/b.txt + head -c 262144 /dev/urandom > /work/deps/blob.bin +' >/dev/null 2>&1 || fail "could not create the source tree in the guest" + +ORIGINAL="$(manifest /work/deps)" +[ -n "$ORIGINAL" ] || fail "the source tree came back empty" + +# --------------------------------------------------------------------------- +# One backend, every format. +# --------------------------------------------------------------------------- +run_backend() { + local label="$1" + echo + echo "e2e: === $label backend ===" + + local fmt key dest restored + for fmt in gzip zstd estargz none; do + key="e2e-$$-$fmt" + dest="/restored-$fmt" + KEYS+=("$key") + + "$BIN" cache save "$ID:/work/deps" --key "$key" --compression "$fmt" >/dev/null 2>&1 \ + || fail "$label: cache save failed for $fmt" + + "$BIN" exec "$ID" rm -rf "$dest" >/dev/null 2>&1 + "$BIN" cache restore "$ID:$dest" --key "$key" >/dev/null 2>&1 \ + || fail "$label: cache restore failed for $fmt" + + restored="$(manifest "$dest")" + if [ "$restored" != "$ORIGINAL" ]; then + echo "--- expected ---" >&2; echo "$ORIGINAL" >&2 + echo "--- restored ---" >&2; echo "$restored" >&2 + fail "$label: $fmt did not round-trip" + fi + echo "e2e: $label/$fmt round-tripped" + done + + # `ls` has to show what we just saved. + "$BIN" cache ls | grep -q "e2e-$$-gzip" \ + || fail "$label: cache ls does not list the entry it just saved" + + # A miss is an ordinary answer, not a failure — a first CI run depends on it. + if ! "$BIN" cache restore "$ID:/miss" --key "definitely-absent-$$" >/dev/null 2>&1; then + fail "$label: a cache miss exited non-zero" + fi + + # Restore-keys: an exact miss falls back to a prefix. + "$BIN" exec "$ID" rm -rf /fallback >/dev/null 2>&1 + "$BIN" cache restore "$ID:/fallback" --key "e2e-$$-nope" --restore-keys "e2e-$$-" >/dev/null 2>&1 \ + || fail "$label: restore-keys fallback failed" + [ -n "$(manifest /fallback)" ] || fail "$label: restore-keys fallback restored nothing" + echo "e2e: $label restore-keys fallback works" + + # Saving over an existing key needs --force. This is the path that issues the + # "does this key exist?" probe, whose 404 must not read as an error. + if "$BIN" cache save "$ID:/work/deps" --key "e2e-$$-gzip" >/dev/null 2>&1; then + fail "$label: saving over an existing key succeeded without --force" + fi + "$BIN" cache save "$ID:/work/deps" --key "e2e-$$-gzip" --force >/dev/null 2>&1 \ + || fail "$label: --force did not replace the entry" + echo "e2e: $label duplicate-key handling works" + + # Removal really removes — and a removed key must also stop resolving, which + # is what proves the archive went with the metadata rather than being orphaned + # in the store where no listing would ever show it again. + "$BIN" cache rm "e2e-$$-gzip" >/dev/null 2>&1 || fail "$label: cache rm failed" + if "$BIN" cache ls | grep -q "e2e-$$-gzip"; then + fail "$label: cache rm left the entry listed" + fi + "$BIN" exec "$ID" rm -rf /gone >/dev/null 2>&1 + "$BIN" cache restore "$ID:/gone" --key "e2e-$$-gzip" >/dev/null 2>&1 \ + || fail "$label: restoring a removed key errored instead of missing" + if [ -n "$(manifest /gone)" ]; then + fail "$label: a removed key still restored content" + fi + echo "e2e: $label removal works" + + # Clear the rest of this run's keys, so the next backend starts clean and the + # store is left as we found it. + for fmt in zstd estargz none; do + "$BIN" cache rm "e2e-$$-$fmt" >/dev/null 2>&1 + done + if "$BIN" cache ls | grep -q "e2e-$$-"; then + fail "$label: entries from this run survived removal" + fi + echo "e2e: $label store is clean" +} + +# --------------------------------------------------------------------------- +# Disk backend (the default). +# --------------------------------------------------------------------------- +unset BSDKRUN_CACHE_BACKEND +run_backend "disk" + +# --------------------------------------------------------------------------- +# S3 backend, against MinIO when one is reachable. +# --------------------------------------------------------------------------- +if [ -z "${S3_ENDPOINT:-}" ]; then + echo + echo "e2e: S3_ENDPOINT unset — skipping the S3 half" +else + export BSDKRUN_CACHE_BACKEND=s3 + export BSDKRUN_CACHE_S3_ENDPOINT="$S3_ENDPOINT" + export BSDKRUN_CACHE_S3_BUCKET="$S3_BUCKET" + export BSDKRUN_CACHE_S3_REGION="${AWS_REGION:-us-east-1}" + export BSDKRUN_CACHE_S3_PREFIX="e2e" + : "${AWS_ACCESS_KEY_ID:?e2e: S3_ENDPOINT is set but AWS_ACCESS_KEY_ID is not}" + : "${AWS_SECRET_ACCESS_KEY:?e2e: S3_ENDPOINT is set but AWS_SECRET_ACCESS_KEY is not}" + + run_backend "s3" +fi + +echo +echo "e2e: PASS" diff --git a/tests/estargz_interop.sh b/tests/estargz_interop.sh new file mode 100755 index 0000000..fc3c8ea --- /dev/null +++ b/tests/estargz_interop.sh @@ -0,0 +1,170 @@ +#!/usr/bin/env bash +# +# Verify that `--compression estargz` produces an archive containerd can read. +# +# Our own tests can only prove the writer agrees with itself. eStargz exists to +# be consumed by stargz-snapshotter, so the test that matters opens the archive +# with *that* library: parse the 51-byte footer, follow it to the TOC, then seek +# to each entry's recorded offset and check the bytes there are the file's — not +# its tar header, which is the mistake the format invites. +# +# Needs Go and network access (it fetches the estargz module). Exit 0 on +# success, 1 on failure, 2 on missing prerequisites. Overridable via: +# BSDKRUN_BIN (default target/debug/bsdkrun) +# IMAGE (default "alpine") +# TIMEOUT (seconds per boot, default 180) +set -uo pipefail + +BIN="${BSDKRUN_BIN:-target/debug/bsdkrun}" +IMAGE="${IMAGE:-alpine}" +TIMEOUT="${TIMEOUT:-180}" +NAME="e2e-estargz-$$" +KEY="e2e-estargz-$$" + +if [ ! -x "$BIN" ]; then + echo "interop: missing bsdkrun binary: $BIN (run 'make build' first)" >&2 + exit 2 +fi +command -v go >/dev/null 2>&1 || { echo "interop: go is not installed" >&2; exit 2; } + +ID="" +WORK="$(mktemp -d -t bsdkrun-estargz.XXXXXX)" + +cleanup() { + "$BIN" cache rm "$KEY" >/dev/null 2>&1 + [ -n "$ID" ] && "$BIN" rm -f "$ID" >/dev/null 2>&1 + rm -rf "$WORK" +} +trap cleanup EXIT INT TERM + +fail() { echo "interop: FAIL — $*" >&2; exit 1; } + +wait_for_agent() { + local deadline=$(( SECONDS + TIMEOUT )) + while [ "$SECONDS" -lt "$deadline" ]; do + "$BIN" exec "$ID" /bin/true >/dev/null 2>&1 && return 0 + sleep 2 + done + fail "guest agent not reachable within ${TIMEOUT}s" +} + +# The disk backend, so the archive is a file we can hand to Go. An S3 store +# would work too but would only add a download to the middle of the test. +unset BSDKRUN_CACHE_BACKEND + +echo "interop: booting $IMAGE" +ID="$("$BIN" linux -d --name "$NAME" "$IMAGE" 2>/dev/null /work/tree/main.txt + printf "deep\n" > /work/tree/nested/deep/notes.txt + head -c 4096 /dev/urandom > /work/tree/data.bin +' >/dev/null 2>&1 || fail "could not create the source tree in the guest" + +"$BIN" cache save "$ID:/work/tree" --key "$KEY" --compression estargz >/dev/null 2>&1 \ + || fail "cache save --compression estargz failed" + +# Find the archive the save just wrote. `cache ls --json` names the key; the +# file is the only .tar.estargz in the store whose name derives from it. +ARCHIVE="$(find "${HOME}/.cache/bsdkrun/caches" -name '*.tar.estargz' -newermt '-5 minutes' 2>/dev/null | head -n 1)" +[ -n "$ARCHIVE" ] || fail "could not find the saved estargz archive" +echo "interop: checking $ARCHIVE" + +cat > "$WORK/main.go" <<'GO' +package main + +import ( + "fmt" + "io" + "os" + + "github.com/containerd/stargz-snapshotter/estargz" +) + +// The three files the shell half wrote, with the sizes it wrote them at. +var want = map[string]string{ + "main.txt": "hello from the cache\n", + "nested/deep/notes.txt": "deep\n", +} + +func main() { + f, err := os.Open(os.Args[1]) + if err != nil { + fmt.Println("FAIL open:", err) + os.Exit(1) + } + defer f.Close() + fi, err := f.Stat() + if err != nil { + fmt.Println("FAIL stat:", err) + os.Exit(1) + } + + // Parses the footer at len-51, follows it to the TOC, and validates it. + r, err := estargz.Open(io.NewSectionReader(f, 0, fi.Size())) + if err != nil { + fmt.Println("FAIL estargz.Open:", err) + os.Exit(1) + } + fmt.Println("ok estargz.Open — footer and TOC parsed") + + for name, body := range want { + e, ok := r.Lookup(name) + if !ok { + fmt.Println("FAIL Lookup:", name) + os.Exit(1) + } + sr, err := r.OpenFile(name) + if err != nil { + fmt.Println("FAIL OpenFile:", name, err) + os.Exit(1) + } + buf := make([]byte, e.Size) + if _, err := sr.ReadAt(buf, 0); err != nil && err != io.EOF { + fmt.Println("FAIL ReadAt:", name, err) + os.Exit(1) + } + if string(buf) != body { + fmt.Printf("FAIL %s: read %q at offset %d, want %q\n", name, buf, e.Offset, body) + os.Exit(1) + } + fmt.Printf("ok %-24s %d bytes at offset %d\n", name, e.Size, e.Offset) + } + + // A binary file too: the seek arithmetic is what breaks, and it breaks the + // same way for text — just less visibly. + if e, ok := r.Lookup("data.bin"); !ok || e.Size != 4096 { + fmt.Println("FAIL data.bin missing or the wrong size in the TOC") + os.Exit(1) + } + fmt.Println("ok data.bin present at 4096 bytes") + + // The landmark the spec requires; without it a verifying reader rejects + // the archive outright. + if _, ok := r.Lookup(".no.prefetch.landmark"); !ok { + fmt.Println("FAIL the required landmark entry is missing") + os.Exit(1) + } + fmt.Println("ok .no.prefetch.landmark present") +} +GO + +cat > "$WORK/go.mod" <<'GO' +module estargzinterop + +go 1.23 +GO + +( cd "$WORK" && go mod tidy >/dev/null 2>&1 && go run . "$ARCHIVE" ) || fail "containerd's reader rejected the archive" + +# ...and it must still be an ordinary tar.gz to anything that has never heard of +# eStargz. That property is the reason the format is safe to make the default +# one day, and it is easy to lose. +tar -tzf "$ARCHIVE" >/dev/null 2>&1 || fail "the archive is not readable as a plain tar.gz" +echo "interop: plain 'tar -tzf' reads it too" + +echo "interop: PASS"