#!/usr/bin/env bash # Drive a local match without holding a terminal. # # ./scripts/arena.sh up # start, wait until it serves, return # ./scripts/arena.sh up 2-Ambush.mms # a different scenario # ./scripts/arena.sh status # JSON, for something that branches on it # ./scripts/arena.sh logs -f # ./scripts/arena.sh down # always safe, even with nothing running # # scripts/run.sh is the same match for a person: it prints a banner and holds # the terminal until Ctrl-C. This is the same match for a caller that wants to # bring it up, measure something and tear it down. They share the manifest, the # caps and the singleton guard - see scripts/lib/common.sh. # # Every start is capped at the production shape, 2 vCPU and 8GB. The caps are # not optional. `arena.sh caps` prints what a start would apply. # # Exit codes, because the caller is usually not a person: # 0 did what was asked # 1 bad usage, or something is missing # 2 a match is already running (up) # 3 it did not become ready in time (up, wait-ready) # 4 nothing is running (logs, wait-ready) set -euo pipefail ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" # shellcheck source=lib/common.sh . "$ROOT/scripts/lib/common.sh" # Named per session, not "arena-dev" for everyone: several sessions work on # this machine at once and a shared name means the second one collides with the # first, or worse, tears it down. NAME="${CONTAINER_NAME:-arena-$ARENA_SESSION}" PORT_WS="${PORT_WS:-}" READY_TIMEOUT="${ARENA_READY_TIMEOUT:-180}" # The manifest is bind-mounted, so it has to outlive this process now that the # container is detached. run.sh can use mktemp and a trap; this cannot. # `down` removes the directory. # Per session, like the container name. A shared path is a shared bind mount: # when another session's `down` removed the manifest, docker recreated it as a # directory for the next mount and every later start failed with "Is a # directory". STATE_DIR="${ARENA_STATE_DIR:-${TMPDIR:-/tmp}/lance-blue-arena-$ARENA_SESSION}" usage() { cat <<'USAGE' usage: arena.sh [options] up [SCENARIO] start a match and wait until it serves -i, --image SELECTOR tag, short sha, or whatever scripts/image.sh takes -s, --scenario PATH scenario, relative to the scenario library -H, --human NAME faction you play (default: the first one) --spectate nobody plays; Princess takes every faction and the browser only watches -p, --port N host port (default 8080) -t, --timeout SEC how long to wait for it to serve (default 180) --live-scripts mount ./container over the baked-in scripts --latency SPEC shape the browser-facing link, e.g. 28ms. A 28ms/8ms form adds jitter, which reorders packets and is usually not what you want - see below. Applied once the match is serving, so it slows play and not startup. down [--all] stop and remove this session's match; safe when idle --all takes every arena container, including other sessions' - ask them first status one JSON object: running, image, caps, port, uptime logs [-f] [-n N] the running match's logs wait-ready [-t S] block until the match serves caps print the caps a start would apply, and exit environment: ARENA_PROFILE set to 1 to profile both JVMs; down collects the stacks ARENA_PROFILE_INTERVAL sampling interval (default 10ms) ARENA_PROFILE_TIMEOUT seconds the client profile covers, from when the client JVM starts (default 180) ARENA_LAZY_PNG set to 1 to load the lazy-png javaagent on the client JVM (default off) ARENA_ARTIFACTS_DIR where down puts them (default .arena-artifacts) ARENA_STOP_TIMEOUT seconds down waits for the exit scripts (default 60) ARENA_LATENCY same as --latency, e.g. 28ms/8ms ARENA_CPUS CPUs, 1 to 8 (default 2) ARENA_MEMORY memory, 2g to 12g (default 8g) ARENA_PIDS pid limit, 128 to 4096 (default 512) ARENA_SESSION who owns the container (default: the checkout's name) ARENA_MAX_MATCHES most matches on this machine at once (default 2) ARENA_AUTO_READY seconds before the host readies the human seats itself (default 0, meaning a person clicks Done). Set it when something other than a person is driving the match - a profile or a latency run needs a seat the server answers, not a lounge. ARENA_AUTO_READY seconds before the host readies the human seats itself (default 0, meaning a person clicks Done). Set it when something other than a person is driving the match - a profile or a latency run needs a seat the server answers, not a lounge. PORT_WS, CONTAINER_NAME, ARENA_READY_TIMEOUT, ARENA_KEEP_ON_FAIL The caps cannot be turned off. Out-of-range values are clamped and the clamp is printed. USAGE } # --- helpers ---------------------------------------------------------------- no_options() { local what="$1"; shift; [ $# -eq 0 ] || die "$what takes no options"; } die_not_running() { printf 'ERROR: no arena container is running\n' >&2 hint "./scripts/arena.sh up starts one" exit 4 } # Shape the browser-facing link with netem, so a local match behaves like one # played across a network instead of over loopback. # # This exists because measuring on loopback has repeatedly produced conclusions # that did not survive production: the paint tick, the image cache and the frame # ack were all judged against a 0ms round trip, and two of the three inverted # when a real one showed up. 28ms is the median ping measured from Boston to # us-east-1; 28ms/8ms adds jitter. # # Applied from a throwaway container that joins this one's network namespace, so # the match image needs no tc and no extra capability of its own. The qdisc # lives in the namespace and outlives the sidecar; `down` removes it with the # container. # # Delay is egress-only, which is the right shape for round-trip measurement: a # frame going out is delayed and the input coming back is not, so one request # and its reply cost one delay - the same total a symmetric half-RTT each way # would give. # # Jitter is off by default and should stay off unless you know why you want it. # netem's jitter delays each packet independently, so packets overtake one # another, and TCP reads reordering as loss: measured here, `delay 28ms 8ms` # turned a steady 59ms request into 44-77ms and took the match's p90 latency to # twelve seconds, against 117ms in production. Flat delay gives 59-60ms with # almost no variance, which is the honest simulation of a fixed-RTT path. Real # jitter needs a rate-limited qdisc underneath to preserve ordering; that is not # built here. NETEM_IMAGE="${NETEM_IMAGE:-arena-netem:local}" arena_netem_image() { docker image inspect "$NETEM_IMAGE" >/dev/null 2>&1 && return 0 say "building $NETEM_IMAGE (once)" printf 'FROM debian:bookworm-slim\nRUN apt-get update && apt-get install -y --no-install-recommends iproute2 && rm -rf /var/lib/apt/lists/*\n' \ | docker build -q -t "$NETEM_IMAGE" - >/dev/null } # arena_netem_apply spec: "28ms" or "28ms/8ms" arena_netem_apply() { local name="$1" spec="$2" delay jitter delay="${spec%%/*}" jitter="" case "$spec" in *?/?*) jitter="${spec##*/}" ;; esac arena_netem_image || { warn "could not build $NETEM_IMAGE; link left unshaped"; return 1; } docker run --rm --net="container:$name" --cap-add=NET_ADMIN "$NETEM_IMAGE" \ tc qdisc replace dev eth0 root netem delay "$delay" ${jitter:+"$jitter" distribution normal} \ >/dev/null 2>&1 \ || { warn "netem could not be applied; link left unshaped"; return 1; } say "latency: $delay${jitter:+ +/- $jitter} on the browser-facing link" } # The one running arena container, or nothing. running_id() { local rows id rows="$(arena_containers running)" [ -n "$rows" ] || return 0 IFS='|' read -r id _ <<<"$rows" printf '%s\n' "$id" } # Port the running container publishes 8080 on, falling back to the default so # a probe still has something to try. # # One published port is two bindings on a dual-stack host - docker maps 0.0.0.0 # and [::] separately - so this range runs twice. Without a separator it # returned the port twice concatenated: 8080 and 8080 became 80808080, which # `status` reported as the port and `wait-ready` then probed, timing out # against a container that was serving perfectly well. `println` gives each # binding its own line and the first one is taken. container_port() { local id="$1" p p="$(docker inspect -f '{{range $p := index .NetworkSettings.Ports "8080/tcp"}}{{println $p.HostPort}}{{end}}' \ "$id" 2>/dev/null | head -1 || true)" printf '%s\n' "${p:-$PORT_WS}" } # wait_ready # 0 serving # 3 still not serving when the timeout ran out # 5 the container stopped # # The stopped case is separate and immediate: a container that exited is never # going to answer, and sitting out the rest of the timeout only delays the logs # that explain why. wait_ready() { local id="$1" port="$2" timeout="$3" deadline deadline=$(( $(date +%s) + timeout )) while :; do if [ "$(docker inspect -f '{{.State.Running}}' "$id" 2>/dev/null || echo false)" != true ]; then return 5 fi if arena_probe "$port"; then return 0 fi if [ "$(date +%s)" -ge "$deadline" ]; then return 3 fi sleep 1 done } # What an automated caller gets instead of a silent hang. report_failure() { local id="$1" why="$2" printf 'ERROR: %s\n' "$why" >&2 printf '\n--- last 50 log lines ---\n' >&2 docker logs --tail 50 "$id" >&2 || true printf -- '--- end ---\n\n' >&2 } # --- up --------------------------------------------------------------------- cmd_up() { local scenario="" image_sel="" human="${HUMAN:-}" live="${LIVE_SCRIPTS:-0}" while [ $# -gt 0 ]; do case "$1" in -i|--image) image_sel="${2:?--image needs a selector}"; shift 2 ;; -s|--scenario) scenario="${2:?--scenario needs a path}"; shift 2 ;; -H|--human) human="${2:?--human needs a faction}"; shift 2 ;; --spectate) human=-; shift ;; -p|--port) PORT_WS="${2:?--port needs a number}"; shift 2 ;; -t|--timeout) READY_TIMEOUT="${2:?--timeout needs seconds}"; shift 2 ;; --live-scripts) live=1; shift ;; --latency) ARENA_LATENCY="${2:?--latency needs a spec like 28ms/8ms}"; shift 2 ;; -h|--help) usage; exit 0 ;; -*) die "up: unknown option '$1'" ;; *) [ -z "$scenario" ] || die "up: more than one scenario" scenario="$1"; shift ;; esac done scenario="${scenario:-TrainingScenarios/1-FirstRun.mms}" require_cmd jq curl docker arena_resolve_caps # A fixed 8080 is another way two sessions collide. An explicit --port or # PORT_WS is honoured as given; otherwise take the first free one, so two # matches can run side by side. if [ -z "$PORT_WS" ]; then PORT_WS="$(arena_free_port 8080)" || die "no free port in 8080-8099" fi local image image="$(arena_resolve_image "$image_sel")" # Both halves of the guard. The name collision is the one run.sh has always # caught; the running check is the one that matters, because two or three # matches at once under different names is what takes the laptop down. arena_require_capacity "./scripts/arena.sh down stops it" || exit 2 if docker ps -a --format '{{.Names}}' | grep -qx "$NAME"; then printf "ERROR: a container named '%s' already exists\n" "$NAME" >&2 hint "./scripts/arena.sh down removes it" exit 2 fi mkdir -p "$STATE_DIR" local manifest="$STATE_DIR/manifest.json" arena_write_manifest "$image" "$scenario" "$human" "$manifest" local mounts=( -v "$manifest:/run/dev-manifest.json:ro" ) # container/ is COPYed into the image, so editing an init, watch or exit # script normally means a rebuild before the change has any effect. # --live-scripts mounts the working copy over them, turning a rebuild into a # restart. Off by default: a plain start should test what was actually built. if [ "$live" = 1 ]; then mounts+=( -v "$ROOT/container:/opt/arena/container:ro" ) fi say "image: $image" say "scenario: $scenario ($ARENA_HUMAN is human, Princess plays the rest)" arena_caps_line say "url: http://localhost:$PORT_WS/ (one client app per seat, at /)" # Instrumentation is off unless one of these is set in this shell. Passed with # `-e NAME` rather than `-e NAME=value` so an unset variable stays unset # inside the container instead of arriving as an empty string - entrypoint.sh # treats "set but empty" and "unset" the same, but 40-render-config.sh's # defaults do not survive an explicit empty value. # # This list is the whole interface. A name the container reads and this loop # does not carry is a switch that silently does nothing from here, which is # how ARENA_LAZY_PNG and ARENA_PROFILE_TIMEOUT both spent their first days: # set on the command line, absent in the container, no error either place. # tests/shell/test-arena.sh checks each of them arrives. local profile_env=() v for v in ARENA_PROFILE ARENA_PROFILE_INTERVAL \ ARENA_PROFILE_CLIENT_EVENT ARENA_PROFILE_HOST_EVENT \ ARENA_PROFILE_TIMEOUT ARENA_LAZY_PNG ARENA_HOST_JIT_ARGS \ ARENA_DRAW_DELAY_MS ARENA_CLIENT_VMARGS; do [ -n "${!v:-}" ] && profile_env+=( -e "$v" ) done if [ -n "${ARENA_PROFILE:-}" ] && [ "$ARENA_PROFILE" != 0 ]; then say "profile: on; ./scripts/arena.sh down collects the stacks" fi # Detached and without --rm on purpose. --rm would take the logs with it the # moment a start failed, which is exactly when they are worth having. `down` # is what removes the container. # # --init so signals reach the entrypoint and zombie JVMs get reaped. local id id="$(docker run -d \ --name "$NAME" \ --init \ "${ARENA_CAP_ARGS[@]}" \ --label "$ARENA_LOCAL_LABEL=1" \ --label "$ARENA_SESSION_LABEL=$ARENA_SESSION" \ -p "$PORT_WS:8080" \ "${mounts[@]}" \ -e ARENA_MANIFEST_FILE=/run/dev-manifest.json \ "${profile_env[@]}" \ "$image")" local started rc=0 started="$(date +%s)" wait_ready "$id" "$PORT_WS" "$READY_TIMEOUT" || rc=$? if [ "$rc" -eq 0 ]; then say "ready: after $(( $(date +%s) - started ))s" # After readiness on purpose: shaping the link during startup would slow the # readiness probe and confuse the phase timings, and the thing worth # simulating is play, not boot. [ -n "${ARENA_LATENCY:-}" ] && arena_netem_apply "$NAME" "$ARENA_LATENCY" return 0 fi if [ "$rc" -eq 5 ]; then report_failure "$id" "the container exited before it served anything" else report_failure "$id" "not serving after ${READY_TIMEOUT}s" fi # A container left behind after a failed start blocks the next `up` and is # the orphan this is meant not to create; its logs are already printed above. # Keep it only when someone asks to poke at it. if [ "${ARENA_KEEP_ON_FAIL:-0}" = 1 ]; then hint "ARENA_KEEP_ON_FAIL=1: $NAME left in place; ./scripts/arena.sh down removes it" else cmd_down >/dev/null fi exit 3 } # --- down ------------------------------------------------------------------- # Removes every arena container, running or stopped, not just $NAME: a leftover # from a differently named run is exactly what blocks the next start. # # SIGTERM first, then remove. It used to be a straight `rm -f`, which SIGKILLs - # and the entrypoint's whole exit path hangs off SIGTERM: on a profiled run it # stops the JVMs so async-profiler flushes, then runs finalize to collect the # stacks and the stats table. A killed container leaves all of that unwritten, # so the first profiled match produced nothing until it was stopped by hand. # # `docker stop` SIGKILLs anyway once its timeout runs out, so a wedged container # is still removed - it just costs the timeout first. -v drops the anonymous # volumes so a repeated cycle does not fill the disk. cmd_down() { local scope=mine while [ $# -gt 0 ]; do case "$1" in --all) scope=any; shift ;; -h|--help) usage; exit 0 ;; *) die "down: unknown option '$1'" ;; esac done local rows removed=0 id name state stop_timeout="${ARENA_STOP_TIMEOUT:-60}" rows="$(arena_containers all "$scope")" if [ -n "$rows" ]; then while IFS='|' read -r id name _ state _; do [ -n "$id" ] || continue # The listing already carries the state, so this costs no extra call. # Stopping something that has already exited is not an error, just a # pointless round trip and a misleading line of output. if [ "$state" = running ]; then say "stopping $name (up to ${stop_timeout}s for the exit scripts)" docker stop -t "$stop_timeout" "$id" >/dev/null 2>&1 || true fi collect_artifacts "$id" "$name" if docker rm -f -v "$id" >/dev/null 2>&1; then say "removed $name" else warn "could not remove $name" fi removed=$((removed + 1)) done <<<"$rows" fi rm -rf "$STATE_DIR" [ "$removed" -gt 0 ] || say "nothing of this session's is running" # Say what was left standing, whether or not we removed anything of our own. # The old down removed these silently, which is how it tore down other # sessions' matches mid-run. if [ "$scope" != any ]; then local others others="$(arena_containers running any)" if [ -n "$others" ]; then while IFS='|' read -r _ line image _ owner; do [ -n "$line" ] && hint "$line ($image) belongs to ${owner:-an older run}" done <<<"$others" hint "./scripts/arena.sh down --all stops those too" fi fi return 0 } # collect_artifacts # # Everything the exit scripts wrote lives on the container's own filesystem and # dies with it. finalize.sh uploads what it can, but a local match has no upload # target - it says "no upload target for 'diagnostics'; keeping ... locally" and # keeps the file inside a container that is about to be removed. So the profiles # and the stats table have to be copied out here or they are lost. # # Quiet when there is nothing worth keeping, which is every unprofiled match: # host.status and identity.json are not artefacts. collect_artifacts() { local id="$1" name="$2" dest tmp dest="${ARENA_ARTIFACTS_DIR:-$ROOT/.arena-artifacts}/$name" tmp="$(mktemp -d)" if ! docker cp "$id:/run/arena/state/." "$tmp/" >/dev/null 2>&1; then rm -rf "$tmp" return 0 fi # Only the things worth keeping. A bare `docker cp` of the state dir would # also drag along host.status and identity.json every time. # *.log covers whatever a run asked the JVMs to write - ARENA_CLIENT_VMARGS # with -Xlog:class+load writes one, and it is gone with the container if it # is not copied out here. local kept=0 f for f in "$tmp"/profile-*.collapsed "$tmp"/stats-summary.txt "$tmp"/result.json \ "$tmp"/*.log; do [ -s "$f" ] || continue mkdir -p "$dest" mv "$f" "$dest/" kept=$((kept + 1)) done rm -rf "$tmp" [ "$kept" -gt 0 ] && say "kept $kept artefact(s) in $dest" return 0 } # --- status ----------------------------------------------------------------- cmd_status() { local id port id="$(running_id)" if [ -z "$id" ]; then jq -n '{running: false, ready: false, container: null, image: null, caps: null, port: null, url: null, uptime_seconds: null}' return 0 fi port="$(container_port "$id")" # The caps reported are the ones docker is enforcing, read back off the # container, rather than the ones this script meant to apply. local fields fields="$(docker inspect -f '{{.Name}} {{.Config.Image}} {{.State.Status}} {{.State.StartedAt}} {{.HostConfig.NanoCpus}} {{.HostConfig.Memory}} {{.HostConfig.PidsLimit}}' "$id")" local name image state started nanocpus memory pids { read -r name; read -r image; read -r state; read -r started read -r nanocpus; read -r memory; read -r pids; } <<<"$fields" name="${name#/}" local uptime=0 start_epoch start_epoch="$(date -d "$started" +%s 2>/dev/null || echo 0)" if [ "$start_epoch" -gt 0 ]; then uptime=$(( $(date +%s) - start_epoch )) fi local ready=false if arena_probe "$port"; then ready=true fi jq -n \ --arg id "$id" --arg name "$name" --arg image "$image" --arg state "$state" \ --arg started "$started" --arg url "http://localhost:$port/" \ --argjson ready "$ready" --argjson port "$port" --argjson uptime "$uptime" \ --argjson nanocpus "${nanocpus:-0}" --argjson memory "${memory:-0}" \ --argjson pids "${pids:-0}" \ '{ running: ($state == "running"), ready: $ready, container: { id: $id, name: $name, state: $state, started_at: $started }, image: $image, caps: { cpus: ($nanocpus / 1000000000), memory_bytes: $memory, memory_mb: ($memory / 1048576), pids_limit: $pids }, port: $port, url: $url, uptime_seconds: $uptime }' } # --- logs, wait-ready, caps ------------------------------------------------- cmd_logs() { local id id="$(running_id)" [ -n "$id" ] || die_not_running docker logs "$@" "$id" } cmd_wait_ready() { local timeout="$READY_TIMEOUT" while [ $# -gt 0 ]; do case "$1" in -t|--timeout) timeout="${2:?--timeout needs seconds}"; shift 2 ;; *) die "wait-ready: unknown option '$1'" ;; esac done require_cmd curl local id rc=0 id="$(running_id)" [ -n "$id" ] || die_not_running wait_ready "$id" "$(container_port "$id")" "$timeout" || rc=$? if [ "$rc" -eq 0 ]; then say "ready" return 0 fi if [ "$rc" -eq 5 ]; then report_failure "$id" "the container exited before it served anything" else report_failure "$id" "not serving after ${timeout}s" fi exit 3 } cmd_caps() { no_options caps "$@" arena_resolve_caps arena_caps_line } # --- dispatch --------------------------------------------------------------- [ $# -gt 0 ] || { usage >&2; exit 1; } cmd="$1"; shift case "$cmd" in up) cmd_up "$@" ;; down) cmd_down "$@" ;; status) no_options status "$@"; cmd_status ;; logs) cmd_logs "$@" ;; wait-ready) cmd_wait_ready "$@" ;; caps) cmd_caps "$@" ;; -h|--help|help) usage ;; *) printf "ERROR: unknown command '%s'\n" "$cmd" >&2; usage >&2; exit 1 ;; esac