diff --git a/scripts/matches/artifacts.sh b/scripts/matches/artifacts.sh new file mode 100755 index 0000000..1c81fc6 --- /dev/null +++ b/scripts/matches/artifacts.sh @@ -0,0 +1,120 @@ +#!/usr/bin/env bash +# Everything one match left in S3: +# +# ./scripts/matches/artifacts.sh [--paths] +# +# The id is what ./scripts/matches/list.sh prints in its match column. A +# leading fragment of one is enough - these are S3 prefixes, so any prefix of +# the id works, and if it matches more than one match this says which ids it +# found rather than merging them. +# +# Full s3:// URIs, so a line can be pasted straight into `aws s3 cp`. --paths +# prints only those, one per line, for piping: +# +# ./scripts/matches/artifacts.sh 5fe10a12 --paths | xargs -n1 -I{} aws s3 cp {} . +# +# The prefixes say what each file is, and the layout is arena's: +# +# artifacts/launch/ what the match was given, written before it starts +# artifacts/raw/ verbatim MegaMek, byte-for-byte as it wrote it +# artifacts/derived/ what arena made of it, including socials/ +# artifacts/diagnostics/ operational, nobody's evidence +# +# Two things sit at the match root rather than under artifacts/, because +# headquarters writes them rather than arena: manifest.json, which is how the +# container is told what to play, and the staged camo. See +# headquarters' docs/match-artifacts.md. +# +# Everything here expires: the bucket ages matches/ out, so an empty listing +# for a match that certainly ran means it is older than that. +set -euo pipefail + +source "$(dirname "$0")/../lib/aws.sh" +require_aws_credentials +ENV_DIR="$(dirname "$0")/../../envs/lance.blue" + +MATCH="${1:-}" +MODE="${2:-}" + +if [ -z "$MATCH" ]; then + echo "usage: $(basename "$0") [--paths]" >&2 + echo "match ids: ./scripts/matches/list.sh" >&2 + exit 2 +fi + +case "$MATCH" in + */*) echo "that is a path, not a match id" >&2; exit 2 ;; +esac + +BUCKET="$(tofu -chdir="$ENV_DIR" output -raw artifacts_bucket)" +PREFIX="matches/$MATCH" + +objects="$(aws s3api list-objects-v2 --bucket "$BUCKET" --prefix "$PREFIX" \ + --query 'Contents[].{Key:Key,Size:Size,Modified:LastModified}' --output json)" + +if [ "$objects" = "null" ] || [ -z "$objects" ]; then + echo "nothing under s3://$BUCKET/$PREFIX" >&2 + echo "either no match has that id, or its artifacts have aged out" >&2 + exit 1 +fi + +# A prefix that spans several matches is ambiguous rather than empty, and +# merging them would print one match's board beside another's result. +ids="$(jq -r '[.[].Key | capture("^matches/(?[^/]+)/").id] | unique | .[]' <<<"$objects")" +if [ "$(wc -l <<<"$ids")" -gt 1 ]; then + echo "'$MATCH' matches more than one match:" >&2 + sed 's/^/ /' <<<"$ids" >&2 + exit 2 +fi + +if [ "$MODE" = "--paths" ]; then + jq -r --arg b "$BUCKET" '.[] | "s3://\($b)/\(.Key)"' <<<"$objects" + exit 0 +fi + +# Ordered the way a match produces them - what it was given, what MegaMek +# wrote, what arena made of it, then the operational leftovers - rather than +# alphabetically, which would put diagnostics second. +jq -r --arg b "$BUCKET" ' + def kind: + if test("/artifacts/launch/") then "launch" + elif test("/artifacts/raw/") then "raw" + elif test("/artifacts/derived/socials/") then "socials" + elif test("/artifacts/derived/") then "derived" + elif test("/artifacts/diagnostics/") then "diagnostics" + elif test("/artifacts/") then "other" + else "root" end; + + def rank: + {"launch": 1, "raw": 2, "derived": 3, "socials": 4, + "diagnostics": 5, "other": 6, "root": 7}[.] // 9; + + def human: + if . >= 1048576 then "\((. / 104857.6 | floor) / 10) MiB" + elif . >= 1024 then "\((. / 102.4 | floor) / 10) KiB" + else "\(.) B" end; + + (["class", "size", "modified", "path"] | @tsv), + (map(. + {class: (.Key | kind)}) + | sort_by([(.class | rank), .Key])[] + | [.class, + (.Size | human), + (.Modified | tostring | .[0:19]), + "s3://\($b)/\(.Key)"] + | @tsv)' <<<"$objects" \ +| column -t -s $'\t' + +echo +total="$(jq '[.[].Size] | add' <<<"$objects")" +count="$(jq 'length' <<<"$objects")" +human="$(numfmt --to=iec --suffix=B "$total" 2>/dev/null || echo "$total bytes")" +echo "$count object(s), $human" + +# The layout arena wrote before launch/ raw/ derived/. A match that predates +# it reads perfectly well - headquarters looks under both - but it is worth +# saying which one you are looking at, because the names mean different +# things: shots/share-card.png is a board MegaMek rendered, and +# derived/socials/share-card.png is the frame arena chose out of thirty-one. +if jq -e 'any(.[].Key; test("/(turns|shots)/|/result\\.json$"))' <<<"$objects" >/dev/null; then + echo "note: this match used the layout that predates launch/ raw/ derived/" +fi diff --git a/scripts/matches/list.sh b/scripts/matches/list.sh index 0872c2c..7545e7c 100755 --- a/scripts/matches/list.sh +++ b/scripts/matches/list.sh @@ -4,6 +4,17 @@ # # The usage numbers need containerInsights enabled on the cluster; without # data it says so and the task list still prints. +# +# The match column is the id everything else is keyed by - the artifacts in +# S3, the row in the api's database, the page a player shares - and it is +# what ./scripts/matches/artifacts.sh takes. A task id only means something +# to ECS, and stops meaning anything once the task is reaped. +# +# It comes off the task itself: headquarters passes the manifest as a +# presigned URL in ARENA_MANIFEST_URL, and that URL's path is +# matches//manifest.json. Only the id is printed. The URL is a bearer +# credential for as long as the signature lives, so it must not reach a +# terminal or a log. set -euo pipefail source "$(dirname "$0")/../lib/aws.sh" @@ -23,9 +34,19 @@ else # shellcheck disable=SC2086 # task arns are space-separated on purpose aws ecs describe-tasks --cluster "$CLUSTER" --tasks $tasks --output json \ | jq -r ' - (["task", "status", "cpu", "mem", "exit", "started", "stopped-reason"] | @tsv), + def match_id: + [.overrides.containerOverrides[]? + | select(.name == "arena") + | .environment[]? + | select(.name == "ARENA_MANIFEST_URL") + | .value + | capture("matches/(?[0-9a-fA-F-]{36})/").id] + | first // "-"; + + (["match", "task", "status", "cpu", "mem", "exit", "started", "stopped-reason"] | @tsv), (.tasks[] - | [(.taskArn | split("/") | last | .[0:8]), + | [match_id, + (.taskArn | split("/") | last | .[0:8]), .lastStatus, .cpu, (.memory + "MB"),