#!/usr/bin/env bash # collection interning storage A/B (hydrant-3kc.2 baseline + hydrant-3kc.6 arm). # # runs one arm per binary over the SAME uniform repo sample into a FRESH database, # so the comparison needs no migration: only the interning read/write paths. # # both arms are driven from this one script, sequentially, never overlapping — # the host is shared with production co-tenants and a single API port is reused. # # usage: # ./collection_interning_ab.sh = [= ...] # # example: # ./collection_interning_ab.sh /root/bench/ci-20260801 /root/bench/sample-30k.ndjson \ # baseline=/root/bench/bin/hydrant-baseline interned=/root/bench/bin/hydrant-interned set -euo pipefail # NixOS bench hosts have neither jq nor python3 in PATH; jq is all this needs if ! command -v jq >/dev/null 2>&1; then if command -v nix-shell >/dev/null 2>&1; then exec nix-shell -p jq --run "$(printf '%q ' "$0" "$@")" fi echo "FATAL: jq is required" >&2 exit 1 fi RUN_DIR=${1:?run dir} SAMPLE=${2:?sample ndjson} shift 2 [ $# -ge 1 ] || { echo "need at least one =" >&2; exit 1; } API_PORT=${API_PORT:-19310} DEBUG_PORT=${DEBUG_PORT:-19311} INTERVAL=${INTERVAL:-5} # /stats sizes counts only FLUSHED segments, so prod-sized memtables leave a # short run reporting near-zero. small memtables push everything to L0 instead. # identical in both arms, so the ratio stays valid even though the absolute # numbers are not prod-tuned. MEMTABLE_MB=${MEMTABLE_MB:-8} # consecutive samples with an empty queue AND a static record count before an # arm counts as drained SETTLE_SAMPLES=${SETTLE_SAMPLES:-12} DRAIN_TIMEOUT=${DRAIN_TIMEOUT:-10800} # plc.klbr.net is dawn's unthrottled mirror; falls back if unreachable PLC_URL=${PLC_URL:-https://plc.klbr.net} mkdir -p "$RUN_DIR" echo "run dir: $RUN_DIR" echo "sample: $SAMPLE ($(wc -l < "$SAMPLE") repos)" if ! curl -s --max-time 10 "$PLC_URL/did:plc:yk4q3id7id6p5z3bypvshc64" >/dev/null; then echo "WARNING: $PLC_URL unreachable, falling back to plc.directory" >&2 PLC_URL=https://plc.directory fi echo "plc: $PLC_URL" # a stray hydrant on our port would silently join the measurement if curl -s --max-time 3 "http://127.0.0.1:$API_PORT/stats" >/dev/null 2>&1; then echo "FATAL: something is already serving :$API_PORT" >&2 exit 1 fi run_arm() { local arm=$1 bin=$2 local dir="$RUN_DIR/$arm" mkdir -p "$dir" local db="$dir/hydrant.db" rm -rf "$db" echo echo "=== arm $arm ===" # NB: do NOT probe the binary with --version. hydrant does not implement it, # and unknown argv silently boots a full instance with default config, which # connects to live relays and hangs the script (hydrant-j2e). md5sum "$bin" | awk '{print " binary md5 " $1}' # filtered mode over an explicit repo list: no crawler, no firehose. # HYDRANT_RELAY_HOSTS and HYDRANT_CRAWLER_URLS must be set EMPTY rather than # unset -- unset falls back to built-in defaults and quietly adds drift. env \ HYDRANT_DATABASE_PATH="$db" \ HYDRANT_API_BIND="127.0.0.1:$API_PORT" \ HYDRANT_ENABLE_DEBUG=true \ HYDRANT_DEBUG_PORT="$DEBUG_PORT" \ HYDRANT_ENABLE_FIREHOSE=false \ HYDRANT_ENABLE_CRAWLER=false \ HYDRANT_RELAY_HOSTS="" \ HYDRANT_CRAWLER_URLS="" \ HYDRANT_FULL_NETWORK=false \ HYDRANT_BACKFILL_STRATEGY=auto \ HYDRANT_PLC_URL="$PLC_URL" \ HYDRANT_DATA_COMPRESSION=zstd \ HYDRANT_JOURNAL_COMPRESSION=zstd \ HYDRANT_DB_RECORDS_MEMTABLE_SIZE_MB="$MEMTABLE_MB" \ HYDRANT_DB_BLOCKS_MEMTABLE_SIZE_MB="$MEMTABLE_MB" \ HYDRANT_DB_EVENTS_MEMTABLE_SIZE_MB="$MEMTABLE_MB" \ HYDRANT_DB_REPOS_MEMTABLE_SIZE_MB="$MEMTABLE_MB" \ "$bin" > "$dir/hydrant.log" 2>&1 & local pid=$! echo "$pid" > "$dir/pid" echo "pid $pid" # shellcheck disable=SC2064 trap "kill $pid 2>/dev/null || true" EXIT for _ in $(seq 60); do if curl -s --max-time 2 "http://127.0.0.1:$API_PORT/stats" >/dev/null 2>&1; then break; fi if ! kill -0 "$pid" 2>/dev/null; then echo "FATAL: $arm died on startup, see $dir/hydrant.log" >&2 tail -20 "$dir/hydrant.log" >&2 exit 1 fi sleep 1 done date +%s > "$dir/start_epoch" # telemetry sampler: rss, db size, and journal size over time ( # a failing probe must not kill the sampler: under pipefail a single # unmatched glob or transient curl would otherwise end telemetry after # the header set +e +o pipefail echo "epoch,rss_kb,vmhwm_kb,db_bytes,journal_bytes,records,pending,collections,collections_bytes" while kill -0 "$pid" 2>/dev/null; do local_rss=$(awk '/^VmRSS:/{print $2}' "/proc/$pid/status" 2>/dev/null || echo 0) local_hwm=$(awk '/^VmHWM:/{print $2}' "/proc/$pid/status" 2>/dev/null || echo 0) dbb=$(du -sb "$db" 2>/dev/null | cut -f1 || echo 0) jrn=$(du -scb "$db"/*.jnl 2>/dev/null | awk '/total/{print $1+0}') st=$(curl -s --max-time 4 "http://127.0.0.1:$API_PORT/stats" 2>/dev/null || echo '{}') rec=$(echo "$st" | jq -r '.counts.records // 0' 2>/dev/null || echo 0) pend=$(echo "$st" | jq -r '.counts.pending // 0' 2>/dev/null || echo 0) col=$(echo "$st" | jq -r '.counts.collections // 0' 2>/dev/null || echo 0) colb=$(echo "$st" | jq -r '.counts.collections_bytes // 0' 2>/dev/null || echo 0) echo "$(date +%s),$local_rss,$local_hwm,$dbb,$jrn,$rec,$pend,$col,$colb" sleep "$INTERVAL" done ) > "$dir/telemetry.csv" & local sampler=$! echo "feeding $(wc -l < "$SAMPLE") repos..." curl -s -X PUT "http://127.0.0.1:$API_PORT/repos" \ -H 'content-type: application/x-ndjson' \ --data-binary "@$SAMPLE" -o "$dir/put-repos.json" \ -w 'put /repos -> %{http_code}\n' # drain. pending is only the QUEUE depth: it reads zero while in-flight # backfills are still writing records, so waiting on it alone stops the arm # mid-flight and the arms end up having indexed different repo sets. require # the record count to have stopped growing as well. echo "draining..." local settled=0 elapsed=0 last_records=-1 while [ "$settled" -lt "$SETTLE_SAMPLES" ]; do sleep "$INTERVAL" elapsed=$((elapsed + INTERVAL)) local st pend recs st=$(curl -s --max-time 10 "http://127.0.0.1:$API_PORT/stats" 2>/dev/null || echo '{}') pend=$(echo "$st" | jq -r '.counts.pending // 1' 2>/dev/null || echo 1) recs=$(echo "$st" | jq -r '.counts.records // -1' 2>/dev/null || echo -1) if [ "$pend" = "0" ] && [ "$recs" = "$last_records" ]; then settled=$((settled + 1)) else settled=0 fi last_records=$recs if [ $((elapsed % 60)) -lt "$INTERVAL" ]; then echo " ${elapsed}s pending=$pend records=$recs settled=$settled" fi if [ "$elapsed" -gt "$DRAIN_TIMEOUT" ]; then echo " giving up after ${DRAIN_TIMEOUT}s with pending=$pend records=$recs" >&2 echo "INCOMPLETE" > "$dir/INCOMPLETE" break fi done echo " drained at records=$last_records after ${elapsed}s" date +%s > "$dir/end_epoch" curl -s --max-time 60 "http://127.0.0.1:$API_PORT/stats" > "$dir/stats-predrain.json" # a clean stop and reopen flushes whatever is still in a memtable, so the # sizes below cover everything written rather than only what happened to # have been flushed. compact after the reopen, since sizes before compaction # are inflated. echo "restarting to flush memtables..." kill "$sampler" 2>/dev/null || true kill "$pid" 2>/dev/null || true wait "$pid" 2>/dev/null || true env \ HYDRANT_DATABASE_PATH="$db" \ HYDRANT_API_BIND="127.0.0.1:$API_PORT" \ HYDRANT_ENABLE_FIREHOSE=false \ HYDRANT_ENABLE_CRAWLER=false \ HYDRANT_RELAY_HOSTS="" \ HYDRANT_CRAWLER_URLS="" \ HYDRANT_ENABLE_BACKFILL=false \ HYDRANT_FULL_NETWORK=false \ HYDRANT_PLC_URL="$PLC_URL" \ HYDRANT_DATA_COMPRESSION=zstd \ HYDRANT_JOURNAL_COMPRESSION=zstd \ "$bin" > "$dir/hydrant-reopen.log" 2>&1 & pid=$! trap "kill $pid 2>/dev/null || true" EXIT for _ in $(seq 120); do curl -s --max-time 2 "http://127.0.0.1:$API_PORT/stats" >/dev/null 2>&1 && break sleep 1 done echo "compacting..." curl -s -X POST --max-time 3600 "http://127.0.0.1:$API_PORT/db/compact" -o "$dir/compact.json" \ -w 'compact -> %{http_code}\n' curl -s --max-time 60 "http://127.0.0.1:$API_PORT/stats" > "$dir/stats-final.json" # per-collection distribution, for reading the bytes/record numbers curl -s --max-time 300 "http://127.0.0.1:$DEBUG_PORT/debug/iter?partition=counts&limit=400000" \ -o "$dir/counts-iter.json" || echo "(counts iter failed, non-fatal)" kill "$pid" 2>/dev/null || true wait "$pid" 2>/dev/null || true trap - EXIT # per-keyspace apparent bytes as a cross-check on /stats sizes du -sb "$db" > "$dir/db-du-bytes.txt" 2>/dev/null || true du -sb "$db"/keyspaces/* > "$dir/keyspace-du-bytes.txt" 2>/dev/null || true echo "arm $arm done" } for spec in "$@"; do arm=${spec%%=*} bin=${spec#*=} [ -x "$bin" ] || { echo "FATAL: $bin is not executable" >&2; exit 1; } run_arm "$arm" "$bin" done echo echo "=== summary ===" for spec in "$@"; do arm=${spec%%=*} f="$RUN_DIR/$arm/stats-final.json" if [ ! -f "$f" ]; then echo "$arm: no stats-final.json"; continue; fi peak=0 if [ -f "$RUN_DIR/$arm/telemetry.csv" ]; then peak=$(awk -F, 'NR>1 && $3+0>m {m=$3+0} END{printf "%.1f", m/1024}' "$RUN_DIR/$arm/telemetry.csv") fi jq -r --arg arm "$arm" --arg peak "$peak" ' (.counts.records // 0) as $r | (.sizes | to_entries | map(.value) | add // 0) as $total | "[\($arm)]", " records \($r)", " collections \(.counts.collections // 0)", " collections_bytes \(.counts.collections_bytes // 0)", " records_bytes \(.sizes.records // 0)", " counts_bytes \(.sizes.counts // 0)", " blocks_bytes \(.sizes.blocks // 0)", " total_bytes \($total)", " records_b_per_rec \(if $r > 0 then ((.sizes.records // 0) / $r * 100 | round / 100) else 0 end)", " counts_b_per_rec \(if $r > 0 then ((.sizes.counts // 0) / $r * 100 | round / 100) else 0 end)", " total_b_per_rec \(if $r > 0 then ($total / $r * 100 | round / 100) else 0 end)", " peak_rss_mb \($peak)", "" ' "$f" # machine-readable row for the comparison below jq -r --arg arm "$arm" --arg peak "$peak" ' (.counts.records // 0) as $r | (.sizes | to_entries | map(.value) | add // 0) as $total | [$arm, $r, (.sizes.records // 0), (.sizes.counts // 0), $total, $peak] | @tsv ' "$f" >> "$RUN_DIR/summary.tsv" done if [ "$(wc -l < "$RUN_DIR/summary.tsv" 2>/dev/null || echo 0)" -eq 2 ]; then echo "=== arm 2 vs arm 1 ===" awk -F'\t' ' NR==1 { a=$1; ar=$2; arec=$3; acnt=$4; atot=$5; arss=$6 } NR==2 { b=$1; br=$2; brec=$3; bcnt=$4; btot=$5; brss=$6 } END { if (ar>0 && br>0) { arb=arec/ar; brb=brec/br acb=acnt/ar; bcb=bcnt/br atb=atot/ar; btb=btot/br printf " records_b_per_rec %.2f -> %.2f (%+.2f%%)\n", arb, brb, (brb-arb)/arb*100 printf " counts_b_per_rec %.2f -> %.2f (%+.2f%%)\n", acb, bcb, (acb>0?(bcb-acb)/acb*100:0) printf " total_b_per_rec %.2f -> %.2f (%+.2f%%)\n", atb, btb, (btb-atb)/atb*100 if (arss>0) printf " peak_rss_mb %.1f -> %.1f (%+.2f%%)\n", arss, brss, (brss-arss)/arss*100 if (ar != br) { d = (br-ar)/ar*100; if (d<0) d=-d printf " !! RECORD COUNTS DIFFER: %d vs %d (%.1f%%)\n", ar, br, d printf " !! the arms did not index the same set, so this is NOT a\n" printf " !! valid paired comparison -- rerun with a longer drain\n" } } } ' "$RUN_DIR/summary.tsv" fi