Something went wrong. Try again.
This repository has no description
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445#!/usr/bin/env bash# Launch a detached training run on the box.## scripts/launch_run.sh <run_name> [extra hydra overrides...]## Defaults to the TRM paper's Sudoku-Extreme recipe (MLP variant) with cheap# subsampled in-training evals; the authoritative full-test-set number comes from# scripts/eval_checkpoint.py afterwards.## NOTE: never `pkill -f dsem.pretrain` from an ssh one-liner — the pattern matches# the ssh command line itself and kills your own session. Use scripts/stop_run.sh.set -euo pipefail
cd "$(dirname "$0")/.."
RUN_NAME="${1:?usage: launch_run.sh <run_name> [overrides...]}"shift || true
mkdir -p runsLOG="runs/${RUN_NAME}.out"
if [ -f "runs/${RUN_NAME}.pid" ] && kill -0 "$(cat "runs/${RUN_NAME}.pid")" 2>/dev/null; then echo "run ${RUN_NAME} already alive (pid $(cat "runs/${RUN_NAME}.pid"))" >&2 exit 1fi
# PYTHONUNBUFFERED: otherwise "Processing batch" prints sit in a block buffer when# stdout is redirected, which once looked exactly like a hung eval. It wasn't.PYTHONUNBUFFERED=1 nohup .venv/bin/python -m dsem.pretrain \ arch=trm \ data_paths="[data/sudoku-extreme-1k-aug-1000]" \ data_paths_test="[data/sudoku-testsub-12k]" \ evaluators="[]" \ epochs=50000 eval_interval=5000 \ global_batch_size=768 \ lr=1e-4 puzzle_emb_lr=1e-4 weight_decay=1.0 puzzle_emb_weight_decay=1.0 \ arch.mlp_t=True arch.pos_encodings=none \ arch.L_layers=2 arch.H_cycles=3 arch.L_cycles=6 \ ema=True \ "+run_name=${RUN_NAME}" \ "$@" > "${LOG}" 2>&1 < /dev/null &
echo $! > "runs/${RUN_NAME}.pid"echo "launched ${RUN_NAME} pid $(cat "runs/${RUN_NAME}.pid") -> ${LOG}"