#!/usr/bin/env bash # Provisions a vast.ai instance and runs the full pipeline: generate -> verify -> # train -> push to the Hub. Written to be re-runnable: a reclaimed instance can be # replaced and this script started again from scratch. # # Expects, in the environment: # HF_TOKEN write token, for checkpoint pushes # HUB_REPO e.g. yourname/tiny-cube-solver # Optional: PAIRS (default 10000000), STEPS, HIDDEN, LAYERS set -euo pipefail PAIRS="${PAIRS:-10000000}" STEPS="${STEPS:-40000}" HIDDEN="${HIDDEN:-384}" LAYERS="${LAYERS:-8}" CORES="$(nproc)" echo "=== $CORES cores, $(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null || echo 'no GPU') ===" # Build against SYSTEM python, not the image's conda python. kociemba ships a C # extension and a pure-Python fallback, and it falls back *silently* when the # extension cannot build -- which is what happens if the venv's interpreter has no # matching dev headers. python3-dev is therefore not optional here. apt-get update -qq # libffi-dev is the one that actually matters and is the least obvious: kociemba's # native extension is built through cffi, which needs ffi.h. Without it the # compile dies with "fatal error: ffi.h: No such file or directory", kociemba # installs its pure-Python fallback instead, and pip still exits 0. apt-get install -y -qq build-essential python3-dev python3-venv libffi-dev git # Prefer python3.11. kociemba 1.2.1 builds its C extension cleanly on 3.11 (the # version this repo is developed against) but silently produces no extension on # the 3.10 that ships as system python in the Ubuntu 22.04 based images -- pip # still reports success, so the only symptom is the fallback. The PyTorch images # carry a conda 3.11, so prefer that and fall back to system python. PYBIN="" for cand in /opt/conda/bin/python3.11 /opt/conda/bin/python "$(command -v python3.11 || true)" /usr/bin/python3; do [ -x "$cand" ] || continue if "$cand" -c 'import sys; sys.exit(0 if sys.version_info[:2] >= (3,11) else 1)' 2>/dev/null; then PYBIN="$cand"; break fi done [ -z "$PYBIN" ] && PYBIN=/usr/bin/python3 echo "building venv from $PYBIN ($($PYBIN -V 2>&1))" "$PYBIN" -m venv /opt/venv /opt/venv/bin/pip install -q --upgrade pip # kociemba is only needed to LABEL data. The eval and value-iteration modes need # neither: value iteration bootstraps its own targets, and the search evaluator # only ever applies moves. Building it anyway costs ~20 minutes of metered time, # because a failed C-extension build falls back to an upstream-git rebuild -- # observed on a run that then never imported the module at all. NEEDS_SOLVER=1 if [ -n "${EVAL_VALUE_ONLY:-}${TRAIN_VALUE_ITER:-}${TRAIN_POLICY_DISTILL:-}" ]; then NEEDS_SOLVER=""; fi # kociemba's setup.py predates modern setuptools' removal of install_layout. /opt/venv/bin/pip install -q "setuptools<60" wheel # Full output, not a tail: when the C extension fails to build, kociemba still # installs successfully (as pure Python), so the compiler error is the only # evidence of what went wrong and it must not be truncated away. if [ -n "$NEEDS_SOLVER" ]; then /opt/venv/bin/pip install --no-build-isolation kociemba 2>&1 | tail -40 # Last resort before giving up: build from the upstream git source, whose build # metadata is newer than the 2017 PyPI sdist. if ! /opt/venv/bin/python -c "import kociemba.ckociembawrapper" 2>/dev/null; then echo "=== sdist build produced no extension; trying upstream git ===" /opt/venv/bin/pip install --force-reinstall --no-build-isolation \ "git+https://github.com/muodov/kociemba.git" 2>&1 | tail -15 || true fi if ! /opt/venv/bin/python -c "import kociemba.ckociembawrapper" 2>/dev/null; then echo "=== NATIVE BUILD FAILED -- diagnostics ===" echo "python: $(/opt/venv/bin/python -V 2>&1) at $(/opt/venv/bin/python -c 'import sys;print(sys.executable)')" echo "gcc: $(gcc --version 2>&1 | head -1)" echo "includes: $(/opt/venv/bin/python -c 'import sysconfig;print(sysconfig.get_paths()["include"])')" ls -l "$(/opt/venv/bin/python -c 'import sysconfig;print(sysconfig.get_paths()["include"])')/Python.h" 2>&1 || echo " Python.h MISSING" dpkg -l | grep -E "python3.*dev|build-essential" || echo " no dev packages found" echo "=== retrying verbosely ===" /opt/venv/bin/pip install --force-reinstall --no-build-isolation -v kociemba 2>&1 | grep -iE "error|gcc|Python.h|fatal" | head -30 fi fi /opt/venv/bin/pip install -q torch --index-url https://download.pytorch.org/whl/cu128 /opt/venv/bin/pip install -q transformers huggingface_hub # Search evaluation only: skip data generation and training entirely, download a # trained value model and score the state search on the canonical scrambles. The # search needs a GPU -- each solve evaluates beam*18 children per step -- so it # cannot be done from the session container. if [ -n "${EVAL_VALUE_ONLY:-}" ]; then cd "${CODE_DIR:-/workspace/code}" echo "=== evaluating value search from ${EVAL_VALUE_ONLY} ===" /opt/venv/bin/python eval_value_search.py --value "$EVAL_VALUE_ONLY" \ --value-file "${VALUE_FILE:-value.pt}" \ --beam "${SEARCH_BEAM:-64}" --limit-per-depth "${SEARCH_N:-25}" \ ${SEARCH_DEPTHS:+--depths "$SEARCH_DEPTHS"} ${SEARCH_LAM:+--lam "$SEARCH_LAM"} echo "=== done ===" exit 0 fi # Policy distillation from a trained value model. Needs no dataset and no solver: # the targets are the teacher's scores for the 18 children of each visited state, # and the auxiliary facelet targets come from applying moves. So, like value # iteration, this goes straight from a GPU check to training. # # DISTILL_ARMS runs several arms back to back on one box, which is the point of # the experiment rather than a convenience -- the arms differ only in the loss, so # comparing them across two rentals would add a hardware variable to a difference # of a few points: # full soft targets + the state-tracking head (the whole proposal) # soft soft targets only -- isolates what the head contributes # hard teacher argmin, no soft mass -- isolates soft-vs-hard from the teacher if [ -n "${TRAIN_POLICY_DISTILL:-}" ]; then cd "${CODE_DIR:-/workspace/code}" mkdir -p checkpoints echo "=== GPU check ===" POLICY_HIDDEN="${HIDDEN:-512}" POLICY_LAYERS="${LAYERS:-10}" POLICY_BATCH="${BATCH:-256}" \ /opt/venv/bin/python - <<'PYPD' import os, sys, torch if not torch.cuda.is_available(): sys.exit("ABORT: no CUDA device") h, l, b = (int(os.environ[k]) for k in ("POLICY_HIDDEN", "POLICY_LAYERS", "POLICY_BATCH")) vram = torch.cuda.get_device_properties(0).total_memory / 1e9 print(f"gpu: {torch.cuda.get_device_name(0)} " f"sm_{''.join(map(str, torch.cuda.get_device_capability(0)))} {vram:.0f}GB " f"| torch {torch.__version__}") # The real config at the real batch, for the reason recorded in the main block: a # scaled-down probe passes on a GPU that cannot hold the model, so the failure # surfaces only once the run has been paid for. try: from transformers import LlamaConfig, LlamaForCausalLM m = LlamaForCausalLM(LlamaConfig( vocab_size=28, hidden_size=h, intermediate_size=h * 4, num_hidden_layers=l, num_attention_heads=8, num_key_value_heads=8, max_position_embeddings=83, tie_word_embeddings=True)).cuda() opt = torch.optim.AdamW(m.parameters(), lr=1e-4) ids = torch.randint(0, 28, (b, 83), device="cuda") with torch.autocast("cuda", dtype=torch.bfloat16 if torch.cuda.is_bf16_supported() else torch.float16): m(input_ids=ids, labels=ids).loss.backward() opt.step() torch.cuda.synchronize() print(f"gpu smoke test: full step at batch {b} OK, " f"peak {torch.cuda.max_memory_allocated()/1e9:.1f}GB of {vram:.0f}GB") except torch.OutOfMemoryError: sys.exit(f"ABORT: batch {b} of a {l}-layer/{h}-wide model does not fit in {vram:.0f}GB") except Exception as e: sys.exit(f"ABORT: GPU cannot run the model ({type(e).__name__}: {e})") PYPD echo "=== fetching teacher ${TEACHER_REPO:-briscoooe/tiny-cube-value-iter} ===" TEACHER="$(TR="${TEACHER_REPO:-briscoooe/tiny-cube-value-iter}" \ TF="${TEACHER_FILE:-checkpoints/value_iter/value_iter_best.pt}" \ /opt/venv/bin/python -c 'import os; from huggingface_hub import hf_hub_download; print(hf_hub_download(os.environ["TR"], os.environ["TF"], token=os.environ.get("HF_TOKEN")))')" echo "teacher: $TEACHER" # The highest score any arm can reach, printed before a single gradient step. # The student copies the teacher's one-step choice while the teacher descends # with the true state supplied by the engine, so the teacher's greedy rate caps # the student. Cheap, and it makes an undertrained result impossible to confuse # with a saturated one. /opt/venv/bin/python train_policy_distill.py --teacher "$TEACHER" --ceiling-only \ --eval-depths "${EVAL_DEPTHS:-3,6,10}" --eval-n "${EVAL_N:-100}" for ARM in $(echo "${DISTILL_ARMS:-full}" | tr ',' ' '); do case "$ARM" in full) ARM_FLAGS="--state-weight ${STATE_WEIGHT:-1.0}" ;; soft) ARM_FLAGS="--state-weight 0" ;; hard) ARM_FLAGS="--state-weight ${STATE_WEIGHT:-1.0} --hard-targets" ;; *) echo "unknown arm: $ARM" >&2; exit 1 ;; esac echo "=== arm $ARM ($ARM_FLAGS) ===" /opt/venv/bin/python train_policy_distill.py \ --teacher "$TEACHER" $ARM_FLAGS \ --hidden "${HIDDEN:-512}" --layers "${LAYERS:-10}" --heads 8 \ --batch-size "${BATCH:-256}" --lr "${LR:-3e-4}" --max-steps "${STEPS:-12000}" \ --rollout-len "${ROLLOUT_LEN:-16}" --scramble-depth "${WALK_DEPTH:-12}" \ --tau "${TAU:-0.35}" --explore-frac "${EXPLORE_FRAC:-0.2}" \ --buffer "${BUFFER:-32}" --updates-per-rollout "${UPDATES_PER_ROLLOUT:-4}" \ --eval-every "${EVAL_EVERY:-1000}" --eval-n "${EVAL_N:-100}" \ --eval-depths "${EVAL_DEPTHS:-3,6,10}" \ --out "checkpoints/policy_distill_$ARM" \ --hub-repo "${HUB_REPO:-}" done echo "=== done ===" exit 0 fi # Value iteration needs no dataset and no solver: it derives its own targets by # bootstrapping from the solved state. So this mode skips both the kociemba gate # and the multi-million-pair generation that dominates every other run's startup # -- it goes straight from a GPU check to training. if [ -n "${TRAIN_VALUE_ITER:-}" ]; then cd "${CODE_DIR:-/workspace/code}" mkdir -p checkpoints echo "=== GPU check ===" /opt/venv/bin/python - <<'PYVI' import sys, torch if not torch.cuda.is_available(): sys.exit("ABORT: no CUDA device") print(f"gpu: {torch.cuda.get_device_name(0)} sm_{''.join(map(str,torch.cuda.get_device_capability(0)))} " f"{torch.cuda.get_device_properties(0).total_memory/1e9:.0f}GB | torch {torch.__version__}") PYVI echo "=== training value function by value iteration ===" /opt/venv/bin/python train_value_iteration.py \ --hidden "${HIDDEN:-512}" --layers "${LAYERS:-8}" --heads 8 \ --batch-size "${BATCH:-1024}" --lr "${LR:-3e-4}" --max-steps "${STEPS:-30000}" \ --scramble-depth "${WALK_DEPTH:-30}" \ --update-threshold "${UPDATE_THRESHOLD:-0.15}" \ --min-update-steps "${MIN_UPDATE_STEPS:-100}" \ --max-update-steps "${MAX_UPDATE_STEPS:-400}" \ --eval-every "${EVAL_EVERY:-1000}" --eval-n "${EVAL_N:-200}" \ --out checkpoints/value_iter --hub-repo "${HUB_REPO:-}" echo "=== done ===" exit 0 fi # Hard gate. The pure-Python fallback is ~50x slower, so generating this dataset # with it would take days rather than hours -- and the only symptom is a warning # buried in the log while the meter runs. Fail here instead. echo "=== verifying native kociemba ===" /opt/venv/bin/python - <<'PYCHECK' import sys, time import kociemba from kociemba import ckociembawrapper # absent entirely when the C build failed SOLVED = "".join("URFDLB"[i // 9] for i in range(54)) kociemba.solve("DRLUUBFBRBLURRLRUBLRDDFDLFUFUFFDBRDUBRUFLLFDDBFLUBLRBD") t = time.time() for _ in range(20): kociemba.solve("DRLUUBFBRBLURRLRUBLRDDFDLFUFUFFDBRDUBRUFLLFDDBFLUBLRBD") ms = (time.time() - t) / 20 * 1000 print(f"kociemba: {ms:.1f} ms/solve (native)") if ms > 200: sys.exit(f"ABORT: {ms:.0f} ms/solve is fallback-speed; refusing to generate") PYCHECK cd "${CODE_DIR:-/workspace/code}" mkdir -p data checkpoints # Smoke-test the GPU *before* generating anything. A torch build without kernels # for this GPU fails only when the first tensor op runs, which used to be training # -- so an unusable GPU was discovered an hour and a full dataset later. This # runs the real model for one forward and backward pass, so it catches missing # kernels rather than merely that a device is visible. echo "=== verifying GPU can run the model ===" /opt/venv/bin/python - <> data/train.jsonl echo "=== training set is now $(wc -l < data/train.jsonl) pairs ===" fi echo "=== verifying (must report 0 FAILED) ===" /opt/venv/bin/python verify_data.py data/train.jsonl --limit 100000 /opt/venv/bin/python verify_data.py data/val.jsonl # TRAIN_VALUE switches the objective from "emit a solution" to "estimate distance # to solved". The policy shape has been shown four times over not to reach past # ~8 moves; this is the heuristic that lets search steer. Same data either way -- # the label is just the solution length instead of the solution. if [ -n "${TRAIN_VALUE:-}" ]; then echo "=== training value function ===" /opt/venv/bin/python train_value.py \ --data data/train.jsonl --val-data data/val.jsonl \ --hidden "$HIDDEN" --layers "$LAYERS" --heads 8 \ --batch-size "${BATCH:-1024}" --lr 6e-4 --max-steps "$STEPS" \ --eval-every 1000 --eval-n 8192 \ --out checkpoints/value --hub-repo "${HUB_REPO:-}" echo "=== done ===" exit 0 fi echo "=== training ===" /opt/venv/bin/python train.py \ --data data/train.jsonl --val-data data/val.jsonl \ --hidden "$HIDDEN" --layers "$LAYERS" --heads 8 \ --batch-size "${BATCH:-1024}" --lr 6e-4 --max-steps "$STEPS" \ --eval-every 1000 --eval-n 512 \ --out checkpoints/cube --hub-repo "${HUB_REPO:-}" ${RESUME:+--resume "$RESUME"} echo "=== done ==="