# vast-gpu-remote: verbs that act on a running instance over ssh.
# Sourced, never executed. Expects EXEC, ROOT, stamp, key_for, life_coords,
# evidence_dir, state, explain and `run` to be in scope.
#
# Why these exist: on 2026-09-22 the Eugeny worker relaunched its training job
# six times by hand, each time as five commands (tar the tree, pipe it through
# ssh, untar, rm the old log, nohup torchrun, then poll the log by hand). One
# mistake in that dance is invisible: a launch that silently did nothing looks
# exactly like a training run that has not printed yet. `launch` makes the
# whole sequence one call and refuses to start a second run on a busy box.

remote_ssh() { # remote_ssh <instance-id> <remote command>
  local iid="$1"; shift
  local key; key="$(key_for "$iid" 2>/dev/null || true)"
  if [ -z "$key" ]; then
    echo "vast-gpu: no ssh identity bound to $iid. Attach one (the method field of its receipt names it); never probe candidate keys against a host." >&2
    return 2
  fi
  local out status host port
  out="$(life_coords "$iid")" || return 7
  IFS=$'\t' read -r _ status host port <<<"$out"
  if [ "$status" != "running" ]; then
    echo "vast-gpu: instance is $status, not running; run 'vast-gpu where $iid --wait'" >&2
    return 7
  fi
  ssh -i "$key" -p "$port" \
      -o StrictHostKeyChecking=accept-new -o ConnectTimeout=20 \
      -o BatchMode=yes -o LogLevel=ERROR \
      root@"$host" "$@"
}

# A run ledger, so the next agent reads what was launched instead of re-deriving
# it from a shell history it does not have.
run_record() { # run_record <instance-id> --name <n> --log <p> --cmd <c> --state <s>
  local iid="$1"; shift
  local name="" log="" cmd="" st=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --name) name="$2"; shift 2 ;;
      --log) log="$2"; shift 2 ;;
      --cmd) cmd="$2"; shift 2 ;;
      --state) st="$2"; shift 2 ;;
      *) shift ;;
    esac
  done
  mkdir -p "$ROOT/instances"
  python3 - "$ROOT/instances/$iid-runs.jsonl" "$(stamp)" "$iid" "$name" "$log" "$cmd" "$st" <<'PY'
import json,sys
path,at,iid,name,log,cmd,st=sys.argv[1:8]
with open(path,"a") as f:
    f.write(json.dumps({"at":at,"instance_id":iid,"name":name,"log":log,
                        "cmd":cmd,"state":st})+"\n")
PY
  printf '%s\n' "$ROOT/instances/$iid-runs.jsonl"
}

life_gpu() { # life_gpu <instance-id>
  [ $# -gt 0 ] || { echo "vast-gpu: gpu needs an instance id" >&2; return 2; }
  local iid="$1"
  local remote_out code=0
  set +e
  remote_out="$(remote_ssh "$iid" '
    nvidia-smi --query-gpu=index,utilization.gpu,memory.used,memory.total,power.draw,temperature.gpu \
      --format=csv,noheader,nounits 2>/dev/null | awk -F", " "
      {u+=\$2; n++; printf \"  gpu %-2s  util %3s%%   mem %6s/%6s MiB   %6s W   %s C\n\", \$1,\$2,\$3,\$4,\$5,\$6}
      END {if (n) printf \"%d GPU(s), mean utilization %.1f%%\n\", n, u/n; else print \"no GPU visible\"}"
    echo "  -- what is using them"
    ps -eo etime,pcpu,args --sort=-pcpu 2>/dev/null \
      | grep -E "[t]orchrun|[t]rain_[a-z0-9_]*\.py|[p]ython3 -u" | head -2 | cut -c1-150 | sed "s/^/  /"
    echo "  -- how long the box has been up"
    uptime | sed "s/^/  /"
  ' 2>&1)"; code=$?
  set -e
  echo "$remote_out"
  if [ $code -ne 0 ]; then
    echo "vast-gpu: reading the box failed (exit $code). The GPUs are still billed." >&2
    return $code
  fi
  if ! printf '%s' "$remote_out" | grep -qE '[1-9][0-9]?%'; then
    echo "vast-gpu: no GPU is above 9% utilization. This box is billing while idle: launch the job or destroy the instance." >&2
  fi
  return 0
}

life_tail() { # life_tail <instance-id> --log <path> [--grep <re>] [--lines <n>] [--follow <secs>]
  [ $# -gt 0 ] || { echo "vast-gpu: tail needs an instance id" >&2; return 2; }
  local iid="$1"; shift
  local log="" pat="" lines=40 follow=0
  while [ $# -gt 0 ]; do
    case "$1" in
      --log|--path) log="$2"; shift 2 ;;
      --grep) pat="$2"; shift 2 ;;
      --lines) lines="$2"; shift 2 ;;
      --follow) follow="$2"; shift 2 ;;
      *) echo "vast-gpu: unknown flag $1 for tail (--log, --grep, --lines, --follow)" >&2; return 2 ;;
    esac
  done
  [ -n "$log" ] || { echo "vast-gpu: tail needs --log <path on the box>. A run's ledger line is 'vast-gpu runs $iid'." >&2; return 2; }
  if [ "$follow" -le 0 ]; then
    remote_ssh "$iid" "tail -n $lines '$log' 2>&1"
    return $?
  fi
  # Context first, then the stream. Scanning the context for the stop pattern
  # would end every follow instantly, because the line an agent wants to see is
  # usually already in it.
  remote_ssh "$iid" "tail -n $lines '$log' 2>&1"
  local tmp; tmp="$(mktemp)"
  local stop="${pat:-MILESTONE|Traceback|CUDA out of memory|RuntimeError|Killed|\[rank[0-9]+\]: Traceback}"
  echo "-- following $log for up to ${follow}s"
  # Follow by streaming one ssh session and bounding it locally: `timeout` does
  # not exist on macOS, and a per-poll round trip spends 2-3 s of every cycle.
  set +e
  remote_ssh "$iid" "tail -n 0 -F '$log' 2>&1" > "$tmp" 2>&1 &
  local pid=$!
  local waited=0
  while [ "$waited" -lt "$follow" ]; do
    sleep 2; waited=$((waited+2))
    grep -qE "$stop" "$tmp" && break
    kill -0 "$pid" 2>/dev/null || break
  done
  kill "$pid" 2>/dev/null; wait "$pid" 2>/dev/null
  set -e
  cat "$tmp"
  local verdict
  if grep -qE 'Traceback|CUDA out of memory|RuntimeError|Killed' "$tmp"; then
    verdict="FAILED: $(grep -m 1 -E 'Traceback|CUDA out of memory|RuntimeError|Killed' "$tmp" | cut -c1-140)"
  elif grep -qE 'MILESTONE' "$tmp"; then
    verdict="progress: $(grep -m 1 -E 'MILESTONE' "$tmp" | cut -c1-120)"
  elif [ -s "$tmp" ]; then
    verdict="output, no milestone in ${waited}s"
  else
    verdict="silent for ${waited}s: the run may be between milestones, or this is not the log being written"
  fi
  echo "vast-gpu tail: $verdict"
  rm -f "$tmp"
  return 0
}

life_push() { # life_push <instance-id> <local-dir> [--to <remote-dir>] [--check-py]
  [ $# -ge 2 ] || { echo "vast-gpu: push needs an instance id and a local directory" >&2; return 2; }
  local iid="$1"; local dir="$2"; shift 2
  local to="/workspace/cluster" check=1
  while [ $# -gt 0 ]; do
    case "$1" in
      --to) to="$2"; shift 2 ;;
      --check-py) check=1; shift ;;
      --no-check-py) check=0; shift ;;
      *) echo "vast-gpu: unknown flag $1 for push (--to, --no-check-py)" >&2; return 2 ;;
    esac
  done
  [ -d "$dir" ] || { echo "vast-gpu: $dir is not a directory" >&2; return 2; }
  # Fail locally before touching the box: a syntax error is the one class of
  # mistake this can catch for free, and it is the one that wastes a rented
  # launch on a traceback.
  if [ "$check" = 1 ]; then
    local pys; pys="$(find "$dir" -maxdepth 2 -name '*.py' -not -path '*/vendor/*' 2>/dev/null | head -60)"
    if [ -n "$pys" ]; then
      local bad
      bad="$(printf '%s\n' "$pys" | xargs -n 40 python3 -c 'import ast,sys
for p in sys.argv[1:]:
    try: ast.parse(open(p).read())
    except SyntaxError as e: print(f"{p}:{e.lineno}: {e.msg}")')"
      if [ -n "$bad" ]; then
        echo "vast-gpu: refusing to push, local python does not parse:" >&2
        printf '%s\n' "$bad" >&2
        return 2
      fi
      echo "local syntax check: $(printf '%s\n' "$pys" | wc -l | tr -d ' ') files parse"
    fi
  fi
  local tgz; tgz="$(mktemp)"
  # COPYFILE_DISABLE keeps macOS from attaching resource forks, which the
  # receiving tar otherwise reports as unknown extended headers, straight into
  # the output an agent is reading.
  COPYFILE_DISABLE=1 tar czf "$tgz" --exclude=__pycache__ --exclude='*.pyc' --exclude=.git -C "$dir" . 2>/dev/null
  local bytes sha
  bytes="$(wc -c <"$tgz" | tr -d ' ')"
  sha="$(shasum -a 256 <"$tgz" | awk '{print $1}')"
  local out code=0
  set +e
  out="$(cat "$tgz" | remote_ssh "$iid" "mkdir -p '$to' && tar xzf - -C '$to' 2>/dev/null && find '$to' -maxdepth 1 -type f | wc -l | tr -d ' '" 2>&1)"; code=$?
  set -e
  rm -f "$tgz"
  if [ $code -ne 0 ]; then
    echo "vast-gpu: push failed (exit $code): $out" >&2
    return $code
  fi
  echo "pushed $bytes bytes (sha256 ${sha:0:12}) to $to on $iid; that directory now holds $out file(s)"
  return 0
}

# Single-quote a string for the remote shell. Commands carry spaces, quotes and
# flags, and a launch that silently mangles its own command is the worst kind
# of failure: it looks like the job started.
shq() { printf "'%s'" "$(printf '%s' "$1" | sed "s/'/'\\''/g")"; }

life_launch() { # life_launch <instance-id> --dir <local-dir> [--script <py>] [--args <s>] [--cmd <s>] [--name <n>] [--log <p>] [--wait <secs>] [--replace] [--no-push]
  [ $# -gt 0 ] || { echo "vast-gpu: launch needs an instance id" >&2; return 2; }
  local iid="$1"; shift
  local dir="" script="" args="" cmd="" name="" log="" wait=0 replace=0 push=1 to="/workspace/cluster"
  while [ $# -gt 0 ]; do
    case "$1" in
      --dir) dir="$2"; shift 2 ;;
      --to) to="$2"; shift 2 ;;
      --script) script="$2"; shift 2 ;;
      --args) args="$2"; shift 2 ;;
      --cmd) cmd="$2"; shift 2 ;;
      --name) name="$2"; shift 2 ;;
      --log) log="$2"; shift 2 ;;
      --wait) wait="$2"; shift 2 ;;
      --replace) replace=1; shift ;;
      --no-push) push=0; shift ;;
      *) echo "vast-gpu: unknown flag $1 for launch (--dir --to --script --args --cmd --name --log --wait --replace --no-push)" >&2; return 2 ;;
    esac
  done
  # Defaults are derived from the launch, never guessed: the name from the
  # script, the log from the name, the command from torchrun and the GPU count
  # the box actually reports.
  [ -n "$script" ] || script="$(basename "${cmd%% *}" 2>/dev/null)"
  [ -n "$name" ] || name="$(basename "${script%.py}")"
  [ -n "$log" ] || log="/workspace/runs/$name.log"

  # Two GPU jobs on one box is the failure this guards: they halve each other's
  # memory bandwidth and one usually dies of OOM. A CPU job beside a training
  # run is legitimate, so the guard only applies when the new command is a GPU
  # launch itself.
  local busy is_gpu=0
  case "$cmd $script" in *torchrun*|*train_*.py*) is_gpu=1 ;; esac
  busy="$(remote_ssh "$iid" "ps -eo args | grep -E '[t]orchrun|[t]rain_[a-z0-9_]*\.py' | head -2" 2>/dev/null || true)"
  if [ -n "$busy" ] && [ "$is_gpu" = 1 ] && [ "$replace" != 1 ]; then
    echo "vast-gpu: a training process is already running on $iid:" >&2
    printf '%s\n' "$busy" | cut -c1-150 | sed 's/^/  /' >&2
    echo "vast-gpu: pass --replace to kill it and start this run, or watch it: vast-gpu gpu $iid" >&2
    return 9
  fi
  [ -n "$busy" ] && [ "$is_gpu" != 1 ] && echo "note: a GPU job is running; this launch is not one, so it proceeds" >&2

  if [ "$push" = 1 ]; then
    if [ -n "$dir" ]; then
      life_push "$iid" "$dir" --to "$to" || return $?
    else
      echo "vast-gpu: no --dir to push; launching whatever is already at $to" >&2
    fi
  fi

  if [ -z "$cmd" ]; then
    [ -n "$script" ] || { echo "vast-gpu: launch needs --script <file.py> or --cmd '<command>'" >&2; return 2; }
    local ngpus
    ngpus="$(remote_ssh "$iid" 'nvidia-smi -L 2>/dev/null | wc -l' | tr -d ' ')"
    case "$ngpus" in ''|0) ngpus=1 ;; esac
    cmd="torchrun --nproc_per_node=$ngpus --standalone $script $args"
    echo "command from defaults ($script, ${ngpus} GPUs): $cmd"
  fi
  if [ "$replace" = 1 ] && [ -n "$busy" ]; then
    # Bracketed patterns, because `pkill -f torchrun` also matches the shell
    # that is running this very line: it kills its own ssh session, the rest of
    # the payload never executes, and the caller sees a silent 255. This is not
    # hypothetical; it happened to the cleanup of this client's own test.
    remote_ssh "$iid" "pkill -9 -f '[t]rain_[a-z0-9_]*\.py' 2>/dev/null; pkill -9 -f '[t]orchrun' 2>/dev/null; sleep 3; ps -eo args | grep -cE '[t]orchrun|[t]rain_[a-z0-9_]*\.py'" 2>&1 | tail -1
    local left; left="$(remote_ssh "$iid" "ps -eo args | grep -cE '[t]orchrun|[t]rain_[a-z0-9_]*\.py'" 2>/dev/null | tr -d ' ')"
    if [ "${left:-1}" -gt 0 ] 2>/dev/null; then
      echo "vast-gpu: --replace asked to kill the previous run but $left process(es) still match; the GPUs are contended. Inspect: vast-gpu gpu $iid" >&2
      return 9
    fi
    echo "killed the previous run (--replace); 0 processes still match"
  fi

  # The job is started by a script piped to the box's bash, so the command may
  # be a shell line (`cd X && torchrun ...`) rather than a single executable.
  # `nohup env ... $cmd` treats the first word of a shell line as the program.
  # The command travels base64-encoded: a command carrying quotes, redirects or
  # a nested shell is exactly what a launch must not mangle, and quoting it into
  # the remote line breaks on the first apostrophe.
  local tmp b64; tmp="$(mktemp)"
  b64="$(printf '%s' "$cmd" | base64 | tr -d '\n')"
  local qlog; qlog="$(shq "$log")"
  local qto; qto="$(shq "$to")"
  cat > "$tmp" <<EOS
set -e
LOG=$qlog
TO=$qto
mkdir -p "\$(dirname "\$LOG")" "\$TO"
cd "\$TO"
rm -f "\$LOG"
CMD=\$(printf '%s' '$b64' | base64 -d)
nohup env OMP_NUM_THREADS=64 TORCH_NCCL_ASYNC_ERROR_HANDLING=1 bash -c "\$CMD" > "\$LOG" 2>&1 < /dev/null &
echo "pid \$!"
EOS
  local started code=0
  set +e
  started="$(remote_ssh "$iid" 'bash -s' < "$tmp" 2>&1)"; code=$?
  set -e
  rm -f "$tmp"
  if [ $code -ne 0 ]; then
    echo "vast-gpu: launch failed (exit $code): $started" >&2
    return $code
  fi
  local rpid; rpid="$(printf '%s\n' "$started" | sed -n 's/^pid \([0-9]*\)$/\1/p' | tail -1)"
  local ledger; ledger="$(run_record "$iid" --name "$name" --log "$log" --cmd "$cmd" --state launched)"
  echo "launched $name on $iid (pid ${rpid:-unknown})"
  echo "log $log"
  echo "ledger $ledger"

  # Liveness is the recorded pid and the log it writes, not a count of
  # processes matching a name: the box may be running a different job already,
  # and that job would make a dead launch look alive.
  sleep 2
  local probe
  probe="$(remote_ssh "$iid" "if [ -n '$rpid' ] && kill -0 '$rpid' 2>/dev/null; then echo alive; else echo gone; fi; wc -c < $(shq "$log") 2>/dev/null || echo 0" 2>/dev/null || echo "unknown 0")"
  local alive bytes
  alive="$(printf '%s\n' "$probe" | head -1)"
  bytes="$(printf '%s\n' "$probe" | tail -1 | tr -d ' ')"
  if [ "$alive" != "alive" ]; then
    echo "vast-gpu: the process is gone two seconds after launch. Exit codes: 10." >&2
    remote_ssh "$iid" "tail -n 15 $(shq "$log") 2>&1" | sed 's/^/  /' >&2
    run_record "$iid" --name "$name" --log "$log" --cmd "$cmd" --state died-at-launch >/dev/null
    return 10
  fi
  echo "alive after 2s, log at $bytes bytes"

  if [ "$wait" -gt 0 ]; then
    echo "-- watching $log for up to ${wait}s"
    life_tail "$iid" --log "$log" --lines 5 --follow "$wait"
    run_record "$iid" --name "$name" --log "$log" --cmd "$cmd" --state watched >/dev/null
  else
    echo "watch it: vast-gpu tail $iid --log $log --follow 120"
  fi
  return 0
}

# The runs ledger, so an agent that inherits a box reads what was launched
# instead of inferring it from a log file it has not found yet.
life_runs() { # life_runs <instance-id>
  [ $# -gt 0 ] || { echo "vast-gpu: runs needs an instance id" >&2; return 2; }
  local iid="$1"
  local f="$ROOT/instances/$iid-runs.jsonl"
  if [ ! -f "$f" ]; then
    echo "vast-gpu: no run recorded for $iid yet; 'vast-gpu launch' writes one."
    return 0
  fi
  python3 - "$f" <<'PY'
import json,sys
for line in open(sys.argv[1]):
    r=json.loads(line)
    print(f"{r['at']}  {r['state']:16s}  {r['name']:24s}  {r['log']}")
    print(f"{'':22s}{r['cmd'][:150]}")
PY
}