#!/bin/bash
# hpcworker: your own Claude Code, as a batch job on your university's cluster, reachable from
# claude.ai/code and the Claude app through Claude Code's Remote Control. This is the one command;
# it runs on the login node with bash, coreutils and the system python3 only. Nothing heavy runs
# here: the container and Claude Code run inside jobs. An account has it one of two ways: the lab's
# shared copy (then ~/.hpcworker/bin/hpcworker is a small launcher for it) or a copy of its own.
#
#   hpcworker install --shared [DIR] [--profile NAME] [--force] [--no-shell-path]
#                     use the lab's shared copy at DIR (default: the profile's shared_install); nothing is copied
#   hpcworker install [--own] [--profile NAME | --discover] [--force] [--no-shell-path]
#                     a copy of your own in ~/.hpcworker (again: refresh or upgrade); --own leaves a shared copy
#                     (--no-shell-path: do not touch ~/.bashrc; profile.json is kept unless --force)
#   hpcworker profile show|path|validate|set KEY VALUE|schema   the site profile
#   hpcworker discover [--full] [--save]                         phase-1 discovery on the login node (under 1 s of CPU; --full adds module avail and QoS)
#   hpcworker probe [--show] [--wait MIN] [--partition P --qos Q] ten-minute probe job; writes probe.json
#   hpcworker worker start [--name N --hours H --cpus C --mem M --partition P --qos Q --permission-mode M --capacity K --no-chain]
#   hpcworker worker stop|status|list|log [--name N] [--json] [--lines N]
#   hpcworker claude login [--plain] | status [--alloc]           your Claude sign-in, in one sitting (--plain: the panel's line contract)
#   hpcworker claude consent                                     the Remote Control answer alone, when login did not cover it
#   hpcworker diag [--json]                                      one line per check
#   hpcworker version                                            the number; "(shared copy at DIR)" on the lab's copy
#
# Rules this command keeps: it never reads, copies or prints a Claude credential (Claude Code keeps
# its own; the framework reads `claude auth status`, and two booleans Claude Code writes into its
# .claude.json: remoteDialogSeen and the workspace's hasTrustDialogAccepted); no token ever goes on
# a command line or into a job file; the login node gets one sbatch/squeue/scancel per action and
# never a loop; state is read from files the job writes; nothing is ever deleted outside the run
# directories under the state directory and a job's own node-local temp; the one subcommand that
# replaces a file of yours (install --force) runs only for a person at a terminal. Versions are integers.
set -uo pipefail
VERSION=6
HPCW_HOME="${HPCW_HOME:-$HOME/.hpcworker}"
# Where the code lives: the directory above this file's bin/. A link to the file itself is followed
# (a ~/bin/hpcworker link still finds lib/), but the directories above stay as spelled: at Wayne
# State /rs/rs_grp_oschome is itself a link to the volume's physical path, which must not end up in
# jobs and messages (readlink -f did that). A shared copy's `current` link is then followed ONE
# level, to its numbered directory: a job names the version it started on, so a later publish never
# changes the code under a running chain, and the older numbered directories are kept for it.
hpcw_follow_file() { local f="$1" t n=0; while [ -L "$f" ] && [ $n -lt 40 ]; do t=$(readlink "$f") || break; case "$t" in /*) f="$t";; *) f="$(dirname "$f")/$t";; esac; n=$((n + 1)); done; printf '%s' "$f"; }
# DIR -> absolute and logical; under a Windows checkout's Git Bash (the tests) the drive spelling, which its native python can open
abs_dir() { (cd "$1" 2>/dev/null || exit 1; case "${OSTYPE:-}" in msys*) pwd -W;; *) pwd;; esac); }
hpcw_code_root() { # DIR -> DIR, or the directory its final link names (one level), still logical
  local d="${1%/}" t; [ -L "$d" ] || { printf '%s' "$d"; return 0; }
  t=$(readlink "$d") || { printf '%s' "$d"; return 0; }
  case "$t" in /*) ;; *) t="$(dirname "$d")/$t";; esac
  abs_dir "$t" || printf '%s' "$d"
}
SELF=$(hpcw_follow_file "${BASH_SOURCE[0]}")
SRC_ROOT=$(hpcw_code_root "$(abs_dir "$(dirname "$SELF")/..")")
# lib, jobs and profiles come from where this file lives: a checkout, the copy in ~/.hpcworker, or the lab's shared copy
LIB="$SRC_ROOT/lib"; JOBS="$SRC_ROOT/jobs"; PROFILES="$SRC_ROOT/profiles"
PROFILE="${HPCW_PROFILE:-$HPCW_HOME/profile.json}"
# An account on the lab's shared copy: ~/.hpcworker/shared (the marker) names the shared directory,
# ~/.hpcworker/bin/hpcworker is a launcher for it, and the profile, the log and the consent marker
# stay in ~/.hpcworker (run directories under shared_state_dir, as always). A full copy of the
# command in ~/.hpcworker itself (an upload from the extension) wins over a marker left behind.
SHARED_DIR=""; [ -s "$HPCW_HOME/shared" ] && { IFS= read -r SHARED_DIR < "$HPCW_HOME/shared" || true; }
LAUNCHER_TAG='# hpcworker launcher for a shared copy (managed: "hpcworker install" rewrites this file)'
PY="${HPCW_PYTHON:-}"; [ -n "$PY" ] || for p in python3 /usr/bin/python3 /usr/libexec/platform-python; do command -v "$p" >/dev/null 2>&1 && { PY=$(command -v "$p"); break; }; done
die() { echo "hpcworker: $*" >&2; exit 1; }
say() { echo "$*"; }
need_py() { [ -n "$PY" ] || die "python3 is needed on this node (the system one is enough; nothing is installed)"; }
prof_py() { need_py; "$PY" "$LIB/profile.py" "$@"; }
# shellcheck disable=SC1090
source "$LIB/adapters/common.sh" || die "$LIB/adapters/common.sh is missing"
json_str() { hpcw_json_str "$1"; }
in_job() { [ -n "${SLURM_JOB_ID:-${PBS_JOBID:-${LSB_JOBID:-}}}" ]; }
same_dir() { local a b; a=$(cd "$1" 2>/dev/null && pwd -P) && b=$(cd "$2" 2>/dev/null && pwd -P) && [ "$a" = "$b" ]; }   # two directories that exist and are one
# this run uses the lab's shared copy: a marker, and the code is not a full copy inside this home
shared_mode() { [ -n "$SHARED_DIR" ] && ! same_dir "$SRC_ROOT" "$HPCW_HOME"; }
read_version() { local v=""; [ -r "$1" ] && v=$(tr -d '[:space:]' 2>/dev/null < "$1"); [[ "$v" =~ ^[0-9]+$ ]] && printf '%s' "$v"; }
load_profile() {
  [ -s "$PROFILE" ] || die "no profile at $PROFILE: run 'hpcworker install --profile wayne' (or 'hpcworker install --discover')"
  local env; env=$(prof_py shellenv "$PROFILE") || die "profile $PROFILE unreadable"
  eval "$env"
  ADAPTER="$LIB/adapters/${P_scheduler}.sh"; [ -s "$ADAPTER" ] || die "no adapter for scheduler '$P_scheduler' (have: $(ls "$LIB/adapters" | sed 's/\.sh$//' | grep -v common | tr '\n' ' '))"
  # shellcheck disable=SC1090
  source "$ADAPTER"
  STATE_DIR="$P_shared_state_dir"; WORKSPACES="$STATE_DIR/workspaces"
  CCD="$P_claude_config_dir"
  export RUNTIME="$P_runtime" RUNTIME_BIN="$P_runtime_bin" RUNTIME_MODULE="$P_runtime_module"
  export HPCW_NODE_TMP_PATTERN="$P_node_tmp" HPCW_NODE_TMP_VAR="$P_node_tmp_var" HPCW_TMPDIR_IS_NODE_LOCAL="$([ "$P_tmpdir_is_node_local" = true ] && echo 1 || echo 0)"
}
# "heavy" = a container or Claude Code itself: only inside a job, never on the login node
refuse_heavy() { in_job || die "$1 would run a container or Claude Code on this node ($(hostname -s)); that is not allowed on a login node, so hpcworker does it inside a job for you: $2"; }
# a destructive action needs a person at a terminal: never an agent (Claude Code, Ashley), never a pipeline
refuse_if_agent() { hpcw_agent_marker && die "$1 would replace or remove something of yours; it runs only for a person at a terminal, not under Claude Code, another agent or a pipeline. Nothing was changed."; return 0; }
consent_markers() { printf '%s\n' "$HPCW_HOME/consent"; local m; for m in ${P_consent_markers:-}; do printf '%s\n' "$m"; done; }
consent_source() { # claude_config | marker | none : Claude Code's own record first, the fallback marker files second
  if hpcw_consent_seen "${CCD:-}"; then echo claude_config; return 0; fi
  local m; while read -r m; do [ -n "$m" ] && [ -e "$m" ] && { echo marker; return 0; }; done < <(consent_markers); echo none
}
consent_given() { [ "$(consent_source)" != none ]; }
fmt_job_state() { # "STATE|NODE|RAW" -> "STATE" or "STATE on NODE"
  local st="${1%%|*}" rest="${1#*|}" node; node="${rest%%|*}"; [ "$rest" = "$1" ] && node=""
  printf '%s%s' "$st" "${node:+ on $node}"
}
clean_env_args() { # the environment Claude Code gets inside the container (matches worker.sbatch.tpl)
  printf '%s\n' env -u ANTHROPIC_API_KEY -u ANTHROPIC_AUTH_TOKEN -u ANTHROPIC_BASE_URL -u CLAUDE_CODE_OAUTH_TOKEN -u CLAUDE_CODE_OAUTH_TOKEN_FILE_DESCRIPTOR -u CLAUDE_CODE_API_KEY_FILE_DESCRIPTOR \
    -u CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC -u DISABLE_GROWTHBOOK HOME="$HOME" CLAUDE_CONFIG_DIR="$CCD" DISABLE_AUTOUPDATER=1 TERM="${TERM:-xterm-256color}" BROWSER=/bin/false TMPDIR=/tmp
  [ "$P_xdg_runtime_fix" = true ] && printf '%s\n' XDG_RUNTIME_DIR="$P_ipc_dir/$USER/xdg"
}

# ---- install -----------------------------------------------------------------------------------
# ---- the shell: ~/.hpcworker/bin on PATH for a person at a terminal -----------------------------
# The extension never needs this (it calls the command by its full path through the portal's shell);
# a person opening the portal's terminal does. One managed block in ~/.bashrc, rewritten in place and
# never duplicated, the way Ashley's grid CLI does it. Only for the default install location: a custom
# HPCW_HOME (or --no-shell-path) gets the one-line hint instead and nothing of theirs is touched.
SHELL_BLOCK_BEGIN='# >>> hpcworker (managed; "hpcworker install" rewrites this block) >>>'
SHELL_BLOCK_END='# <<< hpcworker <<<'
shell_path_block() {
  local rc="$HOME/.bashrc" tmp pf
  if [ ! -e "$rc" ]; then : > "$rc" 2>/dev/null || { say "could not create $rc; add the PATH line yourself: export PATH=\"\$HOME/.hpcworker/bin:\$PATH\""; return 0; }; fi
  [ -w "$rc" ] || { say "$rc is not writable; add the PATH line yourself: export PATH=\"\$HOME/.hpcworker/bin:\$PATH\""; return 0; }
  tmp=$(mktemp "$HOME/.bashrc.hpcw.XXXXXX") || return 0
  awk -v b="$SHELL_BLOCK_BEGIN" -v e="$SHELL_BLOCK_END" '$0==b{skip=1;next} $0==e{skip=0;next} !skip{print}' "$rc" > "$tmp"
  { cat "$tmp"; [ -s "$tmp" ] && [ -n "$(tail -c1 "$tmp")" ] && echo
    printf '%s\n' "$SHELL_BLOCK_BEGIN" '# The hpcworker command for terminals (hpcworker worker status, hpcworker claude login ...).' \
      'case ":$PATH:" in *":$HOME/.hpcworker/bin:"*) ;; *) export PATH="$HOME/.hpcworker/bin:$PATH";; esac' "$SHELL_BLOCK_END"; } > "$rc" && rm -f "$tmp"
  # a login shell that reads only ~/.bash_profile or ~/.profile must be told to read ~/.bashrc
  for pf in "$HOME/.bash_profile" "$HOME/.profile"; do
    [ -f "$pf" ] || continue
    grep -q '\.bashrc' "$pf" || say "note: $pf does not read ~/.bashrc, so login shells will not see the PATH line until it does"
    break
  done
  case ":$PATH:" in *":$HOME/.hpcworker/bin:"*) ;; *) say "PATH: ~/.hpcworker/bin is added by a managed block in ~/.bashrc; new terminals have it, this one needs: export PATH=\"\$HOME/.hpcworker/bin:\$PATH\"";; esac
}

# ---- the lab's shared copy -----------------------------------------------------------------------
# A lab keeps one copy of the framework on its shared storage (a numbered directory per version and
# a `current` link, published by tools/deploy_shared.sh), and each account uses it through a small
# launcher in ~/.hpcworker/bin: none of the code is copied into the home, and a new version reaches
# every account when the lab publishes it. The account's profile, log and consent marker stay in
# ~/.hpcworker; its run directories stay under shared_state_dir.
is_launcher() { [ -f "$1" ] && [ ! -L "$1" ] && head -n 3 "$1" 2>/dev/null | grep -qxF "$LAUNCHER_TAG"; }
shared_problem() { # DIR -> one sentence when DIR is not a shared copy this account can use; nothing when it is
  local d="$1" pd ph
  case "$d" in /*) ;; [A-Za-z]:/*) case "${OSTYPE:-}" in msys*) ;; *) echo "it is not an absolute path"; return 0;; esac;; *) echo "it is not an absolute path"; return 0;; esac
  case "$d" in *$'\n'*) echo "its name has a line break"; return 0;; esac
  { [ -d "$d" ] && [ -x "$d" ] && [ -r "$d" ]; } || { echo "this account cannot open it (it does not exist, or this account is not in the group that may read it)"; return 0; }
  [ -n "$(read_version "$d/VERSION")" ] || { echo "it has no readable VERSION file"; return 0; }
  local f; for f in bin/hpcworker lib/profile.py lib/adapters/common.sh jobs/worker.sbatch.tpl; do [ -r "$d/$f" ] || { echo "$f is missing there or cannot be read"; return 0; }; done
  pd=$(cd "$d" && pwd -P); ph=$(cd "$HPCW_HOME" 2>/dev/null && pwd -P)
  if [ -n "$ph" ]; then case "$pd/" in "$ph"/*) echo "it is inside $HPCW_HOME, the folder the launcher goes into"; return 0;; esac; fi
  return 0
}
shared_default() { # NAME -> the shared copy the site profile names (this copy's profiles/NAME.json, then profile.json), else the one in use
  local v=""
  [ -n "$1" ] && [ -s "$PROFILES/$1.json" ] && v=$(prof_py get "$PROFILES/$1.json" shared_install 2>/dev/null)
  [ -z "$v" ] && [ -s "$PROFILE" ] && v=$(prof_py get "$PROFILE" shared_install 2>/dev/null)
  printf '%s' "${v:-$SHARED_DIR}"
}
own_copy_parts() { # the pieces of a copy of the code inside this home (the launcher is not one)
  local p; for p in bin/hpcworker lib jobs profiles VERSION README.md; do
    [ -e "$HPCW_HOME/$p" ] || [ -L "$HPCW_HOME/$p" ] || continue
    [ "$p" = bin/hpcworker ] && is_launcher "$HPCW_HOME/$p" && continue
    printf '%s\n' "$p"
  done
}
live_run() { # the first run of this account's that may still read code from this home: a queued
  # successor, a heartbeat under three minutes old, a job submitted that has not reported, a probe
  # under way. Files only, no scheduler call. A running job sources lib/ again on every call into
  # the container, so its code must not move while it runs.
  [ -s "$PROFILE" ] || return 0
  local sd d n st now; sd=$(prof_py get "$PROFILE" shared_state_dir 2>/dev/null); { [ -n "$sd" ] && [ -d "$sd" ]; } || return 0
  now=$(date +%s)
  for d in "$sd"/*/; do
    [ -d "$d" ] || continue; n=$(basename "$d"); [ "$n" = workspaces ] && continue
    if [ "$n" = probe ]; then
      [ -s "$d/job.id" ] && [ ! -s "$d/probe.json" ] && [ $(( now - $(stat -c %Y "$d/job.id") )) -lt 900 ] && { echo "probe (job $(cat "$d/job.id"))"; return 0; }
      continue
    fi
    [ -s "$d/next.id" ] && { echo "$n (job $(next_job_of "$d") is queued to take over)"; return 0; }
    if [ -s "$d/status.json" ]; then
      st=$(read_field "$d/status.json" state)
      [ "$st" != stopped ] && [ $(( now - $(stat -c %Y "$d/status.json") )) -lt 180 ] && { echo "$n (job $(read_field "$d/status.json" job), $st)"; return 0; }
    fi
    [ -s "$d/job.id" ] && [ ! -e "$d/stop" ] && [ ! -e "$d/stop.done" ] && { [ ! -e "$d/status.json" ] || [ "$d/job.id" -nt "$d/status.json" ]; } \
      && { echo "$n (job $(cat "$d/job.id") was submitted and has not reported yet)"; return 0; }
  done
  return 0
}
ASIDE=""
move_aside() { # PART...: moved into ~/.hpcworker/replaced/<time>/ (one folder per install run); nothing is ever deleted
  local p n=0 base
  if [ -z "$ASIDE" ]; then base="$HPCW_HOME/replaced/$(date +%Y%m%d-%H%M%S)"; ASIDE="$base"; while [ -e "$ASIDE" ]; do n=$((n + 1)); ASIDE="$base-$n"; done; fi
  for p in "$@"; do mkdir -p "$(dirname "$ASIDE/$p")" && mv -f "$HPCW_HOME/$p" "$ASIDE/$p" || return 1; done
}
write_launcher() { # DIR: ~/.hpcworker/bin/hpcworker as a small regular file that runs the shared copy.
  # Never a link: a copy or an upload onto a link would write into the shared tree itself.
  local dir="$1" f="$HPCW_HOME/bin/hpcworker" home
  # shellcheck disable=SC2016
  if same_dir "$HPCW_HOME" "$HOME/.hpcworker"; then home='"$HOME/.hpcworker"'; else home=$(printf '%q' "$HPCW_HOME"); fi
  # shellcheck disable=SC2016
  { printf '%s\n' '#!/bin/bash' "$LAUNCHER_TAG" "# This account uses the lab's shared copy of hpcworker; its profile, log and consent marker stay in this folder."
    printf 'HPCW_SHARED_ROOT=%q\n' "$dir"
    printf '[ -n "${HPCW_HOME:-}" ] || HPCW_HOME=%s\nexport HPCW_HOME\n' "$home"
    printf '%s\n' '[ -r "$HPCW_SHARED_ROOT/bin/hpcworker" ] || { echo "hpcworker: the shared copy at $HPCW_SHARED_ROOT cannot be read from this account (ask your lab lead to give this account access to it)" >&2; exit 3; }' \
      'exec "${BASH:-bash}" "$HPCW_SHARED_ROOT/bin/hpcworker" "$@"'
  } > "$f.part" && chmod 755 "$f.part" && mv -f "$f.part" "$f"
}
install_profile() { # NAME DISCOVER FORCE: profile.json from a site profile or from discovery; one already there is kept unless --force
  local name="$1" discover="$2" force="$3"
  if [ -n "$name" ]; then
    [ -s "$PROFILES/$name.json" ] || die "no profile named '$name' (have: $(ls "$PROFILES" | sed 's/\.json$//' | tr '\n' ' '))"
    # u+w: a shared copy's files are read-only, and a copy of one keeps its mode
    if [ -s "$PROFILE" ] && [ "$force" = 0 ]; then say "profile.json exists; kept (use --force to replace it with '$name')"; else cp -f "$PROFILES/$name.json" "$PROFILE" && chmod u+w "$PROFILE"; say "profile.json written from '$name'"; fi
  elif [ "$discover" = 1 ]; then
    cmd_discover --full > "$HPCW_HOME/discover.json" || die "discovery failed"
    if [ -s "$PROFILE" ] && [ "$force" = 0 ]; then say "profile.json exists; kept (use --force to replace it with a discovered seed)"
    else prof_py from-discovery "$HPCW_HOME/discover.json" "${HPCW_SITE:-$(hostname -s | tr -cd 'a-z0-9-')}" > "$PROFILE" || die "could not seed a profile from discovery"; say "profile.json seeded from discovery: edit partition_cpu, image_store, claude_bin, then run hpcworker probe"; fi
  fi
  [ -s "$PROFILE" ] && { prof_py validate "$PROFILE" || say "fix the profile before starting a worker: hpcworker profile set KEY VALUE"; }
  return 0
}
install_shared() { # DIR NAME DISCOVER FORCE: the launcher, the marker and the profile; no code is copied
  local dir="$1" name="$2" discover="$3" force="$4"
  [ -n "$dir" ] || dir=$(shared_default "$name")
  [ -n "$dir" ] || die "install --shared: no shared copy is named here; give its folder (hpcworker install --shared DIR) or use a site profile that names one (shared_install)"
  local why; why=$(shared_problem "$dir")
  # exit 3 means "this shared copy cannot be used from this account": the extension then falls back to a copy of the account's own
  if [ -n "$why" ]; then echo "hpcworker: the shared copy at $dir cannot be used from this account: $why. Nothing was changed." >&2; exit 3; fi
  local code sv parts busy own_old prev="$SHARED_DIR"
  code=$(hpcw_code_root "$dir"); sv=$(read_version "$dir/VERSION")
  [ -z "$name" ] || [ -s "$code/profiles/$name.json" ] || die "the shared copy at $dir has no profile named '$name' (have: $(ls "$code/profiles" 2>/dev/null | sed 's/\.json$//' | tr '\n' ' ')). Nothing was changed."
  parts=$(own_copy_parts); own_old=$(read_version "$HPCW_HOME/VERSION")
  if [ -n "$parts" ]; then
    busy=$(live_run)
    [ -z "$busy" ] || die "your worker $busy still runs from the copy in $HPCW_HOME. Stop it first (hpcworker worker stop), then run this again. Nothing was changed."
  fi
  mkdir -p "$HPCW_HOME/bin" "$HPCW_HOME/log" || die "cannot create $HPCW_HOME"
  chmod 700 "$HPCW_HOME"
  # a copy of the account's own is moved aside whole, never deleted; the profile, state and consent stay where they are
  # shellcheck disable=SC2086
  [ -z "$parts" ] || move_aside $parts || die "could not move the copy in $HPCW_HOME aside (see $ASIDE)"
  write_launcher "$dir" || die "could not write the launcher $HPCW_HOME/bin/hpcworker"
  { printf '%s\n' "$dir" > "$HPCW_HOME/shared.part" && mv -f "$HPCW_HOME/shared.part" "$HPCW_HOME/shared"; } || die "could not write $HPCW_HOME/shared"
  SHARED_DIR="$dir"; SRC_ROOT="$code"; LIB="$code/lib"; JOBS="$code/jobs"; PROFILES="$code/profiles"
  install_profile "$name" "$discover" "$force"
  if [ -n "$parts" ]; then say "switched to the shared copy of hpcworker $sv at $dir; your own copy${own_old:+ (version $own_old)} was moved to $ASIDE (nothing deleted)"
  elif [ -z "$prev" ]; then say "installed: this account uses the shared copy of hpcworker $sv at $dir (nothing copied into $HPCW_HOME)"
  elif [ "$prev" != "$dir" ]; then say "this account now uses the shared copy of hpcworker $sv at $dir (it used $prev)"
  else say "hpcworker $sv: shared copy at $dir; launcher and marker refreshed (nothing copied)"; fi
}
install_own() { # NAME DISCOVER FORCE: a copy of the code in this home, as before shared copies existed
  local name="$1" discover="$2" force="$3" hint=""
  # the copy already there, if any: install runs again over it (the same version refreshes the
  # files, a newer one upgrades them); the profile and everything under the state directory stay
  local old; old=$(read_version "$HPCW_HOME/VERSION")
  [ -n "$name" ] && [ -z "$SHARED_DIR" ] && [ -s "$PROFILES/$name.json" ] && hint=$(prof_py get "$PROFILES/$name.json" shared_install 2>/dev/null)
  mkdir -p "$HPCW_HOME"/{bin,lib/adapters,jobs,profiles,log} || die "cannot create $HPCW_HOME"
  chmod 700 "$HPCW_HOME"
  # a link where the command goes is moved aside first: a copy onto it would write through it
  [ -L "$HPCW_HOME/bin/hpcworker" ] && { move_aside bin/hpcworker || die "could not move the link $HPCW_HOME/bin/hpcworker aside"; }
  if ! same_dir "$SRC_ROOT" "$HPCW_HOME"; then
    cp "$SRC_ROOT/bin/hpcworker" "$HPCW_HOME/bin/" && cp "$SRC_ROOT/lib/profile.py" "$SRC_ROOT/lib/login_plain.sh" "$HPCW_HOME/lib/" && cp "$SRC_ROOT"/lib/adapters/*.sh "$HPCW_HOME/lib/adapters/" \
      && cp "$SRC_ROOT"/jobs/* "$HPCW_HOME/jobs/" && cp "$SRC_ROOT"/profiles/*.json "$HPCW_HOME/profiles/" && cp "$SRC_ROOT/VERSION" "$HPCW_HOME/" || die "copy into $HPCW_HOME failed"
    [ -f "$SRC_ROOT/README.md" ] && cp "$SRC_ROOT/README.md" "$HPCW_HOME/"
    # copied from a shared copy (read-only there), the files would stay read-only and the next install or upload could not replace them
    chmod -R u+w "$HPCW_HOME"/bin "$HPCW_HOME"/lib "$HPCW_HOME"/jobs "$HPCW_HOME"/profiles "$HPCW_HOME"/VERSION 2>/dev/null
    [ -f "$HPCW_HOME/README.md" ] && chmod u+w "$HPCW_HOME/README.md"
  fi
  # files uploaded from a Windows checkout may carry CR line endings, which break bash and sbatch
  local f; for f in "$HPCW_HOME"/bin/hpcworker "$HPCW_HOME"/lib/profile.py "$HPCW_HOME"/lib/*.sh "$HPCW_HOME"/lib/adapters/*.sh "$HPCW_HOME"/jobs/* "$HPCW_HOME"/profiles/*.json; do [ -f "$f" ] && sed -i 's/\r$//' "$f"; done
  chmod +x "$HPCW_HOME"/bin/hpcworker "$HPCW_HOME"/jobs/*.sbatch 2>/dev/null
  # this home has its own copy now: a shared copy's marker is moved aside with the rest, so nothing takes it for the copy in use
  local left="$SHARED_DIR"
  if [ -e "$HPCW_HOME/shared" ]; then move_aside shared || die "could not move $HPCW_HOME/shared aside"; SHARED_DIR=""; fi
  SRC_ROOT="$HPCW_HOME"; LIB="$HPCW_HOME/lib"; JOBS="$HPCW_HOME/jobs"; PROFILES="$HPCW_HOME/profiles"
  install_profile "$name" "$discover" "$force"
  if [ -z "$old" ]; then say "installed hpcworker $VERSION at $HPCW_HOME"
  elif [ "$old" = "$VERSION" ]; then say "hpcworker $VERSION at $HPCW_HOME: files refreshed (same version)"
  elif [ "$old" -gt "$VERSION" ]; then say "replaced hpcworker $old with $VERSION at $HPCW_HOME (this checkout is the older one)"
  else say "upgraded $old -> $VERSION at $HPCW_HOME"; fi
  [ -n "$left" ] && say "this account now uses its own copy; the shared copy at $left is no longer used (its marker is in $ASIDE)"
  # only a shared copy this account can open is worth naming (an account outside the lab's group cannot)
  [ -n "$hint" ] && [ -z "$(shared_problem "$hint")" ] && say "note: this site has a shared copy at $hint; 'hpcworker install --shared' uses it instead of a copy of your own"
  return 0
}

cmd_install() {
  local name="" discover=0 force=0 shellpath=1 shared=0 own=0 sdir=""
  while [ $# -gt 0 ]; do case "$1" in
    --profile) name="$2"; shift 2;; --discover) discover=1; shift;; --force) force=1; shift;; --no-shell-path) shellpath=0; shift;;
    --shared) shared=1; if [ $# -ge 2 ] && [ "${2#-}" = "$2" ]; then sdir="$2"; shift 2; else shift; fi;;
    --own) own=1; shift;;
    *) die "install: unknown option $1";; esac; done
  [ $shared = 1 ] && [ $own = 1 ] && die "install: --shared and --own do not go together"
  # --force replaces profile.json, a file of yours: only for a person at a terminal, and only when a profile is in the way
  if [ $force = 1 ] && [ -s "$PROFILE" ]; then refuse_if_agent "install --force"; fi
  # an account on a shared copy stays on it when install runs again (through the launcher, or from a
  # checkout); only --own, or a full copy of the command inside this home (an upload), leaves it
  if [ $shared = 0 ] && [ $own = 0 ] && shared_mode; then shared=1; sdir="$SHARED_DIR"; fi
  if [ $shared = 1 ]; then install_shared "$sdir" "$name" "$discover" "$force"; else install_own "$name" "$discover" "$force"; fi
  if [ $shellpath = 1 ] && same_dir "$HPCW_HOME" "$HOME/.hpcworker"; then shell_path_block
  else case ":$PATH:" in *":$HPCW_HOME/bin:"*) ;; *) say "add it to your PATH:  export PATH=\"$HPCW_HOME/bin:\$PATH\"";; esac; fi
  say "next: hpcworker diag, then hpcworker probe, then hpcworker claude login (once, it covers the Remote Control question too), then hpcworker worker start"
}

# ---- profile -----------------------------------------------------------------------------------
cmd_profile() {
  local sub="${1:-show}"; shift || true
  case "$sub" in
    show) [ -s "$PROFILE" ] || die "no profile at $PROFILE"; prof_py show "$PROFILE";;
    path) echo "$PROFILE";;
    validate) [ -s "$PROFILE" ] || die "no profile at $PROFILE"; prof_py validate "$PROFILE";;
    set) [ $# -ge 2 ] || die "profile set KEY VALUE"; prof_py set "$PROFILE" "$1" "$2";;
    schema) prof_py schema;;
    *) die "profile: show|path|validate|set KEY VALUE|schema";;
  esac
}

# ---- discover: phase 1, on the login node, under a second of CPU, no container, no network ----------
# Measured on warrior (2026-09-29): `module -t avail` alone cost 0.76 s of CPU and sinfo 9 ms, so the
# module list comes from a glob of $MODULEPATH by default and the authoritative `module avail`, the
# QoS query (sacctmgr) and the Pyxis check run only with --full (install --discover uses --full).
# Values are escaped in place (hpcw_json_esc) rather than through subshells: forks were the other cost.
cmd_discover() {
  local full=0 save=0 a; for a in "$@"; do case "$a" in --full) full=1;; --save) save=1;; *) die "discover: unknown option $a";; esac; done
  local fam; fam=$(hpcw_detect_scheduler || true)
  local when host os glibc bins="" parts="" qos="" line c name tl av def first
  local tmux_p="" script_p="" curl_p="" home_fs home_dev nproc tmpdir="${TMPDIR:-}" stmp="${SLURM_TMPDIR:-}" lscr="${L_SCRATCH:-}" proxy="${https_proxy:-${HTTPS_PROXY:-}}" noproxy="${no_proxy:-${NO_PROXY:-}}"
  when=$(date -u +%FT%TZ); host=$(hostname -s 2>/dev/null); host=${host:-unknown}
  os=$(. /etc/os-release 2>/dev/null; echo "${PRETTY_NAME:-unknown}")
  glibc=$(ldd --version 2>/dev/null | head -1 | awk '{print $NF}')
  while IFS= read -r line; do [ -n "$line" ] || continue; c=${line##*/}; hpcw_json_esc line; bins="$bins${bins:+,}\"$c\":\"$line\""; done < <(command -v sbatch squeue scancel sacct sinfo srun qsub qstat qdel bsub bjobs bkill condor_submit 2>/dev/null)
  if [ "$fam" = slurm ]; then
    first=1
    while IFS='|' read -r name tl av; do [ -n "$name" ] || continue; def=false; case "$name" in *\*) def=true; name=${name%\*};; esac
      hpcw_json_esc name; hpcw_json_esc tl; hpcw_json_esc av
      parts="$parts$([ $first = 1 ] || printf ',')"; first=0; parts="$parts{\"name\":\"$name\",\"timelimit\":\"$tl\",\"avail\":\"$av\",\"default\":$def}"; done < <(timeout 10 sinfo -h -o '%P|%l|%a' 2>/dev/null | sort -u)
    [ $full = 1 ] && command -v sacctmgr >/dev/null 2>&1 && qos=$(timeout 10 sacctmgr -n -P show assoc where user="$USER" format=qos 2>/dev/null | tr ',' '\n' | sort -u | tr '\n' ' ')
  fi
  while IFS= read -r line; do case "${line##*/}" in tmux) tmux_p="$line";; script) script_p="$line";; curl) curl_p="$line";; esac; done < <(command -v tmux script curl 2>/dev/null)
  home_fs=$(stat -f -c %T "$HOME" 2>/dev/null); home_dev=$(df -P "$HOME" 2>/dev/null | awk 'NR==2{print $1}')
  nproc=$(ps -u "$USER" --no-headers 2>/dev/null | wc -l | tr -d ' '); nproc=${nproc:-0}
  proxy=$(printf '%s' "$proxy" | sed 's#//[^/@]*@#//#')   # a proxy URL may carry a password: never record it
  local v; for v in when host os glibc qos tmpdir stmp lscr proxy noproxy tmux_p script_p curl_p home_fs home_dev; do hpcw_json_esc "$v"; done
  local pyq="$PY"; hpcw_json_esc pyq; local usr="$USER"; hpcw_json_esc usr
  local out
  out=$(printf '{"version":%s,"full":%s,"when":"%s","host":"%s","user":"%s","os":"%s","glibc":"%s","scheduler":{"family":"%s","bins":{%s}},"runtime":%s,"partitions":[%s],"qos":"%s","env":{"TMPDIR":"%s","SLURM_TMPDIR":"%s","L_SCRATCH":"%s","https_proxy":"%s","no_proxy":"%s"},"tools":{"tmux":"%s","script":"%s","python3":"%s","curl":"%s"},"home_fs":"%s","home_dev":"%s","process_count":%s}' \
    "$VERSION" "$([ $full = 1 ] && echo true || echo false)" "$when" "$host" "$usr" "$os" "$glibc" "$fam" "$bins" "$(hpcw_detect_runtime "$([ $full = 1 ] && echo full)")" "$parts" "$qos" "$tmpdir" "$stmp" "$lscr" "$proxy" "$noproxy" "$tmux_p" "$script_p" "$pyq" "$curl_p" "$home_fs" "$home_dev" "$nproc")
  printf '%s\n' "$out"
  [ $save = 1 ] && { mkdir -p "$HPCW_HOME"; printf '%s\n' "$out" > "$HPCW_HOME/discover.json"; say "saved $HPCW_HOME/discover.json" >&2; }
  return 0
}

# ---- probe: one ten-minute job; results in a file, read from a file ----------------------------------
probe_env() { # what probe.sbatch needs, as shell assignments (no secrets; paths only); HPCW_CODE is where lib/ is (a shared copy's numbered directory, or ~/.hpcworker)
  printf 'HPCW_HOME=%q\nHPCW_CODE=%q\nRUNTIME=%q\nRUNTIME_BIN=%q\nRUNTIME_MODULE=%q\nIMAGE=%q\nIMAGE_TYPE=%q\nCLAUDE_BIN=%q\nBINDS=%q\nCLAUDE_CONFIG_DIR=%q\nCONSENT_MARKERS=%q\nHPCW_NODE_TMP_PATTERN=%q\nHPCW_NODE_TMP_VAR=%q\nHPCW_TMPDIR_IS_NODE_LOCAL=%q\nIPC_DIR=%q\nWORKSPACE=%q\n' \
    "$HPCW_HOME" "$SRC_ROOT" "$P_runtime" "$P_runtime_bin" "$P_runtime_module" "$P_image_store_path" "$P_image_store_type" "$P_claude_bin" "$(bind_args)" "$CCD" "$(consent_markers | paste -sd: -)" \
    "$P_node_tmp" "$P_node_tmp_var" "$([ "$P_tmpdir_is_node_local" = true ] && echo 1 || echo 0)" "$P_ipc_dir" "$WORKSPACES/worker"
}
bind_args() { local b out=""; for b in $P_binds; do out="$out${out:+ }-B $b"; done; printf '%s' "$out"; }
cmd_probe() {
  load_profile
  local show=0 wait_min=0 part="$P_partition_cpu" qos="$P_qos_cpu" time="00:10:00"
  while [ $# -gt 0 ]; do case "$1" in --show) show=1; shift;; --wait) wait_min="$2"; shift 2;; --partition) part="$2"; shift 2;; --qos) qos="$2"; shift 2;; --time) time="$2"; shift 2;; *) die "probe: unknown option $1";; esac; done
  local pd="$STATE_DIR/probe"
  if [ $show = 1 ]; then [ -s "$pd/probe.json" ] || die "no probe result yet at $pd/probe.json (submit one with: hpcworker probe)"; if [ -n "$PY" ]; then "$PY" -c 'import json,sys; print(json.dumps(json.load(open(sys.argv[1])), indent=1, sort_keys=True))' "$pd/probe.json"; else cat "$pd/probe.json"; fi; return 0; fi
  mkdir -p "$pd" "$WORKSPACES/worker" || die "cannot create $pd (is shared_state_dir right?)"
  chmod 700 "$pd" 2>/dev/null   # probe.json carries the plan and the e-mail of the sign-in
  # a probe already queued or running in this one directory: nothing submitted (two probes ran side
  # by side there on 2026-09-29 and neither finished); ONE squeue, on the recorded id only
  local id="" prev line; prev=$(cat "$pd/job.id" 2>/dev/null || true)
  if [ -n "$prev" ]; then
    line=$(adapter_status_many "$prev" 2>/dev/null | head -1)
    case "$(printf '%s' "$line" | cut -d'|' -f2)" in PENDING|RUNNING) id="$prev"; say "probe job $prev is still $(fmt_job_state "${line#*|}"); nothing submitted. It writes $pd/probe.json; read it with: hpcworker probe --show";; esac
  fi
  if [ -z "$id" ]; then
    probe_env > "$pd/probe.env"
    rm -f "$pd/probe.json"
    id=$(cd "$pd" && adapter_submit "$JOBS/probe.sbatch" -J hpcw_probe -p "$part" ${qos:+-q "$qos"} -t "$time" -D "$pd" --export=ALL,HPCW_PROBE_DIR="$pd" 2>&1) || die "the scheduler refused the probe: $id"
    echo "$id" > "$pd/job.id"
    say "probe job $id submitted ($part${qos:+/$qos}, $time). It writes $pd/probe.json; read it with: hpcworker probe --show"
    say "(the file appears when the job has run; do not poll the scheduler: 'hpcworker probe --wait 15' checks the file once a minute)"
  fi
  if [ "$wait_min" -gt 0 ] 2>/dev/null; then
    local i; for ((i = 0; i < wait_min; i++)); do [ -s "$pd/probe.json" ] && break; sleep 60; done
    [ -s "$pd/probe.json" ] && cmd_probe --show || say "no probe.json after $wait_min min; the job may still be queued (hpcworker worker list shows nothing for it; check $pd/probe_$id.out later)"
  fi
}

# ---- worker: start/stop submit and cancel once; status/list/log read files ---------------------------
run_dir() { echo "$STATE_DIR/$1"; }
stop_worker_run() { # RUN_DIR NAME: the stop file, then a wait for the job's own word, then the scheduler once
  # The job notices the stop file within a few seconds, ends its session (Remote Control
  # deregisters), cancels its queued successor, writes "stopped" and, last, touches stop.done. This
  # side waits up to HPCW_STOP_WAIT_S (45 s) for that file, reading the directory only; then ONE
  # squeue over both ids, and ONE scancel for whatever the scheduler still holds. An scancel that
  # lands while the job is leaving is harmless (its trap keeps the state stopped), but it is not
  # sent on a guess: measured on job 40501099, an scancel sent 15 s in overwrote a stop with "preempted".
  local rd="$1" name="$2" wait_s="${HPCW_STOP_WAIT_S:-45}" poll_s="${HPCW_STOP_POLL_S:-5}"
  local jid nid; jid=$(cat "$rd/job.id" 2>/dev/null || true); nid=$(next_job_of "$rd")
  local ids="$jid${jid:+${nid:+,}}$nid" waited=0 left=false already=false
  # a stop that went through before (stopped on record, its stop.done still there, no successor): no wait
  [ -e "$rd/stop.done" ] && [ -z "$nid" ] && [ "$(read_field "$rd/status.json" state 2>/dev/null)" = stopped ] && already=true
  # a worker already stopped keeps its stop.done, so a second "worker stop" is the same no-op
  if [ $already = false ]; then rm -f "$rd/stop.done"; touch "$rd/stop"; fi
  if [ -n "$ids" ] && [ $already = false ]; then
    while [ ! -e "$rd/stop.done" ] && [ "$waited" -lt "$wait_s" ]; do sleep "$poll_s"; waited=$((waited + poll_s)); done
  fi
  [ -e "$rd/stop.done" ] && left=true
  local alive="" line id st raw
  if [ -n "$ids" ]; then
    while IFS= read -r line; do
      [ -n "$line" ] || continue; id=${line%%|*}; st=$(printf '%s' "$line" | cut -d'|' -f2); raw=$(printf '%s' "$line" | cut -d'|' -f4)
      case "$st:$raw" in RUNNING:COMPLETING|RUNNING:CG|RUNNING:E) ;; PENDING:*|RUNNING:*) alive="$alive${alive:+ }$id";; esac   # leaving already: not cancelled again
    done < <(adapter_status_many "$ids" 2>/dev/null)
  fi
  [ -n "$alive" ] && adapter_cancel $alive
  rm -f "$rd/next.id"
  local how; if [ $already = true ]; then how="it had stopped before"; elif [ $left = true ]; then how="the job left on its own within ${waited}s"; elif [ -z "$ids" ]; then how="no job on record"; else how="no word from job ${jid} within ${wait_s}s"; fi
  say "worker $name: stopped ($how${alive:+; cancelled at the scheduler: $alive}; the chain ended)"
}
read_field() { # FILE KEY -> string value (python-free; strings and simple scalars)
  [ -s "$1" ] || return 1
  tr -d '\n' < "$1" | grep -o "\"$2\": *\(\"[^\"]*\"\|[^,}]*\)" | head -1 | sed -e "s/^\"$2\": *//" -e 's/^"//' -e 's/"$//'
}
next_job_of() { [ -s "$1/next.id" ] && awk '{print $1; exit}' "$1/next.id"; return 0; }
cmd_worker() {
  load_profile
  local sub="${1:-list}"; shift || true
  local name="worker" hours="$P_worker_defaults_hours" cpus="$P_worker_defaults_cpus" mem="$P_worker_defaults_mem" part="" qos="" pm="$P_worker_defaults_permission_mode" cap="$P_worker_defaults_capacity" as_json=0 lines=60 requeue="" chain=1
  while [ $# -gt 0 ]; do case "$1" in
    --name) name="$2"; shift 2;; --hours) hours="$2"; shift 2;; --cpus) cpus="$2"; shift 2;; --mem) mem="$2"; shift 2;;
    --partition) part="$2"; shift 2;; --qos) qos="$2"; shift 2;; --permission-mode) pm="$2"; shift 2;; --capacity) cap="$2"; shift 2;;
    --requeue) requeue=1; shift;; --no-requeue) requeue=0; shift;; --no-chain) chain=0; shift;; --chain) chain=1; shift;; --json) as_json=1; shift;; --lines) lines="$2"; shift 2;; *) die "worker: unknown option $1";; esac; done
  [[ "$name" =~ ^[A-Za-z0-9_-]{1,32}$ ]] || die "worker name must be letters, digits, _ or -"
  local rd; rd=$(run_dir "$name")
  case "$sub" in
    start)
      [[ "$hours" =~ ^[0-9]+$ ]] || die "--hours must be an integer"
      mkdir -p "$rd" "$WORKSPACES/$name" || die "cannot create $rd (is shared_state_dir right and writable?)"
      chmod 700 "$rd" 2>/dev/null
      # the running link and its queued successor, in ONE scheduler call: either alive means nothing to submit
      local jid nid ids line; jid=$(cat "$rd/job.id" 2>/dev/null || true); nid=$(next_job_of "$rd"); ids="$jid${jid:+${nid:+,}}$nid"
      if [ -n "$ids" ]; then
        while IFS= read -r line; do [ -n "$line" ] || continue; case "$(echo "$line" | cut -d'|' -f2)" in PENDING|RUNNING)
          say "worker $name already has job ${line%%|*} ($(fmt_job_state "${line#*|}")); nothing submitted"; return 0;; esac; done < <(adapter_status_many "$ids" 2>/dev/null)
      fi
      rm -f "$rd/stop" "$rd/stop.done" "$rd/next.id"
      # first-time notice, in plain words (the report's disclosure text, shortened for a terminal)
      if [ ! -e "$rd/.notice-shown" ]; then
        cat <<'EOT'
Before your first worker starts, know this: files the worker reads and every command's output go to
Anthropic; while you are connected through claude.ai or the app the transcript is also stored on
Anthropic's servers (a personal plan keeps it 30 days, or five years if you left model training on);
no BAA covers a personal plan, so keep patient data, FERPA records, IRB data and export-controlled
material out of the worker's directory; and Remote Control draws from your plan's shared usage limit.
EOT
        touch "$rd/.notice-shown"
      fi
      consent_given || say "note: Claude Code has not recorded your Remote Control answer yet (hpcworker claude login does that in one sitting); the job will start and wait for it"
      # HPCW_CODE: the job sources lib/ from where this command's code is; on a shared copy that is the
      # numbered directory, so a chain keeps the version it started on until the next worker start
      local kv=(RUN="$name" HOURS="$hours" CPUS="$cpus" MEM="$mem" PERMISSION_MODE="$pm" CAPACITY="$cap" HPCW_HOME="$HPCW_HOME" HPCW_CODE="$SRC_ROOT" CHAIN="$chain")
      [ -n "$part" ] && kv+=(PARTITION="$part"); [ -n "$qos" ] && kv+=(QOS="$qos"); [ -n "$requeue" ] && kv+=(REQUEUE="$requeue")
      prof_py render "$JOBS/worker.sbatch.tpl" "$PROFILE" "${kv[@]}" > "$rd/job.sbatch.part" || die "could not render the job (see the message above)"
      mv -f "$rd/job.sbatch.part" "$rd/job.sbatch"
      printf '%s\n' "${kv[@]}" > "$rd/params"
      local out; out=$(cd "$rd" && adapter_submit "$rd/job.sbatch" -D "$rd" 2>&1) || die "the scheduler refused the job: $out"
      echo "$out" > "$rd/job.id"; rm -f "$rd/node"
      say "worker $name submitted as job $out ($(grep -m1 -o -- '--partition=[^ ]*' "$rd/job.sbatch" | cut -d= -f2)${qos:+/$qos}, ${hours} h$([ "$chain" = 1 ] && echo ', always on: each job queues the next before it ends')). When it runs, open the link in 'hpcworker worker status --name $name' (or claude.ai/code, or Code in the Claude app)."
      say "watch it with: hpcworker worker status --name $name   (reads $rd/status.json; no scheduler polling)";;
    stop)
      [ -d "$rd" ] || die "no worker named $name"
      stop_worker_run "$rd" "$name";;
    status)
      [ -d "$rd" ] || die "no worker named $name (hpcworker worker start)"
      local sj="$rd/status.json" cj="$rd/connection.json"
      if [ $as_json = 1 ]; then printf '{"run":%s,"job":%s,"next_job":%s,"status":%s,"connection":%s}\n' "$(json_str "$name")" "$(json_str "$(cat "$rd/job.id" 2>/dev/null)")" "$(json_str "$(next_job_of "$rd")")" "$([ -s "$sj" ] && cat "$sj" || echo null)" "$([ -s "$cj" ] && cat "$cj" || echo null)"; return 0; fi
      local jid; jid=$(cat "$rd/job.id" 2>/dev/null || true)
      if [ ! -s "$sj" ]; then say "worker $name: job ${jid:-none} submitted, no status yet (queued, or not started); slurm output: $(ls -t "$rd"/slurm_*.out 2>/dev/null | head -1)"; return 0; fi
      local st upd age nid; st=$(read_field "$sj" state); upd=$(read_field "$sj" updated_at); nid=$(next_job_of "$rd")
      age=$(( $(date +%s) - $(stat -c %Y "$sj") ))
      say "worker $name: $st (job $(read_field "$sj" job) on $(read_field "$sj" node); heartbeat ${age}s ago at $upd)"
      say "  $(read_field "$sj" message)"
      say "  session: $(read_field "$sj" session_name)   open in Claude: $(read_field "$sj" environment_url)"
      say "  signed in: $(tr -d '\n' < "$sj" | grep -o '"loggedIn": *\(true\|false\)' | head -1 | grep -o 'true\|false') $(tr -d '\n' < "$sj" | grep -o '"subscriptionType": *"[^"]*"' | head -1 | cut -d'"' -f4); consent recorded: $(read_field "$sj" consent) ($(read_field "$sj" consent_source)); restarts: $(read_field "$sj" restarts)"
      say "  always on: $(read_field "$sj" chain); next job: ${nid:-none} $(read_field "$sj" chain_note)"
      [ "$age" -gt 120 ] && case "$st" in stopped) ;; preempted) [ -n "$nid" ] && say "  (this job ended; job $nid is queued to take over)" || say "  (this job ended with no successor; hpcworker worker start --name $name starts it again)";; *) say "  (no heartbeat for ${age}s: the job may have ended; ${nid:+job $nid is queued to take over; }hpcworker worker start --name $name starts it again if not)";; esac
      return 0;;
    list)
      local d any=0; for d in "$STATE_DIR"/*/; do [ -s "$d/job.id" ] || continue; local n; n=$(basename "$d"); [ "$n" = probe ] && continue; any=1
        local sj="$d/status.json" nid; nid=$(next_job_of "$d"); if [ -s "$sj" ]; then printf '%-16s job %-10s %-12s %s (%ss ago)%s\n' "$n" "$(cat "$d/job.id")" "$(read_field "$sj" state)" "$(read_field "$sj" node)" "$(( $(date +%s) - $(stat -c %Y "$sj") ))" "${nid:+  next $nid}"; else printf '%-16s job %-10s %s\n' "$n" "$(cat "$d/job.id")" "no status yet"; fi; done
      [ $any = 1 ] || say "no workers yet (hpcworker worker start)";;
    log)
      [ -d "$rd" ] || die "no worker named $name"
      say "--- rc.log (Claude Code's terminal, last $lines lines, codes stripped)"
      [ -s "$rd/rc.log" ] && tail -c 20000 "$rd/rc.log" | hpcw_strip_codes | grep -v '^[[:space:]]*$' | tail -n "$lines"
      say "--- job output"; ls -t "$rd"/slurm_*.out "$rd"/pbs.out "$rd"/lsf_*.out 2>/dev/null | head -1 | xargs -r tail -n 30;;
    *) die "worker: start|stop|status|list|log";;
  esac
}

# ---- claude: the student's own sign-in and consent, through Claude Code's own flow ---------------------
# The seed adds three settings to the student's .claude.json (never a credential): onboarding done, a
# theme, and trust for the worker's workspace. It only ADDS keys, writes nothing when they are already
# there, refuses to touch a file it cannot parse, and runs only in the login and consent sittings.
# shellcheck disable=SC2016
SEED='const fs=require("fs"),p=process.env.CLAUDE_CONFIG_DIR+"/.claude.json";let j={},had=false;try{const t=fs.readFileSync(p,"utf8");had=true;j=JSON.parse(t)}catch(e){if(had){console.log("unreadable");process.exit(0)}}if(!j||typeof j!=="object")j={};let ch=false;if(j.hasCompletedOnboarding!==true){j.hasCompletedOnboarding=true;ch=true}if(!j.theme){j.theme="dark";ch=true}j.projects=j.projects||{};const d=process.cwd();const pr=j.projects[d]||{};if(pr.hasTrustDialogAccepted!==true){pr.hasTrustDialogAccepted=true;j.projects[d]=pr;ch=true}if(ch){fs.writeFileSync(p+".part",JSON.stringify(j,null,1),{mode:0o600});fs.renameSync(p+".part",p)}console.log(ch?"seeded":"already");'
# the one sitting for a person at a terminal: `claude auth login`, then, if neither Claude Code nor a
# marker file records the Remote Control answer, `claude remote-control` once so the student answers y
# and leaves with Ctrl-C. Runs inside the image. $5 is the colon-separated marker list.
# shellcheck disable=SC2016
LOGIN_SITTING='cd "$1" || exit 1; claude="$2"; cfg="$CLAUDE_CONFIG_DIR/.claude.json"; markers="${5:-}"
seen() { grep -q -E "\"remoteDialogSeen\": *true" "$cfg" 2>/dev/null; }
marked() { local IFS=":" m; for m in $markers; do [ -n "$m" ] && [ -e "$m" ] && return 0; done; return 1; }
if [ "$4" = login ]; then
  "$claude" auth login; rc=$?
  [ $rc = 0 ] || { echo; echo "Claude Code'"'"'s sign-in did not complete (exit $rc). Run hpcworker claude login again."; exit $rc; }
else
  "$claude" auth status 2>/dev/null | grep -q "\"loggedIn\": *true" || { echo "You are not signed in on the cluster yet; run: hpcworker claude login (it asks this question too)"; exit 2; }
fi
node -e "$3" >/dev/null 2>&1 || true
if seen; then echo; echo "Your Remote Control answer is already on record with Claude Code; nothing more to do."; exit 0; fi
if marked; then echo; echo "Your Remote Control answer is already on record (the panel or hpcworker claude consent recorded it); nothing more to do."; exit 0; fi
echo; echo "One more question, once. Claude Code will now ask \"Enable Remote Control? (y/n)\": type y."
echo "When you see \"Connected\" and a session line, press Ctrl+C to leave; your worker takes it from here."; echo; sleep 2
"$claude" remote-control
echo; if seen; then echo "Recorded by Claude Code: your workers may host Remote Control. Next: hpcworker worker start"; else echo "Claude Code did not record the answer this time; run: hpcworker claude consent"; fi'
in_job_exec() { # MODE [args]: inside an allocation, run claude inside the image in the worker workspace. MODE login|consent|run
  load_profile
  refuse_heavy "claude" "hpcworker claude login|consent|status"
  local mode="${1:-run}"; shift || true
  local ws="$WORKSPACES/${HPCW_WS:-worker}"; mkdir -p "$ws"
  export HPCW_BINDS; HPCW_BINDS=$(bind_args)
  # node-local temp for this short allocation, removed when it ends (the guarded helper: per-job directories only)
  HPCW_NODE_TMP=$(hpcw_node_tmp); export HPCW_NODE_TMP; trap 'hpcw_rm_own_tmp "$HPCW_NODE_TMP"' EXIT
  [ "$P_xdg_runtime_fix" = true ] && { mkdir -p "$P_ipc_dir/$USER/xdg" && chmod 700 "$P_ipc_dir/$USER/xdg"; }
  local -a ce; mapfile -t ce < <(clean_env_args)
  case "$mode" in
    login|consent)
      # the student asked for a sign-in: Claude Code's own folder may be created for it (settings, not credentials)
      mkdir -p "$CCD" && chmod 700 "$CCD" 2>/dev/null
      if [ "$mode" = login ] && [ "${1:-}" = "--plain" ]; then
        # the panel's sitting (lib/login_plain.sh): Claude Code under a pty inside the image, a line
        # contract on stdout, the transcript under /dev/shm for this sitting only, removed with it
        local lp="$LIB/login_plain.sh" sd="$P_ipc_dir/$USER/hpcw_login_$(hpcw_job_id)"
        [ -s "$lp" ] || { echo "HPCW-LOGIN-FAILED: the sign-in helper is missing from the installed copy; run hpcworker install again"; return 3; }
        mkdir -p "$sd" && chmod 700 "$sd" || { echo "HPCW-LOGIN-FAILED: this compute node has no writable $P_ipc_dir for the sitting"; return 3; }
        trap "hpcw_rm_own_tmp '$HPCW_NODE_TMP' '$sd'" EXIT
        hpcw_runtime_exec "$P_image_store_path" "${ce[@]}" bash -c "$(cat "$lp")" _ "$ws" "$P_claude_bin" "$sd" "$SEED" 600
        return $?
      fi
      hpcw_runtime_exec "$P_image_store_path" "${ce[@]}" bash -c "$LOGIN_SITTING" _ "$ws" "$P_claude_bin" "$SEED" "$mode" "$(consent_markers | paste -sd: -)";;
    run) hpcw_runtime_exec "$P_image_store_path" "${ce[@]}" bash -c 'cd "$1" && exec "$2" "${@:3}"' _ "$ws" "$P_claude_bin" "$@";;
    *) die "_exec: login|consent|run ...";;
  esac
}
# The panel's sign-in: no prose, a line contract the extension reads over the portal's shell. Prints
# "HPCW-LOGIN: ..." progress lines, "AUTH-URL: <link>" once Claude Code shows Anthropic's authorize
# page, "SIGNED-IN <email> <plan>" and exits 0 when `claude auth status` says loggedIn, or one
# "HPCW-LOGIN-FAILED: <sentence>" and exits 1. The code the student pastes goes through the shell to
# Claude Code's terminal and is never echoed. A worker heartbeat younger than 15 min that already
# says loggedIn answers without an allocation.
plain_login() {
  local self="$HPCW_HOME/bin/hpcworker"; [ -x "$self" ] || self="$SELF"
  local best="" bt=0 d m age
  for d in "$STATE_DIR"/*/; do [ -s "$d/status.json" ] || continue; m=$(stat -c %Y "$d/status.json"); [ "$m" -gt "$bt" ] && { bt=$m; best="$d/status.json"; }; done
  if [ -n "$best" ]; then
    age=$(( $(date +%s) - bt ))
    if [ "$age" -lt 900 ] && tr -d '\n' < "$best" | grep -q '"loggedIn": *true'; then
      say "HPCW-LOGIN: already signed in (a worker confirmed it ${age}s ago)"
      say "SIGNED-IN $(tr -d '\n' < "$best" | grep -o '"email": *"[^"]*"' | head -1 | cut -d'"' -f4) $(tr -d '\n' < "$best" | grep -o '"subscriptionType": *"[^"]*"' | head -1 | cut -d'"' -f4)"
      return 0
    fi
  fi
  say "HPCW-LOGIN: requesting a compute node"
  # a bounded wait for the node (the panel must get a sentence, never a hang); 20 min of wall time for a 10 min sitting
  HPCW_ALLOC_WAIT_S=600 adapter_interactive "$P_partition_cpu" "$P_qos_cpu" 00:20:00 hpcw_claude 2G -- "$self" _exec login --plain
  plain_login_outcome $?
}
plain_login_outcome() { # the sitting's exit: 0 signed in; 3 it already printed its own failure line; anything else is the scheduler's
  case "$1" in
    0) return 0;;
    3) return 1;;
    *) say "HPCW-LOGIN-FAILED: the cluster did not give a compute node in time, or the session ended early; try again in a minute"; return 1;;
  esac
}
cmd_claude() {
  load_profile
  local sub="${1:-status}"; shift || true
  local self="$HPCW_HOME/bin/hpcworker"; [ -x "$self" ] || self="$SELF"
  case "$sub" in
    login)
      if [ "${1:-}" = "--plain" ]; then plain_login; return $?; fi
      cat <<'EOT'
Sign in to Claude on the cluster, in one sitting.

Claude Code itself does the sign-in, in your own account on a compute node, and keeps its
credential in its own folder; hpcworker never sees, copies or stores it. Claude will print a link:
open it in your browser, sign in with your own Claude account (Pro or Max), press Authorize, then
paste the code the page shows back into this terminal. Use `claude auth login` only: a long-lived
token from the other command cannot host Remote Control.

Then, unless you answered it before, Claude Code asks once "Enable Remote Control? (y/n)": type y,
wait for "Connected", press Ctrl+C. Claude Code records that answer itself; your workers use it.
While connected, the session transcript (your messages, Claude's answers, tool activity) is stored
on Anthropic's servers: keep confidential data out of these sessions.
EOT
      say "(asking the scheduler for a short session on a compute node: $P_partition_cpu${P_qos_cpu:+/$P_qos_cpu}, 30 min; this can take a minute)"
      adapter_interactive "$P_partition_cpu" "$P_qos_cpu" 00:30:00 hpcw_claude 2G -- "$self" _exec login || true
      say
      case "$(consent_source)" in
        claude_config) say "Done: signed in, and Claude Code recorded your Remote Control answer. Next: hpcworker worker start";;
        marker) say "Done: signed in; your Remote Control answer is on record (marker file). Next: hpcworker worker start";;
        *) say "Signed in (if the steps above completed). Claude Code has not recorded the Remote Control answer yet: run hpcworker claude consent";;
      esac;;
    consent)
      case "$(consent_source)" in
        claude_config) say "Claude Code already recorded your Remote Control answer ($CCD/.claude.json); nothing to do. (hpcworker claude login covers this question in the same sitting as the sign-in.)"; return 0;;
        marker) say "your Remote Control answer is already recorded ($(consent_markers | while read -r m; do [ -e "$m" ] && echo "$m"; done | head -1)); nothing to do"; return 0;;
      esac
      cat <<'EOT'
Remote Control: Claude Code asks once whether to enable it. `hpcworker claude login` asks this in the
same sitting as the sign-in; this command repeats only that question, for a login that skipped it.

When Claude Code asks "Enable Remote Control? (y/n)" type y; if it asks to trust the folder, type y.
When you see "Connected" and the session line, press Ctrl+C to leave. Claude Code records the answer
itself; from then on your worker hosts Remote Control for you. While connected, the transcript is
stored on Anthropic's servers: keep confidential data out of these sessions.
EOT
      say "(asking the scheduler for a short session on a compute node; this can take a minute)"
      adapter_interactive "$P_partition_cpu" "$P_qos_cpu" 00:30:00 hpcw_claude 2G -- "$self" _exec consent || true
      if hpcw_consent_seen "$CCD"; then say "Recorded by Claude Code: your workers may now host Remote Control."; return 0; fi
      # the fallback marker, only when Claude Code did not write its own record and only on the student's word
      local a; printf '\nClaude Code did not record the answer. Did you answer y to "Enable Remote Control?" [y/N] '; read -r a
      if [[ "$a" =~ ^[Yy] ]]; then mkdir -p "$HPCW_HOME"; date -u +%FT%TZ > "$HPCW_HOME/consent"; say "recorded in $HPCW_HOME/consent (fallback marker): your workers may now host Remote Control"; else say "not recorded; run this again when you want it"; fi;;
    status)
      local alloc=0; [ "${1:-}" = "--alloc" ] && alloc=1
      local present=false; [ -d "$CCD" ] && present=true
      local src; src=$(consent_source)
      # the newest worker heartbeat carries Claude Code's own answer; otherwise ask inside a short allocation
      local best="" bt=0 d m; for d in "$STATE_DIR"/*/; do [ -s "$d/status.json" ] || continue; m=$(stat -c %Y "$d/status.json"); [ "$m" -gt "$bt" ] && { bt=$m; best="$d/status.json"; }; done
      local age=$(( $(date +%s) - bt ))
      if [ -n "$best" ] && [ "$age" -lt 900 ] && [ $alloc = 0 ]; then
        say "sign-in folder present: $present; consent recorded: $src; a worker said ${age}s ago: $(tr -d '\n' < "$best" | grep -o '"claude_login": *{[^}]*}' | sed 's/^"claude_login": *//')"
      elif [ $alloc = 1 ] || [ -z "$best" ]; then
        say "sign-in folder present: $present; consent recorded: $src; asking Claude Code inside a short allocation..."
        local out; out=$(adapter_run_short "$P_partition_cpu" "$P_qos_cpu" 00:10:00 hpcw_claude 2G -- "$self" _exec run auth status 2>/dev/null)
        say "claude auth status: loggedIn=$(printf '%s' "$out" | grep -o '"loggedIn": *\(true\|false\)' | head -1 | grep -o 'true\|false') $(printf '%s' "$out" | grep -o '"subscriptionType": *"[^"]*"' | head -1) $(printf '%s' "$out" | grep -o '"email": *"[^"]*"' | head -1)"
      else say "sign-in folder present: $present; consent recorded: $src; last worker heartbeat is ${age}s old (hpcworker claude status --alloc asks Claude Code now)"; fi;;
    *) die "claude: login|consent|status [--alloc]";;
  esac
}

# ---- diag: every assumption as one line ------------------------------------------------------------
DIAG_FIRST=1; DIAG_JSON=0
emit() { # ID OK(true|false) WARN(true|false) SKIP(true|false) DETAIL
  if [ $DIAG_JSON = 1 ]; then [ $DIAG_FIRST = 1 ] || printf ','; DIAG_FIRST=0; printf '{"id":%s,"ok":%s,"warn":%s,"skip":%s,"detail":%s}' "$(json_str "$1")" "$2" "$3" "$4" "$(json_str "$5")"
  else local tag=FAIL; [ "$2" = true ] && tag=ok; [ "$3" = true ] && tag=warn; [ "$4" = true ] && tag=skip; printf '%-5s %-24s %s\n' "$tag" "$1" "$5"; fi
}
cmd_diag() {
  [ "${1:-}" = "--json" ] && DIAG_JSON=1
  [ $DIAG_JSON = 1 ] && printf '{"version":%s,"host":%s,"user":%s,"nodes":[' "$VERSION" "$(json_str "$(hostname -s)")" "$(json_str "$USER")"
  local miss="" f; for f in bin/hpcworker lib/profile.py lib/login_plain.sh lib/adapters/common.sh lib/adapters/slurm.sh jobs/worker.sbatch.tpl jobs/probe.sbatch VERSION; do [ -s "$SRC_ROOT/$f" ] || miss="$miss $f"; done
  grep -q $'\r' "$SRC_ROOT/bin/hpcworker" "$SRC_ROOT"/lib/*.sh "$SRC_ROOT"/lib/adapters/*.sh "$SRC_ROOT"/jobs/* 2>/dev/null && miss="$miss (CR line endings: run hpcworker install)"
  if [ -n "$miss" ]; then emit hpcw.scripts false false false "missing or damaged:$miss"
  elif shared_mode && ! is_launcher "$HPCW_HOME/bin/hpcworker"; then emit hpcw.scripts true true false "hpcworker $VERSION, the shared copy at $SHARED_DIR ($SRC_ROOT); $HPCW_HOME/bin/hpcworker is not its launcher: hpcworker install"
  elif shared_mode; then emit hpcw.scripts true false false "hpcworker $VERSION, the shared copy at $SHARED_DIR ($SRC_ROOT); profile and log in $HPCW_HOME"
  else emit hpcw.scripts true false false "hpcworker $VERSION in $SRC_ROOT"; fi
  if [ -n "$PY" ]; then emit hpcw.python3 true false false "$PY ($("$PY" -c 'import sys; print("%d.%d" % sys.version_info[:2])' 2>/dev/null))"; else emit hpcw.python3 false false false "no python3 on this node; the system package is enough"; fi
  if [ ! -s "$PROFILE" ]; then emit hpcw.profile false false false "no profile at $PROFILE (hpcworker install --profile NAME or --discover)"; [ $DIAG_JSON = 1 ] && echo ']}'; return 1; fi
  local v; if v=$(prof_py validate "$PROFILE" 2>&1); then emit hpcw.profile true false false "$v"; else emit hpcw.profile false false false "$(echo "$v" | tr '\n' ' ')"; [ $DIAG_JSON = 1 ] && echo ']}'; return 1; fi
  load_profile
  local fam; fam=$(hpcw_detect_scheduler || true)
  if [ "$fam" = "$P_scheduler" ]; then emit hpcw.scheduler true false false "$fam client present"; else emit hpcw.scheduler false false false "profile says $P_scheduler, this node shows '$fam'"; fi
  if in_job; then emit hpcw.login-node true true false "running inside job $(hpcw_job_id) (fine for a check; the CLI is meant for the login node)"
  else local n; n=$(ps -u "$USER" --no-headers 2>/dev/null | wc -l | tr -d ' '); local cap="$P_login_node_rules_max_processes"
    if [ "$n" -ge "$cap" ]; then emit hpcw.login-node false false false "$n processes of yours on $(hostname -s); the site's cap is $cap"; elif [ "$n" -ge $((cap * 3 / 4)) ]; then emit hpcw.login-node true true false "$n processes of yours on $(hostname -s) (cap $cap)"; else emit hpcw.login-node true false false "$n processes of yours on $(hostname -s) (cap $cap)"; fi; fi
  if [ "$fam" = slurm ]; then
    local parts; parts=$(adapter_partitions | cut -d'|' -f1 | sed 's/\*$//' | tr '\n' ' ')
    local want="$P_partition_cpu $P_partition_cpu_batch $P_partitions_gpu" bad="" p; for p in $want; do case " $parts " in *" $p "*) ;; *) bad="$bad $p";; esac; done
    if [ -z "$bad" ]; then emit hpcw.partitions true false false "profile partitions exist: $want"; else emit hpcw.partitions false false false "not in sinfo:$bad (have: $parts)"; fi
    local q; q=$(adapter_my_qos); local qb="" x; for x in $P_qos_cpu $P_qos_batch; do case " $q " in *" $x "*) ;; *) qb="$qb $x";; esac; done
    if [ -z "$q" ]; then emit hpcw.qos true true false "QoS list not readable (sacctmgr); profile wants $P_qos_cpu/$P_qos_batch"; elif [ -z "$qb" ]; then emit hpcw.qos true false false "QoS available: $q"; else emit hpcw.qos false false false "QoS not on your account:$qb (have: $q)"; fi
  fi
  if [ -n "$P_runtime_bin" ] && [ -x "$P_runtime_bin" ]; then emit hpcw.runtime true false false "$P_runtime_bin"
  elif [ -n "$P_runtime_module" ] && hpcw_detect_runtime | grep -q "\"$P_runtime_module\""; then emit hpcw.runtime true false false "module $P_runtime_module listed (binary not on this node's PATH; the probe checks the compute node)"
  else emit hpcw.runtime true true false "runtime $P_runtime not visible from here (bin '$P_runtime_bin', module '$P_runtime_module'); login nodes may hide it: the probe decides"; fi
  if [ "$P_image_store_type" = sandbox ]; then if [ -x "$P_image_store_path/usr/local/bin/node" ]; then emit hpcw.image true false false "sandbox readable: $P_image_store_path"; else emit hpcw.image false false false "sandbox not readable or has no node: $P_image_store_path (build one: jobs/build_image.sbatch, or ask your lead)"; fi
  else if [ -r "$P_image_store_path" ]; then emit hpcw.image true false false "SIF readable: $P_image_store_path"; else emit hpcw.image false false false "SIF missing: $P_image_store_path (jobs/build_image.sbatch builds it)"; fi; fi
  if [ -r "$P_claude_bin" ] || [ "${P_claude_bin#/usr/local/bin/}" != "$P_claude_bin" ]; then emit hpcw.claude-bin true false false "$P_claude_bin"; else emit hpcw.claude-bin false false false "not readable from here: $P_claude_bin"; fi
  if mkdir -p "$STATE_DIR" 2>/dev/null && [ -w "$STATE_DIR" ]; then emit hpcw.state-dir true false false "$STATE_DIR ($(stat -f -c %T "$STATE_DIR" 2>/dev/null))"; else emit hpcw.state-dir false false false "cannot write $STATE_DIR"; fi
  case "$P_node_tmp" in \$TMPDIR*) [ "$P_tmpdir_is_node_local" = true ] && emit hpcw.node-tmp true false false "node_tmp under \$TMPDIR (profile says it is node-local)" || emit hpcw.node-tmp false false false "node_tmp uses \$TMPDIR but the profile does not trust it";; *) emit hpcw.node-tmp true false false "$P_node_tmp (TMPDIR here is ${TMPDIR:-unset}, which is not trusted)";; esac
  # the sign-in: presence of Claude Code's folder only; the credential is never read
  if [ -d "$CCD" ]; then emit hpcw.claude-config-dir true false false "$CCD present (Claude Code's own; the credential is not read)"; else emit hpcw.claude-config-dir false false false "no Claude Code folder at $CCD yet: hpcworker claude login"; fi
  case "$(consent_source)" in
    claude_config) emit hpcw.consent true false false "Remote Control answer recorded by Claude Code";;
    marker) emit hpcw.consent true false false "Remote Control answer recorded (fallback marker file)";;
    *) emit hpcw.consent false false false "not recorded yet: hpcworker claude login asks it in the same sitting as the sign-in (a worker waits for it)";;
  esac
  local left; left=$(find /tmp -maxdepth 1 -user "$USER" \( -name 'hpcworker*' -o -name 'rootfs-*' -o -name 'build-temp-*' -o -name 'apptainer*' -o -name 'singularity*' \) 2>/dev/null | head -5 | tr '\n' ' ')
  if in_job; then emit hpcw.login-tmp true false true "not a login node"; elif [ -z "$left" ]; then emit hpcw.login-tmp true false false "nothing of ours under /tmp on $(hostname -s)"; else emit hpcw.login-tmp false false false "files of yours under /tmp on the login node: $left (remove them yourself; the framework never deletes outside its run directories, and the center asks that /tmp here stays free)"; fi
  local pj="$STATE_DIR/probe/probe.json"
  if [ -s "$pj" ]; then local eg li; eg=$(read_field "$pj" class); li=$(tr -d '\n' < "$pj" | grep -o '"loggedIn": *\(true\|false\)' | head -1 | grep -o 'true\|false')
    local age=$(( ( $(date +%s) - $(stat -c %Y "$pj") ) / 86400 ))
    if [ "$eg" = none ]; then emit hpcw.probe false false false "probe ${age}d old: compute nodes cannot reach api.anthropic.com/platform.claude.com; ask your center for an HTTP proxy"
    else emit hpcw.probe true "$([ "$age" -gt 30 ] && echo true || echo false)" false "probe ${age}d old on $(read_field "$pj" hostname): egress $eg, claude $(read_field "$pj" claude_version), loggedIn=$li, tmux=$(tr -d '\n' < "$pj" | grep -o '"tmux": *{[^}]*}' | grep -o '"present": *\(true\|false\)' | grep -o 'true\|false'), sbatch on node=$(read_field "$pj" sbatch_on_node)"; fi
  else emit hpcw.probe false true false "no probe yet: hpcworker probe (up to ten minutes on a compute node)"; fi
  local d any=0; for d in "$STATE_DIR"/*/; do [ -s "$d/status.json" ] || continue; local n; n=$(basename "$d"); [ "$n" = probe ] && continue; any=1
    local st age nid; st=$(read_field "$d/status.json" state); age=$(( $(date +%s) - $(stat -c %Y "$d/status.json") )); nid=$(next_job_of "$d")
    case "$st" in
      online) [ "$age" -lt 120 ] && emit "hpcw.worker.$n" true false false "online: job $(read_field "$d/status.json" job) on $(read_field "$d/status.json" node), session $(read_field "$d/status.json" session_name) $(read_field "$d/status.json" environment_url)${nid:+; next job $nid queued}" || emit "hpcw.worker.$n" false false false "last said online ${age}s ago; the job has probably ended (${nid:+job $nid is queued to take over; }hpcworker worker start --name $n)";;
      starting|running) [ "$age" -lt 120 ] && emit "hpcw.worker.$n" true true false "$st: $(read_field "$d/status.json" message)" || emit "hpcw.worker.$n" false false false "stale ($st, ${age}s): $(read_field "$d/status.json" message)";;
      need_consent|need_login) emit "hpcw.worker.$n" false false false "$st: $(read_field "$d/status.json" message)";;
      stopped) emit "hpcw.worker.$n" true false true "$st ${age}s ago";;
      preempted) emit "hpcw.worker.$n" true "$([ -n "$nid" ] && echo true || echo false)" "$([ -n "$nid" ] && echo false || echo true)" "$st ${age}s ago${nid:+; job $nid is queued to take over}";;
      *) emit "hpcw.worker.$n" false false false "$st: $(read_field "$d/status.json" message)";;
    esac; done
  [ $any = 1 ] || emit hpcw.worker true false true "no worker started yet"
  [ $DIAG_JSON = 1 ] && echo ']}'
  return 0
}

# sourced by the tests for its functions; dispatch only when run
if [ "${BASH_SOURCE[0]}" = "$0" ]; then
case "${1:-help}" in
  install) shift; cmd_install "$@";;
  profile) shift; cmd_profile "$@";;
  discover) shift; cmd_discover "$@";;
  probe) shift; cmd_probe "$@";;
  worker) shift; cmd_worker "$@";;
  claude) shift; cmd_claude "$@";;
  diag) shift; cmd_diag "$@";;
  _exec) shift; in_job_exec "$@";;
  version) if shared_mode; then printf '%s (shared copy at %s)\n' "$(read_version "$SHARED_DIR/VERSION" || echo "$VERSION")" "$SHARED_DIR"; else echo "$VERSION"; fi;;
  help|-h|--help) awk 'NR > 1 && /^# Rules/ { exit } NR > 1' "$SELF";;
  *) die "unknown command '$1' (hpcworker help)";;
esac
fi
