#!/usr/bin/env bash
# tf-goal.sh — unattended GOAL runner / supervisor for TechieFlow (rule: .tfcore/tasks/_yolo-mode.md)
#
# Runs a goal headless in YOLO mode and keeps it running until the agent itself
# declares the goal done — surviving the subscription 5-hour / weekly usage limits (moves to the
# next model in the tier's fallback chain, or sleeps until the stated reset + a buffer when there
# is none, then RESUMES THE SAME SESSION), crashes, and "I stopped early to ask you something".
#
#   bash .tfcore/utils/tf-goal.sh [options] <app-dir> "<goal text>"
#   bash .tfcore/utils/tf-goal.sh [options] <app-dir> @goal.md
#
# Options
#   --harness claude|opencode   default: claude
#   --model <id>                claude: --model; opencode: -m
#   --tier <tier>               which routing tier this run belongs to, for the fallback chain
#                               (default standard) — .tfcore/routing.yaml `fallbacks:`
#   --no-fallback               never switch model on a usage limit; sleep until the reset instead
#                               (the behaviour before 2026-09-10)
#   --buffer-min <n>            minutes added after a stated limit-reset time (default 15)
#   --probe-min <n>             limit hit but NO reset time parseable → fire a one-turn probe every n
#                               minutes until the API answers again, then resume (default 15)
#   --probe-max-hours <n>       give up probing and resume anyway after n hours (default 8)
#   --default-wait-min <n>      (legacy) only used to stamp resume_at in goal.json when probing
#   --max-cycles <n>            give up after n launches (default 60)
#   --idle-retry-sec <n>        pause before re-prompting an agent that stopped without finishing (default 30)
#   --stall-min <n>             a cycle whose output has not grown for n minutes is killed and
#                               re-prompted (default 15; 0 disables) — MISS-TechieFlow-20260905-22
#   --resume                    continue the last goal run recorded in .tfcore/.session/goal.json
#   --fresh                     with --resume: keep the goal and the cycle count but start a NEW
#                               harness session (the old one hung on continue). The supervisor does
#                               this by itself after two stalled resumes in a row.
#   --dry-run                   print the commands, run nothing
#
# Exit codes: 0 goal complete · 3 agent declared the goal BLOCKED (owner input
# needed) · 4 max cycles reached · 5 the provider refuses the model (a monthly or
# balance limit OpenCode reports only in its own log) · 2 usage error · 130 stopped
# by Ctrl-C / kill (the harness child is stopped too; `--resume` continues the same session).
#
# Test-only knobs (tests/goal/run.sh): TF_GOAL_FAKE_CMD="<shell>" replaces the harness
# command; TF_GOAL_STALL_SEC / TF_GOAL_STALL_TICK set the stall clock in seconds.
#
# Files (all under <app-dir>/.tfcore/.session/, never committed):
#   yolo.json        YOLO flag (tf-yolo.sh on --source goal) — the hook reads it
#   goal.json        supervisor state: cycle, session_id, last_reason, resume_at
#   goal.log         everything the harness printed, all cycles, timestamped
#   goal-done.json   the agent's completion sentinel (tf-yolo.sh done [complete|blocked])
#
# WHEN A MODEL RUNS OUT (added 2026-09-10, MISS-TechieFlow-20260910-01). A usage limit used to mean
# one thing: sleep until the reset. It now means: park that model in the per-machine cooldown file
# (tf-model-pick.sh), ask the tier for the next model in its `fallbacks:` chain, re-bind so the
# sub-agents move too, and start the next cycle straight away on that model. Only when the whole
# chain is parked does the supervisor sleep, exactly as before. If the fallback is limited as well
# the next cycle parks that one too, so an account-wide limit — every Claude model at once — costs
# one wasted cycle and then behaves as it always did. `--no-fallback` turns all of it off.
#
# How the agent is told to finish: the prompt preamble (below) instructs it to run
# `bash .tfcore/utils/tf-yolo.sh done complete "<summary>"` when the goal is met,
# or `... done blocked "<why>"` when only the owner can unblock it. A cycle that
# ends with neither sentinel is classified from its output: USAGE LIMIT → sleep
# until reset+buffer; CRASH/API error → exponential backoff (2m → 30m);
# otherwise the agent simply stopped (asked a question / summarised) → re-prompt.
#
# Harness invocation (verified flags — docs/Capability-Matrix.md):
#   claude   -p "<prompt>" --permission-mode bypassPermissions --output-format stream-json --verbose
#            resume: claude -p --resume <session_id> "<continue>"   (fallback: --continue)
#   opencode run --auto "<prompt>"      resume: opencode run --auto -c "<continue>"
#            exit 2). The approval policy is set as a config override instead.
#   Override command lines with TF_GOAL_CLAUDE_FLAGS / TF_GOAL_OPENCODE_FLAGS /

set -u
# GNU-only commands (setsid, stat -c, date -d) go through the shim, so a stock Mac runs this too.
source "$(dirname "${BASH_SOURCE[0]}")/tf-portable.sh"

HARNESS="claude"; MODEL=""; BUFFER_MIN=15; DEFAULT_WAIT_MIN=60; MAX_CYCLES=60; IDLE_RETRY=30
PROBE_MIN=15; PROBE_MAX_H=8; STALL_MIN=15
TIER="standard"; FALLBACK=1
RESUME=0; FRESH=0; DRY=0
while [[ $# -gt 0 ]]; do
  case "$1" in
    --harness) HARNESS="$2"; shift 2 ;;
    --model) MODEL="$2"; shift 2 ;;
    --tier) TIER="$2"; shift 2 ;;
    --no-fallback) FALLBACK=0; shift ;;
    --buffer-min) BUFFER_MIN="$2"; shift 2 ;;
    --default-wait-min) DEFAULT_WAIT_MIN="$2"; shift 2 ;;
    --probe-min) PROBE_MIN="$2"; shift 2 ;;
    --probe-max-hours) PROBE_MAX_H="$2"; shift 2 ;;
    --max-cycles) MAX_CYCLES="$2"; shift 2 ;;
    --idle-retry-sec) IDLE_RETRY="$2"; shift 2 ;;
    --stall-min) STALL_MIN="$2"; shift 2 ;;
    --resume) RESUME=1; shift ;;
    --fresh) FRESH=1; shift ;;
    --dry-run) DRY=1; shift ;;
    -h|--help) sed -n '2,48p' "$0"; exit 0 ;;
    --) shift; break ;;
    -*) echo "unknown option $1" >&2; exit 2 ;;
    *) break ;;
  esac
done
APP_DIR="${1:-}"; GOAL_ARG="${2:-}"
[[ -n "${TF_GOAL_CLASSIFY:-}" ]] && DRY=1   # the classify debug path touches no state
if [[ -z "$APP_DIR" || ( -z "$GOAL_ARG" && $RESUME -eq 0 ) ]]; then
  echo "usage: tf-goal.sh [options] <app-dir> \"<goal>\" | @goal.md   (or --resume <app-dir>)" >&2; exit 2
fi
APP_DIR="$(cd "$APP_DIR" 2>/dev/null && pwd)" || { echo "no such dir: $1" >&2; exit 2; }
[[ -d "$APP_DIR/.tfcore" ]] || { echo "$APP_DIR has no .tfcore/ — scaffold it first" >&2; exit 2; }
case "$HARNESS" in claude|opencode) ;; *) echo "--harness must be claude|opencode" >&2; exit 2 ;; esac
case "$TIER" in frontier|standard|economy) ;; *) echo "--tier must be frontier|standard|economy" >&2; exit 2 ;; esac
command -v python3 >/dev/null 2>&1 || { echo "python3 is required" >&2; exit 2; }

STATE_DIR="$APP_DIR/.tfcore/.session"; mkdir -p "$STATE_DIR"
STATE="$STATE_DIR/goal.json"; LOG="$STATE_DIR/goal.log"; DONE="$STATE_DIR/goal-done.json"
YOLO_SH="$APP_DIR/.tfcore/utils/tf-yolo.sh"
PICK_SH="$APP_DIR/.tfcore/utils/tf-model-pick.sh"
BIND_SH="$APP_DIR/.tfcore/utils/tf-routing-bind.sh"
[[ -f "$PICK_SH" ]] || FALLBACK=0   # a project that predates tf-model-pick.sh keeps the old behaviour

if [[ "$GOAL_ARG" == @* ]]; then
  GOAL_FILE="${GOAL_ARG#@}"; [[ -f "$GOAL_FILE" ]] || { echo "goal file not found: $GOAL_FILE" >&2; exit 2; }
  GOAL="$(cat "$GOAL_FILE")"
else
  GOAL="$GOAL_ARG"
fi

ts() { date -u +%Y-%m-%dT%H:%M:%SZ; }
# nap: a sleep the INT/TERM trap can interrupt at once (bash defers a trap while a
# foreground command runs, so a plain `sleep 300` held a kill for up to five minutes).
nap() { sleep "$1" & wait $!; }
log() { printf '[%s] tf-goal: %s\n' "$(ts)" "$*" | tee -a "$LOG" >&2; }

state_get() { python3 -c 'import json,sys
try: print(json.load(open(sys.argv[1])).get(sys.argv[2],""))
except Exception: print("")' "$STATE" "$1" 2>/dev/null; }
state_set() { # key value [key value ...]
  python3 - "$STATE" "$@" <<'PY' 2>/dev/null
import json, sys
p = sys.argv[1]; kv = sys.argv[2:]
try: d = json.load(open(p))
except Exception: d = {}
for k, v in zip(kv[::2], kv[1::2]):
    d[k] = int(v) if v.isdigit() else v
json.dump(d, open(p, "w"), indent=1)
PY
}

if [[ $RESUME -eq 1 ]]; then
  [[ -f "$STATE" ]] || { echo "--resume: no $STATE" >&2; exit 2; }
  [[ -z "$GOAL" ]] && GOAL="$(state_get goal)"
  HARNESS="$(state_get harness)"; HARNESS="${HARNESS:-claude}"
  # A resume keeps the model a fallback moved to, and the tier its chain came from; an explicit
  # --model on the resume command still wins.
  _t="$(state_get tier)"; [[ -n "$_t" ]] && TIER="$_t"
  _m="$(state_get model)"; [[ -n "$_m" && -z "$MODEL" ]] && MODEL="$_m"
  CYCLE="$(state_get cycle)"; CYCLE="${CYCLE:-0}"
  SESSION_ID="$(state_get session_id)"
  STALLS="$(state_get stalls)"; STALLS="${STALLS:-0}"
  if [[ $FRESH -eq 1 ]]; then log "resuming goal (cycle $CYCLE) in a FRESH session — the old one (${SESSION_ID:-none}) is left behind"; SESSION_ID=""; STALLS=0; state_set session_id "" stalls 0
  else log "resuming goal (cycle $CYCLE, session ${SESSION_ID:-none})"; fi
else
  # Refuse to start over a run that looks active in this folder: a goal.json written or
  # touched in the last three hours whose last_reason is not a finished state. A second
  # supervisor here would clobber the first one's state and flag (MISS-TechieFlow-20260905-21).
  if [[ $DRY -eq 0 && -f "$STATE" ]] && [[ -n "$(find "$STATE" -mmin -180 2>/dev/null)" ]]; then
    _lr="$(state_get last_reason)"
    case "$_lr" in
      done:*|max-cycles|stopped|"") ;;
      *) echo "tf-goal: a run looks active in $APP_DIR (goal.json last_reason=$_lr, touched $(tf_date_from -u "@$(tf_stat_mtime "$STATE")" +%H:%MZ)). Wait for it, or continue it with --resume." >&2; exit 2 ;;
    esac
  fi
  CYCLE=0; SESSION_ID=""; STALLS=0
  if [[ $DRY -eq 0 ]]; then   # a dry run touches nothing: no state, no flag, no log line
    state_set goal "$GOAL" harness "$HARNESS" tier "$TIER" model "$MODEL" cycle 0 session_id "" stalls 0 started "$(ts)" last_reason "start"
    rm -f "$DONE"
  fi
fi

# ---------------------------------------------------------------- prompts
# PREAMBLE IS HARNESS-NEUTRAL. Every claim in it must hold for claude AND opencode
# because both are sent this text verbatim. Anything true of only
# one harness goes in that harness's block below — never in here. The 2026-08-28 review put its
# strict no-git policy into this shared text, which then told Claude and OpenCode
# goal runs to avoid read-only git that their own hook allows; nothing written down
# had said the preamble was shared, so this comment is that rule.
read -r -d '' PREAMBLE <<'TXT'
UNATTENDED GOAL RUN — YOLO MODE IS ON (TechieFlow rule .tfcore/tasks/_yolo-mode.md; read it first).
- Nobody is watching. NEVER ask a question, NEVER pause for confirmation, NEVER end your turn with a plan, options, or "shall I…". Decide the sensible default yourself and record the decision in the checklist Remarks.
- Permissions: deletes are allowed when they are necessary and precisely scoped. Git WRITES (commit/push/add/reset/checkout/stash/tag) are blocked in every mode — never attempt them; the owner commits. Read-only git (status/log/diff/blame) is decided by the harness policy stated below, if any; where nothing further is stated it is available in this mode as supplementary evidence only — the checklist Requirements Status table and the working tree stay primary.
- Re-entry: start from PROJECT-STATUS.md + docs/*-Checklist.md (Requirements Status table) — continue from the weakest open REQ; do not redo terminal rows.
- A build, a test run or any long command runs in the FOREGROUND and you wait for it (a ten-minute timeout is fine). Never start it as a background job and end your turn to wait for it: the turn ending kills the job, and nobody will wake you.
- A build pass means the WHOLE checklist: every open REQ reaches at least `Implemented` in this pass, then the verifier is chained inline, then FIX mode loops on FAIL rows until they pass. Never stop with "run build-phase again for the remaining REQs".
- When the goal is met (every in-scope REQ terminal, PROJECT-STATUS.md + .html updated, run record emitted): run
      bash .tfcore/utils/tf-yolo.sh done complete "<one-line summary>"
  If, and only if, something ONLY THE OWNER can resolve blocks every remaining REQ (credentials, a physical device, a paid account, a product decision you must not make): finish everything else, write the blocker into PROJECT-STATUS "Known blockers", then run
      bash .tfcore/utils/tf-yolo.sh done blocked "<what the owner must do>"
  Do not run either command before that point — the supervisor stops the moment you do.
TXT

CONTINUE_PROMPT="Continue the UNATTENDED GOAL RUN (YOLO ON). The previous turn ended without the goal-done sentinel — pick up from PROJECT-STATUS.md + the checklist Requirements Status table and keep going. Do not summarise, do not ask; work until the goal is met, then run: bash .tfcore/utils/tf-yolo.sh done complete \"<summary>\". Builds and tests run in the FOREGROUND and you wait for them; never end the turn to wait for a background job (the turn ending kills it). The goal, again:
$GOAL"

FIRST_PROMPT="$PREAMBLE

THE GOAL:
$GOAL"
# ---------------------------------------------------------------- harness command
harness_cmd() { # $1 = first|resume ; prints the argv via NUL-separated echo
  local kind="$1" prompt
  if [[ "$kind" == first ]]; then prompt="$FIRST_PROMPT"; else prompt="$CONTINUE_PROMPT"; fi
  if [[ "$HARNESS" == claude ]]; then
    CMD=(claude -p --permission-mode bypassPermissions --output-format stream-json --verbose)
    [[ -n "$MODEL" ]] && CMD+=(--model "$MODEL")
    # shellcheck disable=SC2206
    [[ -n "${TF_GOAL_CLAUDE_FLAGS:-}" ]] && CMD+=($TF_GOAL_CLAUDE_FLAGS)
    if [[ "$kind" == resume ]]; then
      if [[ -n "$SESSION_ID" ]]; then CMD+=(--resume "$SESSION_ID"); else CMD+=(--continue); fi
    fi
    CMD+=("$prompt")
  elif [[ "$HARNESS" == opencode ]]; then
    CMD=(opencode run --auto)
    [[ -n "$MODEL" ]] && CMD+=(-m "$MODEL")
    # shellcheck disable=SC2206
    [[ -n "${TF_GOAL_OPENCODE_FLAGS:-}" ]] && CMD+=($TF_GOAL_OPENCODE_FLAGS)
    if [[ "$kind" == resume ]]; then
      if [[ -n "$SESSION_ID" ]]; then CMD+=(-s "$SESSION_ID"); else CMD+=(-c); fi
    fi
    CMD+=("$prompt")
  else
    CMD=(opencode run --auto)
    [[ -n "$MODEL" ]] && CMD+=(-m "$MODEL")
    if [[ "$kind" == resume ]]; then
      if [[ -n "$SESSION_ID" ]]; then CMD+=(-s "$SESSION_ID"); else CMD+=(-c); fi
    fi
    CMD+=("$prompt")
  fi
}

# ---------------------------------------------------------------- classification
# Reads the cycle's output file; prints one line: KIND<TAB>detail
#   LIMIT <epoch-to-resume>   CRASH <text>   IDLE <text>
classify_output() {
  TF_OUT="$1" TF_BUFFER_MIN="$BUFFER_MIN" TF_DEFAULT_WAIT_MIN="$DEFAULT_WAIT_MIN" TF_RC="$2" python3 - <<'PY'
import os, re, sys, json, time, datetime, calendar
try:
    from zoneinfo import ZoneInfo
except Exception:
    ZoneInfo = None

path = os.environ["TF_OUT"]; buffer_min = int(os.environ["TF_BUFFER_MIN"]); default_wait = int(os.environ["TF_DEFAULT_WAIT_MIN"])
rc = int(os.environ.get("TF_RC") or 0)
try:
    text = open(path, errors="replace").read()
except Exception:
    text = ""
tail = text[-20000:]
now = time.time()

def out(kind, detail):
    print(f"{kind}\t{detail}"); sys.exit(0)

# ---- 0. Claude Code stream-json: the last `result` line says how the turn ended, and a
# clean one (is_error false) is a STOP, never a crash. Until 2026-09-06 the crash regex
# below matched the harmless field "api_error_status":null that every clean result carries,
# so every early stop was called a harness error and backed off 2m, 4m, 8m … instead of
# being re-prompted after 30s (MISS-TechieFlow-20260905-23, MISS-TechieFlow-20260906-01).
last_result = None
for line in text.splitlines():
    line = line.strip()
    if not line.startswith("{") or '"result"' not in line:
        continue
    try:
        d = json.loads(line)
    except Exception:
        continue
    if isinstance(d, dict) and d.get("type") == "result":
        last_result = d
if last_result is not None and rc == 0 and not last_result.get("is_error"):
    out("IDLE", "clean stop: result is_error=false, %s turn(s)" % last_result.get("num_turns", "?"))
# a rejected rate_limit_event carries the exact reset epoch — better than parsing "9am"
if last_result is not None and last_result.get("is_error"):
    ev = re.findall(r'"rate_limit_info"\s*:\s*\{[^{}]*"status"\s*:\s*"rejected"[^{}]*"resetsAt"\s*:\s*(\d{10})', text)
    if not ev:
        ev = re.findall(r'"status"\s*:\s*"rejected"[^{}]*?"resetsAt"\s*:\s*(\d{10})', text)
    if ev:
        out("LIMIT", f"{int(ev[-1]) + buffer_min*60}\tparsed")

# ---- 1. usage limit?
LIMIT_PAT = re.compile(
    r"(usage limit|hit your limit|exceeded your .*limit|you've hit your (5-hour|weekly|usage) limit|rate[ _-]?limit(ed)?(?!_event)|"
    r"limit (has been )?(reached|exceeded)|too many requests|\b429\b|overloaded_error|quota exceeded|"
    r"out of extra usage|resets? (at|in)\b|weekly limit|session limit)", re.I)
m = LIMIT_PAT.search(tail)
if m:
    resume_at = None
    # a) legacy "Claude AI usage limit reached|<epoch>"
    e = re.search(r"limit reached\|(\d{10})", tail)
    if e:
        resume_at = int(e.group(1))
    # b) ISO timestamp near the match
    if resume_at is None:
        iso = re.search(r"(20\d\d-\d\d-\d\dT\d\d:\d\d(?::\d\d)?(?:\.\d+)?(?:Z|[+-]\d\d:?\d\d)?)", tail[m.start()-200:m.end()+300])
        if iso:
            s = iso.group(1).replace("Z", "+00:00")
            try:
                resume_at = datetime.datetime.fromisoformat(s).timestamp()
            except Exception:
                pass
    # c) "resets in 2h 14m" / "resets in 45 minutes" / "retry after 120"
    if resume_at is None:
        r = re.search(r"(?:resets?|try again|retry)\s+(?:in|after)[:\s]+(?:(\d+)\s*h(?:ours?|rs?)?)?\s*(?:(\d+)\s*m(?:in(?:utes?)?)?)?\s*(?:(\d+)\s*s(?:ec(?:onds?)?)?)?", tail, re.I)
        if r and any(r.groups()):
            h, mi, s = (int(x) if x else 0 for x in r.groups())
            resume_at = now + h*3600 + mi*60 + s
        else:
            r = re.search(r"retry[- ]after[:=\s]+(\d+)", tail, re.I)
            if r:
                resume_at = now + int(r.group(1))
    # d) "resets 3pm (Asia/Kolkata)" / "resets at 14:30" / "resets Tue 3pm"
    if resume_at is None:
        r = re.search(r"resets?\s+(?:at\s+)?(?:(mon|tue|wed|thu|fri|sat|sun)[a-z]*\s+)?(\d{1,2})(?::(\d{2}))?\s*(am|pm)?(?:\s*\(([^)]+)\))?", tail, re.I)
        if r:
            dow, hh, mm, ampm, tzname = r.groups()
            hh = int(hh); mm = int(mm or 0)
            if ampm:
                if ampm.lower() == "pm" and hh != 12: hh += 12
                if ampm.lower() == "am" and hh == 12: hh = 0
            tz = None
            if tzname and ZoneInfo:
                try: tz = ZoneInfo(tzname.strip())
                except Exception: tz = None
            base = datetime.datetime.now(tz) if tz else datetime.datetime.now().astimezone()
            cand = base.replace(hour=hh, minute=mm, second=0, microsecond=0)
            if dow:
                want = ["mon","tue","wed","thu","fri","sat","sun"].index(dow.lower()[:3])
                delta = (want - cand.weekday()) % 7
                cand = cand + datetime.timedelta(days=delta)
            if cand.timestamp() <= now:
                cand = cand + datetime.timedelta(days=1 if not dow else 7)
            resume_at = cand.timestamp()
    if resume_at is None:
        # no parseable reset time → the supervisor PROBES instead of guessing
        out("LIMIT", f"{int(now + default_wait*60)}\tprobe")
    resume_at = max(resume_at, now + 60) + buffer_min*60
    out("LIMIT", f"{int(resume_at)}\tparsed")

# ---- 2. a crash / API error / auth problem?
if rc != 0 or re.search(r"(api_error(?!_status)|internal server error|\b5\d\d\b .*error|ECONNRESET|ETIMEDOUT|ENOTFOUND|socket hang up|"
                        r"authentication_error|invalid api key|not logged in|please run /login|error_during_execution|"
                        r"\"is_error\"\s*:\s*true|Error: .*(fetch|network|connect))", tail, re.I):
    out("CRASH", f"rc={rc}")

# ---- 3. otherwise the agent just stopped
out("IDLE", "no sentinel")
PY
}

extract_session_id() { # from a cycle's output file (Claude/OpenCode JSONL)
  python3 - "$1" <<'PY' 2>/dev/null
import sys, json, re
sid = ""
for line in open(sys.argv[1], errors="replace"):
    line = line.strip()
    if not line.startswith("{"): continue
    try: d = json.loads(line)
    except Exception: continue
    if isinstance(d, dict) and d.get("session_id"): sid = d["session_id"]
    if isinstance(d, dict) and d.get("type") == "thread.started" and d.get("thread_id"): sid = d["thread_id"]
if not sid:
    m = re.findall(r'"(?:session_id|thread_id)"\s*:\s*"([0-9A-Za-z_-]{8,})"', open(sys.argv[1], errors="replace").read())
    sid = m[-1] if m else ""
print(sid)
PY
}

extract_model() { # the model the harness reported for this cycle, from its own output
  python3 - "$1" <<'PY' 2>/dev/null
import sys, json, re
m = ""
try:
    for line in open(sys.argv[1], errors="replace"):
        line = line.strip()
        if not line.startswith("{") or '"model"' not in line:
            continue
        try: d = json.loads(line)
        except Exception: continue
        if not isinstance(d, dict): continue
        if isinstance(d.get("model"), str) and d["model"]:
            m = d["model"]
        msg = d.get("message")
        if isinstance(msg, dict) and isinstance(msg.get("model"), str) and msg["model"]:
            m = msg["model"]
except Exception:
    pass
print(m)
PY
}

# Park the model this cycle ran on and move to the next one in its tier. Prints nothing and
# returns 1 when there is nowhere to go — the caller then sleeps, which is what it always did.
#   $1 = epoch the limit is stated to reset at (or empty when it could not be parsed)
#   $2 = why, for the cooldown record
try_fallback() {
  [[ $FALLBACK -eq 1 ]] || return 1
  local until="${1:-}" why="${2:-usage limit}" cur next
  cur="$MODEL"
  [[ -z "$cur" ]] && cur="$(extract_model "$OUT")"
  if [[ -z "$cur" ]]; then
    cur="$(TF_PROJECT_DIR="$APP_DIR" bash "$PICK_SH" pick "$TIER" "$HARNESS" 2>/dev/null | tail -1)"
    [[ "$cur" == inherit ]] && cur=""
  fi
  if [[ -z "$cur" ]]; then
    log "no fallback: nothing here says which model this cycle ran on (no --model, no model id in the output, no routing answer for tier $TIER)"
    return 1
  fi
  [[ -z "$until" ]] && until="probe"
  TF_PROJECT_DIR="$APP_DIR" bash "$PICK_SH" cooldown "$HARNESS" "$cur" "$until" "$why" >>"$LOG" 2>&1
  next="$(TF_PROJECT_DIR="$APP_DIR" bash "$PICK_SH" pick "$TIER" "$HARNESS" 2>/dev/null | tail -1)"
  if [[ -z "$next" || "$next" == "inherit" ]]; then
    log "no fallback left: every model in the $TIER chain is on cooldown — waiting for the reset instead"
    return 1
  fi
  MODEL="$next"
  state_set model "$MODEL" fallback_from "$cur" last_reason "fallback:$cur->$MODEL"
  log "FALLBACK: $cur is limited ($why) — the next cycle runs on $MODEL (tier $TIER). Re-binding so the sub-agents move too."
  [[ -f "$BIND_SH" ]] && bash "$BIND_SH" "$APP_DIR" 2>&1 | sed 's/^/    /' | tee -a "$LOG" >&2
  return 0
}

# Fallback when a limit message carries no parseable reset time: fire a one-turn
# probe every PROBE_MIN minutes until it comes back clean (exit 0 AND no limit
# wording in its output). Gives up after PROBE_MAX_H hours and resumes anyway.
probe_until_clear() {
  local deadline=$(( $(date +%s) + PROBE_MAX_H * 3600 )) n=0 pout prc
  pout="$STATE_DIR/goal-probe.out"
  while :; do
    n=$(( n + 1 ))
    if [[ "$HARNESS" == claude ]]; then
      ( cd "$APP_DIR" && claude -p --max-turns 1 --output-format text "Reply with the single word OK." ) > "$pout" 2>&1; prc=$?
    else
      ( cd "$APP_DIR" && opencode run --auto "Reply with the single word OK." ) > "$pout" 2>&1; prc=$?
    fi
    if [[ $prc -eq 0 ]] && ! grep -qiE 'usage limit|hit your limit|rate[ _-]?limit|limit (has been )?(reached|exceeded)|too many requests|(^|[^[:alnum:]_])429([^[:alnum:]_]|$)|overloaded|weekly limit|resets? (at|in)([^[:alnum:]_]|$)' "$pout"; then
      log "probe #$n OK"; return 0
    fi
    log "probe #$n still limited (rc=$prc: $(head -c 120 "$pout" | tr '\n' ' ')) — next in ${PROBE_MIN}m"
    state_set resume_at "probe #$((n+1)) at $(tf_date_from "+${PROBE_MIN} min" '+%H:%M' 2>/dev/null)"
    if [[ $(date +%s) -ge $deadline ]]; then log "probe window (${PROBE_MAX_H}h) exhausted — resuming anyway"; return 1; fi
    nap $(( PROBE_MIN * 60 ))
  done
}

sleep_until() { # epoch
  local target="$1" now left
  while :; do
    now=$(date +%s); left=$(( target - now )); [[ $left -le 0 ]] && break
    state_set resume_at "$(tf_date_from -u "@$target" +%Y-%m-%dT%H:%M:%SZ 2>/dev/null || echo "$target")"
    if [[ $left -gt 300 ]]; then nap 300; else nap "$left"; fi
  done
  state_set resume_at ""
}

# Debug: `TF_GOAL_CLASSIFY=<output-file> [TF_GOAL_CLASSIFY_RC=n] tf-goal.sh <app-dir> x` prints the classification and exits.
if [[ -n "${TF_GOAL_CLASSIFY:-}" ]]; then classify_output "$TF_GOAL_CLASSIFY" "${TF_GOAL_CLASSIFY_RC:-0}"; exit 0; fi

# ---------------------------------------------------------------- the harness child
# The harness runs as a background child in its own process group, so the supervisor can
# (a) watch its output file grow and kill it after STALL_MIN silent minutes — the OpenCode
# run that printed nothing for 43 minutes while the supervisor waited forever
# (MISS-TechieFlow-20260905-22) — and (b) take the child down with it on Ctrl-C / kill,
# instead of leaving an orphaned agent working in the folder (the previous trap only
# cleared the YOLO flag and did not even exit, so a killed supervisor carried on to the
# next cycle: MISS-TechieFlow-20260906-02).
CHILD=""; STALLED=0
STALL_SEC="${TF_GOAL_STALL_SEC:-$(( STALL_MIN * 60 ))}"; STALL_TICK="${TF_GOAL_STALL_TICK:-30}"
kill_child() {
  [[ -n "$CHILD" ]] || return 0
  kill -0 "$CHILD" 2>/dev/null || { CHILD=""; return 0; }
  kill -TERM -- "-$CHILD" 2>/dev/null || kill -TERM "$CHILD" 2>/dev/null || true
  local i=0
  while kill -0 "$CHILD" 2>/dev/null && [[ $i -lt 10 ]]; do sleep 1; i=$((i+1)); done
  kill -KILL -- "-$CHILD" 2>/dev/null || kill -KILL "$CHILD" 2>/dev/null || true
  CHILD=""
}
run_cycle() { # runs "${CMD[@]}" in $APP_DIR, output to $OUT and $LOG; sets RC, STALLED
  STALLED=0
  local quiet=0 prev=-1 size
  if [[ -n "${TF_GOAL_FAKE_CMD:-}" ]]; then CMD=(bash -c "$TF_GOAL_FAKE_CMD"); fi
  # Its own session and process group (tf_setsid: setsid(1), or python3/perl on a Mac), so
  # kill_child can stop the whole tree.
  ( cd "$APP_DIR" && TF_YOLO=1 tf_setsid --exec "${CMD[@]}" ) > >(tee -a "$LOG" > "$OUT") 2>&1 &
  CHILD=$!
  while kill -0 "$CHILD" 2>/dev/null; do
    nap "$STALL_TICK"
    kill -0 "$CHILD" 2>/dev/null || break
    size="$(tf_stat_size "$OUT" 2>/dev/null || echo 0)"
    if [[ "$size" == "$prev" ]]; then quiet=$(( quiet + STALL_TICK )); else quiet=0; prev="$size"; fi
    if [[ "$STALL_SEC" -gt 0 && $quiet -ge "$STALL_SEC" ]]; then
      log "STALL: no output for $(( quiet / 60 ))m$(( quiet % 60 ))s (cycle $CYCLE, ${size} bytes) — stopping the harness, re-prompting"
      kill_child; STALLED=1; break
    fi
  done
  wait "$CHILD" 2>/dev/null; RC=$?; CHILD=""
  sleep 1  # let tee flush
}
# OpenCode prints NOTHING on a provider refusal — `opencode run` shows its header and waits
# while ~/.local/share/opencode/log/opencode.log says "Monthly usage limit reached. Resets in
# 6 days" (2026-09-06, three silent 15-minute stalls on MyDiary-oc; MISS-TechieFlow-20260906-11).
# After a stall the supervisor reads that log for stream errors newer than the cycle start.
OPENCODE_LOG="${TF_GOAL_OPENCODE_LOG:-$HOME/.local/share/opencode/log/opencode.log}"
opencode_log_error() { # $1 = cycle start epoch; prints the newest provider error text since then, or nothing
  [[ "$HARNESS" == opencode && -f "$OPENCODE_LOG" ]] || return 0
  python3 - "$OPENCODE_LOG" "$1" <<'PY2' 2>/dev/null
import re, sys, datetime
path, since = sys.argv[1], float(sys.argv[2])
last = ""
try:
    for line in open(path, errors="replace"):
        if "stream error" not in line: continue
        m = re.search(r"timestamp=(\S+)", line)
        try:
            ts = datetime.datetime.fromisoformat(m.group(1).replace("Z", "+00:00")).timestamp()
        except Exception:
            continue
        if ts < since: continue
        e = re.search(r'error\.error="([^"]*)"', line)
        last = (e.group(1) if e else line.strip())[:300]
except Exception:
    pass
print(last)
PY2
}
on_signal() {
  trap - INT TERM
  log "supervisor stopped by signal (cycle $CYCLE) — stopping the harness child; continue with --resume"
  kill_child
  state_set last_reason "stopped" ended "$(ts)"
  exit 130
}

# ---------------------------------------------------------------- main loop
export TF_YOLO=1
# Pin the state dir to the APP (not the caller's cwd / a parent repo's CLAUDE_PROJECT_DIR).
[[ $DRY -eq 0 ]] && { CLAUDE_PROJECT_DIR="$APP_DIR" TF_PROJECT_DIR="$APP_DIR" bash "$YOLO_SH" on --source goal --goal "$GOAL" >/dev/null 2>&1 || true; }

# The flag we just wrote is OURS — clear it on EVERY exit path (goal done, max
# cycles, Ctrl-C, kill). Without this it survives the run and silently puts every
# later interactive session in this repo into YOLO: rm/rmdir/sudo stop prompting
# and read-only git opens up, with nothing on screen saying so. Only a flag whose
# source is `goal` is removed — a `*yolo` the owner turned on by hand is left alone.
clear_goal_yolo() {
  local f="$STATE_DIR/yolo.json"
  [[ -f "$f" ]] || return 0
  grep -q '"source":[[:space:]]*"goal"' "$f" 2>/dev/null || return 0
  CLAUDE_PROJECT_DIR="$APP_DIR" TF_PROJECT_DIR="$APP_DIR" bash "$YOLO_SH" off >/dev/null 2>&1 || true
}
[[ $DRY -eq 0 ]] && { trap clear_goal_yolo EXIT; trap on_signal INT TERM; }
BACKOFF=120
KIND=first; [[ $RESUME -eq 1 && $CYCLE -gt 0 && $FRESH -eq 0 ]] && KIND=resume

while :; do
  if [[ -f "$DONE" ]]; then
    OUTCOME="$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1])).get("outcome","complete"))' "$DONE" 2>/dev/null || echo complete)"
    log "goal-done sentinel present (outcome=$OUTCOME): $(cat "$DONE")"
    state_set last_reason "done:$OUTCOME" ended "$(ts)"
    [[ "$OUTCOME" == blocked ]] && exit 3 || exit 0
  fi
  if [[ $CYCLE -ge $MAX_CYCLES ]]; then log "max cycles ($MAX_CYCLES) reached — stopping"; state_set last_reason "max-cycles"; exit 4; fi
  CYCLE=$((CYCLE + 1))
  harness_cmd "$KIND"
  OUT="$STATE_DIR/goal-cycle-$CYCLE.out"
  if [[ $DRY -eq 1 ]]; then echo "dry run — would launch cycle $CYCLE ($KIND):" >&2; printf '  %q' "${CMD[@]}"; echo; exit 0; fi
  state_set cycle "$CYCLE"; CYCLE_START="$(date +%s)"
  log "cycle $CYCLE ($KIND) → ${CMD[*]:0:6} … (prompt ${#CMD[${#CMD[@]}-1]} chars)"

  # The run's start is now, not when the agent reaches step 0: write an unclaimed marker the
  # first command's `tf-phase.sh start` claims (MISS-TechieFlow-20260905-20).
  [[ $CYCLE -eq 1 && -f "$APP_DIR/.tfcore/utils/tf-phase.sh" ]] && bash "$APP_DIR/.tfcore/utils/tf-phase.sh" goal "$(basename "$APP_DIR")" >/dev/null 2>&1 || true
  run_cycle
  SID="$(extract_session_id "$OUT")"; [[ -n "$SID" ]] && { SESSION_ID="$SID"; state_set session_id "$SID"; }
  KIND=resume

  if [[ -f "$DONE" ]]; then continue; fi

  if [[ $STALLED -eq 1 ]]; then
    CLASS=IDLE; DETAIL="stalled ${STALL_MIN}m"; HOW=""
    PERR="$(opencode_log_error "$CYCLE_START")"
    if [[ -n "$PERR" ]] && grep -qiE "usage limit|monthly|balance|quota|rate limit|429|insufficient|billing|unauthorized|api key" <<<"$PERR"; then
      log "the provider refused the model while the harness printed nothing (cycle $CYCLE): $PERR"
      # A monthly limit or an empty balance states no reset time, so the model is parked for the
      # probe window and the run continues on the next model in the tier. Only when the chain is
      # exhausted does this stay what it was before 2026-09-10: the owner's problem, exit 5.
      if try_fallback "" "provider refused the model: ${PERR:0:120}"; then
        STALLS=0; state_set stalls 0; BACKOFF=120; continue
      fi
      log "no other model in the $TIER chain is available — the owner must act (enable balance, wait for the reset, or pick another model); stopping"
      state_set last_reason "provider-limit" ended "$(ts)"
      exit 5
    fi
    STALLS=$(( STALLS + 1 )); state_set stalls "$STALLS"
    # A resumed session that hangs twice running is not coming back (OpenCode `-c` printed its
    # header and nothing else for 15 minutes, twice, on 2026-09-06): the next cycle starts a
    # fresh session with the full goal; the checklist and PROJECT-STATUS carry the state.
    if [[ "$KIND" == resume && $STALLS -ge 2 ]]; then
      log "two stalled resumes in a row — the next cycle starts a FRESH session (the old one, ${SESSION_ID:-none}, is left behind)"
      KIND=first; SESSION_ID=""; STALLS=0; state_set session_id "" stalls 0
    fi
  else
    IFS=$'\t' read -r CLASS DETAIL HOW < <(classify_output "$OUT" "$RC")
    STALLS=0; state_set stalls 0
  fi
  case "$CLASS" in
    LIMIT)
      # Another model in the same tier beats waiting, when there is one.
      if [[ "$HOW" == probe ]]; then _FB_UNTIL=""; else _FB_UNTIL="$DETAIL"; fi
      if try_fallback "$_FB_UNTIL" "usage limit"; then BACKOFF=120; continue; fi
      if [[ "$HOW" == probe ]]; then
        log "USAGE LIMIT hit (cycle $CYCLE) but no reset time could be parsed from the message — probing every ${PROBE_MIN}m until the API answers again (max ${PROBE_MAX_H}h). Tail of the message:"
        tail -c 400 "$OUT" | tr '\n' ' ' | sed 's/^/    /' | tee -a "$LOG" >&2; echo >&2
        state_set last_reason "limit-probe" resume_at "probing every ${PROBE_MIN}m"
        probe_until_clear
        log "probe succeeded — resuming session ${SESSION_ID:-(--continue)} after a ${BUFFER_MIN}m buffer"
        nap $(( BUFFER_MIN * 60 ))
      else
        WHEN="$(tf_date_from "@$DETAIL" '+%Y-%m-%d %H:%M %Z' 2>/dev/null || echo "$DETAIL")"
        log "USAGE LIMIT hit (cycle $CYCLE) — RETRY AT $WHEN (stated reset + ${BUFFER_MIN}m buffer). Sleeping."
        state_set last_reason "limit" resume_at "$WHEN"
        sleep_until "$DETAIL"
        log "limit window over — resuming session ${SESSION_ID:-(--continue)}"
      fi
      BACKOFF=120 ;;
    CRASH)
      log "harness/API error (cycle $CYCLE, $DETAIL) — backing off ${BACKOFF}s then resuming"
      state_set last_reason "crash:$DETAIL"
      nap "$BACKOFF"; BACKOFF=$(( BACKOFF * 2 )); [[ $BACKOFF -gt 1800 ]] && BACKOFF=1800 ;;
    IDLE|*)
      log "agent stopped without the goal-done sentinel (cycle $CYCLE, ${DETAIL:-no sentinel}) — re-prompting in ${IDLE_RETRY}s"
      state_set last_reason "idle"
      nap "$IDLE_RETRY"; BACKOFF=120 ;;
  esac
done
