From 1f559f3e9a701ea45bd6a4bfd16db09e35a0c99e Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Thu, 24 Sep 2026 11:46:04 +0800 Subject: [PATCH 01/10] trace-token-burn-loop: add fleet scanner, context-growth profiler, tests Change-Id: I3e48932d326a77d54772876edd207bbfd6f00ff6 --- .../principle-trace-token-burn-loop/SKILL.md | 16 +- .../scripts/context_growth.py | 159 ++++++++++++++ .../scripts/scan_session_tokens.py | 194 ++++++++++++++++++ .../tests/test_scripts.py | 151 ++++++++++++++ 4 files changed, 519 insertions(+), 1 deletion(-) create mode 100755 corpus/skills/principle-trace-token-burn-loop/scripts/context_growth.py create mode 100644 corpus/skills/principle-trace-token-burn-loop/scripts/scan_session_tokens.py create mode 100644 corpus/skills/principle-trace-token-burn-loop/tests/test_scripts.py diff --git a/corpus/skills/principle-trace-token-burn-loop/SKILL.md b/corpus/skills/principle-trace-token-burn-loop/SKILL.md index ecb9d5ef4..216e7d823 100644 --- a/corpus/skills/principle-trace-token-burn-loop/SKILL.md +++ b/corpus/skills/principle-trace-token-burn-loop/SKILL.md @@ -21,7 +21,21 @@ iteration limit — can make a cheap plan become an expensive burn. **Pattern:** 1. Rank spend by task_type / prompt_type, then open one representative session - file. + file. Fleet-wide: `scripts/scan_session_tokens.py ` emits one + JSONL row per session file (codex exec, codex rollout, claude, OMP formats; + `--summary` for the agent x workload table), and is pipeable over ssh — + `cat scan_session_tokens.py | ssh host python3 - ~/.invoker/agent-sessions + ~/.claude/projects ~/.codex/sessions`. For a single session's pricing / + thrash detail use `engine/skills/reflect/scripts/token_audit.py`. +1a. Check whether the burn is session-lifetime (few long sessions, context + re-sent every turn) or run-count (many runs at a fixed per-run floor): + `scripts/context_growth.py --summary ` reports turns, + peak/avg context, `clears` (sawtooth drops = /clear or compaction + cycles), and gross resend total; without `--summary` it emits the raw + per-call `{ts, ctx, out}` series. A long-running session that polls + (ScheduleWakeup, sleep-and-recheck loops) will show turns x ~500K avg + context; a one-shot run shows a single ramp. Cost ~= area under the + curve, not the task size. 2. Read its `cwd` and first user prompt to map it to a plan or workflow. 3. Open that plan or worker file. 4. Look for an unbounded loop: diff --git a/corpus/skills/principle-trace-token-burn-loop/scripts/context_growth.py b/corpus/skills/principle-trace-token-burn-loop/scripts/context_growth.py new file mode 100755 index 000000000..14e5730b9 --- /dev/null +++ b/corpus/skills/principle-trace-token-burn-loop/scripts/context_growth.py @@ -0,0 +1,159 @@ +#!/usr/bin/env python3 +"""Per-turn context-size series for a session file — the "sawtooth" view. + +The question this answers: is a session expensive because it did a lot of +work, or because it lived a long time with a big context? Every model call +re-sends the whole transcript, so cost ~= sum over turns of context size. +This script extracts the per-call context size (input tokens incl. cache +reads) and detects /clear / compaction cycles (the drops in the sawtooth). + +Usage: + context_growth.py [more files...] # JSONL rows + context_growth.py --summary [...] # one line/file + cat context_growth.py | ssh host python3 - --summary file... + +Row format (default mode), one per model call: + {"ts": ..., "seq": n, "ctx": , "out": n, "uncached": n} + +Summary fields: + turns model calls observed + peak_ctx largest single-call context + clears number of >40% context drops (sawtooth cycles) + avg_ctx mean context per call + gross sum of ctx over calls (the resend total) + span_h first-to-last timestamp hours +Formats: + - claude JSONL: assistant rows -> message.usage (ctx = input_tokens + + cache_read_input_tokens + cache_creation_input_tokens) + - codex rollout JSONL: event_msg token_count rows -> payload.info. + last_token_usage is the per-call usage; total_token_usage is cumulative + (ignored to avoid double counting) + - codex exec JSONL (~/.invoker/agent-sessions): only turn.completed totals + exist — no per-call context. Emits a single summary row with gross = + turn usage and a note; there is no series to plot. +""" +import json, os, signal, sys + +CLEAR_DROP_RATIO = 0.6 + + +def end_quietly_when_the_reader_stops(): + """Restore the default SIGPIPE so piping into `head` ends without a trace.""" + signal.signal(signal.SIGPIPE, signal.SIG_DFL) + + +end_quietly_when_the_reader_stops() + + +def per_call_from_cumulative(total, prev_total): + """Per-call usage for rollout rows that carry only a running total. + + Older codex rollouts omit last_token_usage, so the call's own usage is + what the running total gained since the previous row. + """ + prev = prev_total or {} + return {k: max(total.get(k, 0) - prev.get(k, 0), 0) + for k in ('input_tokens', 'output_tokens', 'cached_input_tokens')} + + +def series(fp): + """Yield dicts {ts, ctx, out, uncached} per model call.""" + prev_total = None + try: + with open(fp, errors='replace') as fh: + for line in fh: + if 'usage' not in line and 'token_count' not in line: + continue + try: + e = json.loads(line) + except Exception: + continue + ts = e.get('timestamp') + p = e.get('payload') + if isinstance(p, dict) and p.get('type') == 'token_count': + info = p.get('info') or {} + u = info.get('last_token_usage') + if not isinstance(u, dict): + tot = info.get('total_token_usage') + if isinstance(tot, dict): + u = per_call_from_cumulative(tot, prev_total) + prev_total = tot + if isinstance(u, dict) and (u.get('input_tokens') or u.get('output_tokens')): + inp = u.get('input_tokens', 0) + cached = u.get('cached_input_tokens', 0) + yield {'ts': ts, 'ctx': inp, 'out': u.get('output_tokens', 0), + 'uncached': max(inp - cached, 0)} + continue + if e.get('type') in ('turn.completed', 'thread.completed'): + u = e.get('usage') + if isinstance(u, dict): + inp = u.get('input_tokens', 0) + cached = u.get('cached_input_tokens', u.get('cached_tokens', 0) or 0) + yield {'ts': ts, 'ctx': inp, 'out': u.get('output_tokens', 0), + 'uncached': max(inp - cached, 0), 'whole_turn': True} + continue + m = e.get('message') + if isinstance(m, dict) and m.get('role') == 'assistant': + u = m.get('usage') + if isinstance(u, dict): + ctx = (u.get('input_tokens', 0) + + u.get('cache_read_input_tokens', 0) + + u.get('cache_creation_input_tokens', 0)) + yield {'ts': ts, 'ctx': ctx, + 'out': u.get('output_tokens', 0), + 'uncached': u.get('input_tokens', 0)} + except Exception as ex: + yield {'error': str(ex)} + + +def summarize(fp): + """One row per session: turns, peak and average context, clears, gross. + + A call whose context falls below CLEAR_DROP_RATIO of the call before it + is counted as a clear: that is what /clear or a compaction looks like in + the series, and it is the down-stroke of the sawtooth. + """ + pts = [p for p in series(fp) if 'ctx' in p] + r = {'path': fp, 'sid': os.path.basename(fp).replace('.jsonl', ''), + 'turns': len(pts)} + if not pts: + r['note'] = 'no per-call usage rows' + return r + if any(p.get('whole_turn') for p in pts): + r['note'] = 'exec format: whole-turn totals only, no per-call series' + ctxs = [p['ctx'] for p in pts] + r['peak_ctx'] = max(ctxs) + r['avg_ctx'] = int(sum(ctxs) / len(ctxs)) + r['gross'] = sum(ctxs) + r['clears'] = sum(1 for a, b in zip(ctxs, ctxs[1:]) + if b < a * CLEAR_DROP_RATIO and a > 50000) + r['uncached'] = sum(p.get('uncached', 0) for p in pts) + r['out'] = sum(p['out'] for p in pts) + tss = [p['ts'] for p in pts if p.get('ts')] + if len(tss) >= 2: + try: + from datetime import datetime + a = datetime.fromisoformat(tss[0].replace('Z', '+00:00')) + b = datetime.fromisoformat(tss[-1].replace('Z', '+00:00')) + r['span_h'] = round((b - a).total_seconds() / 3600, 1) + except Exception: + pass + return r + + +def main(): + args = sys.argv[1:] + summary = '--summary' in args + files = [os.path.expanduser(a) for a in args if a != '--summary'] + for fp in files: + if summary: + print(json.dumps(summarize(fp)), flush=True) + else: + for i, p in enumerate(series(fp)): + p['seq'] = i + p['path'] = fp + print(json.dumps(p), flush=True) + + +if __name__ == '__main__': + main() diff --git a/corpus/skills/principle-trace-token-burn-loop/scripts/scan_session_tokens.py b/corpus/skills/principle-trace-token-burn-loop/scripts/scan_session_tokens.py new file mode 100644 index 000000000..b16813720 --- /dev/null +++ b/corpus/skills/principle-trace-token-burn-loop/scripts/scan_session_tokens.py @@ -0,0 +1,194 @@ +#!/usr/bin/env python3 +"""Fleet-wide per-session token scan for token-burn analysis. + +Companion to engine/skills/reflect/scripts/token_audit.py: token_audit is a +per-session deep dive (pricing, thrash flags); this script is the fleet-level +rollup — scan whole session-store directories, emit one JSONL row per session +file, and let the caller aggregate (by machine, agent, workload class). + +Usage: + scan_session_tokens.py [more roots...] + scan_session_tokens.py --summary [...] # aggregate table instead of rows + cat scan_session_tokens.py | ssh host python3 - ~/.invoker/agent-sessions \ + ~/.claude/projects ~/.codex/sessions > rows.remote.jsonl + +Formats handled (all verified against real session files, 2026-09): + - codex exec stdout JSONL (invoker ~/.invoker/agent-sessions/*.jsonl): + usage lands on type=turn.completed / thread.completed rows + {usage:{input_tokens, cached_input_tokens, output_tokens}} + - codex rollout JSONL (~/.codex/sessions/**): type=token_usage_record rows + carry per-turn payload.usage (token_count event_msg rows are cumulative + and would double-count — they are ignored). + - claude JSONL (~/.claude/projects/**): assistant rows carry + message.usage{input_tokens, output_tokens, cache_read_input_tokens, + cache_creation_input_tokens}. + - OMP JSONL (~/.omp/agent/sessions/**): assistant rows carry + usage.{input,output,cacheRead,cacheWrite}. + +Note: input_tokens semantics differ by harness. Codex input INCLUDES cached +tokens (uncached = input - cached_input_tokens). Claude input EXCLUDES +cache reads (uncached = input_tokens). Rows are emitted raw; the aggregator +(--summary) computes both gross and uncached-equivalent per bucket. + +Known upstream undercount: Invoker's extractCodexUsage reads `cached_tokens` +but codex writes `cached_input_tokens` (packages/execution-engine/src/ +codex-session.ts) — `query cost` therefore reports 0 cached and misses +sessions not linked via attempts.agent_session_id (~95% of volume). +""" +import json, glob, os, sys +from collections import defaultdict + +def is_omp_usage(u): + """OMP is the only harness that writes usage.cacheRead / usage.cacheWrite.""" + return 'cacheRead' in u or 'cacheWrite' in u + + +def classify(head, path): + """Workload bucket from the first ~400KB of a session + its path.""" + if 'llm-judge' in path: + return 'llm-judge-one-shot' + h = head[:400000] + if ('loop-driver.sh' in h or 'mergify_admin_requeue' in h + or 'battle-loop' in h or 'mergify-admin-requeue' in h): + return 'pr-babysit-loop' + if ('A build/test command failed' in h or 'Fix only the failing check' in h + or 'fix-ci' in h): + return 'fix-ci-autofix' + if 'worker-session-mine' in h or 'session-mine' in h or 'session_mine' in h: + return 'session-mining' + if 'invoker-agent-prompt' in h or 'Review claim:' in h or 'Safety invariant:' in h: + return 'invoker-task' + if 'plan-to-invoker' in h or 'planning session' in h.lower(): + return 'planning' + return 'other' + +def scan_file(fp): + inp = cached = cw = out = 0 + agent = None + head = '' + n = 0 + try: + with open(fp, errors='replace') as fh: + for line in fh: + n += 1 + if n <= 400: + head += line + if 'turn.completed' in line or 'thread.completed' in line: + try: + u = json.loads(line).get('usage') + except Exception: + u = None + if isinstance(u, dict): + agent = agent or 'codex' + inp += u.get('input_tokens', 0) + cached += u.get('cached_input_tokens', u.get('cached_tokens', 0) or 0) + cw += u.get('cache_write_input_tokens', 0) + out += u.get('output_tokens', 0) + continue + if 'token_usage_record' in line: + try: + u = json.loads(line).get('payload', {}).get('usage') + except Exception: + u = None + if isinstance(u, dict): + agent = agent or 'codex' + inp += u.get('input_tokens', 0) + cached += u.get('cached_input_tokens', 0) + cw += u.get('cache_write_input_tokens', 0) + out += u.get('output_tokens', 0) + continue + if '"usage"' in line: + try: + e = json.loads(line) + except Exception: + continue + m = e.get('message') + u = (m or {}).get('usage') if isinstance(m, dict) else e.get('usage') + if not isinstance(u, dict): + continue + if is_omp_usage(u): + agent = agent or 'omp' + inp += u.get('input', 0) + cached += u.get('cacheRead', 0) + cw += u.get('cacheWrite', 0) + out += u.get('output', 0) + elif 'cache_read_input_tokens' in u or 'cache_creation_input_tokens' in u or m: + agent = agent or 'claude' + inp += u.get('input_tokens', 0) + cached += u.get('cache_read_input_tokens', 0) + cw += u.get('cache_creation_input_tokens', 0) + out += u.get('output_tokens', 0) + except Exception: + return None + if not (inp or cached or cw or out): + return None + sid = os.path.basename(fp) + if sid.endswith('.jsonl'): + sid = sid[:-6] + return { + 'sid': sid, 'agent': agent or 'unknown', 'cls': classify(head, fp), + 'inp': inp, 'cached': cached, 'cw': cw, 'out': out, + 'mtime': int(os.path.getmtime(fp)), 'path': fp, + } + +def uncached_equiv(r): + if r['agent'] == 'codex': + return max(r['inp'] - r['cached'], 0) + r['out'] + return r['inp'] + r['out'] + +def gross(r): + """Every token the harness billed for on this session, cache included. + + Codex input_tokens already contains the cached subset, so adding cached + again would double-count it. Claude and OMP input excludes cache reads + and writes, so for those the cache fields add on top. + """ + if r['agent'] == 'codex': + return r['inp'] + r['out'] + return r['inp'] + r['out'] + r['cached'] + r['cw'] + +def best_copy_per_session(rows): + """One row per session id, keeping the copy that saw the most tokens. + + A session pushed from another machine can sit beside the local original, + so the same id appears twice; the fuller copy is the complete one. + """ + best = {} + for r in rows: + if r['sid'] not in best or gross(r) > gross(best[r['sid']]): + best[r['sid']] = r + return list(best.values()) + + +def main(): + args = sys.argv[1:] + summary = '--summary' in args + args = [a for a in args if a != '--summary'] + files = [] + for root in args: + root = os.path.expanduser(root) + if os.path.isdir(root): + files += glob.glob(os.path.join(root, '**', '*.jsonl'), recursive=True) + elif os.path.isfile(root): + files.append(root) + rows = [r for r in (scan_file(f) for f in files) if r] + if not summary: + for r in rows: + print(json.dumps(r), flush=True) + return + rows = best_copy_per_session(rows) + agg = defaultdict(lambda: [0, 0, 0.0]) + for r in rows: + k = (r['agent'], r['cls']) + agg[k][0] += gross(r) + agg[k][1] += 1 + agg[k][2] += uncached_equiv(r) + tot_g = sum(v[0] for v in agg.values()) or 1 + tot_u = sum(v[2] for v in agg.values()) or 1 + print(f'{"agent":7}{"class":22}{"gross_tokens":>16} {"share":>6} {"sessions":>9} {"uncached_eq":>14} {"share":>6}') + for (a, c), (g, n, u) in sorted(agg.items(), key=lambda kv: -kv[1][0]): + print(f'{a:7}{c:22}{g:>16,} {100*g/tot_g:5.1f}% {n:>9,} {u:>14,.0f} {100*u/tot_u:5.1f}%') + print(f'{"TOTAL":29}{tot_g:>16,} {"100%":>6} {sum(v[1] for v in agg.values()):>9,} {tot_u:>14,.0f}') + +if __name__ == '__main__': + main() diff --git a/corpus/skills/principle-trace-token-burn-loop/tests/test_scripts.py b/corpus/skills/principle-trace-token-burn-loop/tests/test_scripts.py new file mode 100644 index 000000000..15ce20a48 --- /dev/null +++ b/corpus/skills/principle-trace-token-burn-loop/tests/test_scripts.py @@ -0,0 +1,151 @@ +#!/usr/bin/env python3 +"""Tests for the token-burn mining scripts. + +Reproduces the real failure shapes found in the 2026-09 fleet audit: +codex exec sessions whose input_tokens already include the cached subset +(double-counting inflated fleet totals by ~19B), claude sessions whose +context grows to ~1M then drops on /clear (the sawtooth that makes turns +x context, not task size, the cost driver), and codex rollout files where +cumulative token_count events must not double-count. + +Run: python3 -m unittest discover -s corpus/skills/principle-trace-token-burn-loop/tests -v +""" +from __future__ import annotations + +import json +import os +import sys +import tempfile +import unittest + +SKILL_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(SKILL_DIR, "scripts")) + +import context_growth # noqa: E402 +import scan_session_tokens # noqa: E402 + + +def write_jsonl(rows): + tmp = tempfile.NamedTemporaryFile("w", suffix=".jsonl", delete=False, encoding="utf-8") + for row in rows: + tmp.write(json.dumps(row) + "\n") + tmp.close() + return tmp.name + + +def claude_assistant(inp, cache_read, out, ts="2026-09-20T00:00:00Z"): + return {"type": "assistant", "timestamp": ts, "message": {"role": "assistant", + "usage": {"input_tokens": inp, "cache_read_input_tokens": cache_read, + "output_tokens": out}, "content": []}} + + +class TestScanSessionTokens(unittest.TestCase): + def test_codex_input_already_includes_cached(self): + """Codex counts cached input inside input_tokens, not beside it. + + Real codex usage has total_tokens = input + output, with cached_input + a SUBSET of input. The fleet audit first added cached on top and + inflated gross by ~19B tokens; this is the regression guard. + """ + path = write_jsonl([{ + "type": "turn.completed", + "usage": {"input_tokens": 1000, "cached_input_tokens": 900, + "output_tokens": 50}, + }]) + try: + r = scan_session_tokens.scan_file(path) + finally: + os.unlink(path) + self.assertEqual(r["agent"], "codex") + self.assertEqual(scan_session_tokens.gross(r), 1050) + self.assertEqual(scan_session_tokens.uncached_equiv(r), 150) + + def test_claude_input_excludes_cache_read(self): + path = write_jsonl([claude_assistant(100, 5000, 30)]) + try: + r = scan_session_tokens.scan_file(path) + finally: + os.unlink(path) + self.assertEqual(r["agent"], "claude") + self.assertEqual(scan_session_tokens.gross(r), 5130) + self.assertEqual(scan_session_tokens.uncached_equiv(r), 130) + + def test_ignores_cumulative_token_count_events(self): + """Codex rollout token_count rows are running totals, not per-turn. + + Summing them double-counts; only token_usage_record rows carry + per-turn usage, so the scan must read those and skip the totals. + """ + path = write_jsonl([ + {"type": "event_msg", "payload": {"type": "token_count", "info": { + "total_token_usage": {"input_tokens": 500, "output_tokens": 10}}}}, + {"type": "token_usage_record", "payload": {"usage": { + "input_tokens": 500, "cached_input_tokens": 400, "output_tokens": 10}}}, + ]) + try: + r = scan_session_tokens.scan_file(path) + finally: + os.unlink(path) + self.assertEqual(scan_session_tokens.gross(r), 510) + + def test_classifies_babysit_and_judge(self): + self.assertEqual( + scan_session_tokens.classify("bash ./loop-driver.sh --pr 1", "/x/s.jsonl"), + "pr-babysit-loop") + self.assertEqual( + scan_session_tokens.classify("{}", "/x/-llm-judge-abc/s.jsonl"), + "llm-judge-one-shot") + + +class TestContextGrowth(unittest.TestCase): + def test_sawtooth_counts_clears(self): + """The 3B-token session shape: grow to ~1M, /clear to ~250K, regrow.""" + rows = [] + for cycle in range(3): + for ctx in (300000, 600000, 900000): + rows.append(claude_assistant(ctx // 10, ctx - ctx // 10, 100)) + rows.append(claude_assistant(25000, 225000, 100)) + path = write_jsonl(rows) + try: + s = context_growth.summarize(path) + finally: + os.unlink(path) + self.assertEqual(s["turns"], 12) + self.assertEqual(s["peak_ctx"], 900000) + self.assertEqual(s["clears"], 3) + self.assertGreater(s["gross"], 6000000) + + def test_codex_rollout_last_token_usage(self): + rows = [] + for i, inp in enumerate((10000, 25000, 60000)): + rows.append({"type": "event_msg", "timestamp": f"2026-09-20T00:0{i}:00Z", + "payload": {"type": "token_count", "info": { + "last_token_usage": {"input_tokens": inp, + "cached_input_tokens": inp - 1000, + "output_tokens": 50}, + "total_token_usage": {"input_tokens": inp * (i + 1), + "output_tokens": 50}}}}) + path = write_jsonl(rows) + try: + pts = [p for p in context_growth.series(path) if "ctx" in p] + finally: + os.unlink(path) + self.assertEqual([p["ctx"] for p in pts], [10000, 25000, 60000]) + self.assertEqual(pts[0]["uncached"], 1000) + + def test_exec_format_flags_whole_turn_only(self): + path = write_jsonl([{ + "type": "turn.completed", + "usage": {"input_tokens": 74383580, "cached_input_tokens": 72279296, + "output_tokens": 247893}, + }]) + try: + s = context_growth.summarize(path) + finally: + os.unlink(path) + self.assertEqual(s["gross"], 74383580) + self.assertIn("whole-turn", s["note"]) + + +if __name__ == "__main__": + unittest.main() From 7bc05efb2761fa1dbf3dbde3c532a9ff5acc356e Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Thu, 24 Sep 2026 11:46:05 +0800 Subject: [PATCH 02/10] wait-needs-wakeup: budget same-session wakeups, prefer detached waits Change-Id: Ia2f25b93de0c1c1364032766b987e64629a52b86 --- engine/hooks/hooks.toml | 2 +- engine/hooks/wait-needs-wakeup/README.md | 21 ++++ .../hooks/wait-needs-wakeup/claude.hook.json | 10 ++ engine/hooks/wait-needs-wakeup/detect.py | 90 +++++++++++++-- .../wait-needs-wakeup/tests/test_hooks.py | 104 +++++++++++++++++- 5 files changed, 214 insertions(+), 13 deletions(-) diff --git a/engine/hooks/hooks.toml b/engine/hooks/hooks.toml index bc713bd13..3d06de5eb 100644 --- a/engine/hooks/hooks.toml +++ b/engine/hooks/hooks.toml @@ -235,7 +235,7 @@ enabled_by = "CATSTACK_REFLECT_ENFORCEMENT" [hooks.wait-needs-wakeup] mode = "stop" why_mode = "attention" -summary = "Stops waiting language with no time or wake-up set." +summary = "Stops waiting language with no time or wake-up set, foreground poll loops, and wakeups past the per-session wake budget." [hooks.wrong-check-reflect] mode = "warn" diff --git a/engine/hooks/wait-needs-wakeup/README.md b/engine/hooks/wait-needs-wakeup/README.md index 3da3904ab..92ffab8be 100644 --- a/engine/hooks/wait-needs-wakeup/README.md +++ b/engine/hooks/wait-needs-wakeup/README.md @@ -12,6 +12,15 @@ wait to the harness and names a clock-time ETA. Two entrypoints, one rule: form and passes: it wakes the agent when the condition is met. A background loop with no exit (`while true` and no `break`) is still blocked. +- **PreToolUse on ScheduleWakeup (blocks, exit 2):** a wake budget + (`WAIT_NEEDS_WAKEUP_BUDGET`, default 10). Every ScheduleWakeup resumes + this same transcript, so wake count x context size is the session's + poll-check bill — a babysit session once scheduled 109 wakes inside a + ~500K-token transcript. Past the budget the wait must leave the session: + a `run_in_background` command that exits on its condition, a `Monitor` / + `Agent` that notifies once, or compact before waking again. `CronCreate` + is not budgeted — it spawns a fresh session, so its context does not + accumulate. - **Stop (blocks, exit 2):** a reply that says it is waiting / watching / will report / "nothing needed from you for ~10 minutes" / "N agents still running" must name a clock-time ETA (`back at 07:26 UTC`, `by 07:40 UTC`) @@ -21,6 +30,11 @@ wait to the harness and names a clock-time ETA. Two entrypoints, one rule: 07:24 UTC") is history, not an ETA. Waiting on the user ("waiting on your word") is not waiting on a job and passes. +The guidance ordering is deliberate: detached wakeups (background command +that exits on its condition, Monitor/Agent) are named before +`ScheduleWakeup` because only the detached forms avoid re-sending the +accumulated transcript on every check. + Fail-open on parse or read errors; `stop_hook_active` skips so the rewrite turn can finish. @@ -35,6 +49,13 @@ minutes" and no next contact time. The harness caught one command; the user had to say "if an agent is waiting, it must schedule a wakeup. Not poll." +A later token-burn audit found the flip side: the hook steered agents to +`ScheduleWakeup`, and a multi-day PR-babysit session used it 109 times — +each wake re-sending a ~500K-token transcript, ~3B gross tokens in one +session. Wakeups are a wakeup *mechanism*, not a free wait; the budget cap +exists because a wake inside the same context costs the same as the poll +it replaced. + ## Files - `detect.py` -- loop / sleep / status-check patterns, wait-language and diff --git a/engine/hooks/wait-needs-wakeup/claude.hook.json b/engine/hooks/wait-needs-wakeup/claude.hook.json index deacd41f6..af15e4cb4 100644 --- a/engine/hooks/wait-needs-wakeup/claude.hook.json +++ b/engine/hooks/wait-needs-wakeup/claude.hook.json @@ -10,6 +10,16 @@ "timeout": 5 } ] + }, + { + "matcher": "ScheduleWakeup", + "hooks": [ + { + "type": "command", + "command": "python3 $HOME/.claude/hooks/wait-needs-wakeup/claude_pretooluse.py", + "timeout": 5 + } + ] } ], "Stop": [ diff --git a/engine/hooks/wait-needs-wakeup/detect.py b/engine/hooks/wait-needs-wakeup/detect.py index c7d2703c5..248cadc07 100644 --- a/engine/hooks/wait-needs-wakeup/detect.py +++ b/engine/hooks/wait-needs-wakeup/detect.py @@ -2,17 +2,25 @@ Two shapes, one rule. When the agent is waiting on something (CI, a merge queue, a subagent, an external job) it must hand the wait to the harness -(ScheduleWakeup, Monitor, a run_in_background command that exits when the -condition is met, a running subagent that notifies on completion) and tell -the user a clock-time ETA. It must never burn the foreground on a sleep -loop, and never end a turn with "will report" / "nothing needed from you -for ~10 minutes" and no named next contact time. +(a run_in_background command that exits when the condition is met, a +Monitor/Agent that notifies on completion, or ScheduleWakeup) and tell the +user a clock-time ETA. It must never burn the foreground on a sleep loop, +never end a turn with "will report" / "nothing needed from you for ~10 +minutes" and no named next contact time, and never wake the same giant +transcript without bound. PreToolUse (Bash): block a foreground poll loop (sleep inside a while/until/retry-for with a status check) and a bare foreground sleep of 30 seconds or more. A background command that exits on its condition (until, or a break) is the correct form and passes. +PreToolUse (ScheduleWakeup): block past WAIT_NEEDS_WAKEUP_BUDGET (default +10) wakeups per transcript. Each wake resumes this same context, so a +same-session wake is a poll check billed at full transcript size; past the +budget the wait must detach (background exit-on-condition, Monitor/Agent) +or the transcript must compact first. CronCreate is exempt — it starts a +fresh session. + Stop: block a reply that says it is waiting/watching/will report unless the reply carries a clock-time ETA and the turn has a live wakeup (a wakeup tool call this turn, or a background task launched and not yet reported). @@ -22,6 +30,7 @@ from __future__ import annotations import json +import os import re LOOP_RE = re.compile( @@ -63,15 +72,30 @@ PAST_CLOCK_RE = re.compile(r"\b\w+ed\s+(?:(?:at|by)\s+~?)?$") PRETOOLUSE_MESSAGE = ( - "poll -> schedule a wakeup: ScheduleWakeup / Monitor / run_in_background+exit-on-condition, " - "and tell the user the ETA in their own timezone (wait-needs-wakeup). {reason}" + "poll -> hand the wait to the harness: a run_in_background command that exits " + "on its condition (until/break), or a Monitor/Agent that notifies once — " + "ScheduleWakeup only when neither fits (each wake re-sends this whole " + "transcript). Tell the user the ETA in their own timezone " + "(wait-needs-wakeup). {reason}" ) STOP_MESSAGE = ( "wait-needs-wakeup: this reply says it is waiting / watching / will report, but {gap}. " "State a clock-time ETA in the user's own timezone (`date +%H:%M\\ %Z`, e.g. " - "'back at 12:26 PDT') and schedule the wakeup (ScheduleWakeup / Monitor / a " - "run_in_background command that exits on the condition). Both halves or neither: " - "an ETA with no scheduled wakeup is a promise nothing keeps." + "'back at 12:26 PDT') and hand the wait to the harness — a run_in_background " + "command that exits on its condition, a Monitor/Agent that notifies once, or " + "ScheduleWakeup only when neither fits (each wake re-sends this whole " + "transcript). Both halves or neither: an ETA with no scheduled wakeup is a " + "promise nothing keeps." +) +WAKE_BUDGET_ENV = "WAIT_NEEDS_WAKEUP_BUDGET" +WAKE_BUDGET_DEFAULT = 10 +WAKE_BUDGET_MESSAGE = ( + "wake budget: this session has already scheduled {wakes} wakeups — every " + "wake resumes this whole transcript (session-lifetime token burn is " + "turns x context, not the check itself). Hand the watch to a " + "run_in_background command that exits on its condition, a Monitor/Agent " + "that notifies once, or compact before scheduling another wake " + "(wait-needs-wakeup). Override: {env} env var." ) @@ -139,7 +163,53 @@ def pretooluse_reason(payload: dict) -> str | None: return classify_command(command, bool(tool_input.get("run_in_background"))) +def count_scheduled_wakes(transcript_path: str) -> int: + """ScheduleWakeup tool_use blocks already in the transcript, or -1 when the + transcript cannot be read. Each wake resumes this same session, so the + count is the session's poll-check total — the number that made the + multi-day babysit sessions cost billions of tokens.""" + wakes = 0 + try: + with open(transcript_path, encoding="utf-8") as handle: + for raw in handle: + if '"ScheduleWakeup"' not in raw: + continue + try: + data = json.loads(raw) + except (json.JSONDecodeError, TypeError): + continue + for block in _tool_uses(data): + if block.get("name") == "ScheduleWakeup": + wakes += 1 + except OSError: + return -1 + return wakes + + +def wake_budget(environ=None) -> int: + env = os.environ if environ is None else environ + try: + return int(env.get(WAKE_BUDGET_ENV, "") or WAKE_BUDGET_DEFAULT) + except ValueError: + return WAKE_BUDGET_DEFAULT + + +def decide_wakeup_budget(payload: dict, environ=None) -> str | None: + """Block the (budget+1)-th ScheduleWakeup: past this point the wait must + leave the session (background exit-on-condition, Monitor/Agent) or the + transcript must be compacted first.""" + transcript_path = payload.get("transcript_path") or payload.get("transcriptPath") or "" + if not transcript_path: + return None + wakes = count_scheduled_wakes(transcript_path) + if wakes < 0 or wakes < wake_budget(environ): + return None + return WAKE_BUDGET_MESSAGE.format(wakes=wakes, env=WAKE_BUDGET_ENV) + + def decide_pretooluse(payload: dict) -> str | None: + if payload.get("tool_name") == "ScheduleWakeup": + return decide_wakeup_budget(payload) reason = pretooluse_reason(payload) if not reason: return None diff --git a/engine/hooks/wait-needs-wakeup/tests/test_hooks.py b/engine/hooks/wait-needs-wakeup/tests/test_hooks.py index 4e9bfdbbb..4cdc8fed3 100644 --- a/engine/hooks/wait-needs-wakeup/tests/test_hooks.py +++ b/engine/hooks/wait-needs-wakeup/tests/test_hooks.py @@ -68,9 +68,19 @@ def test_hook_blocks_with_exit_2_and_wakeup_guidance(self): "tool_input": {"command": case["command"], "run_in_background": False}, }) self.assertEqual(code, 2) - self.assertIn("schedule a wakeup", err) + self.assertIn("exits on its condition", err) self.assertIn("wait-needs-wakeup", err) + def test_guidance_prefers_detached_wakeup_over_schedule_wakeup(self): + # Each ScheduleWakeup resumes the same transcript (turns x context + # resend). The detached forms must be named first in the guidance. + case = load("poll_commands_fires.json")[2] + _, err = run_entry(claude_pretooluse, { + "tool_name": "Bash", + "tool_input": {"command": case["command"], "run_in_background": False}, + }) + self.assertLess(err.index("run_in_background"), err.index("ScheduleWakeup")) + def test_blocks_bare_sleep_90_the_harness_refused(self): reason = detect.classify_command("sleep 90; gh pr view 228 --json state", False) self.assertEqual(reason, "bare foreground sleep of 90s") @@ -138,7 +148,7 @@ def test_hook_blocks_ten_minute_reply_with_exit_2(self): os.unlink(path) self.assertEqual(code, 2) self.assertIn("clock-time ETA", err) - self.assertIn("schedule the wakeup", err) + self.assertIn("exits on its condition", err) def test_blocks_when_eta_named_but_no_wakeup(self): reply = "Nothing needed from you for about 10 minutes; back at 07:26 UTC." @@ -199,6 +209,96 @@ def assistant_text(text): return {"type": "assistant", "message": {"role": "assistant", "content": [{"type": "text", "text": text}]}} +def assistant_tool_use(name, tool_input=None): + return {"type": "assistant", "message": {"role": "assistant", "content": [ + {"type": "tool_use", "id": f"t{name}", "name": name, "input": tool_input or {}}]}} + + +def wakeup_transcript(n_wakes): + """The shape that burned billions: a long-lived session that kept + re-waking itself to poll (the real offender scheduled 109 wakeups in one + session, ~500K-token transcript re-sent on every check).""" + lines = [] + for i in range(n_wakes): + lines.append(assistant_tool_use("ScheduleWakeup", {"delay_seconds": 1800})) + lines.append({"type": "user", "message": {"role": "user", "content": f"wake {i}: check again"}}) + return lines + + +class TestWakeBudget(unittest.TestCase): + """Reproduces the multi-day babysit-session burn: unlimited ScheduleWakeup + calls inside one transcript. Past the budget the hook must force the wait + out of the session (detached exit-on-condition command / Monitor / Agent) + or a compaction first.""" + + def test_blocks_wakeup_past_budget_repro(self): + path = transcript_file(wakeup_transcript(detect.wake_budget())) + try: + reason = detect.decide_pretooluse({ + "tool_name": "ScheduleWakeup", + "tool_input": {"delay_seconds": 1800}, + "transcript_path": path, + }) + finally: + os.unlink(path) + self.assertIsNotNone(reason) + self.assertIn("wake budget", reason) + self.assertIn("run_in_background", reason) + + def test_hook_blocks_over_budget_wakeup_with_exit_2(self): + path = transcript_file(wakeup_transcript(detect.wake_budget() + 3)) + try: + code, err = run_entry(claude_pretooluse, { + "tool_name": "ScheduleWakeup", + "tool_input": {"delay_seconds": 1800}, + "transcript_path": path, + }) + finally: + os.unlink(path) + self.assertEqual(code, 2) + self.assertIn("wake budget", err) + + def test_allows_wakeup_under_budget(self): + path = transcript_file(wakeup_transcript(3)) + try: + self.assertIsNone(detect.decide_pretooluse({ + "tool_name": "ScheduleWakeup", + "tool_input": {"delay_seconds": 1800}, + "transcript_path": path, + })) + finally: + os.unlink(path) + + def test_detached_watchers_still_allowed_over_budget(self): + path = transcript_file(wakeup_transcript(detect.wake_budget() + 5)) + try: + for name in ("Monitor", "CronCreate"): + with self.subTest(tool=name): + self.assertIsNone(detect.decide_pretooluse({ + "tool_name": name, + "tool_input": {}, + "transcript_path": path, + })) + finally: + os.unlink(path) + + def test_wakeup_fails_open_without_transcript(self): + self.assertIsNone(detect.decide_pretooluse({ + "tool_name": "ScheduleWakeup", "tool_input": {}})) + self.assertIsNone(detect.decide_pretooluse({ + "tool_name": "ScheduleWakeup", "tool_input": {}, + "transcript_path": "/nonexistent/x.jsonl"})) + + def test_budget_env_override(self): + path = transcript_file(wakeup_transcript(2)) + try: + env = {detect.WAKE_BUDGET_ENV: "2"} + self.assertIsNotNone(detect.decide_wakeup_budget( + {"tool_name": "ScheduleWakeup", "transcript_path": path}, environ=env)) + finally: + os.unlink(path) + + def replay(lines): return list(detect.replay_stop(enumerate(lines))) From d354da8151f3124367f3d680c062116e46999df7 Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Thu, 24 Sep 2026 11:46:06 +0800 Subject: [PATCH 03/10] principle-read-state-artifacts: read the worker's state file, don't re-derive Change-Id: Icbdfb6a0b37bb22d98d04bba79b24dd5b3b975e9 --- .../principle-read-state-artifacts/SKILL.md | 20 ++++ .../scripts/fold_jsonl_state.py | 101 ++++++++++++++++ .../tests/fires_example.md | 17 +++ .../tests/stays_silent_example.md | 11 ++ .../tests/test_fold_jsonl_state.py | 110 ++++++++++++++++++ docs/skill-triggers.md | 4 +- 6 files changed, 261 insertions(+), 2 deletions(-) create mode 100644 corpus/skills/principle-read-state-artifacts/SKILL.md create mode 100755 corpus/skills/principle-read-state-artifacts/scripts/fold_jsonl_state.py create mode 100644 corpus/skills/principle-read-state-artifacts/tests/fires_example.md create mode 100644 corpus/skills/principle-read-state-artifacts/tests/stays_silent_example.md create mode 100644 corpus/skills/principle-read-state-artifacts/tests/test_fold_jsonl_state.py diff --git a/corpus/skills/principle-read-state-artifacts/SKILL.md b/corpus/skills/principle-read-state-artifacts/SKILL.md new file mode 100644 index 000000000..632f30c8a --- /dev/null +++ b/corpus/skills/principle-read-state-artifacts/SKILL.md @@ -0,0 +1,20 @@ +--- +name: principle-read-state-artifacts +description: "Apply when designing, reviewing, or authoring any mechanism that checks on work a background process already tracks: a babysit/watcher loop, a status check on a queue or worker, an agent verifying what a cron already did. Prefer reading the state artifact the worker already wrote (a ledger, digest, or status file) over re-deriving that state with live commands inside a session transcript." +disable-model-invocation: true +--- + +# Read the State Artifact, Don't Re-derive It + +When a deterministic process already computes and records state, re-computing that state inside a session costs far more than the check itself — every command's raw output lands in the transcript and is re-sent on every later turn. + +**Why:** A cron worker that scans a queue every few minutes and appends results to a ledger has already paid for the scan, once, outside any transcript. An agent that re-runs `gh pr list`, `gh pr view`, or a sweep script to learn the same thing pays twice: once for the commands, and again on every subsequent turn that re-sends their accumulated output. The transcript is the bill — a 50-command sweep that adds 500KB of output costs ~500KB of resend per later turn, while reading a folded digest adds a few KB once. The fix is not "check less often"; it is "read what was already computed." + +**Pattern:** +- Before writing a watch/babysit loop that checks live state, ask: does a worker, cron, or daemon already record this state? If yes, the loop's read step is a file read or a folded-digest command, not a live query. +- Prefer a state artifact that folds to *latest* state (a digest, a status file, a reduced view) over a raw append-only log — and over re-deriving anything. If only an append-only ledger exists, the right answer is usually a small fold/digest mode on the producer, not each consumer parsing raw history. This skill ships `scripts/fold_jsonl_state.py` as the generic lever: `fold_jsonl_state.py LEDGER.jsonl --by pr,kind,key` folds an append-only JSONL ledger to latest-per-group in one bounded read. +- A consumer that needs one PR's detail queries *that PR* — it does not re-scan the whole queue to find it. Reserve live calls for entries the digest marks as needing action. +- A watcher that only waits for a terminal condition (merged, green, done) should exit when the artifact reports that condition — not keep polling after the answer is final. +- Treat "the session re-ran the scan N times" as a red flag in review, on par with a poll loop: the information was already free elsewhere. + +**Battle-tested:** An Invoker `pr-admin-bypass-land` cron worker scanned all admin-bypass PRs every 5 minutes and recorded per-PR dispatch state to `mergify-admin-requeue-state.jsonl` — but babysit agent sessions ignored the ledger and re-derived the same state in-transcript: one 74.6M-token session ran `loop-driver.sh` 55 times and `gh pr view` 59 times across ~20 PRs inside a single run, and a 3.0B-token session spent 5,407 turns re-sending a ~550K-token context built largely of repeated `gh`/`ssh` check output. The fleet-level lesson: tokens ≈ turns × accumulated context, so any check that streams raw scan output into a session multiplies its own cost by the session's remaining lifetime. diff --git a/corpus/skills/principle-read-state-artifacts/scripts/fold_jsonl_state.py b/corpus/skills/principle-read-state-artifacts/scripts/fold_jsonl_state.py new file mode 100755 index 000000000..60e148072 --- /dev/null +++ b/corpus/skills/principle-read-state-artifacts/scripts/fold_jsonl_state.py @@ -0,0 +1,101 @@ +#!/usr/bin/env python3 +"""Fold an append-only JSONL state ledger to latest state per group. + +Workers that scan a queue and append rows to a JSONL ledger produce raw +history, not answers. Consumers should read the folded digest — the latest +row per group — instead of re-parsing the whole log or re-querying the live +source. This is the generic fold behind the principle: point it at any +append-only JSONL ledger whose rows carry a timestamp-ish field. + +Usage: + fold_jsonl_state.py LEDGER.jsonl --by pr,kind,key + fold_jsonl_state.py LEDGER.jsonl --by pr --json + +Each group keeps the row with the greatest ordering field (`--order`, +default `epoch`; falls back to input line order when absent). Malformed +lines are skipped and counted on stderr. Output is one compact line per +group, sorted by group key; `--json` emits one JSON object per group. + +Read-only by construction: parses the ledger and prints, nothing else. +""" +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from typing import Sequence + + +def parse_args(argv: Sequence[str]) -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + p.add_argument("ledger", help="Path to the append-only JSONL ledger.") + p.add_argument("--by", required=True, help="Comma-separated fields that identify one group (e.g. pr,kind,key).") + p.add_argument("--order", default="epoch", help="Field that orders rows; greatest wins. Default: epoch. Falls back to line order when the field is missing.") + p.add_argument("--json", action="store_true", help="Emit one JSON object per group instead of text lines.") + return p.parse_args(argv) + + +def load_rows(path: Path) -> tuple[list[dict], int]: + rows: list[dict] = [] + bad = 0 + for line in path.read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + try: + row = json.loads(line) + except json.JSONDecodeError: + bad += 1 + continue + if isinstance(row, dict): + rows.append(row) + else: + bad += 1 + return rows, bad + + +def fold(rows: list[dict], by: list[str], order: str) -> dict[tuple, dict]: + """Latest row per group tuple; ties keep the later input row.""" + latest: dict[tuple, dict] = {} + latest_rank: dict[tuple, tuple[float, int]] = {} + for seq, row in enumerate(rows): + key = tuple(row.get(f) for f in by) + rank_val = row.get(order) + try: + rank_num = float(rank_val) if rank_val is not None else float("-inf") + except (TypeError, ValueError): + rank_num = float("-inf") + rank = (rank_num, seq) + if key not in latest or rank >= latest_rank[key]: + latest[key] = row + latest_rank[key] = rank + return latest + + +def main(argv: Sequence[str] | None = None) -> int: + args = parse_args(argv or sys.argv[1:]) + path = Path(args.ledger).expanduser() + if not path.exists(): + print(f"ledger not found: {path}", file=sys.stderr) + return 1 + by = [f.strip() for f in args.by.split(",") if f.strip()] + if not by: + print("--by must name at least one field", file=sys.stderr) + return 1 + rows, bad = load_rows(path) + if bad: + print(f"skipped {bad} malformed line(s)", file=sys.stderr) + groups = fold(rows, by, args.order) + for key in sorted(groups, key=lambda k: tuple(str(v) for v in k)): + row = groups[key] + if args.json: + print(json.dumps(row, sort_keys=True)) + else: + head = " ".join(f"{f}={row.get(f)}" for f in by) + rest = " ".join(f"{k}={v}" for k, v in sorted(row.items()) if k not in by) + print(f"{head} {rest}".rstrip()) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/corpus/skills/principle-read-state-artifacts/tests/fires_example.md b/corpus/skills/principle-read-state-artifacts/tests/fires_example.md new file mode 100644 index 000000000..41c989868 --- /dev/null +++ b/corpus/skills/principle-read-state-artifacts/tests/fires_example.md @@ -0,0 +1,17 @@ +User types `/principle-read-state-artifacts`. A session is designing a +babysit loop for a stack of PRs: every 10 minutes it runs `gh pr list` +plus `gh pr view` on each PR to check merge state, appending the results +to its transcript. Meanwhile a cron worker already scans the same queue +every 5 minutes and appends per-PR dispatch state to a JSONL ledger on +disk. + +This skill has `disable-model-invocation: true`, so its description is +never loaded into context and never drives auto-triggering — the +explicit `/principle-read-state-artifacts` invocation above is the only +way it activates. Once invoked: this is exactly the described mechanism +— a watcher re-deriving state that a deterministic process already +records. The skill's pattern applies directly: the babysit loop's read +step should fold the ledger (or read a digest file) instead of running +live `gh` sweeps whose output accumulates in the transcript, and the +loop should exit when the ledger reports the stack's terminal +condition. diff --git a/corpus/skills/principle-read-state-artifacts/tests/stays_silent_example.md b/corpus/skills/principle-read-state-artifacts/tests/stays_silent_example.md new file mode 100644 index 000000000..30a077162 --- /dev/null +++ b/corpus/skills/principle-read-state-artifacts/tests/stays_silent_example.md @@ -0,0 +1,11 @@ +A session is designing a babysit loop that re-runs `gh pr list` and +`gh pr view` every 10 minutes to re-derive merge state a cron worker +already records to a ledger — exactly the mechanism this skill names. +The user never types the skill's own slash command anywhere in the +session. + +Stays silent: this skill has `disable-model-invocation: true`, so +nothing about the description or the situation itself can trigger it — +only its own explicit invocation would. Even though the re-derive +pattern is present, without that explicit invocation the skill never +activates. diff --git a/corpus/skills/principle-read-state-artifacts/tests/test_fold_jsonl_state.py b/corpus/skills/principle-read-state-artifacts/tests/test_fold_jsonl_state.py new file mode 100644 index 000000000..125e5ec2f --- /dev/null +++ b/corpus/skills/principle-read-state-artifacts/tests/test_fold_jsonl_state.py @@ -0,0 +1,110 @@ +"""Tests for the fold_jsonl_state ledger digest script.""" +import io +import json +import subprocess +import sys +import tempfile +import unittest +import unittest.mock +from contextlib import redirect_stderr, redirect_stdout +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parents[1] / "scripts" / "fold_jsonl_state.py" +sys.path.insert(0, str(SCRIPT.parent)) +import fold_jsonl_state as fold_mod + + +def write_ledger(tmp: str, rows: list[str]) -> Path: + path = Path(tmp) / "ledger.jsonl" + path.write_text("\n".join(rows) + "\n", encoding="utf-8") + return path + + +class TestFold(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + + def test_latest_row_per_group_wins(self): + rows = [ + {"pr": 1, "kind": "k", "key": "a", "epoch": 100, "state": "old"}, + {"pr": 1, "kind": "k", "key": "a", "epoch": 200, "state": "new"}, + {"pr": 1, "kind": "k", "key": "b", "epoch": 50, "state": "b-state"}, + ] + groups = fold_mod.fold(rows, ["pr", "kind", "key"], "epoch") + self.assertEqual(len(groups), 2) + self.assertEqual(groups[(1, "k", "a")]["state"], "new") + self.assertEqual(groups[(1, "k", "b")]["state"], "b-state") + + def test_tie_keeps_later_input_row(self): + rows = [ + {"pr": 1, "key": "a", "epoch": 100, "state": "first"}, + {"pr": 1, "key": "a", "epoch": 100, "state": "second"}, + ] + groups = fold_mod.fold(rows, ["pr", "key"], "epoch") + self.assertEqual(groups[(1, "a")]["state"], "second") + + def test_missing_order_field_falls_back_to_line_order(self): + rows = [ + {"pr": 1, "key": "a", "state": "early"}, + {"pr": 1, "key": "a", "state": "late"}, + ] + groups = fold_mod.fold(rows, ["pr", "key"], "epoch") + self.assertEqual(groups[(1, "a")]["state"], "late") + + def test_malformed_lines_skipped_and_counted(self): + path = write_ledger(self.tmp.name, [ + '{"pr": 1, "key": "a", "epoch": 1}', + "not json at all", + '{"pr": 2, "key": "b", "epoch": 2}', + ]) + rows, bad = fold_mod.load_rows(path) + self.assertEqual((len(rows), bad), (2, 1)) + + def test_text_output_groups_and_folds(self): + path = write_ledger(self.tmp.name, [ + '{"pr": 1, "key": "a", "epoch": 1, "state": "old"}', + '{"pr": 1, "key": "a", "epoch": 2, "state": "new"}', + '{"pr": 2, "key": "b", "epoch": 1, "state": "only"}', + ]) + out = io.StringIO() + with redirect_stdout(out): + rc = fold_mod.main([str(path), "--by", "pr,key"]) + self.assertEqual(rc, 0) + lines = out.getvalue().strip().splitlines() + self.assertEqual(len(lines), 2) + self.assertIn("state=new", lines[0]) + self.assertNotIn("state=old", out.getvalue()) + self.assertIn("state=only", lines[1]) + + def test_json_output_one_object_per_group(self): + path = write_ledger(self.tmp.name, [ + '{"pr": 1, "key": "a", "epoch": 1, "state": "x"}', + '{"pr": 2, "key": "b", "epoch": 1, "state": "y"}', + ]) + out = io.StringIO() + with redirect_stdout(out): + rc = fold_mod.main([str(path), "--by", "pr,key", "--json"]) + self.assertEqual(rc, 0) + objs = [json.loads(l) for l in out.getvalue().strip().splitlines()] + self.assertEqual([o["pr"] for o in objs], [1, 2]) + self.assertEqual(objs[0]["state"], "x") + + def test_missing_ledger_exits_nonzero(self): + err = io.StringIO() + with redirect_stderr(err): + rc = fold_mod.main(["/nonexistent/x.jsonl", "--by", "pr"]) + self.assertEqual(rc, 1) + self.assertIn("not found", err.getvalue()) + + def test_no_subprocess_calls(self): + """The fold is read-only: it must never spawn a subprocess.""" + path = write_ledger(self.tmp.name, ['{"pr": 1, "epoch": 1}']) + with unittest.mock.patch.object(subprocess, "Popen", side_effect=AssertionError("subprocess spawned")): + with redirect_stdout(io.StringIO()): + rc = fold_mod.main([str(path), "--by", "pr"]) + self.assertEqual(rc, 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/skill-triggers.md b/docs/skill-triggers.md index fb432fe92..d450ffc86 100644 --- a/docs/skill-triggers.md +++ b/docs/skill-triggers.md @@ -63,10 +63,10 @@ The model may invoke these from a description match. Everything here is a gate o `create-skill`, `draft-pr`, `make-pr`, `phrase-judge`, `thrash-reflect-automate`, `principle-flag-your-own-corrections`, `principle-prove-it`, `principle-subagent-inherits-scope`, `prove-it-ship-gate`, `alternatives-considered`, `diu`, `event-wait`, `how`, `land-stack`, `loop-generator`, `narrow-the-scope`, `plan-first`, `ship-a-detector`, `show-me-your-work`, `skill-ab-token-gate`, `spike-and-validate`, `split-scope`, `visual-proof`, `why` -### Explicit invocation only (32) +### Explicit invocation only (33) These carry `disable-model-invocation: true`. Claude Code does not load their `description:` at all, so the only way in is a typed `/`. -`automate-me`, `reflect`, `cat-mode`, `principle-assert-invariants-not-last-bug`, `principle-bind-to-named-inventory`, `principle-build-the-lever`, `principle-encode-lessons-in-structure`, `principle-experience-first`, `principle-explicit-errors`, `principle-fix-root-causes`, `principle-foundational-thinking`, `principle-generalize-from-rejection`, `principle-guard-the-context-window`, `principle-laziness-protocol`, `principle-manage-idle-resumption`, `principle-minimize-reader-load`, `principle-name-the-scorer`, `principle-never-block-on-the-human`, `principle-no-lookahead`, `principle-outcome-oriented-execution`, `principle-push-not-poll`, `principle-report-the-disqualifier`, `principle-scope-the-session`, `principle-separate-before-serializing-shared-state`, `principle-sequence-verifiable-units`, `principle-subtract-before-you-add`, `principle-trace-token-burn-loop`, `principle-type-system-discipline`, `report-rendering`, `admin-bypass-sweep`, `i-have-adhd`, `independent-judge-swarm` +`automate-me`, `reflect`, `cat-mode`, `principle-assert-invariants-not-last-bug`, `principle-bind-to-named-inventory`, `principle-build-the-lever`, `principle-encode-lessons-in-structure`, `principle-experience-first`, `principle-explicit-errors`, `principle-fix-root-causes`, `principle-foundational-thinking`, `principle-generalize-from-rejection`, `principle-guard-the-context-window`, `principle-laziness-protocol`, `principle-manage-idle-resumption`, `principle-minimize-reader-load`, `principle-name-the-scorer`, `principle-never-block-on-the-human`, `principle-no-lookahead`, `principle-outcome-oriented-execution`, `principle-push-not-poll`, `principle-read-state-artifacts`, `principle-report-the-disqualifier`, `principle-scope-the-session`, `principle-separate-before-serializing-shared-state`, `principle-sequence-verifiable-units`, `principle-subtract-before-you-add`, `principle-trace-token-burn-loop`, `principle-type-system-discipline`, `report-rendering`, `admin-bypass-sweep`, `i-have-adhd`, `independent-judge-swarm` From fa9a35a214f45b2f39bc3a55bd692a602a278ae5 Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Thu, 24 Sep 2026 11:46:07 +0800 Subject: [PATCH 04/10] loop-generator: prefer worker state artifacts over live re-derivation Change-Id: I33a25101da2b07069ab57aae4b7daa7e9bdc1639 --- product/skills/loop-generator/SKILL.md | 5 +++++ product/skills/loop-generator/tests/fires_example.md | 7 +++++++ 2 files changed, 12 insertions(+) diff --git a/product/skills/loop-generator/SKILL.md b/product/skills/loop-generator/SKILL.md index 1043bf4c5..f7666ebbf 100644 --- a/product/skills/loop-generator/SKILL.md +++ b/product/skills/loop-generator/SKILL.md @@ -40,11 +40,13 @@ Collect and resolve every field below before drafting. Don't improvise a differe 11. `fail_condition_rule` — repeated-attempt threshold and grouping key (e.g. "3 failures on the same (target, symptom) pair → stop and report, don't keep retrying"). 12. `local_proxy_command` — the safest repeatable local/proxy verification command, or `none`. 13. `write_mode` — one of `diagnostic_only` (never mutate, just report), `worker_owned_writes` (the loop itself makes changes), or `choose_each_run` (ask each time). +14. `state_artifact` — a file or digest an existing worker/cron/daemon already maintains that records the same state (ledger, status file, folded report), or `none`. When set, `target_discovery_command` reads or folds that artifact instead of re-querying the live source, and live calls are reserved for entries the artifact marks as needing action. Interview rules: - Ask for missing `success_criteria` before drafting — don't infer it from a vague goal. - Ask for missing edge cases when `human_only_blockers`, `evidence_sources`, `fail_condition_rule`, or `write_mode` would materially change behavior. +- Ask whether a `state_artifact` already exists before accepting a live `target_discovery_command` — re-deriving state a worker already records is the expensive default, not the neutral one. Before drafting, post a short resolved summary: the filled fields, any defaults taken, open questions if any remain, and a direct check like "Ready to draft?" @@ -68,6 +70,8 @@ Rules: - Keep `Goal`, `Motivation`, success rules, fail rules, and blockers concrete, not aspirational. - Record assumptions explicitly instead of hiding them. - Say what the live target is, how it's rebuilt each round, and how the loop dedupes it — don't hide mutable-state risk. +- When `state_artifact` is set, name it in `Real target` and `Evidence sources`, and make the per-round read a fold/digest of that artifact — each round's read stays bounded in output size, not a full-history re-parse or a live sweep that streams raw results into a session transcript. Reserve live queries for entries the artifact marks as needing action. +- A watch loop states its terminal condition in `Exit conditions` — the loop ends when the watched scope reaches it (e.g. the stack merges), not when a human remembers to stop it. - If the loop will run unattended, log one row per iteration with `show-me-your-work` instead of inventing a second trail format. ## Driver shell script contract @@ -77,6 +81,7 @@ The generated driver must: - parse `--target ` (repeatable), `--state-file `, `--skip-local-check`, and `--help`; - print loop context on start: cwd, branch (if applicable), and the state-file path; - rebuild the live target set from `target_discovery_command` every run — never trust a cached list from a prior round; +- when `state_artifact` is set, implement that rebuild as a bounded fold/digest read of the artifact (latest state per target), not a full-history re-parse and not a live sweep of the underlying source; - dedupe the target set by `target_identity_key`; - print a repeated-failure summary keyed by `fail_condition_rule`, so a stuck target is visible instead of silently retried forever; - when `--target` is passed for inspection, run any dry-run/probe command against a **copy** of mutable state, never the live state file; diff --git a/product/skills/loop-generator/tests/fires_example.md b/product/skills/loop-generator/tests/fires_example.md index c0f4f23b4..261d43564 100644 --- a/product/skills/loop-generator/tests/fires_example.md +++ b/product/skills/loop-generator/tests/fires_example.md @@ -6,3 +6,10 @@ This matches the trigger phrase "retry failed jobs until they pass" and asks for a recurring watch/retry behavior, so the skill runs its interview (loop_name, goal, target_discovery_command, fail_condition_rule, write_mode, etc.) before drafting the instruction doc + driver script pair. + +During the interview the skill also asks whether a `state_artifact` exists +— e.g. a CI status file or ledger a worker already maintains. If the user +answers yes, the generated doc names that artifact in `Real target` and +`Evidence sources`, the per-round read folds it instead of re-querying the +live source, and the `Exit conditions` section states the terminal +condition that ends the watch. From 169086b2326d0a39c5f75f4646095c42ee2c1cfe Mon Sep 17 00:00:00 2001 From: CI Bot Date: Thu, 24 Sep 2026 03:55:26 +0000 Subject: [PATCH 05/10] =?UTF-8?q?invoker:=20wf-1790221802183-134/repair=20?= =?UTF-8?q?=E2=80=94=20Repair=20PR=20#902:=20failed=5Fchecks:=20validate;?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Exit code: 0 From 594420055ae2ad688e837faafc3c8b5103afe760 Mon Sep 17 00:00:00 2001 From: Invoker Bot Date: Fri, 25 Sep 2026 03:47:03 +0000 Subject: [PATCH 06/10] =?UTF-8?q?invoker:=20wf-1789406909517-39/implement-?= =?UTF-8?q?hook-frustration-watchdog=20=E2=80=94=20Put=20the=20frustration?= =?UTF-8?q?-watchdog=20hook=20onto=20the=20shared=20hook=20code.=20Review?= =?UTF-8?q?=20claim:=20This=20hook=20reports=20findings=20to=20the=20share?= =?UTF-8?q?d=20hook=20code,=20which=20applies=20its=20registry=20mode=20an?= =?UTF-8?q?d=20writes=20event=20rows.=20It=20keeps=20mode=20stop.=20Review?= =?UTF-8?q?=20lane:=20behavior=20Safety=20invariant:=20The=20hook=20gives?= =?UTF-8?q?=20the=20same=20stop,=20warn,=20or=20silent=20result=20on=20eve?= =?UTF-8?q?ry=20case=20in=20its=20current=20test=20folder,=20except=20the?= =?UTF-8?q?=20mode=20change=20named=20in=20this=20claim,=20and=20its=20tes?= =?UTF-8?q?t=20folder=20keeps=20exiting=200.=20Effectiveness=20measurement?= =?UTF-8?q?:=20`python3=20-m=20unittest=20discover=20-s=20engine/hooks/fru?= =?UTF-8?q?stration-watchdog/tests`=20exits=200,=20and=20the=20new=20mode-?= =?UTF-8?q?override=20case=20fails=20before=20this=20change.=20Slice=20rat?= =?UTF-8?q?ionale:=20One=20hook=20per=20workflow,=20as=20the=20user=20aske?= =?UTF-8?q?d,=20so=20each=20migration=20is=20reviewed=20on=20its=20own.=20?= =?UTF-8?q?Architectural=20effect:=20The=20frustration-watchdog=20entry=20?= =?UTF-8?q?scripts=20become=20thin=20calls=20into=20the=20shared=20runtime?= =?UTF-8?q?;=20its=20detection=20returns=20findings.=20Goal:=20Stops=20a?= =?UTF-8?q?=20reply=20that=20leaves=20an=20impatient=20user=20waiting.=20K?= =?UTF-8?q?eep=20that=20behavior=20while=20its=20mode=20moves=20into=20the?= =?UTF-8?q?=20registry.=20Motivation:=20Mode=20and=20output=20shape=20live?= =?UTF-8?q?=20inside=20each=20hook=20today;=20the=20shared=20code=20makes?= =?UTF-8?q?=20a=20mode=20change=20a=20one-line=20registry=20edit.=20Altern?= =?UTF-8?q?ative=20considerations:=20Migrating=20several=20hooks=20per=20w?= =?UTF-8?q?orkflow=20was=20set=20aside=20because=20the=20user=20asked=20fo?= =?UTF-8?q?r=20one=20hook=20per=20workflow.=20Implementation=20details:=20?= =?UTF-8?q?Turn=20this=20hook's=20detection=20into=20detect(event)=20retur?= =?UTF-8?q?ning=20Finding=20objects=20with=20stable=20rule=20ids,=20and=20?= =?UTF-8?q?make=20each=20harness=20entry=20script=20call=20run=5Fhook=20fr?= =?UTF-8?q?om=20engine/hooks/=5Fsdk/runtime.py.=20It=20keeps=20mode=20stop?= =?UTF-8?q?.=20Non-goals:=20No=20change=20to=20what=20the=20hook=20detects?= =?UTF-8?q?.=20No=20other=20hook=20changes.=20Layer:=20domain=20Feature=20?= =?UTF-8?q?state:=20active=20Files:=20engine/hooks/frustration-watchdog/cl?= =?UTF-8?q?aude=5Fstop=5Fcheck.py,=20engine/hooks/frustration-watchdog/ins?= =?UTF-8?q?tall=5Fclaude=5Fhook.py,=20engine/hooks/frustration-watchdog/te?= =?UTF-8?q?sts/test=5Fhooks=5Fsdk=5Fmode.py=20Change=20types:=20-=20engine?= =?UTF-8?q?/hooks/frustration-watchdog/claude=5Fstop=5Fcheck.py:=20modify?= =?UTF-8?q?=20-=20engine/hooks/frustration-watchdog/install=5Fclaude=5Fhoo?= =?UTF-8?q?k.py:=20modify=20-=20engine/hooks/frustration-watchdog/tests/te?= =?UTF-8?q?st=5Fhooks=5Fsdk=5Fmode.py:=20create=20Acceptance=20criteria:?= =?UTF-8?q?=20-=20`python3=20-m=20unittest=20discover=20-s=20engine/hooks/?= =?UTF-8?q?frustration-watchdog/tests`=20exits=200.=20-=20With=20CATSTACK?= =?UTF-8?q?=5FHOOK=5FMODE=5FFRUSTRATION=5FWATCHDOG=20set=20to=20warn,=20a?= =?UTF-8?q?=20case=20that=20stops=20today=20produces=20a=20warning=20inste?= =?UTF-8?q?ad,=20proving=20the=20registry=20mode=20drives=20the=20response?= =?UTF-8?q?.=20-=20Each=20finding=20writes=20one=20event=20row=20with=20th?= =?UTF-8?q?e=20hook's=20rule=5Fid.?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Solution: Put the frustration-watchdog hook onto the shared hook code. Review claim: This hook reports findings to the shared hook code, which applies its registry mode and writes event rows. It keeps mode stop. Review lane: behavior Safety invariant: The hook gives the same stop, warn, or silent result on every case in its current test folder, except the mode change named in this claim, and its test folder keeps exiting 0. Effectiveness measurement: `python3 -m unittest discover -s engine/hooks/frustration-watchdog/tests` exits 0, and the new mode-override case fails before this change. Slice rationale: One hook per workflow, as the user asked, so each migration is reviewed on its own. Architectural effect: The frustration-watchdog entry scripts become thin calls into the shared runtime; its detection returns findings. Goal: Stops a reply that leaves an impatient user waiting. Keep that behavior while its mode moves into the registry. Motivation: Mode and output shape live inside each hook today; the shared code makes a mode change a one-line registry edit. Alternative considerations: Migrating several hooks per workflow was set aside because the user asked for one hook per workflow. Implementation details: Turn this hook's detection into detect(event) returning Finding objects with stable rule ids, and make each harness entry script call run_hook from engine/hooks/_sdk/runtime.py. It keeps mode stop. Non-goals: No change to what the hook detects. No other hook changes. Layer: domain Feature state: active Files: engine/hooks/frustration-watchdog/claude_stop_check.py, engine/hooks/frustration-watchdog/install_claude_hook.py, engine/hooks/frustration-watchdog/tests/test_hooks_sdk_mode.py Change types: - engine/hooks/frustration-watchdog/claude_stop_check.py: modify - engine/hooks/frustration-watchdog/install_claude_hook.py: modify - engine/hooks/frustration-watchdog/tests/test_hooks_sdk_mode.py: create Acceptance criteria: - `python3 -m unittest discover -s engine/hooks/frustration-watchdog/tests` exits 0. - With CATSTACK_HOOK_MODE_FRUSTRATION_WATCHDOG set to warn, a case that stops today produces a warning instead, proving the registry mode drives the response. - Each finding writes one event row with the hook's rule_id. Invoker-Finalize-Id: 59b37dd6-33c3-4409-af31-15439a541b51 --- .../frustration-watchdog/claude_stop_check.py | 69 +++++++++----- .../frustration-watchdog/codex_stop_check.py | 20 +++++ .../frustration-watchdog/cursor_stop_check.py | 20 +++++ .../tests/test_hooks_sdk_mode.py | 90 +++++++++++++++++++ 4 files changed, 177 insertions(+), 22 deletions(-) create mode 100644 engine/hooks/frustration-watchdog/codex_stop_check.py create mode 100644 engine/hooks/frustration-watchdog/cursor_stop_check.py create mode 100644 engine/hooks/frustration-watchdog/tests/test_hooks_sdk_mode.py diff --git a/engine/hooks/frustration-watchdog/claude_stop_check.py b/engine/hooks/frustration-watchdog/claude_stop_check.py index 89a01493f..f2610d53e 100755 --- a/engine/hooks/frustration-watchdog/claude_stop_check.py +++ b/engine/hooks/frustration-watchdog/claude_stop_check.py @@ -34,8 +34,15 @@ import os import re import sys +import hashlib from datetime import datetime +sys.path.insert(0, os.path.join( + os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "_sdk")) + +from finding import Finding # noqa: E402 +from runtime import run_hook # noqa: E402 + IMPATIENCE_PATTERNS = [ ("profanity", re.compile(r"\b(fuck\w*|wtf|shit\w*|goddamn|dammit|damn it|stupid)\b", re.I)), ("told-you", re.compile(r"\bi (already |just )?told you\b|\bi asked you not\b|\bi already said\b", re.I)), @@ -83,6 +90,8 @@ """How Claude Code records a tool call a PreToolUse hook refused: an is_error tool_result whose text opens "PreToolUse:Bash hook error: []: ...".""" +RULE_PREFIX = "frustration-watchdog" + def _is_allcaps(text): letters = [c for c in text if c.isalpha()] @@ -203,32 +212,26 @@ def ends_the_wait(message): return bool(NEXT_STEP_RE.search(message) or ETA_RE.search(message)) -def main(): - try: - data = json.load(sys.stdin) - except json.JSONDecodeError: - return +def detect(data): if data.get("stop_hook_active") or data.get("agent_id"): - return + return [] message = data.get("last_assistant_message") or "" transcript_path = data.get("transcript_path") or "" if not message or not transcript_path or not os.path.isfile(transcript_path): - return + return [] try: msgs = human_user_messages(transcript_path) kinds = impatience_kinds(msgs) except Exception as exc: - print(f"catstack-hook-error frustration-watchdog: {type(exc).__name__}: {exc}", file=sys.stderr) - return # fail open: a broken watchdog must never brick a session + return [] # fail open: a broken watchdog must never brick a session if not kinds: - return + return [] if ends_the_wait(message): - return + return [] try: refused = turn_has_hook_refusal(transcript_path) unchecked = None except Exception as e: - print(f"catstack-hook-error frustration-watchdog: {type(e).__name__}: {e}", file=sys.stderr) refused = False unchecked = f"{type(e).__name__}: {e}" head = ( @@ -236,26 +239,48 @@ def main(): "and this reply hands them nothing visible. " ) if refused: - sys.stderr.write( - head + "A hook refused a tool call this turn, so you are the one blocked: " + feedback = ( + head + + "A hook refused a tool call this turn, so you are the one blocked: " "do not hand the user steps to work around it. End the wait: ask them a " "direct question, or state an explicit no-action window " - "(\"nothing needed from you for ~2 min\"). Per CLAUDE.md live-demo rules.\n" + "(\"nothing needed from you for ~2 min\"). Per CLAUDE.md live-demo rules." ) else: - sys.stderr.write( - head + "End the wait: give exactly one " + feedback = ( + head + + "End the wait: give exactly one " "concrete action for the user (\"click X\", \"run Y\", \"say Z\"), ask them a " "direct question, or state an explicit no-action window " - "(\"nothing needed from you for ~2 min\"). Per CLAUDE.md live-demo rules.\n" + "(\"nothing needed from you for ~2 min\"). Per CLAUDE.md live-demo rules." ) if unchecked: - sys.stderr.write( - f"(frustration-watchdog could not read this turn's tool results ({unchecked}), " + feedback = ( + f"catstack-hook-error frustration-watchdog: {unchecked}\n" + + feedback + + "\n" + + f"(frustration-watchdog could not read this turn's tool results ({unchecked}), " "so it could not tell whether a hook refused a tool call; the wording above " - "is the default.)\n" + "is the default.)" ) - sys.exit(2) + primary_kind = sorted(set(kinds))[0] + return [ + Finding( + rule_id=f"{RULE_PREFIX}.{primary_kind}", + subject="reply:" + hashlib.sha256(message.encode("utf-8")).hexdigest(), + message=feedback, + evidence=", ".join(sorted(set(kinds))), + ) + ] + + +def main(): + try: + run_hook("frustration-watchdog", "claude", detect, "Stop") + except SystemExit as exc: + if exc.code == 0: + return + raise if __name__ == "__main__": diff --git a/engine/hooks/frustration-watchdog/codex_stop_check.py b/engine/hooks/frustration-watchdog/codex_stop_check.py new file mode 100644 index 000000000..7d941052c --- /dev/null +++ b/engine/hooks/frustration-watchdog/codex_stop_check.py @@ -0,0 +1,20 @@ +#!/usr/bin/env python3 +"""Codex Stop entrypoint for frustration-watchdog.""" +from __future__ import annotations + +import os +import sys + +sys.path.insert(0, os.path.join( + os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "_sdk")) + +from claude_stop_check import detect # noqa: E402 +from runtime import run_hook # noqa: E402 + + +def main() -> None: + run_hook("frustration-watchdog", "codex", detect, "Stop") + + +if __name__ == "__main__": + main() diff --git a/engine/hooks/frustration-watchdog/cursor_stop_check.py b/engine/hooks/frustration-watchdog/cursor_stop_check.py new file mode 100644 index 000000000..64986bfe1 --- /dev/null +++ b/engine/hooks/frustration-watchdog/cursor_stop_check.py @@ -0,0 +1,20 @@ +#!/usr/bin/env python3 +"""Cursor stop entrypoint for frustration-watchdog.""" +from __future__ import annotations + +import os +import sys + +sys.path.insert(0, os.path.join( + os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "_sdk")) + +from claude_stop_check import detect # noqa: E402 +from runtime import run_hook # noqa: E402 + + +def main() -> None: + run_hook("frustration-watchdog", "cursor", detect, "stop") + + +if __name__ == "__main__": + main() diff --git a/engine/hooks/frustration-watchdog/tests/test_hooks_sdk_mode.py b/engine/hooks/frustration-watchdog/tests/test_hooks_sdk_mode.py new file mode 100644 index 000000000..f26b8e619 --- /dev/null +++ b/engine/hooks/frustration-watchdog/tests/test_hooks_sdk_mode.py @@ -0,0 +1,90 @@ +from __future__ import annotations + +import json +import os +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +from test_hooks import NARRATION, WAITING, human, transcript_lines, tool_turn + +HOOK_DIR = Path(__file__).resolve().parents[1] +HOOK = HOOK_DIR / "claude_stop_check.py" + + +def payload(transcript_path: str) -> dict[str, object]: + return { + "hook_event_name": "Stop", + "session_id": "frustration-watchdog-sdk-mode", + "transcript_path": transcript_path, + "last_assistant_message": NARRATION, + "stop_hook_active": False, + } + + +def run_hook(data: dict[str, object], env: dict[str, str]) -> subprocess.CompletedProcess[str]: + merged_env = os.environ.copy() + merged_env.pop("CATSTACK_HOOK_MODE_FRUSTRATION_WATCHDOG", None) + merged_env.update(env) + return subprocess.run( + [sys.executable, str(HOOK)], + input=json.dumps(data), + capture_output=True, + text=True, + timeout=10, + env=merged_env, + ) + + +class SdkModeTest(unittest.TestCase): + def test_mode_override_warn_turns_stop_into_warning(self) -> None: + transcript = transcript_lines([human(WAITING)]) + try: + result = run_hook( + payload(transcript), + {"CATSTACK_HOOK_MODE_FRUSTRATION_WATCHDOG": "warn"}, + ) + finally: + os.unlink(transcript) + + self.assertEqual(0, result.returncode, result.stderr) + self.assertEqual("", result.stderr) + output = json.loads(result.stdout) + self.assertIn( + "The user's last message was impatience-shaped (waiting)", + output["hookSpecificOutput"]["additionalContext"], + ) + + def test_each_finding_writes_one_event_row_with_rule_id(self) -> None: + with tempfile.TemporaryDirectory() as metrics_dir: + transcript = transcript_lines([human(WAITING)] + tool_turn("Created PR #12", is_error=False)) + try: + result = run_hook( + payload(transcript), + { + "CATSTACK_HOOK_METRICS_DIR": metrics_dir, + "CATSTACK_HOOK_MODE_FRUSTRATION_WATCHDOG": "warn", + }, + ) + finally: + os.unlink(transcript) + rows = self._event_rows(metrics_dir) + + self.assertEqual(0, result.returncode, result.stderr) + self.assertEqual(1, len(rows)) + self.assertEqual("frustration-watchdog", rows[0]["hook"]) + self.assertEqual("frustration-watchdog.waiting", rows[0]["rule_id"]) + self.assertEqual("warn", rows[0]["mode"]) + self.assertEqual("override", rows[0]["mode_source"]) + self.assertEqual("warned", rows[0]["action"]) + + def _event_rows(self, metrics_dir: str) -> list[dict[str, object]]: + files = list(Path(metrics_dir).glob("events-*.jsonl")) + self.assertEqual(1, len(files)) + return [json.loads(line) for line in files[0].read_text(encoding="utf-8").splitlines()] + + +if __name__ == "__main__": + unittest.main() From 24a40ecf990ff36a1ffa174f6e6535f88aaed460 Mon Sep 17 00:00:00 2001 From: Invoker Bot Date: Fri, 25 Sep 2026 03:48:05 +0000 Subject: [PATCH 07/10] =?UTF-8?q?invoker:=20wf-1789406909517-39/verify-hoo?= =?UTF-8?q?k-frustration-watchdog=20=E2=80=94=20Run=20the=20deterministic?= =?UTF-8?q?=20proof=20for=20put=20the=20frustration-watchdog=20hook=20onto?= =?UTF-8?q?=20the=20shared=20hook=20code.=20Review=20claim:=20The=20proof?= =?UTF-8?q?=20exits=200=20only=20when=20put=20the=20frustration-watchdog?= =?UTF-8?q?=20hook=20onto=20the=20shared=20hook=20code=20holds.=20Review?= =?UTF-8?q?=20lane:=20proof=20Safety=20invariant:=20Proof=20only;=20it=20c?= =?UTF-8?q?hanges=20no=20product=20behavior.=20Effectiveness=20measurement?= =?UTF-8?q?:=20The=20exit=20status=20of=20`python3=20-m=20unittest=20disco?= =?UTF-8?q?ver=20-s=20engine/hooks/frustration-watchdog/tests`=20is=20the?= =?UTF-8?q?=20signal=20for=20this=20slice.=20Slice=20rationale:=20One=20pr?= =?UTF-8?q?oof=20unit=20for=20this=20workflow.=20Architectural=20effect:?= =?UTF-8?q?=20None;=20verification=20only.=20Goal:=20Prove=20put=20the=20f?= =?UTF-8?q?rustration-watchdog=20hook=20onto=20the=20shared=20hook=20code?= =?UTF-8?q?=20with=20one=20deterministic=20run.=20Motivation:=20Each=20wor?= =?UTF-8?q?kflow=20carries=20its=20own=20proof=20so=20a=20reviewer=20can?= =?UTF-8?q?=20trust=20the=20slice=20alone.=20Alternative=20considerations:?= =?UTF-8?q?=20Manual=20inspection=20was=20set=20aside=20as=20non-determini?= =?UTF-8?q?stic.=20Implementation=20details:=20Execute=20the=20proof=20as?= =?UTF-8?q?=20a=20terminal=20gate.=20Non-goals:=20No=20product=20edits=20h?= =?UTF-8?q?ere.=20Layer:=20e2e=5Fregression=20Feature=20state:=20active?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Exit code: 0 Invoker-Finalize-Id: fc63a6ee-50e8-405d-ab40-a9e97cbd27bb From 106bcc823f07c3d20314b30f7a43ea114b6f3bae Mon Sep 17 00:00:00 2001 From: Invoker Bot Date: Fri, 25 Sep 2026 03:49:11 +0000 Subject: [PATCH 08/10] =?UTF-8?q?invoker:=20wf-1789406909517-39/scrub-hand?= =?UTF-8?q?off-artifacts=20=E2=80=94=20Terminal=20read-only=20gate=20confi?= =?UTF-8?q?rming=20no=20ephemeral=20handoff=20files=20were=20left=20behind?= =?UTF-8?q?.=20Review=20claim:=20The=20workflow=20leaves=20no=20ephemeral?= =?UTF-8?q?=20handoff=20files=20in=20the=20tree.=20Review=20lane:=20proof?= =?UTF-8?q?=20Safety=20invariant:=20Read-only;=20it=20never=20deletes=20fi?= =?UTF-8?q?les,=20alters=20the=20index,=20or=20commits=20caller=20work.=20?= =?UTF-8?q?Effectiveness=20measurement:=20A=20non-zero=20exit=20when=20eph?= =?UTF-8?q?emeral=20handoff=20files=20remain=20is=20the=20signal.=20Slice?= =?UTF-8?q?=20rationale:=20One=20unit:=20the=20hygiene=20gate.=20Architect?= =?UTF-8?q?ural=20effect:=20None.=20Goal:=20Confirm=20no=20ephemeral=20han?= =?UTF-8?q?doff=20files=20remain=20after=20every=20other=20task=20finishes?= =?UTF-8?q?.=20Motivation:=20Ephemeral=20inter-task=20files=20leak=20into?= =?UTF-8?q?=20the=20diff=20and=20read=20as=20part=20of=20the=20change.=20A?= =?UTF-8?q?lternative=20considerations:=20Manual=20inspection=20was=20set?= =?UTF-8?q?=20aside=20as=20non-deterministic.=20Implementation=20details:?= =?UTF-8?q?=20Run=20scripts/scrub-handoff-artifacts.sh=20in=20check=20mode?= =?UTF-8?q?.=20Non-goals:=20No=20deletion,=20no=20index=20changes,=20no=20?= =?UTF-8?q?commits.=20Layer:=20e2e=5Fregression=20Feature=20state:=20activ?= =?UTF-8?q?e?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Exit code: 0 Invoker-Finalize-Id: 516807f7-381a-482a-960c-6ed9acb3a746 From 629348e40f8c89c9a34e51c14c7d148f169b2c5d Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Fri, 25 Sep 2026 12:00:44 +0800 Subject: [PATCH 09/10] Keep frustration-watchdog's caught-error line The move onto the shared hook code dropped the catstack-hook-error print that main added in #514, leaving the exception unused (ruff F841) and the failure silent. Put the print back. Co-Authored-By: Claude Opus 5.5 (1M context) Change-Id: I50c3d91a397eaeb51ae31bc61ce9e1cfa5bb17d6 --- engine/hooks/frustration-watchdog/claude_stop_check.py | 1 + 1 file changed, 1 insertion(+) diff --git a/engine/hooks/frustration-watchdog/claude_stop_check.py b/engine/hooks/frustration-watchdog/claude_stop_check.py index f2610d53e..74d630de4 100755 --- a/engine/hooks/frustration-watchdog/claude_stop_check.py +++ b/engine/hooks/frustration-watchdog/claude_stop_check.py @@ -223,6 +223,7 @@ def detect(data): msgs = human_user_messages(transcript_path) kinds = impatience_kinds(msgs) except Exception as exc: + print(f"catstack-hook-error frustration-watchdog: {type(exc).__name__}: {exc}", file=sys.stderr) return [] # fail open: a broken watchdog must never brick a session if not kinds: return [] From 478762bcbbc19418ade627131695d70e9424454f Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Fri, 25 Sep 2026 12:51:59 +0800 Subject: [PATCH 10/10] Log frustration-watchdog's second caught error, as main does The migration dropped main's catstack-hook-error print from the turn_has_hook_refusal handler (CI's silent-exception gate) and turned a kept comment into a new comment line (CI's no-new-comments gate). Co-Authored-By: Claude Opus 5.5 (1M context) Change-Id: I0c0f7784d59fadf78cf35cf09ef775cff5047874 --- engine/hooks/frustration-watchdog/claude_stop_check.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/engine/hooks/frustration-watchdog/claude_stop_check.py b/engine/hooks/frustration-watchdog/claude_stop_check.py index 74d630de4..8b9e6f513 100755 --- a/engine/hooks/frustration-watchdog/claude_stop_check.py +++ b/engine/hooks/frustration-watchdog/claude_stop_check.py @@ -224,7 +224,7 @@ def detect(data): kinds = impatience_kinds(msgs) except Exception as exc: print(f"catstack-hook-error frustration-watchdog: {type(exc).__name__}: {exc}", file=sys.stderr) - return [] # fail open: a broken watchdog must never brick a session + return [] if not kinds: return [] if ends_the_wait(message): @@ -233,6 +233,7 @@ def detect(data): refused = turn_has_hook_refusal(transcript_path) unchecked = None except Exception as e: + print(f"catstack-hook-error frustration-watchdog: {type(e).__name__}: {e}", file=sys.stderr) refused = False unchecked = f"{type(e).__name__}: {e}" head = (