Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
25 commits
Select commit Hold shift + click to select a range
97cb800
pr-schema-gate: find the validator in catstack, report UNCHECKED when…
EdbertChan Sep 23, 2026
513bdd7
invoker: wf-1790182316514-5/implement-hook-finds-catstack-checker — R…
EdbertChan Sep 23, 2026
25c09db
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-3 — Re…
EdbertChan Sep 23, 2026
ca78ea9
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-1 — Re…
EdbertChan Sep 23, 2026
70c9a89
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-4 — Re…
EdbertChan Sep 23, 2026
76ef88b
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-2 — Re…
EdbertChan Sep 23, 2026
7610a3d
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-2 — Re…
EdbertChan Sep 23, 2026
d89942d
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-4 — Re…
EdbertChan Sep 23, 2026
20ece5b
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-2 — Re…
EdbertChan Sep 23, 2026
4deb746
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-4 — Re…
EdbertChan Sep 23, 2026
8801be9
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-2 — Re…
EdbertChan Sep 23, 2026
09a8f4b
invoker: wf-1790182316514-5/verify-hook-finds-catstack-checker-4 — Re…
EdbertChan Sep 23, 2026
f2cc6db
Invoker: merge experiment/wf-1790182316514-5/verify-hook-finds-catsta…
EdbertChan Sep 23, 2026
24e8a2d
Invoker: merge experiment/wf-1790182316514-5/verify-hook-finds-catsta…
EdbertChan Sep 23, 2026
848d884
Invoker: merge experiment/wf-1790182316514-5/verify-hook-finds-catsta…
EdbertChan Sep 23, 2026
fecb746
invoker: wf-1790182316514-5/scrub-handoff-artifacts — Review claim: N…
EdbertChan Sep 23, 2026
ea7d365
Merge experiment/wf-1790182316514-5/scrub-handoff-artifacts/g0.t2.a-a…
EdbertChan Sep 23, 2026
ac88c4e
wrong-check-reflect: one shot per reply, and scope the reflect scan t…
EdbertChan Sep 23, 2026
aa4dd73
prove-it-ship-gate: the user's machine is a live surface, a PR link i…
EdbertChan Sep 23, 2026
78178ca
llm-judge: leave out a runner that cannot answer (#799)
EdbertChan Sep 23, 2026
75f2e5d
Merge PR #840 head (ea7d365) to repair review thread PRRT_kwDOT3uYWs6…
Sep 23, 2026
eb0fa2b
pr-schema-gate: a skipped sub-check is not a vacuous pass
Sep 23, 2026
faab129
invoker: wf-1790193889047-345/repair — Resolve bot review thread on P…
Sep 23, 2026
ae99cb3
draft-pr tests: pin the changed-folder-name half of the code-name check
Sep 23, 2026
b310b7f
invoker: wf-1790200429814-416/repair — Repair PR #840 (failed check v…
Sep 23, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions docs/ecosystem.md
Original file line number Diff line number Diff line change
Expand Up @@ -81,6 +81,7 @@ again.
| `frustration-watchdog` | hook |
| `named-verb-guard` | hook |
| `plan-discipline` | hook (not always installed) |
| `prove-it-ship-gate` | hook (Stop; blocks a done/shipped claim about a live surface -- an external service, or the user's own machine, session, or screen -- when the message shows no receipt the run itself emitted) |
| `pr-schema-gate` | hook (advisory; PreToolUse on shell tools; checks direct PR text writes with the repo's own `scripts/validate-pr-body.mjs` and reminds about the stack follow-up; never blocks) |
| `reflect-on-thrash` | hook (off unless `CATSTACK_REFLECT_ENFORCEMENT=1`) |
| `scope-lock` | hook (off unless `CATSTACK_REFLECT_ENFORCEMENT=1`; stops every tool after a second scope correction) |
Expand Down
8 changes: 8 additions & 0 deletions engine/hooks/llm-judge/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,14 @@ A runner fails, and the next one is tried, when its binary is not on `PATH`
parses as a JSON object. Each try is recorded in `attempts` with a reason of at
most 300 characters, taken from the end of stderr or the error text.

A runner that is not installed or exits non-zero is left out of the table for
the next 6 hours, so a runner this account cannot use (a usage limit, a login
it does not have) stops costing every later verdict a failed try. The marker
lives in `unavailable/` under the state folder, and `judge.log` gets a line
saying which runner was left out and why. A timeout or a reply with no JSON
does not leave a runner out. If every runner is left out, the whole table is
tried anyway, and a runner that answers is put back at once.

`CATSTACK_LLM_JUDGE_RUNNERS` replaces the three runners. It is a JSON list of
`[name, argv]` pairs, and any argv item equal to `{prompt}` becomes the prompt.
Tests use it to plug in small fake runners. If it is set but not that shape,
Expand Down
60 changes: 58 additions & 2 deletions engine/hooks/llm-judge/judge.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,8 @@
TIMEOUT_SECONDS = 60
INVESTIGATE_TIMEOUT_CAP = 600
KILL_GRACE_SECONDS = 5
UNAVAILABLE_SECONDS = 6 * 3600
NOT_INSTALLED = "not installed"
REASON_LIMIT = 300
PROMPT_SLOT = "{prompt}"
CHILD_ENV = "CATSTACK_LLM_JUDGE_CHILD"
Expand Down Expand Up @@ -130,7 +132,7 @@ def bounded_timeout(timeout_seconds: object) -> int | float:

def run_runner(name: str, argv: list[str], prompt: str, timeout_seconds: object = TIMEOUT_SECONDS, cwd: object = None) -> tuple[dict, dict | None]:
if shutil.which(argv[0]) is None:
return failed(name, "not installed"), None
return failed(name, NOT_INSTALLED), None
command = [prompt if item == PROMPT_SLOT else item for item in argv]
env = dict(os.environ)
env[CHILD_ENV] = "1"
Expand Down Expand Up @@ -167,15 +169,69 @@ def run_runner(name: str, argv: list[str], prompt: str, timeout_seconds: object
return {"runner": name, "ok": True, "reason": "answered"}, answer


def unavailable_path(name: str) -> str:
digest = hashlib.sha1(name.encode("utf-8")).hexdigest()[:16]
return os.path.join(state_root(), "unavailable", f"{digest}.json")


def unavailable_until(name: str) -> float:
path = unavailable_path(name)
try:
with open(path, encoding="utf-8") as handle:
data = json.load(handle)
except FileNotFoundError:
return 0.0
except (OSError, ValueError) as exc:
log(f"runner {name}: unreadable unavailable marker {path}, keeping the runner: {exc}")
return 0.0
until = data.get("until") if isinstance(data, dict) else None
if isinstance(until, bool) or not isinstance(until, (int, float)):
return 0.0
return float(until)


def shows_unavailable(attempt: dict) -> bool:
reason = attempt.get("reason") or ""
return not attempt.get("ok") and (reason == NOT_INSTALLED or reason.startswith("exit "))


def mark_unavailable(name: str, reason: str) -> None:
try:
write_json_atomic(unavailable_path(name), {"runner": name, "until": time.time() + UNAVAILABLE_SECONDS, "reason": reason})
log(f"runner {name}: left out of the judge table for {UNAVAILABLE_SECONDS}s after: {reason}")
except OSError as exc:
print(f"catstack-hook-error llm-judge: could not mark runner {name} unavailable: {exc}", file=sys.stderr)


def mark_available(name: str) -> None:
path = unavailable_path(name)
if not os.path.exists(path):
return
try:
os.remove(path)
except OSError as exc:
print(f"catstack-hook-error llm-judge: could not clear unavailable marker for {name}: {exc}", file=sys.stderr)


def available_runners(mode: object = None) -> list[tuple[str, list[str]]]:
table = runners(mode)
now = time.time()
kept = [(name, argv) for name, argv in table if unavailable_until(name) <= now]
return kept or table


def ask(prompt: str, mode: object = None, timeout_seconds: object = None, cwd: object = None) -> dict:
if timeout_seconds is None:
timeout_seconds = TIMEOUT_SECONDS
attempts = []
for name, argv in runners(mode):
for name, argv in available_runners(mode):
attempt, answer = run_runner(name, argv, prompt, timeout_seconds=timeout_seconds, cwd=cwd)
attempts.append(attempt)
if answer is not None:
mark_available(name)
return {"outcome": "answered", "runner": name, "answer": answer, "attempts": attempts}
if shows_unavailable(attempt):
mark_unavailable(name, attempt["reason"])
return {"outcome": "unchecked", "runner": None, "answer": None, "attempts": attempts}


Expand Down
36 changes: 36 additions & 0 deletions engine/hooks/llm-judge/tests/test_judge.py
Original file line number Diff line number Diff line change
Expand Up @@ -127,6 +127,42 @@ def test_long_stderr_reason_is_capped_at_300_characters(self):
self.assertLessEqual(len(reason), 300)
self.assertTrue(reason.startswith("exit 1: eee"))

def test_runner_that_exits_non_zero_is_left_out_of_the_next_ask(self):
self.use_runners(EXIT_NONZERO, ANSWER_MATCH)
judge.ask("first")
result = judge.ask("second")
self.assertEqual([a["runner"] for a in result["attempts"]], ["answers"])
self.assertEqual(result["runner"], "answers")

def test_missing_binary_is_left_out_of_the_next_ask(self):
self.use_runners(MISSING_BINARY, ANSWER_MATCH)
judge.ask("first")
self.assertEqual([a["runner"] for a in judge.ask("second")["attempts"]], ["answers"])

def test_left_out_runner_comes_back_after_the_window(self):
self.use_runners(EXIT_NONZERO, ANSWER_MATCH)
judge.ask("first")
with patch.object(judge.time, "time", return_value=time.time() + judge.UNAVAILABLE_SECONDS + 1):
result = judge.ask("later")
self.assertEqual([a["runner"] for a in result["attempts"]], ["crashes", "answers"])

def test_timeout_and_prose_do_not_leave_a_runner_out(self):
self.use_runners(runner("hangs", "import time; time.sleep(30)"), PROSE_ONLY, ANSWER_MATCH)
with patch.object(judge, "TIMEOUT_SECONDS", 1):
judge.ask("first")
result = judge.ask("second")
self.assertEqual([a["runner"] for a in result["attempts"]], ["hangs", "rambles", "answers"])

def test_when_every_runner_is_left_out_the_whole_table_is_tried_and_an_answer_clears_it(self):
flaky = os.path.join(self.state.name, "flaky-ok")
script = f"import json, os, sys; sys.exit(3) if not os.path.exists({flaky!r}) else print(json.dumps({{'match': True}}))"
self.use_runners(runner("flaky", script))
self.assertEqual(judge.ask("first")["outcome"], "unchecked")
open(flaky, "w").close()
result = judge.ask("second")
self.assertEqual(result["runner"], "flaky")
self.assertFalse(os.path.exists(judge.unavailable_path("flaky")))

def test_malformed_runners_env_refuses_instead_of_running_defaults(self):
os.environ[judge.RUNNERS_ENV] = "not json"
with self.assertRaises(ValueError):
Expand Down
38 changes: 31 additions & 7 deletions engine/hooks/pr-schema-gate/README.md
Original file line number Diff line number Diff line change
@@ -1,10 +1,23 @@
# pr-schema-gate

Keeps PR text in the repo's own style without blocking anything. In any repo
that has `scripts/create-pr.mjs`, when a shell tool call writes PR text
directly, the hook checks that text with the repo's own
`scripts/validate-pr-body.mjs` and tells the agent the result. The command
always runs.
Keeps PR text in the repo's own style without blocking anything. When a
shell tool call writes PR text directly, the hook checks that text with the
repo's own validator and tells the agent the result. The command always runs.

## Which repos are in scope

The repo is the directory holding `.git` (a directory, or a worktree's
`.git` file), found by walking up from the command's working directory. Its
validator is the first of these that exists:

1. `scripts/validate-pr-body.mjs` (Invoker)
2. `engine/skills/draft-pr/scripts/validate-pr-body.mjs` (catstack)

A repo with either is fully in scope. A PR-publishing command (`gh pr create`,
`gh pr edit` with a body, `gh api` on `pulls` with a body, `mergify stack
push`) in a git repo with neither reports UNCHECKED, naming both paths it
looked for. It is never silent. Outside any git repo the hook says nothing.
`scripts/create-pr.mjs` plays no part in scope.

## What counts as a direct PR text write

Expand All @@ -31,11 +44,18 @@ Three outcomes, never two:
|---|---|---|
| clean | the validator exits 0 | nothing |
| failed | the validator exits 1 | `pr-schema-gate: the PR text in <file> does not follow this repo's PR style ... The command is not blocked.` plus the validator's error lines (up to 20) |
| unchecked | inline or piped text, a missing or unreadable file, no validator, `node` missing, a crash (any other exit code), a timeout (3s), or a command the parser cannot read | `pr-schema-gate: could not check this PR text against the repo's PR style: <reason>. The command is not blocked.` |
| unchecked | inline or piped text, a missing or unreadable file, no validator at either path, `node` missing, a crash (any other exit code), a timeout (3s), an exit 0 that states no verdict and prints UNCHECKED/SKIPPED/not-installed, or a command the parser cannot read | `pr-schema-gate: could not check this PR text against the repo's PR style: <reason>. The command is not blocked.` |

An unchecked write is never reported as clean. The rules live only in the
repo's validator, so this hook carries no copy of them to drift.

A run that prints `PR body validation passed.` has judged the body, so it is
clean even when the same run names a sub-check it skipped. The catstack
validator says `Summary reading grade unchecked: ...` for a Summary too short
to grade and still accepts the body; reading that note as a vacuous pass told
the agent an accepted body was unchecked and left an owed stack follow-up
armed.

Claude Code gets the message as `additionalContext` on its `Bash` tool.
Every harness also gets it on stderr. Whether Cursor and Codex show a
stderr line from an exit-0 `preToolUse` hook to the agent is unverified.
Expand All @@ -45,6 +65,8 @@ stderr line from an exit-0 `preToolUse` hook to the agent is unverified.
`mergify stack push` publishes PRs with a bare body. The push is told which
follow-up is owed (`node scripts/create-pr.mjs ... --update-existing`, or a
direct body write whose file passes the validator) and arms a pending flag.
In a repo with no validator the push reports UNCHECKED instead and arms
nothing.
A later push while the flag is armed repeats the reminder. Either follow-up
clears it; an unchecked or failing direct write does not.

Expand All @@ -68,7 +90,9 @@ The hook never blocks, so every failure fails open, and says so:
with no local checkout under
`PR_SCHEMA_GATE_CHECKOUTS_ROOT` (default `~/Documents/GitHub`): out of
scope;
- a repo with no `scripts/create-pr.mjs`: out of scope.
- a directory outside any git repo: out of scope;
- an unparseable command in a repo with no validator: silent, since it may
not be a PR write at all.

There is no escape hatch because there is nothing to escape.

Expand Down
15 changes: 12 additions & 3 deletions engine/hooks/pr-schema-gate/claude_pretooluse.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,15 +16,18 @@

from detect import ( # noqa: E402
UNPARSEABLE_MESSAGE,
VALIDATOR_RELATIVE_PATHS,
check_body_file,
classify_pr_text_write,
clear_pending,
find_validator,
followup_message,
is_create_pr_followup,
is_stack_push,
mark_pending,
read_pending,
scope_root,
stack_push_unchecked_message,
style_message,
)
from shell_model import parse_commands, shell_call_from_tool_input # noqa: E402
Expand Down Expand Up @@ -66,7 +69,8 @@ def evaluate(payload: dict) -> list[str]:
commands = parse_commands(call, session_cwd)
if commands is None:
base = call.workdir or session_cwd
return [UNPARSEABLE_MESSAGE] if scope_root(base, None) else []
root = scope_root(base, None)
return [UNPARSEABLE_MESSAGE] if root and find_validator(root) else []

messages: list[str] = []
for command in commands:
Expand All @@ -83,7 +87,8 @@ def evaluate(payload: dict) -> list[str]:
outcome, detail = check_body_file(root, write.body_file, write.cwd)
if outcome == "clean":
clear_pending(root)
message = style_message(outcome, detail, write.body_file)
message = style_message(outcome, detail, write.body_file,
find_validator(root) or VALIDATOR_RELATIVE_PATHS[0])
if message:
messages.append(message)
continue
Expand All @@ -93,7 +98,11 @@ def evaluate(payload: dict) -> list[str]:
if is_create_pr_followup(command):
clear_pending(root)
elif is_stack_push(command):
messages.append(followup_message(read_pending(root) is not None))
validator = find_validator(root)
if validator is None:
messages.append(stack_push_unchecked_message())
continue
messages.append(followup_message(read_pending(root) is not None, validator))
mark_pending(root)
return messages

Expand Down
Loading
Loading