From 6bf9040753bd6a337184ec44ce878b734fa527de Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Tue, 22 Sep 2026 00:43:56 -0700 Subject: [PATCH] unverified-tag-ledger: read the turn's tools from the transcript, not a key Claude Code never sends detect.py read `payload["tools_used"]` (or `tool_names`). A real Claude Code Stop payload has neither key -- captured live and committed as tests/fixtures/claude-stop-payload.json. So `_tools_used` always returned the empty set, the discharge branch in record_turn was unreachable, and every Stop turn that emitted a tag was judged as "ran no verification tool". Live state of the ledger before this change: 35 rows across 25 session caches, 0 resolved, oldest 72 turns old. The suite was green over the dead path because every test supplied `tools_used` itself. The tests now build their payloads from the captured fixture's key set and drive a real captured transcript, and a contract test walks detect.py's AST for every `payload.get("k")` and fails if the key is absent from the captured payload. Tools now come from transcript_path, the way scope-lock/detect.py already reads them. The read has three outcomes: a set, an empty set, and None for unchecked. Unchecked neither refuses the turn nor discharges a row, and says why on stderr. Co-Authored-By: Claude Opus 5 (1M context) Change-Id: I0ec5bbd9e302c882c2f4aa68df4fa4e0e4d33595 --- engine/hooks/unverified-tag-ledger/README.md | 8 + engine/hooks/unverified-tag-ledger/detect.py | 174 +++++++++++++++--- .../tests/fixtures/claude-stop-payload.json | 1 + .../tests/fixtures/claude-transcript.jsonl | 5 + .../unverified-tag-ledger/tests/test_hooks.py | 138 ++++++++++++-- 5 files changed, 286 insertions(+), 40 deletions(-) create mode 100644 engine/hooks/unverified-tag-ledger/tests/fixtures/claude-stop-payload.json create mode 100644 engine/hooks/unverified-tag-ledger/tests/fixtures/claude-transcript.jsonl diff --git a/engine/hooks/unverified-tag-ledger/README.md b/engine/hooks/unverified-tag-ledger/README.md index 150247b4..ccfa1d8c 100644 --- a/engine/hooks/unverified-tag-ledger/README.md +++ b/engine/hooks/unverified-tag-ledger/README.md @@ -43,6 +43,14 @@ fixtures in `tests/test_hooks.py`. behaviour without preventing the turn from ending at all. - **Discharge** — a claim is resolved when a later turn runs a verification tool (`Bash`, `Read`, `Grep`, `Glob`, `NotebookRead`) and stops re-emitting it. +- **Where the turn's tool list comes from** — the transcript named by + `transcript_path`, read the way `scope-lock/detect.py` reads it. A Claude Code + Stop payload carries no tool list of any kind; the captured one in + `tests/fixtures/claude-stop-payload.json` is the record of that. The read has + three outcomes, not two: a set of names, an empty set, and *unchecked* when + the transcript is missing or unreadable. Unchecked is not "no tools" — an + unchecked turn is neither refused nor allowed to discharge a row, and the + reason is written to stderr. - **Escalation** — a claim outstanding `ESCALATE_AFTER_TURNS` (3) turns or more is reported as a reflect trigger rather than accumulating quietly. diff --git a/engine/hooks/unverified-tag-ledger/detect.py b/engine/hooks/unverified-tag-ledger/detect.py index c5b70f9a..8a47c2ef 100644 --- a/engine/hooks/unverified-tag-ledger/detect.py +++ b/engine/hooks/unverified-tag-ledger/detect.py @@ -96,12 +96,17 @@ def outstanding(rows: list[dict]) -> list[dict]: return [row for row in rows if not row.get("resolved")] -def record_turn(session_id: str, message: str, tools_used: set[str], now=None) -> list[dict]: - """Log new tags, discharge ones this turn verified and dropped.""" +def record_turn(session_id: str, message: str, tools_used: set[str] | None, now=None) -> list[dict]: + """Log new tags, discharge ones this turn verified and dropped. + + `tools_used` is None when the turn's tool calls could not be read. That is + not the same as "ran no tools": an unchecked turn discharges nothing, so a + row stays outstanding rather than being retired on no evidence. + """ stamp = now() if now else time.time() rows = read_ledger(session_id) present = {tag["claim"] for tag in parse_tags(message)} - verified = bool(tools_used & VERIFY_TOOLS) + verified = tools_used is not None and bool(tools_used & VERIFY_TOOLS) for row in rows: if row.get("resolved"): @@ -171,13 +176,20 @@ def evaluate(payload: dict) -> dict: """ session_id = str(payload.get("session_id") or "") message = _last_assistant_text(payload) - tools = _tools_used(payload) + tools = tools_used_this_turn(payload) rows = record_turn(session_id, message, tools) new_claims = {tag["claim"] for tag in parse_tags(message)} if not new_claims: return {"note": "", "block": ""} + if tools is None: + return {"note": ( + f"unverified-tag-ledger: logged {len(new_claims)} CAT-UNVERIFIED claim(s), but this " + "turn's tool calls could not be read from transcript_path (see the line above), so " + "whether a check was attempted is UNCHECKED, not clean. Nothing was discharged and " + "the turn was not refused."), "block": ""} + if not tools & VERIFY_TOOLS and not payload.get("stop_hook_active"): claims = "; ".join(sorted(new_claims)[:MAX_LISTED]) return {"note": "", "block": ( @@ -202,24 +214,136 @@ def decide_stop(payload: dict) -> str: def _last_assistant_text(payload: dict) -> str: - for key in ("last_assistant_message", "assistant_message", "message"): - value = payload.get(key) - if isinstance(value, str) and value.strip(): - return value - transcript = payload.get("transcript") or [] - if isinstance(transcript, list): - for entry in reversed(transcript): - if isinstance(entry, dict) and entry.get("role") == "assistant": - content = entry.get("content") - if isinstance(content, str): - return content - return "" - - -def _tools_used(payload: dict) -> set[str]: - raw = payload.get("tools_used") or payload.get("tool_names") or [] - if isinstance(raw, str): - return {raw} - if isinstance(raw, list): - return {str(item) for item in raw} - return set() + """The reply this Stop event is about. + + `last_assistant_message` is the key a real Claude Code Stop payload + carries -- see tests/fixtures/claude-stop-payload.json, captured from a + live run. The transcript is the fallback for a payload that omits it. + """ + value = payload.get("last_assistant_message") + if isinstance(value, str) and value.strip(): + return value + path = payload.get("transcript_path") + if not isinstance(path, str) or not path: + return "" + text = "" + try: + with open(path, encoding="utf-8") as handle: + for line in handle: + line = line.strip() + if not line: + continue + try: + entry = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(entry, dict) and _assistant_text(entry): + text = _assistant_text(entry) + except (OSError, UnicodeError) as exc: + sys.stderr.write( + f"unverified-tag-ledger: cannot read transcript {path}: {exc!r}\n") + return "" + return text + + +def _entry_content(entry: dict): + message = entry.get("message") + if isinstance(message, dict): + return str(message.get("role") or ""), message.get("content") + return str(entry.get("role") or ""), entry.get("content") + + +def _assistant_text(entry: dict) -> str: + role, content = _entry_content(entry) + if entry.get("type") != "assistant" and role != "assistant": + return "" + if isinstance(content, str): + return content + if not isinstance(content, list): + return "" + return "\n".join( + str(block.get("text") or "") + for block in content + if isinstance(block, dict) and block.get("type") in {"text", "output_text"} + ).strip() + + +def _tool_names(entry: dict) -> list[str]: + role, content = _entry_content(entry) + if entry.get("type") != "assistant" and role != "assistant": + return [] + if not isinstance(content, list): + return [] + return [ + str(block.get("name") or "") + for block in content + if isinstance(block, dict) and block.get("type") == "tool_use" + ] + + +def _starts_a_new_turn(entry: dict) -> bool: + """True for a real user prompt -- the boundary this turn's tools start at. + + A `tool_result` arrives as a user entry too, and so does the hook's own + feedback (`isMeta`). Neither is the user speaking, so neither ends the + turn whose tool calls we are counting. + """ + role, content = _entry_content(entry) + if entry.get("type") != "user" and role != "user": + return False + if entry.get("isMeta"): + return False + if isinstance(content, str): + return bool(content.strip()) + if not isinstance(content, list): + return False + return not any( + isinstance(block, dict) and block.get("type") == "tool_result" + for block in content + ) + + +def tools_used_this_turn(payload: dict) -> set[str] | None: + """Tool names this turn actually called, or None when that cannot be read. + + Claude Code's Stop payload has no tool list of any kind -- see the + captured fixture. The turn's tool calls live in the transcript, which is + how `scope-lock/detect.py` reads the same thing. + + Three outcomes, not two: a set (checked), an empty set (checked, no tools), + and None (unchecked). None is not "no tools" -- a caller that collapses it + to an empty set would block a turn it never managed to inspect, and would + discharge ledger rows on no evidence. + """ + path = payload.get("transcript_path") + if not isinstance(path, str) or not path: + sys.stderr.write( + "unverified-tag-ledger: payload carries no transcript_path, " + "this turn's tool list is unchecked\n") + return None + names: set[str] = set() + try: + with open(path, encoding="utf-8") as handle: + for line in handle: + line = line.strip() + if not line: + continue + try: + entry = json.loads(line) + except json.JSONDecodeError as exc: + sys.stderr.write( + f"unverified-tag-ledger: {path} has a non-JSON line, " + f"tool list is unchecked: {exc}\n") + return None + if not isinstance(entry, dict): + continue + if _starts_a_new_turn(entry): + names = set() + continue + names.update(name for name in _tool_names(entry) if name) + except (OSError, UnicodeError) as exc: + sys.stderr.write( + f"unverified-tag-ledger: cannot read transcript {path}, " + f"tool list is unchecked: {exc!r}\n") + return None + return names diff --git a/engine/hooks/unverified-tag-ledger/tests/fixtures/claude-stop-payload.json b/engine/hooks/unverified-tag-ledger/tests/fixtures/claude-stop-payload.json new file mode 100644 index 00000000..f4ee4582 --- /dev/null +++ b/engine/hooks/unverified-tag-ledger/tests/fixtures/claude-stop-payload.json @@ -0,0 +1 @@ +{"session_id":"f4d915a9-b685-4a03-b2fb-63f560eb4a21","transcript_path":"/Users/edbertchan/.claude/projects/-private-tmp-claude-501--Users-edbertchan-Documents-GitHub-catstack-667f4e30-1169-4c52-8310-c66a5261dc95-scratchpad-capture-work/f4d915a9-b685-4a03-b2fb-63f560eb4a21.jsonl","cwd":"/private/tmp/claude-501/-Users-edbertchan-Documents-GitHub-catstack/667f4e30-1169-4c52-8310-c66a5261dc95/scratchpad/capture/work","prompt_id":"82dad848-caee-4ebd-9834-90f4c68ff6bf","permission_mode":"default","effort":{"level":"high"},"hook_event_name":"Stop","stop_hook_active":false,"last_assistant_message":"done.","background_tasks":[],"session_crons":[]} diff --git a/engine/hooks/unverified-tag-ledger/tests/fixtures/claude-transcript.jsonl b/engine/hooks/unverified-tag-ledger/tests/fixtures/claude-transcript.jsonl new file mode 100644 index 00000000..cca6675e --- /dev/null +++ b/engine/hooks/unverified-tag-ledger/tests/fixtures/claude-transcript.jsonl @@ -0,0 +1,5 @@ +{"parentUuid":null,"isSidechain":false,"promptId":"82dad848-caee-4ebd-9834-90f4c68ff6bf","type":"user","message":{"role":"user","content":"Run the Bash tool once with the command: echo hello-capture. Then reply with exactly: done."},"uuid":"452f4560-6849-4545-87a9-8ee5b12868fe","timestamp":"2026-09-22T07:31:27.890Z","permissionMode":"default","promptSource":"sdk","turnOrigin":"sdk","userType":"external","entrypoint":"sdk-cli","cwd":"/private/tmp/claude-501/-Users-edbertchan-Documents-GitHub-catstack/667f4e30-1169-4c52-8310-c66a5261dc95/scratchpad/capture/work","sessionId":"f4d915a9-b685-4a03-b2fb-63f560eb4a21","version":"2.1.278","gitBranch":"HEAD"} +{"parentUuid":"88d4942c-23c1-4d6f-b02c-ae421e99230a","isSidechain":false,"message":{"model":"claude-sonnet-5","id":"msg_011CfJ62w9A2M9rhdykcDMGG","type":"message","role":"assistant","content":[{"type":"thinking","thinking":"","signature":"Eq4CCqgBCBIYAipA2/IPIRV/4tdfFDHn4quDINZ8tue8+3Fj2/AgycscQWBKZV/XcakNCrrVXNeTTjC6c+POgm7o2J1rWLjPeSBK7zIPY2xhdWRlLXNvbm5ldC01OABCCHRoaW5raW5nWiRiMThhNmYzZC01MDA4LTQyMGEtOTk1OC03OGFiNTU5MTI3NDFyEBYjjRHMxvzl8Lo/e0L/MyiIAQGoAdTdyNUGsAECEgzBpTQOAnMHDRdJLaEaDLuZ8BH+Kypl/xbkdCIwkDJ0GVthbchFvnzH5NR5DUcqkiR4TEcodkhqlTZVjI/8qlE4H/zGZLVtgJIcYDyeKjNCswG9GnINJGdyE6nGOZnhuqJT3scDBSpJqf4HCvkExe0Ho2DGSsYSYLtNZEJxayZ1M6sYAQ=="}],"container":null,"stop_reason":"tool_use","stop_sequence":null,"stop_details":null,"usage":{"input_tokens":2,"cache_creation_input_tokens":31990,"cache_read_input_tokens":18639,"output_tokens":95,"output_tokens_details":{"thinking_tokens":14},"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"service_tier":"standard","cache_creation":{"ephemeral_1h_input_tokens":31990,"ephemeral_5m_input_tokens":0},"inference_geo":"not_available","iterations":[{"input_tokens":2,"output_tokens":95,"cache_read_input_tokens":18639,"cache_creation_input_tokens":31990,"cache_creation":{"ephemeral_5m_input_tokens":0,"ephemeral_1h_input_tokens":31990},"type":"message"}],"speed":"standard"},"input_transformations":[],"diagnostics":null,"context_management":null},"apiBlockIndex":0,"requestId":"req_011CfJ62vq3wQ499FxAMXYuK","type":"assistant","uuid":"75efffd8-901b-4a8c-9bfd-faf3062e94d3","timestamp":"2026-09-22T07:31:33.025Z","effort":"high","perTurnEffort":null,"userType":"external","entrypoint":"sdk-cli","cwd":"/private/tmp/claude-501/-Users-edbertchan-Documents-GitHub-catstack/667f4e30-1169-4c52-8310-c66a5261dc95/scratchpad/capture/work","sessionId":"f4d915a9-b685-4a03-b2fb-63f560eb4a21","version":"2.1.278","gitBranch":"HEAD"} +{"parentUuid":"75efffd8-901b-4a8c-9bfd-faf3062e94d3","isSidechain":false,"message":{"model":"claude-sonnet-5","id":"msg_011CfJ62w9A2M9rhdykcDMGG","type":"message","role":"assistant","content":[{"type":"tool_use","id":"toolu_01BWL7hmHoxVvT55xngKgrWA","name":"Bash","input":{"command":"echo hello-capture","description":"Print hello-capture"},"caller":{"type":"direct"}}],"container":null,"stop_reason":"tool_use","stop_sequence":null,"stop_details":null,"usage":{"input_tokens":2,"cache_creation_input_tokens":31990,"cache_read_input_tokens":18639,"output_tokens":95,"output_tokens_details":{"thinking_tokens":14},"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"service_tier":"standard","cache_creation":{"ephemeral_1h_input_tokens":31990,"ephemeral_5m_input_tokens":0},"inference_geo":"not_available","iterations":[{"input_tokens":2,"output_tokens":95,"cache_read_input_tokens":18639,"cache_creation_input_tokens":31990,"cache_creation":{"ephemeral_5m_input_tokens":0,"ephemeral_1h_input_tokens":31990},"type":"message"}],"speed":"standard"},"input_transformations":[],"diagnostics":null,"context_management":null},"wireToolInputs":{"toolu_01BWL7hmHoxVvT55xngKgrWA":{"command":"echo hello-capture","description":"Print hello-capture"}},"apiBlockIndex":1,"requestId":"req_011CfJ62vq3wQ499FxAMXYuK","type":"assistant","uuid":"7f6bd8f3-9a45-4334-bde4-9e16bb3fd52e","timestamp":"2026-09-22T07:31:33.027Z","effort":"high","perTurnEffort":null,"userType":"external","entrypoint":"sdk-cli","cwd":"/private/tmp/claude-501/-Users-edbertchan-Documents-GitHub-catstack/667f4e30-1169-4c52-8310-c66a5261dc95/scratchpad/capture/work","sessionId":"f4d915a9-b685-4a03-b2fb-63f560eb4a21","version":"2.1.278","gitBranch":"HEAD"} +{"parentUuid":"7f6bd8f3-9a45-4334-bde4-9e16bb3fd52e","isSidechain":false,"promptId":"82dad848-caee-4ebd-9834-90f4c68ff6bf","type":"user","message":{"role":"user","content":[{"tool_use_id":"toolu_01BWL7hmHoxVvT55xngKgrWA","type":"tool_result","content":"hello-capture","is_error":false}]},"uuid":"922b472c-ec57-49eb-b784-8fd6606b6ee5","timestamp":"2026-09-22T07:31:35.475Z","toolUseResult":{"stdout":"hello-capture","stderr":"","interrupted":false,"isImage":false,"noOutputExpected":false},"sourceToolAssistantUUID":"7f6bd8f3-9a45-4334-bde4-9e16bb3fd52e","userType":"external","entrypoint":"sdk-cli","cwd":"/private/tmp/claude-501/-Users-edbertchan-Documents-GitHub-catstack/667f4e30-1169-4c52-8310-c66a5261dc95/scratchpad/capture/work","sessionId":"f4d915a9-b685-4a03-b2fb-63f560eb4a21","version":"2.1.278","gitBranch":"HEAD"} +{"parentUuid":"f6dd2d75-5353-4d8a-98ea-868abdd245e9","isSidechain":false,"message":{"model":"claude-sonnet-5","id":"msg_011CfJ63Hi9yPWxCGLJniyzr","type":"message","role":"assistant","content":[{"type":"text","text":"done."}],"container":null,"stop_reason":"end_turn","stop_sequence":null,"stop_details":null,"usage":{"input_tokens":2,"cache_creation_input_tokens":145,"cache_read_input_tokens":50629,"output_tokens":4,"output_tokens_details":{"thinking_tokens":0},"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"service_tier":"standard","cache_creation":{"ephemeral_1h_input_tokens":145,"ephemeral_5m_input_tokens":0},"inference_geo":"not_available","iterations":[{"input_tokens":2,"output_tokens":4,"cache_read_input_tokens":50629,"cache_creation_input_tokens":145,"cache_creation":{"ephemeral_5m_input_tokens":0,"ephemeral_1h_input_tokens":145},"type":"message"}],"speed":"standard"},"input_transformations":[],"diagnostics":null,"context_management":null},"apiBlockIndex":0,"requestId":"req_011CfJ63HNa5AVatjNvpR5eJ","type":"assistant","uuid":"8e60555e-f456-4803-b26e-9b4caa1cf68a","timestamp":"2026-09-22T07:31:36.925Z","effort":"high","perTurnEffort":null,"userType":"external","entrypoint":"sdk-cli","cwd":"/private/tmp/claude-501/-Users-edbertchan-Documents-GitHub-catstack/667f4e30-1169-4c52-8310-c66a5261dc95/scratchpad/capture/work","sessionId":"f4d915a9-b685-4a03-b2fb-63f560eb4a21","version":"2.1.278","gitBranch":"HEAD"} diff --git a/engine/hooks/unverified-tag-ledger/tests/test_hooks.py b/engine/hooks/unverified-tag-ledger/tests/test_hooks.py index 88f03839..12ab9931 100644 --- a/engine/hooks/unverified-tag-ledger/tests/test_hooks.py +++ b/engine/hooks/unverified-tag-ledger/tests/test_hooks.py @@ -2,16 +2,31 @@ """The positive fixtures are the two real tags from the session that motivated this hook (2026-09-11, NiceSpeak streaming): both were emitted, both ended the turn, neither left a trace anywhere. + +`fixtures/claude-stop-payload.json` and `fixtures/claude-transcript.jsonl` are +captured from a live Claude Code run, not written by hand. Hand-written +payloads are what let this hook read `tools_used` -- a key Claude Code has +never sent -- for its whole life: every test supplied the key itself, so the +suite stayed green over a branch that could not run. `test_payload_contract...` +below is the catch: every key `detect.py` reads off the payload has to exist +in the captured one. """ from __future__ import annotations +import ast +import io +import json import os import sys import tempfile import unittest +from contextlib import redirect_stderr HERE = os.path.dirname(os.path.abspath(__file__)) HOOK = os.path.dirname(HERE) +FIXTURES = os.path.join(HERE, "fixtures") +REAL_PAYLOAD = os.path.join(FIXTURES, "claude-stop-payload.json") +REAL_TRANSCRIPT = os.path.join(FIXTURES, "claude-transcript.jsonl") sys.path.insert(0, HOOK) REAL_TAG_1 = ( @@ -24,6 +39,11 @@ MALFORMED = "{{CAT-UNVERIFIED: something I did not check}}" +def _real_transcript_lines() -> list[str]: + with open(REAL_TRANSCRIPT, encoding="utf-8") as handle: + return [line for line in handle if line.strip()] + + class LedgerTests(unittest.TestCase): def setUp(self) -> None: self.tmp = tempfile.TemporaryDirectory() @@ -37,6 +57,83 @@ def tearDown(self) -> None: os.environ.pop("CATSTACK_TAG_LEDGER_DIR", None) self.tmp.cleanup() + def transcript(self, *, tools: bool, name: str = "transcript.jsonl") -> str: + """A real captured transcript, optionally with its tool_use line cut.""" + lines = _real_transcript_lines() + if not tools: + lines = [line for line in lines if '"tool_use"' not in line] + path = os.path.join(self.tmp.name, name) + with open(path, "w", encoding="utf-8") as handle: + handle.writelines(lines) + return path + + def payload(self, message: str, *, tools: bool, **extra) -> dict: + """A Stop payload with exactly the keys the captured one has.""" + data = { + "session_id": "s1", + "last_assistant_message": message, + "transcript_path": self.transcript(tools=tools), + } + data.update(extra) + return data + + def test_payload_contract_every_key_detect_reads_exists_in_a_real_payload(self) -> None: + """Every `payload.get("k")` in detect.py must be a key Claude Code sends.""" + with open(os.path.join(HOOK, "detect.py"), encoding="utf-8") as handle: + tree = ast.parse(handle.read()) + keys = set() + for node in ast.walk(tree): + if (isinstance(node, ast.Call) + and isinstance(node.func, ast.Attribute) + and node.func.attr == "get" + and isinstance(node.func.value, ast.Name) + and node.func.value.id == "payload" + and node.args + and isinstance(node.args[0], ast.Constant) + and isinstance(node.args[0].value, str)): + keys.add(node.args[0].value) + self.assertTrue(keys, "found no payload.get() calls to check") + with open(REAL_PAYLOAD, encoding="utf-8") as handle: + real = json.load(handle) + missing = sorted(key for key in keys if key not in real) + self.assertEqual( + missing, [], + f"detect.py reads {missing} but a real Claude Code Stop payload has " + f"only {sorted(real)}") + + def test_real_payload_carries_no_tool_list_of_any_kind(self) -> None: + """The regression this hook shipped with: there is no tools_used key.""" + with open(REAL_PAYLOAD, encoding="utf-8") as handle: + real = json.load(handle) + self.assertNotIn("tools_used", real) + self.assertNotIn("tool_names", real) + self.assertIn("transcript_path", real) + + def test_tools_are_read_from_the_real_transcript(self) -> None: + tools = self.detect.tools_used_this_turn( + {"transcript_path": self.transcript(tools=True)}) + self.assertEqual(tools, {"Bash"}) + + def test_a_turn_with_no_tool_use_reads_as_an_empty_set_not_none(self) -> None: + tools = self.detect.tools_used_this_turn( + {"transcript_path": self.transcript(tools=False)}) + self.assertEqual(tools, set()) + + def test_a_missing_transcript_reads_as_unchecked_and_says_so(self) -> None: + buffer = io.StringIO() + with redirect_stderr(buffer): + tools = self.detect.tools_used_this_turn( + {"transcript_path": os.path.join(self.tmp.name, "nope.jsonl")}) + self.assertIsNone(tools) + self.assertIn("unchecked", buffer.getvalue()) + + def test_a_payload_with_no_transcript_path_reads_as_unchecked(self) -> None: + buffer = io.StringIO() + with redirect_stderr(buffer): + tools = self.detect.tools_used_this_turn({"session_id": "s1"}) + self.assertIsNone(tools) + self.assertIn("unchecked", buffer.getvalue()) + def test_real_tag_is_parsed_into_claim_and_reason(self) -> None: tags = self.detect.parse_tags(f"Some prose.\n\n{REAL_TAG_1}") self.assertEqual(len(tags), 1) @@ -53,41 +150,45 @@ def test_message_with_no_tag_leaves_ledger_empty(self) -> None: self.assertEqual(self.detect.reminder("s1"), "") def test_tag_after_a_real_attempt_is_logged_and_allowed(self) -> None: - verdict = self.detect.evaluate( - {"session_id": "s1", "message": REAL_TAG_1, "tools_used": ["Bash"]}) + verdict = self.detect.evaluate(self.payload(REAL_TAG_1, tools=True)) self.assertEqual(verdict["block"], "") self.assertIn("deferred, not discharged", verdict["note"]) self.assertEqual(len(self.detect.outstanding(self.detect.read_ledger("s1"))), 1) def test_tag_with_no_attempt_is_blocked(self) -> None: - verdict = self.detect.evaluate( - {"session_id": "s1", "message": REAL_TAG_1, "tools_used": []}) + verdict = self.detect.evaluate(self.payload(REAL_TAG_1, tools=False)) self.assertIn("ran no verification tool", verdict["block"]) self.assertIn("cat-mode/SKILL.md:269", verdict["block"]) self.assertIn("widened scope", verdict["block"]) def test_blocked_turn_is_still_recorded(self) -> None: - self.detect.evaluate({"session_id": "s1", "message": REAL_TAG_1, "tools_used": []}) + self.detect.evaluate(self.payload(REAL_TAG_1, tools=False)) self.assertEqual(len(self.detect.outstanding(self.detect.read_ledger("s1"))), 1) def test_rewrite_turn_is_released_so_the_block_cannot_loop(self) -> None: - verdict = self.detect.evaluate({ - "session_id": "s1", "message": REAL_TAG_1, - "tools_used": [], "stop_hook_active": True}) + verdict = self.detect.evaluate( + self.payload(REAL_TAG_1, tools=False, stop_hook_active=True)) self.assertEqual(verdict["block"], "") def test_untagged_turn_with_no_tools_is_never_blocked(self) -> None: verdict = self.detect.evaluate( - {"session_id": "s1", "message": "Short answer, nothing claimed.", "tools_used": []}) + self.payload("Short answer, nothing claimed.", tools=False)) self.assertEqual(verdict["block"], "") self.assertEqual(verdict["note"], "") def test_both_real_session_turns_would_have_been_blocked(self) -> None: for tag in (REAL_TAG_1, REAL_TAG_2): verdict = self.detect.evaluate( - {"session_id": "replay", "message": f"prose\n\n{tag}", "tools_used": []}) + self.payload(f"prose\n\n{tag}", tools=False, session_id="replay")) self.assertIn("ran no verification tool", verdict["block"]) + def test_an_unreadable_turn_is_neither_blocked_nor_called_clean(self) -> None: + verdict = self.detect.evaluate({ + "session_id": "s1", "last_assistant_message": REAL_TAG_1, + "transcript_path": os.path.join(self.tmp.name, "gone.jsonl")}) + self.assertEqual(verdict["block"], "") + self.assertIn("UNCHECKED", verdict["note"]) + def test_reminder_names_the_claim_and_cites_the_rule(self) -> None: self.detect.record_turn("s1", REAL_TAG_1, set()) text = self.detect.reminder("s1") @@ -100,12 +201,21 @@ def test_two_tags_in_one_session_both_tracked(self) -> None: self.detect.record_turn("s1", REAL_TAG_2, set()) self.assertEqual(len(self.detect.outstanding(self.detect.read_ledger("s1"))), 2) - def test_verified_and_dropped_tag_is_discharged(self) -> None: - self.detect.record_turn("s1", REAL_TAG_1, set()) - self.detect.record_turn("s1", "Here is the pasted output proving it.", {"Bash"}) + def test_verified_and_dropped_tag_is_discharged_through_the_real_payload(self) -> None: + """The branch that could never be reached: a live payload discharges.""" + self.detect.evaluate(self.payload(REAL_TAG_1, tools=True)) + self.assertEqual(len(self.detect.outstanding(self.detect.read_ledger("s1"))), 1) + self.detect.evaluate(self.payload("Here is the pasted output proving it.", tools=True)) self.assertEqual(self.detect.outstanding(self.detect.read_ledger("s1")), []) self.assertEqual(self.detect.reminder("s1"), "") + def test_unchecked_turn_does_not_discharge_a_row(self) -> None: + self.detect.evaluate(self.payload(REAL_TAG_1, tools=True)) + self.detect.evaluate({ + "session_id": "s1", "last_assistant_message": "Moving on.", + "transcript_path": os.path.join(self.tmp.name, "gone.jsonl")}) + self.assertEqual(len(self.detect.outstanding(self.detect.read_ledger("s1"))), 1) + def test_redropping_without_verifying_keeps_it_outstanding(self) -> None: self.detect.record_turn("s1", REAL_TAG_1, set()) self.detect.record_turn("s1", "Moving on to something else.", set()) @@ -131,8 +241,6 @@ def test_corrupt_ledger_row_is_reported_not_swallowed(self) -> None: os.makedirs(os.path.dirname(path), exist_ok=True) with open(path, "w", encoding="utf-8") as handle: handle.write("{not json}\n") - import io - from contextlib import redirect_stderr buffer = io.StringIO() with redirect_stderr(buffer): rows = self.detect.read_ledger("s3")