diff --git a/engine/hooks/diu-stop/README.md b/engine/hooks/diu-stop/README.md index 1c931080..ff4a8c66 100644 --- a/engine/hooks/diu-stop/README.md +++ b/engine/hooks/diu-stop/README.md @@ -31,6 +31,11 @@ Three outcomes, not two. When the transcript cannot be read, whether the path was read is *unchecked*: the citation does not buy silence, and the block says which path it could not check and that the ref would settle it. +The marker check reads prose only. A marker inside a fence or a pair of +backticks is being shown, not used, so explaining the tag, quoting the rule +that defines it, or relaying this gate's own refusal word for word all stay +silent. A marker in running prose is a use and still counts. + Not one file per harness, because there is no single "stop" mechanism shared by every harness -- each one has a genuinely different amount of power at that point: diff --git a/engine/hooks/diu-stop/claude_stop_check.py b/engine/hooks/diu-stop/claude_stop_check.py index 44b2270a..840eded1 100755 --- a/engine/hooks/diu-stop/claude_stop_check.py +++ b/engine/hooks/diu-stop/claude_stop_check.py @@ -148,15 +148,37 @@ def _opening_word(message): return match.group(0).lower() if match else "" +def prose_only(message): + """`message` with fenced blocks and inline code removed. + + A marker inside a fence or a pair of backticks is being shown, not used: + explaining the tag, quoting the rule that defines it, or pasting a gate's + own message back to the user all put the token on screen without claiming + anything. `find_unverified_claims` has stripped both for a while; the + marker check read the raw message, so the gate fired on the sentence that + taught the reader how not to trip it. + + An unterminated fence leaves a `\u0060\u0060\u0060` behind after the + substitution. Everything from that marker on is inside a code block that + never closed, so it is dropped too. + """ + prose = INLINE_CODE_RE.sub("", FENCED_BODY_RE.sub("", message or "")) + if FENCE_MARKER in prose: + prose = prose[:prose.rindex(FENCE_MARKER)] + return prose + + def find_marker_problems(message): """Return the marker complaints this message earns, in report order. A tag that names no blocker, and the retired bare `UNVERIFIED:`, each - draw their own message. Both can be present at once.""" + draw their own message. Both can be present at once. Only prose counts -- + see `prose_only`.""" + prose = prose_only(message) problems = [] - if markers.malformed_tags(message): + if markers.malformed_tags(prose): problems.append(markers.MALFORMED_TAG_MESSAGE) - if markers.has_legacy_marker(message): + if markers.has_legacy_marker(prose): problems.append(markers.LEGACY_MARKER_MESSAGE) return problems diff --git a/engine/hooks/diu-stop/tests/test_hooks.py b/engine/hooks/diu-stop/tests/test_hooks.py index abc4f9e9..69cf6775 100644 --- a/engine/hooks/diu-stop/tests/test_hooks.py +++ b/engine/hooks/diu-stop/tests/test_hooks.py @@ -24,9 +24,11 @@ FIXTURES_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "fixtures") LLM_JUDGE_DIR = os.path.join(os.path.dirname(HOOKS_DIR), "llm-judge") sys.path.insert(0, LLM_JUDGE_DIR) +sys.path.insert(0, os.path.join(os.path.dirname(HOOKS_DIR), "_markers")) sys.path.insert(0, HOOKS_DIR) import claude_prompt_reminder # noqa: E402 +import markers # noqa: E402 import claude_stop_check # noqa: E402 import codex_notify # noqa: E402 import install_claude_hook # noqa: E402 @@ -323,6 +325,53 @@ def test_legacy_bare_marker_is_blocked_and_names_the_new_tag(self): self.assertIn("CAT-UNVERIFIED", err) self.assertIn("prove-it", err.lower()) + def test_a_backticked_mention_of_the_tag_is_silent(self): + """Explaining the mechanism is not using it.""" + message = ( + "The escape hatch is `{{CAT-UNVERIFIED}}` and it has to name a blocker " + "after the colon, or the gate rejects it.") + self.assertEqual(claude_stop_check.find_marker_problems(message), []) + + def test_a_mention_inside_a_fence_is_silent(self): + message = ( + "Here is the shape the gate wants:\n" + "```\n" + "{{CAT-UNVERIFIED}}\n" + "UNVERIFIED: the old one\n" + "```\n" + "Use the first form and name the blocker.") + self.assertEqual(claude_stop_check.find_marker_problems(message), []) + + def test_quoting_the_cat_mode_rule_verbatim_is_silent(self): + """The line that defines the rule must not trip the gate enforcing it.""" + message = ( + "The rule is: **Unhedged root-cause or fix claims about live system " + "behavior need instrument-level proof in the same message, or a " + "`{{CAT-UNVERIFIED}}` tag naming the blocker.**") + self.assertEqual(claude_stop_check.find_marker_problems(message), []) + blocked, err = run_claude_check({"last_assistant_message": message}) + self.assertNotIn("names no blocker", err) + self.assertFalse(blocked) + + def test_relaying_the_gates_own_refusal_is_silent(self): + """cat-mode asks for a gate's message word for word; that must be safe.""" + message = "The gate said:\n\n" + markers.MALFORMED_TAG_MESSAGE + self.assertEqual(claude_stop_check.find_marker_problems(message), []) + + def test_an_unclosed_fence_does_not_leak_a_mention_back_into_prose(self): + message = "Example:\n```\n{{CAT-UNVERIFIED}}\n" + self.assertEqual(claude_stop_check.find_marker_problems(message), []) + + def test_a_real_tag_with_no_blocker_in_prose_still_fires(self): + message = "Confirmed the crash loop. {{CAT-UNVERIFIED: the loop is real}}" + self.assertIn( + markers.MALFORMED_TAG_MESSAGE, claude_stop_check.find_marker_problems(message)) + + def test_a_bare_legacy_marker_in_prose_still_fires(self): + message = "UNVERIFIED: the crash loop is real." + self.assertIn( + markers.LEGACY_MARKER_MESSAGE, claude_stop_check.find_marker_problems(message)) + def test_retry_still_checks_a_new_claim(self): message = ( "Correction on scope: that only covers the root chain. "