diff --git a/corpus/skills/cat-mode/SKILL.md b/corpus/skills/cat-mode/SKILL.md index e056db7c..351c8278 100644 --- a/corpus/skills/cat-mode/SKILL.md +++ b/corpus/skills/cat-mode/SKILL.md @@ -288,7 +288,7 @@ agent switch, or resubmit is a fix, and none comes before the repro. **A factual or technical claim gets a real repro script, not a history search.** Judging an old comment or a "probably confabulated" suspicion needs an actual attempt under the claimed conditions, not a `git log` sweep. No citation means "never verified," not "false." -**Unhedged root-cause or fix claims about live system behavior need instrument-level proof in the same message, or a `{{CAT-UNVERIFIED}}` tag naming the blocker.** The gate is the claim type, not a hedge word. Invoking `/prove-it` once does not arm it for later claims. Any hedge auto-runs prove-it in the same turn — a hedge is a trigger to verify, never a place to stop. +**Unhedged root-cause or fix claims about live system behavior need instrument-level proof in the same message, or a `{{CAT-UNVERIFIED: -- cannot verify: }}` tag naming the blocker.** The gate is the claim type, not a hedge word. Invoking `/prove-it` once does not arm it for later claims. Any hedge auto-runs prove-it in the same turn — a hedge is a trigger to verify, never a place to stop. Outputs carry failures explicitly (a status column, an error row), never dropped — [[principle-explicit-errors]]. diff --git a/corpus/skills/cat-mode/references/verify.md b/corpus/skills/cat-mode/references/verify.md index 8f384542..09ab952a 100644 --- a/corpus/skills/cat-mode/references/verify.md +++ b/corpus/skills/cat-mode/references/verify.md @@ -67,8 +67,10 @@ place immediately, not left in chat until asked again. ## Unhedged causal claims about live system behavior Unhedged root-cause or fix claims about live system behavior need -instrument-level proof in the same message, or a `{{CAT-UNVERIFIED}}` tag naming the blocker. The gate is the -claim type ("this is why it's slow," "this is the bug"), not a hedge word. +instrument-level proof in the same message, or a +`{{CAT-UNVERIFIED: -- cannot verify: }}` tag naming the +blocker. The gate is the claim type ("this is why it's slow," "this is the +bug"), not a hedge word. Log-reading and code-reading aren't enough: attach with `strace`/a debugger, or query live state (raw SQLite `PRAGMA`). Take a second sample before calling a hang. Invoking `/prove-it` once does not arm it for later claims — each new diff --git a/tests/test_cat_mode.py b/tests/test_cat_mode.py index 6a040ca5..9e4e5014 100644 --- a/tests/test_cat_mode.py +++ b/tests/test_cat_mode.py @@ -1691,5 +1691,67 @@ def test_hook_blocks_the_same_estimate_once_the_job_already_exited(self): self.assertIsNotNone(detect.decide_stop_from_lines(ESTIMATE_REPLY, lines)) +VERIFY_REF = os.path.join(REFERENCE_DIR, "verify.md") + + +def cat_mode_markdown_paths(): + """Every markdown file a cat-mode reader loads: SKILL.md and all of + references/. Listing the directory rather than naming files keeps a new + reference inside the scan the day it lands.""" + paths = [SKILL_PATH] + paths.extend(sorted( + os.path.join(REFERENCE_DIR, name) + for name in os.listdir(REFERENCE_DIR) + if name.endswith(".md") + )) + return paths + + +class TestEscapeHatchTemplateIsWellFormed(unittest.TestCase): + """Every tag this skill shows a reader must be one the gate accepts. + + cat-mode carried a bare `{{CAT-UNVERIFIED}}` in the sentence that tells + the reader to use the tag, so quoting the rule tripped the gate that + enforces it. The template has to name a blocker, the same one + engine/CLAUDE.core.md already shows. + + The scan covers references/, not just SKILL.md. That sentence is stored + twice on purpose -- SKILL.md keeps the one-line form and verify.md holds + the full text (TestCatModeReferencePackage pins that split) -- so a + SKILL.md-only scan reports clean while the copy a reader is pointed at + still shows the bare tag. + """ + + def markers(self): + import importlib.util as util + path = os.path.join(REPO_ROOT, "engine", "hooks", "_markers", "markers.py") + spec = util.spec_from_file_location("markers_for_test", path) + module = util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + def read(self, path): + with open(path, encoding="utf-8") as handle: + return handle.read() + + def test_no_tag_anywhere_in_the_skill_names_no_blocker(self): + """A scan over an empty or mis-rooted list reports clean, which reads + the same as a pass. Naming the two files that carry the rule keeps the + sweep from silently covering nothing.""" + paths = cat_mode_markdown_paths() + self.assertIn(SKILL_PATH, paths) + self.assertIn(VERIFY_REF, paths) + for path in paths: + with self.subTest(path=os.path.relpath(path, REPO_ROOT)): + malformed = self.markers().malformed_tags(self.read(path)) + self.assertEqual(malformed, [], f"tags naming no blocker: {malformed}") + + def test_both_copies_of_the_rule_still_show_the_tag_at_all(self): + """So the fix cannot be "delete the example" in either file.""" + for path in (SKILL_PATH, VERIFY_REF): + with self.subTest(path=os.path.relpath(path, REPO_ROOT)): + self.assertTrue(self.markers().well_formed_tags(self.read(path))) + + if __name__ == "__main__": unittest.main()