diff --git a/engine/hooks/llm-judge/phrases/wrong-check-reflect.json b/engine/hooks/llm-judge/phrases/wrong-check-reflect.json index 173dbcbe..cf121c91 100644 --- a/engine/hooks/llm-judge/phrases/wrong-check-reflect.json +++ b/engine/hooks/llm-judge/phrases/wrong-check-reflect.json @@ -1,7 +1,7 @@ { "checker": "wrong-check-reflect", "reads": "reply", - "meaning": "The latest assistant reply admits that something it told the user earlier was wrong, misread, or answered the wrong question.", + "meaning": "The latest assistant reply admits that something it told the user earlier was wrong, misread, or answered the wrong question, OR that an earlier claim of its own went out before the check backing it had run. A reply saying the claim turned out correct but was stated before it was checked IS a match: the failure is the order, not the answer, so no word in the reply has to say anything was wrong.", "match": [ "my earlier check was wrong", "You're right, I misread that", @@ -10,7 +10,9 @@ "I was wrong about the path", "Correction: the file I pointed you to earlier is not the one in use; the real one is src/b.py.", "Good catch. The earlier number was off; the real count is 12.", - "My earlier read of the config was wrong \u2014 the default is 4, not 8." + "My earlier read of the config was wrong \u2014 the default is 4, not 8.", + "Correcting one claim and arming the check I implied:", + "I was right - but I said it a turn before I checked it" ], "not_match": [ "You're right. Let's go with option B.", diff --git a/engine/hooks/unverified-tag-ledger/README.md b/engine/hooks/unverified-tag-ledger/README.md index ccfa1d8c..121cd94a 100644 --- a/engine/hooks/unverified-tag-ledger/README.md +++ b/engine/hooks/unverified-tag-ledger/README.md @@ -53,6 +53,12 @@ fixtures in `tests/test_hooks.py`. reason is written to stderr. - **Escalation** — a claim outstanding `ESCALATE_AFTER_TURNS` (3) turns or more is reported as a reflect trigger rather than accumulating quietly. +- **Discharge is itself a reflect trigger** — a row going outstanding -> + discharged is the record of a claim that went out first and was checked + after. That is an evidence-order miss, and it carries no wrongness word, so + the phrase scanners (`engine/skills/reflect/scripts/self_retraction_scan.py`, + and the `wrong-check-reflect` dictionary) cannot see it from the text. This + hook sees it from state instead, and says so on the Stop that discharges. Malformed tags are deliberately ignored here; `diu-stop` already rejects those. diff --git a/engine/hooks/unverified-tag-ledger/detect.py b/engine/hooks/unverified-tag-ledger/detect.py index 8a47c2ef..1f54902d 100644 --- a/engine/hooks/unverified-tag-ledger/detect.py +++ b/engine/hooks/unverified-tag-ledger/detect.py @@ -34,6 +34,15 @@ ESCALATE_AFTER_TURNS = 3 MAX_LISTED = 5 +DISCHARGE_REFLECT = ( + "unverified-tag-ledger: {count} claim(s) went from unverified to checked this turn: " + "{claims}. That transition is the whole event: the claim went out first and the check " + "ran after. No wording has to admit anything for this to be true, which is why the " + "phrase scanners miss it -- an evidence-order miss carries no wrongness word. " + "Treat it as a reflect trigger, not a milestone: run reflect on this transcript, or " + "say plainly why this one does not need it." +) + CLAIM_RE = re.compile( r"\{\{\s*CAT-UNVERIFIED\s*:?\s*(?P.*?)(?:--|—)\s*cannot\s+verify\s*:\s*(?P[^}]*)\}\}", re.IGNORECASE | re.DOTALL, @@ -177,18 +186,26 @@ def evaluate(payload: dict) -> dict: session_id = str(payload.get("session_id") or "") message = _last_assistant_text(payload) tools = tools_used_this_turn(payload) + was_open = {row["claim"] for row in outstanding(read_ledger(session_id))} rows = record_turn(session_id, message, tools) + notes = [] + discharged = sorted( + row["claim"] for row in rows if row.get("resolved") and row["claim"] in was_open) + if discharged: + notes.append(DISCHARGE_REFLECT.format( + count=len(discharged), claims="; ".join(discharged[:MAX_LISTED]))) new_claims = {tag["claim"] for tag in parse_tags(message)} if not new_claims: - return {"note": "", "block": ""} + return {"note": "\n".join(notes), "block": ""} if tools is None: - return {"note": ( + notes.append( f"unverified-tag-ledger: logged {len(new_claims)} CAT-UNVERIFIED claim(s), but this " "turn's tool calls could not be read from transcript_path (see the line above), so " "whether a check was attempted is UNCHECKED, not clean. Nothing was discharged and " - "the turn was not refused."), "block": ""} + "the turn was not refused.") + return {"note": "\n".join(notes), "block": ""} if not tools & VERIFY_TOOLS and not payload.get("stop_hook_active"): claims = "; ".join(sorted(new_claims)[:MAX_LISTED]) @@ -201,12 +218,12 @@ def evaluate(payload: dict) -> dict: fresh = [row for row in rows if row["claim"] in new_claims and not row.get("resolved") and row.get("turns", 0) == 0] - if not fresh: - return {"note": "", "block": ""} - return {"note": ( - f"unverified-tag-ledger: logged {len(fresh)} CAT-UNVERIFIED claim(s) against this session. " - "They are deferred, not discharged, and will be raised again next turn " - "(cat-mode/SKILL.md:269)."), "block": ""} + if fresh: + notes.append( + f"unverified-tag-ledger: logged {len(fresh)} CAT-UNVERIFIED claim(s) against this " + "session. They are deferred, not discharged, and will be raised again next turn " + "(cat-mode/SKILL.md:269).") + return {"note": "\n".join(notes), "block": ""} def decide_stop(payload: dict) -> str: diff --git a/engine/hooks/unverified-tag-ledger/tests/test_hooks.py b/engine/hooks/unverified-tag-ledger/tests/test_hooks.py index 12ab9931..251cede6 100644 --- a/engine/hooks/unverified-tag-ledger/tests/test_hooks.py +++ b/engine/hooks/unverified-tag-ledger/tests/test_hooks.py @@ -38,6 +38,9 @@ "-- cannot verify: my own reasoning isn't observable by any command}}") MALFORMED = "{{CAT-UNVERIFIED: something I did not check}}" +EVIDENCE_ORDER_1 = "Correcting one claim and arming the check I implied:" +EVIDENCE_ORDER_2 = "I was right - but I said it a turn before I checked it" + def _real_transcript_lines() -> list[str]: with open(REAL_TRANSCRIPT, encoding="utf-8") as handle: @@ -209,6 +212,33 @@ def test_verified_and_dropped_tag_is_discharged_through_the_real_payload(self) - self.assertEqual(self.detect.outstanding(self.detect.read_ledger("s1")), []) self.assertEqual(self.detect.reminder("s1"), "") + def test_a_discharged_claim_fires_the_reflect_trigger(self) -> None: + self.detect.evaluate(self.payload(REAL_TAG_1, tools=True)) + verdict = self.detect.evaluate( + self.payload("Here is the pasted output proving it.", tools=True)) + self.assertIn("reflect trigger", verdict["note"]) + self.assertIn("widened scope", verdict["note"]) + + def test_an_evidence_order_correction_triggers_with_no_wrongness_word(self) -> None: + """The transition fires; the reply's wording is not consulted at all.""" + for index, reply in enumerate((EVIDENCE_ORDER_1, EVIDENCE_ORDER_2)): + session = f"evidence-order-{index}" + self.detect.evaluate(self.payload(REAL_TAG_1, tools=True, session_id=session)) + verdict = self.detect.evaluate( + self.payload(reply, tools=True, session_id=session)) + self.assertIn("reflect trigger", verdict["note"]) + self.assertIn("carries no wrongness word", verdict["note"]) + + def test_a_turn_that_discharges_nothing_stays_silent_about_reflect(self) -> None: + verdict = self.detect.evaluate( + self.payload("Ran the tests, all green.", tools=True)) + self.assertEqual(verdict["note"], "") + + def test_a_reemitted_tag_is_not_reported_as_discharged(self) -> None: + self.detect.evaluate(self.payload(REAL_TAG_1, tools=True)) + verdict = self.detect.evaluate(self.payload(REAL_TAG_1, tools=True)) + self.assertNotIn("reflect trigger", verdict["note"]) + def test_unchecked_turn_does_not_discharge_a_row(self) -> None: self.detect.evaluate(self.payload(REAL_TAG_1, tools=True)) self.detect.evaluate({ diff --git a/engine/hooks/wrong-check-reflect/eval_dictionary.py b/engine/hooks/wrong-check-reflect/eval_dictionary.py index 258dabb2..9a9e8927 100644 --- a/engine/hooks/wrong-check-reflect/eval_dictionary.py +++ b/engine/hooks/wrong-check-reflect/eval_dictionary.py @@ -17,6 +17,9 @@ (HIT_TEXT, True), ("You're right. Let's go with option B.", False), ("I double-checked my earlier count and it holds; nothing in it was wrong.", False), + ("Correcting one claim and arming the check I implied:", True), + ("I was right - but I said it a turn before I checked it", True), + ("I ran the check first and then said it, so the order was right.", False), ) diff --git a/engine/skills/reflect/scripts/tests/test_self_retraction_scan.py b/engine/skills/reflect/scripts/tests/test_self_retraction_scan.py index 72266f87..85258277 100644 --- a/engine/skills/reflect/scripts/tests/test_self_retraction_scan.py +++ b/engine/skills/reflect/scripts/tests/test_self_retraction_scan.py @@ -74,6 +74,30 @@ def test_third_party_blame_inside_a_reported_clause_stays_clean(self): self.assertIsNone(self_retraction_scan.find_admission(text)) +class TestEvidenceOrderIsOutOfReach(unittest.TestCase): + """Two real corrections this scan cannot see, and the reason it cannot. + + Both are corrections about evidence ORDER: the claim was true, and it was + asserted before the check ran. Nothing in either sentence says anything was + wrong, so every pattern here misses them by construction. Pinned so the + next author widens the regex knowingly rather than by accident: the catch + for this class is the unverified-tag-ledger discharge transition, which + reads state rather than wording. + """ + + def test_arming_the_implied_check_is_not_reachable_by_wording(self): + text = "Correcting one claim and arming the check I implied:" + self.assertIsNone(self_retraction_scan.find_admission(text)) + + def test_right_but_asserted_early_is_not_reachable_by_wording(self): + text = "I was right - but I said it a turn before I checked it" + self.assertIsNone(self_retraction_scan.find_admission(text)) + + def test_the_same_sentence_with_a_wrongness_word_does_fire(self): + text = "I was wrong about the path; I said it a turn before I checked it." + self.assertIsNotNone(self_retraction_scan.find_admission(text)) + + class TestScanAssistantTexts(unittest.TestCase): def test_collects_one_hit_per_admission(self): hits = self_retraction_scan.scan_assistant_texts(