Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 4 additions & 2 deletions engine/hooks/llm-judge/phrases/wrong-check-reflect.json
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
{
"checker": "wrong-check-reflect",
"reads": "reply",
"meaning": "The latest assistant reply admits that something it told the user earlier was wrong, misread, or answered the wrong question.",
"meaning": "The latest assistant reply admits that something it told the user earlier was wrong, misread, or answered the wrong question, OR that an earlier claim of its own went out before the check backing it had run. A reply saying the claim turned out correct but was stated before it was checked IS a match: the failure is the order, not the answer, so no word in the reply has to say anything was wrong.",
"match": [
"my earlier check was wrong",
"You're right, I misread that",
Expand All @@ -10,7 +10,9 @@
"I was wrong about the path",
"Correction: the file I pointed you to earlier is not the one in use; the real one is src/b.py.",
"Good catch. The earlier number was off; the real count is 12.",
"My earlier read of the config was wrong \u2014 the default is 4, not 8."
"My earlier read of the config was wrong \u2014 the default is 4, not 8.",
"Correcting one claim and arming the check I implied:",
"I was right - but I said it a turn before I checked it"
],
"not_match": [
"You're right. Let's go with option B.",
Expand Down
6 changes: 6 additions & 0 deletions engine/hooks/unverified-tag-ledger/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,12 @@ fixtures in `tests/test_hooks.py`.
reason is written to stderr.
- **Escalation** — a claim outstanding `ESCALATE_AFTER_TURNS` (3) turns or more
is reported as a reflect trigger rather than accumulating quietly.
- **Discharge is itself a reflect trigger** — a row going outstanding ->
discharged is the record of a claim that went out first and was checked
after. That is an evidence-order miss, and it carries no wrongness word, so
the phrase scanners (`engine/skills/reflect/scripts/self_retraction_scan.py`,
and the `wrong-check-reflect` dictionary) cannot see it from the text. This
hook sees it from state instead, and says so on the Stop that discharges.

Malformed tags are deliberately ignored here; `diu-stop` already rejects those.

Expand Down
35 changes: 26 additions & 9 deletions engine/hooks/unverified-tag-ledger/detect.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,15 @@
ESCALATE_AFTER_TURNS = 3
MAX_LISTED = 5

DISCHARGE_REFLECT = (
"unverified-tag-ledger: {count} claim(s) went from unverified to checked this turn: "
"{claims}. That transition is the whole event: the claim went out first and the check "
"ran after. No wording has to admit anything for this to be true, which is why the "
"phrase scanners miss it -- an evidence-order miss carries no wrongness word. "
"Treat it as a reflect trigger, not a milestone: run reflect on this transcript, or "
"say plainly why this one does not need it."
)

CLAIM_RE = re.compile(
r"\{\{\s*CAT-UNVERIFIED\s*:?\s*(?P<claim>.*?)(?:--|—)\s*cannot\s+verify\s*:\s*(?P<reason>[^}]*)\}\}",
re.IGNORECASE | re.DOTALL,
Expand Down Expand Up @@ -177,18 +186,26 @@ def evaluate(payload: dict) -> dict:
session_id = str(payload.get("session_id") or "")
message = _last_assistant_text(payload)
tools = tools_used_this_turn(payload)
was_open = {row["claim"] for row in outstanding(read_ledger(session_id))}
rows = record_turn(session_id, message, tools)
notes = []
discharged = sorted(
row["claim"] for row in rows if row.get("resolved") and row["claim"] in was_open)
if discharged:
notes.append(DISCHARGE_REFLECT.format(
count=len(discharged), claims="; ".join(discharged[:MAX_LISTED])))

new_claims = {tag["claim"] for tag in parse_tags(message)}
if not new_claims:
return {"note": "", "block": ""}
return {"note": "\n".join(notes), "block": ""}
Comment thread
cursor[bot] marked this conversation as resolved.

if tools is None:
return {"note": (
notes.append(
f"unverified-tag-ledger: logged {len(new_claims)} CAT-UNVERIFIED claim(s), but this "
"turn's tool calls could not be read from transcript_path (see the line above), so "
"whether a check was attempted is UNCHECKED, not clean. Nothing was discharged and "
"the turn was not refused."), "block": ""}
"the turn was not refused.")
return {"note": "\n".join(notes), "block": ""}

if not tools & VERIFY_TOOLS and not payload.get("stop_hook_active"):
claims = "; ".join(sorted(new_claims)[:MAX_LISTED])
Expand All @@ -201,12 +218,12 @@ def evaluate(payload: dict) -> dict:

fresh = [row for row in rows
if row["claim"] in new_claims and not row.get("resolved") and row.get("turns", 0) == 0]
if not fresh:
return {"note": "", "block": ""}
return {"note": (
f"unverified-tag-ledger: logged {len(fresh)} CAT-UNVERIFIED claim(s) against this session. "
"They are deferred, not discharged, and will be raised again next turn "
"(cat-mode/SKILL.md:269)."), "block": ""}
if fresh:
notes.append(
f"unverified-tag-ledger: logged {len(fresh)} CAT-UNVERIFIED claim(s) against this "
"session. They are deferred, not discharged, and will be raised again next turn "
"(cat-mode/SKILL.md:269).")
return {"note": "\n".join(notes), "block": ""}


def decide_stop(payload: dict) -> str:
Expand Down
30 changes: 30 additions & 0 deletions engine/hooks/unverified-tag-ledger/tests/test_hooks.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,9 @@
"-- cannot verify: my own reasoning isn't observable by any command}}")
MALFORMED = "{{CAT-UNVERIFIED: something I did not check}}"

EVIDENCE_ORDER_1 = "Correcting one claim and arming the check I implied:"
EVIDENCE_ORDER_2 = "I was right - but I said it a turn before I checked it"


def _real_transcript_lines() -> list[str]:
with open(REAL_TRANSCRIPT, encoding="utf-8") as handle:
Expand Down Expand Up @@ -209,6 +212,33 @@ def test_verified_and_dropped_tag_is_discharged_through_the_real_payload(self) -
self.assertEqual(self.detect.outstanding(self.detect.read_ledger("s1")), [])
self.assertEqual(self.detect.reminder("s1"), "")

def test_a_discharged_claim_fires_the_reflect_trigger(self) -> None:
self.detect.evaluate(self.payload(REAL_TAG_1, tools=True))
verdict = self.detect.evaluate(
self.payload("Here is the pasted output proving it.", tools=True))
self.assertIn("reflect trigger", verdict["note"])
self.assertIn("widened scope", verdict["note"])

def test_an_evidence_order_correction_triggers_with_no_wrongness_word(self) -> None:
"""The transition fires; the reply's wording is not consulted at all."""
for index, reply in enumerate((EVIDENCE_ORDER_1, EVIDENCE_ORDER_2)):
session = f"evidence-order-{index}"
self.detect.evaluate(self.payload(REAL_TAG_1, tools=True, session_id=session))
verdict = self.detect.evaluate(
self.payload(reply, tools=True, session_id=session))
self.assertIn("reflect trigger", verdict["note"])
self.assertIn("carries no wrongness word", verdict["note"])

def test_a_turn_that_discharges_nothing_stays_silent_about_reflect(self) -> None:
verdict = self.detect.evaluate(
self.payload("Ran the tests, all green.", tools=True))
self.assertEqual(verdict["note"], "")

def test_a_reemitted_tag_is_not_reported_as_discharged(self) -> None:
self.detect.evaluate(self.payload(REAL_TAG_1, tools=True))
verdict = self.detect.evaluate(self.payload(REAL_TAG_1, tools=True))
self.assertNotIn("reflect trigger", verdict["note"])

def test_unchecked_turn_does_not_discharge_a_row(self) -> None:
self.detect.evaluate(self.payload(REAL_TAG_1, tools=True))
self.detect.evaluate({
Expand Down
3 changes: 3 additions & 0 deletions engine/hooks/wrong-check-reflect/eval_dictionary.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,9 @@
(HIT_TEXT, True),
("You're right. Let's go with option B.", False),
("I double-checked my earlier count and it holds; nothing in it was wrong.", False),
("Correcting one claim and arming the check I implied:", True),
("I was right - but I said it a turn before I checked it", True),
("I ran the check first and then said it, so the order was right.", False),
)


Expand Down
24 changes: 24 additions & 0 deletions engine/skills/reflect/scripts/tests/test_self_retraction_scan.py
Original file line number Diff line number Diff line change
Expand Up @@ -74,6 +74,30 @@ def test_third_party_blame_inside_a_reported_clause_stays_clean(self):
self.assertIsNone(self_retraction_scan.find_admission(text))


class TestEvidenceOrderIsOutOfReach(unittest.TestCase):
"""Two real corrections this scan cannot see, and the reason it cannot.

Both are corrections about evidence ORDER: the claim was true, and it was
asserted before the check ran. Nothing in either sentence says anything was
wrong, so every pattern here misses them by construction. Pinned so the
next author widens the regex knowingly rather than by accident: the catch
for this class is the unverified-tag-ledger discharge transition, which
reads state rather than wording.
"""

def test_arming_the_implied_check_is_not_reachable_by_wording(self):
text = "Correcting one claim and arming the check I implied:"
self.assertIsNone(self_retraction_scan.find_admission(text))

def test_right_but_asserted_early_is_not_reachable_by_wording(self):
text = "I was right - but I said it a turn before I checked it"
self.assertIsNone(self_retraction_scan.find_admission(text))

def test_the_same_sentence_with_a_wrongness_word_does_fire(self):
text = "I was wrong about the path; I said it a turn before I checked it."
self.assertIsNotNone(self_retraction_scan.find_admission(text))


class TestScanAssistantTexts(unittest.TestCase):
def test_collects_one_hit_per_admission(self):
hits = self_retraction_scan.scan_assistant_texts(
Expand Down
Loading