From 0ace8936a286d146591a1b9dc6b7ac39b3f5b2ab Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Mon, 28 Sep 2026 07:44:13 +0800 Subject: [PATCH 1/3] cat-mode: require Expected predicates before Visual Proof capture Codify authenticity defaults so agents match proof surface to claim, declare Expected before capture, and reject synthesized UI as proof. Pin the standing wording with structural tests in test_cat_mode. Co-authored-by: Cursor Change-Id: Ib9421a359f38c44ab8f1f342f83dd783711b8dfc --- corpus/skills/cat-mode/SKILL.md | 6 ++++ corpus/skills/cat-mode/references/verify.md | 21 +++++++++++ .../tests/fires_visual_proof_wrong_surface.md | 16 +++++++++ ...tays_silent_visual_proof_expected_match.md | 17 +++++++++ tests/test_cat_mode.py | 35 +++++++++++++++++++ 5 files changed, 95 insertions(+) create mode 100644 corpus/skills/cat-mode/tests/fires_visual_proof_wrong_surface.md create mode 100644 corpus/skills/cat-mode/tests/stays_silent_visual_proof_expected_match.md diff --git a/corpus/skills/cat-mode/SKILL.md b/corpus/skills/cat-mode/SKILL.md index 909fac070..b4d7a14de 100644 --- a/corpus/skills/cat-mode/SKILL.md +++ b/corpus/skills/cat-mode/SKILL.md @@ -144,6 +144,12 @@ CLAUDE.md's evidence rules already apply here. Also, don't declare something fix **UI testing must not disrupt the user's own session.** Prove a UI or surface change somewhere disposable — a test channel or workspace, a throwaway profile, a second display, a VM, a headless run. Driving the user's real keyboard, mouse, or screen is a last resort needing an explicit hands-off window first: state the acceptance test in one line, get the yes, `touch /tmp/.ui-input-window`, and remove it when the window closes; a PreToolUse hook (`engine/hooks/ui-input-guard/`) blocks synthetic input and screen recording while no window is open, the screen is locked, or the user is still typing. Stop at the first sign the session is theirs again (idle time drops, the frontmost app changes, the screen locks), and leave no residue: undo stray messages, pins, or reactions, or say what was left behind. +**Visual Proof authenticity** (extends [[visual-proof]] / [[principle-prove-it]]): + +- **The Visual Proof surface must match the Review Claim surface.** A claim about one product surface needs pixels from that surface (e.g. a Slack-thread claim → Slack-thread pixels). A different product's screen, a provider login page, or an adjacent flow is not that proof. +- **Declare Expected surface and Expected predicates before capture.** Write what must be visible and what must not appear; only then capture. `Manually inspected:` checks claim↔pixels against that Expected list by reading the image — a marker-only line is not a check. +- **Never submit synthesized UI as Visual Proof** unless the user asked for a mockup: generated text slides, HTML mock surfaces, reconstructed controls, or redrawn UI do not count. + **A factual or technical claim gets a real repro script, not a history search.** Judging an old comment or a "probably confabulated" suspicion needs an actual attempt under the claimed conditions, not a `git log` sweep. No citation means "never verified," not "false." **Unhedged root-cause or fix claims about live system behavior need instrument-level proof in the same message, or a `{{CAT-UNVERIFIED: -- cannot verify: }}` tag naming the blocker.** The gate is the claim type, not a hedge word. Invoking `/prove-it` once does not arm it for later claims. Any hedge auto-runs prove-it in the same turn — a hedge is a trigger to verify, never a place to stop. diff --git a/corpus/skills/cat-mode/references/verify.md b/corpus/skills/cat-mode/references/verify.md index ff0a8ad4a..b4ffb1957 100644 --- a/corpus/skills/cat-mode/references/verify.md +++ b/corpus/skills/cat-mode/references/verify.md @@ -89,3 +89,24 @@ hang. Invoking `/prove-it` once does not arm it for later claims — each new causal claim needs its own same-message evidence. Any hedge — "I think," "probably," a retired bare `UNVERIFIED:` — auto-runs prove-it in the same turn; a hedge is a trigger to verify, never a place to stop. + +## Visual Proof authenticity + +Personal standing rules on top of [[visual-proof]] and [[principle-prove-it]]. +These extend those skills; they harden authenticity when a near-neighbor +capture would otherwise stand in for the claimed surface. + +- **The Visual Proof surface must match the Review Claim surface.** The claim + names which product surface must be proved. Capture that surface's pixels — + not a different product, not an upstream provider login, not an adjacent + step that "looks related." A Slack-thread claim needs Slack-thread pixels; a Claude or OpenAI login page is not Slack UI proof. +- **Declare Expected surface and Expected predicates before capture.** Before + any screenshot or frame grab, write (1) the Expected surface and (2) the + Expected predicates: what must be visible, and what must not appear. Capture + only after that list exists. Then `Manually inspected:` walks claim↔pixels + against that list by reading the image (or extracted frames). A bare + `Manually inspected:` marker with no predicate check is not a check. +- **Never submit synthesized UI as Visual Proof** unless the user explicitly + asked for a mockup. Generated text slides (e.g. ffmpeg lavfi/drawtext), HTML mock surfaces, + reconstructed controls, and redrawn UI prove only that the generator ran — + not that the claimed surface showed the claimed state. diff --git a/corpus/skills/cat-mode/tests/fires_visual_proof_wrong_surface.md b/corpus/skills/cat-mode/tests/fires_visual_proof_wrong_surface.md new file mode 100644 index 000000000..ed45beba9 --- /dev/null +++ b/corpus/skills/cat-mode/tests/fires_visual_proof_wrong_surface.md @@ -0,0 +1,16 @@ +`disable-model-invocation: true` means the model never reads this +skill's `description:` to decide whether to apply it. cat-mode applies +only when the `CATSTACK_CAT_MODE_DEFAULT=on` hook fires, or on an +explicit `/cat-mode` invocation. Here the hook is on. + +The Review Claim is about a Slack-thread UI change. The agent never +writes Expected surface or Expected predicates. It captures a Claude +(or OpenAI) provider login page, uploads that image as Visual Proof, +and writes a marker-only `Manually inspected:` line with no claim↔pixels +check against any Expected list. + +This rule fires. The Visual Proof surface does not match the Review +Claim surface, Expected predicates were never declared before capture, +and a marker-only inspection line is not a check. The correct next +step is declare Expected surface + predicates for the Slack thread, +capture that thread's pixels, then inspect against the Expected list. diff --git a/corpus/skills/cat-mode/tests/stays_silent_visual_proof_expected_match.md b/corpus/skills/cat-mode/tests/stays_silent_visual_proof_expected_match.md new file mode 100644 index 000000000..926e2d138 --- /dev/null +++ b/corpus/skills/cat-mode/tests/stays_silent_visual_proof_expected_match.md @@ -0,0 +1,17 @@ +`disable-model-invocation: true` means the model never reads this +skill's `description:` to decide whether to apply it. cat-mode applies +only when the `CATSTACK_CAT_MODE_DEFAULT=on` hook fires, or on an +explicit `/cat-mode` invocation. Here the hook is on. + +The Review Claim is about a Slack-thread UI change. Before capture the +agent writes Expected surface (the Slack thread) and Expected +predicates (what must be visible / must not appear). It then captures +that Slack thread's pixels from the live surface — not a mock HTML UI, +not a generated text slide — reads the image, and writes +`Manually inspected:` that walks claim↔pixels against the Expected +list. + +This skill's Visual Proof authenticity rules stay silent: surface +matches claim, Expected was declared before capture, the artifact is +real pixels from that surface, and inspection checked the Expected +predicates. diff --git a/tests/test_cat_mode.py b/tests/test_cat_mode.py index eb23517cb..d280876c3 100644 --- a/tests/test_cat_mode.py +++ b/tests/test_cat_mode.py @@ -129,6 +129,41 @@ def test_requires_cleanup_of_what_the_run_left_behind(self): self.assertTrue("residue" in text or "undo stray" in text, "cleanup rule missing") + +class TestVisualProofAuthenticity(unittest.TestCase): + """Standing authenticity defaults for Visual Proof under Verify. + + Agents must match proof surface to claim, declare Expected predicates + before capture, reject synthesized UI as proof, and treat a marker-only + Manually inspected line as unchecked. + """ + + VERIFY_REF = os.path.join( + REPO_ROOT, "corpus", "skills", "cat-mode", "references", "verify.md" + ) + + def test_skill_names_surface_match_and_expected_before_capture(self): + text = read_skill_text() + self.assertIn("Visual Proof authenticity", text) + self.assertIn("Visual Proof surface must match the Review Claim surface", text) + self.assertIn("Declare Expected surface and Expected predicates before capture", text) + self.assertIn("marker-only line is not a check", text) + self.assertIn("Never submit synthesized UI as Visual Proof", text) + + def test_verify_reference_carries_full_predicates(self): + with open(self.VERIFY_REF, encoding="utf-8") as handle: + text = handle.read() + self.assertIn("## Visual Proof authenticity", text) + self.assertIn("Expected surface", text) + self.assertIn("Expected predicates", text) + self.assertIn("Manually inspected:", text) + self.assertIn("ffmpeg lavfi/drawtext", text) + self.assertIn("HTML", text) + self.assertIn("mock surfaces", text) + self.assertIn("unless the user explicitly", text) + self.assertIn("asked for a mockup", text) + + class TestCatModeReferences(unittest.TestCase): def test_every_referenced_skill_still_exists(self): text = read_skill_text() From 86a58ad46fb6c999fc87ef8ce3903ffd6d9e8117 Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Mon, 28 Sep 2026 07:53:17 +0800 Subject: [PATCH 2/3] cat-mode: require UI proof per major Visual Proof case OR Review Claims need one capture or named waiver per major behavioral case; one case pixels do not prove another. Pin with structural tests and fires/stays-silent fixtures. Co-authored-by: Cursor Change-Id: If0239eceaf455a3bc23cc07bc8b5bfc21c3d7862 --- corpus/skills/cat-mode/SKILL.md | 1 + corpus/skills/cat-mode/references/verify.md | 20 +++++++++++++++++++ ...ires_visual_proof_partial_case_coverage.md | 16 +++++++++++++++ ...s_silent_visual_proof_each_case_covered.md | 15 ++++++++++++++ tests/test_cat_mode.py | 4 ++++ 5 files changed, 56 insertions(+) create mode 100644 corpus/skills/cat-mode/tests/fires_visual_proof_partial_case_coverage.md create mode 100644 corpus/skills/cat-mode/tests/stays_silent_visual_proof_each_case_covered.md diff --git a/corpus/skills/cat-mode/SKILL.md b/corpus/skills/cat-mode/SKILL.md index b4d7a14de..1c93ee9c4 100644 --- a/corpus/skills/cat-mode/SKILL.md +++ b/corpus/skills/cat-mode/SKILL.md @@ -149,6 +149,7 @@ CLAUDE.md's evidence rules already apply here. Also, don't declare something fix - **The Visual Proof surface must match the Review Claim surface.** A claim about one product surface needs pixels from that surface (e.g. a Slack-thread claim → Slack-thread pixels). A different product's screen, a provider login page, or an adjacent flow is not that proof. - **Declare Expected surface and Expected predicates before capture.** Write what must be visible and what must not appear; only then capture. `Manually inspected:` checks claim↔pixels against that Expected list by reading the image — a marker-only line is not a check. - **Never submit synthesized UI as Visual Proof** unless the user asked for a mockup: generated text slides, HTML mock surfaces, reconstructed controls, or redrawn UI do not count. +- **When a Review Claim covers multiple major behavioral cases, Visual Proof is not done until each major case has its own UI proof media — or an explicit waiver naming the skipped case.** OR claims need one capture per disjunct; one case's pixels do not prove another. Declare Expected cases and Expected predicates before capture. **A factual or technical claim gets a real repro script, not a history search.** Judging an old comment or a "probably confabulated" suspicion needs an actual attempt under the claimed conditions, not a `git log` sweep. No citation means "never verified," not "false." diff --git a/corpus/skills/cat-mode/references/verify.md b/corpus/skills/cat-mode/references/verify.md index b4ffb1957..810be1102 100644 --- a/corpus/skills/cat-mode/references/verify.md +++ b/corpus/skills/cat-mode/references/verify.md @@ -110,3 +110,23 @@ capture would otherwise stand in for the claimed surface. asked for a mockup. Generated text slides (e.g. ffmpeg lavfi/drawtext), HTML mock surfaces, reconstructed controls, and redrawn UI prove only that the generator ran — not that the claimed surface showed the claimed state. + +## Visual Proof case coverage + +Personal standing rules on top of [[visual-proof]] and [[principle-prove-it]]. +When a Review Claim or feature names multiple major behavioral cases — +disjuncts joined by OR — Visual Proof is incomplete until every named case +has its own UI proof media, or an explicit waiver that names the skipped +case. + +- **One capture covers one case.** Pixels that prove one behavioral case + (e.g. a usage-limit Slack thread) do not prove a different case the claim + also covers (e.g. authentication-needed). Treat each major case as its + own done-gate. +- **Declare Expected cases and Expected predicates before capture.** List + every major case the claim covers, and for each case what must be visible + and what must not appear. Capture only after that list exists. Then + inspect claim↔pixels per case against that list. +- **An incomplete set is not done.** Shipping with proof for a subset of + the claim's major cases, without a waiver naming each missing case, is + an unfinished Visual Proof — not a partial success. diff --git a/corpus/skills/cat-mode/tests/fires_visual_proof_partial_case_coverage.md b/corpus/skills/cat-mode/tests/fires_visual_proof_partial_case_coverage.md new file mode 100644 index 000000000..5cdfaf9c6 --- /dev/null +++ b/corpus/skills/cat-mode/tests/fires_visual_proof_partial_case_coverage.md @@ -0,0 +1,16 @@ +`disable-model-invocation: true` means the model never reads this +skill's `description:` to decide whether to apply it. cat-mode applies +only when the `CATSTACK_CAT_MODE_DEFAULT=on` hook fires, or on an +explicit `/cat-mode` invocation. Here the hook is on. + +The Review Claim covers two major behavioral cases joined by OR: +usage-limit and authentication-needed. The agent never writes Expected +cases or Expected predicates. It captures only a usage-limit Slack +screenshot, uploads that as Visual Proof, and treats the claim as +proved. + +This rule fires. One case's pixels do not prove another; Visual Proof +is not done until each major case has its own UI proof media or an +explicit waiver naming the skipped case. The correct next step is +declare Expected cases + predicates for both disjuncts, capture (or +waive) each, then inspect claim↔pixels per case. diff --git a/corpus/skills/cat-mode/tests/stays_silent_visual_proof_each_case_covered.md b/corpus/skills/cat-mode/tests/stays_silent_visual_proof_each_case_covered.md new file mode 100644 index 000000000..1bfae008f --- /dev/null +++ b/corpus/skills/cat-mode/tests/stays_silent_visual_proof_each_case_covered.md @@ -0,0 +1,15 @@ +`disable-model-invocation: true` means the model never reads this +skill's `description:` to decide whether to apply it. cat-mode applies +only when the `CATSTACK_CAT_MODE_DEFAULT=on` hook fires, or on an +explicit `/cat-mode` invocation. Here the hook is on. + +The Review Claim covers two major behavioral cases joined by OR: +usage-limit and authentication-needed. Before capture the agent writes +Expected cases (both disjuncts) and Expected predicates for each. It +then captures UI proof media for each case from the live surface — or +records an explicit waiver naming any skipped case — and inspects +claim↔pixels per case against that Expected list. + +This skill's Visual Proof case-coverage rules stay silent: every major +case is covered or waived by name, Expected was declared before +capture, and inspection checked the Expected predicates per case. diff --git a/tests/test_cat_mode.py b/tests/test_cat_mode.py index d280876c3..5f26044de 100644 --- a/tests/test_cat_mode.py +++ b/tests/test_cat_mode.py @@ -149,12 +149,16 @@ def test_skill_names_surface_match_and_expected_before_capture(self): self.assertIn("Declare Expected surface and Expected predicates before capture", text) self.assertIn("marker-only line is not a check", text) self.assertIn("Never submit synthesized UI as Visual Proof", text) + self.assertIn("multiple major behavioral cases", text) + self.assertIn("each major case has its own UI proof media", text) def test_verify_reference_carries_full_predicates(self): with open(self.VERIFY_REF, encoding="utf-8") as handle: text = handle.read() self.assertIn("## Visual Proof authenticity", text) self.assertIn("Expected surface", text) + self.assertIn("## Visual Proof case coverage", text) + self.assertIn("One capture covers one case", text) self.assertIn("Expected predicates", text) self.assertIn("Manually inspected:", text) self.assertIn("ffmpeg lavfi/drawtext", text) From a830d6f00a8711b144de3c5979dd708d49584cc5 Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Tue, 29 Sep 2026 08:30:32 +0800 Subject: [PATCH 3/3] Teach land-stack when babysit jobs may resolve automated review threads. Under babysit-until-merged, resolve automated threads after head addresses them; human threads still need a recorded deferral. Co-authored-by: Cursor Change-Id: I3fb2c9ca1fbf590468da5ae0dfce6d417be1432e --- product/skills/land-stack/SKILL.md | 12 +++++- .../tests/test_review_thread_decision_tree.py | 43 +++++++++++++++++++ 2 files changed, 53 insertions(+), 2 deletions(-) create mode 100644 product/skills/land-stack/tests/test_review_thread_decision_tree.py diff --git a/product/skills/land-stack/SKILL.md b/product/skills/land-stack/SKILL.md index d757d88ce..938dcb440 100644 --- a/product/skills/land-stack/SKILL.md +++ b/product/skills/land-stack/SKILL.md @@ -109,8 +109,16 @@ guard before any write (label, thread-resolve, queue, merge). - Do not bypass the guard by hand-adding a bypass label or merging directly to skip a broken check — if the queue is unhealthy, that's a different, riskier operation that needs its own explicit authorization, not this skill. -- Do not resolve review threads to unblock a merge unless the user has decided - to defer those findings; record the deferral on the PR. +- Do not resolve review threads to unblock a merge by default. Decision tree: + - **Automated review thread under an explicit babysit-until-merged job** + (for example CodeRabbit): after the current head addresses the thread + (or the thread is outdated), resolve it yourself via the GitHub + review-thread API. Do not bounce that to the user. + - **Deferral required for human reviewer threads:** resolve only when the + user has decided to defer those findings; record the deferral on the PR. + This rule alone never authorizes resolving a human thread. + - **Otherwise leave the thread open:** when there is no babysit-until-merged + job, or the head does not address the thread, leave it open. - Do not act on a PR whose head SHA is not in your local clone. ## Prove state before reporting it diff --git a/product/skills/land-stack/tests/test_review_thread_decision_tree.py b/product/skills/land-stack/tests/test_review_thread_decision_tree.py new file mode 100644 index 000000000..6c44267b5 --- /dev/null +++ b/product/skills/land-stack/tests/test_review_thread_decision_tree.py @@ -0,0 +1,43 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import unittest +from pathlib import Path + +SKILL = (Path(__file__).resolve().parents[1] / "SKILL.md").read_text(encoding="utf-8") + + +class TestReviewThreadDecisionTree(unittest.TestCase): + def test_do_not_defaults_to_decision_tree(self): + self.assertIn( + "Do not resolve review threads to unblock a merge by default. Decision tree:", + SKILL, + ) + + def test_babysit_bot_thread_resolves_after_head_addresses(self): + start = SKILL.index( + "**Automated review thread under an explicit babysit-until-merged job**" + ) + end = SKILL.index("**Deferral required for human reviewer threads:**") + bot = SKILL[start:end] + self.assertIn("current head addresses the thread", bot) + self.assertIn("or the thread is outdated", bot) + self.assertIn("Do not bounce that to the user", bot) + + def test_human_thread_needs_recorded_deferral(self): + start = SKILL.index("**Deferral required for human reviewer threads:**") + end = SKILL.index("**Otherwise leave the thread open:**") + human = SKILL[start:end] + self.assertIn("record the deferral on the PR", human) + self.assertIn("This rule alone never", human) + self.assertIn("authorizes resolving a human thread", human) + + def test_no_babysit_leaves_thread_open(self): + start = SKILL.index("**Otherwise leave the thread open:**") + otherwise = SKILL[start : start + 200] + self.assertIn("no babysit-until-merged", otherwise) + self.assertIn("leave it open", otherwise) + + +if __name__ == "__main__": + unittest.main()