diff --git a/engine/skills/reflect/SKILL.md b/engine/skills/reflect/SKILL.md index 5c47964a2..a7bbf10fd 100644 --- a/engine/skills/reflect/SKILL.md +++ b/engine/skills/reflect/SKILL.md @@ -85,16 +85,37 @@ Token usage is exact data sitting in every transcript. Don't have an LLM reviewe Read [references/cost-audit.md](references/cost-audit.md) for CLI (`token_audit.py`, including `--out`), thrash detectors, model-tier backtest, `top_sessions.py`, and tests. +If `CATSTACK_REFLECT_LENS_BUDGET` is set to a positive integer token count, +read the audit report's `total` field after step 2. When `total` is over the +configured budget, step 3 uses the ordered required lens set from +[references/lenses.md](references/lenses.md) and omits the other lenses. An +unset variable leaves the historical fan-out unchanged. A malformed or +non-positive value is an error: do not guess at a budget. + ### 3. Spawn parallel reviewers Read [references/lenses.md](references/lenses.md) for the five lenses and the fix hierarchy. Prefer the cheapest check that still catches the mistake — do not write a skill line when a hook or test would do. Before fanning out, check for sibling passes on the same incident: `git branch --all | grep -E "(reflect-ci|fix-ci)-"` for concurrently dispatched fix/reflect branches, and `ls ~/.claude/projects/ | grep -F ` for a sibling reflect's surviving transcript. A crashed sibling commits nothing — its synthesis lives only in its transcript tail; read that as prior art instead of re-deriving the same facts from zero. +The normal expected set is `Judgment Tooling Cost History Divergent Frustration`. +If the budget gate reduced the pass, launch only the ordered required set and +record that exact set before launching: +`python3 scripts/fanout_complete.py --expected `. The expected +set is the contract for this pass; it is not silently inferred from returned +agents. If the harness supports per-agent model selection, use the cheaper +configured lens model for the reduced pass; otherwise keep the normal model. +Pass the following status to synthesis when reduced: +`reduced reflect: ran , omitted (session tokens over budget )`. + ### 4. Synthesize One more `Agent` call, given all reviewers' output, merges overlapping findings, writes each one in three parts (below), and sorts them into: +When the budget gate reduced step 3, synthesis must include the exact reduced +status line above before findings. It must name every omitted lens, even when +all required lenses returned. + - **Accepted** — real, durable, worth acting on. Apply the elimination hierarchy from step 3 before slotting a finding here as a skill edit: if a reviewer proposed a skill/rule fix but a categorical or lint/test fix was actually available, bump it to Backlog with the stronger fix named instead, or split it. - **Backlog** — real, but the right fix is higher up the hierarchy than a skill edit. Note which tier (1: categorical, 2: lint/test, 3: hook) each backlog item is. - **Grounding gate for skill prose.** Before an Accepted item becomes skill prose, name the established principle it instantiates — author, title, year, and a checkable URL — or write "no known prior art". An incident-shaped rule with neither goes to Backlog for grounding, not to Accepted. A rule that restates one session's bug in fresh words reads as invented and drifts into an incident log; the field's own name for it (fail fast, invariant, completeness check, reconciliation) is what the skill should say. diff --git a/engine/skills/reflect/references/lenses.md b/engine/skills/reflect/references/lenses.md index 7d72815ce..aa99bfc89 100644 --- a/engine/skills/reflect/references/lenses.md +++ b/engine/skills/reflect/references/lenses.md @@ -4,6 +4,17 @@ Read this when running reflect step 3 (parallel reviewers). ## Lenses +The normal ordered fan-out is: +`Judgment`, `Tooling`, `Cost`, `History`, `Divergent`, `Frustration`. + +When `CATSTACK_REFLECT_LENS_BUDGET` is set and the step-2 +`token_audit.py` report has `total` tokens over that budget, the required +reduced set is the ordered subset `Judgment`, `Cost`, `Frustration`. These are +required because they preserve root-cause analysis, measured cost analysis, +and user-impact/failure detection. The omitted set is `Tooling`, `History`, +and `Divergent`; synthesis must name each one in the required reduced status +line. An unset budget keeps the normal full set. + One message, parallel `Agent` calls (`subagent_type: general-purpose`), each given the transcript path (plus the cost-audit output for the Cost lens, the git log for the History lens, and the mechanical frustration-signal list from `token_audit.py` for the Frustration lens) and a distinct lens: | Lens | Looks for | diff --git a/engine/skills/reflect/scripts/fanout_complete.py b/engine/skills/reflect/scripts/fanout_complete.py index 222b22933..3de61ff3e 100755 --- a/engine/skills/reflect/scripts/fanout_complete.py +++ b/engine/skills/reflect/scripts/fanout_complete.py @@ -23,6 +23,7 @@ import argparse import json +import os import sys COMPLETE = "complete" @@ -31,6 +32,52 @@ EXIT = {COMPLETE: 0, INCOMPLETE: 1, UNCHECKED: 2} +ALL_LENSES = ("Judgment", "Tooling", "Cost", "History", "Divergent", "Frustration") +REQUIRED_LENSES = ("Judgment", "Cost", "Frustration") + + +def lens_plan(session_tokens, budget=None): + """Return the deterministic step-3 lens plan for a measured session. + + ``budget=None`` is the feature-disabled state and deliberately returns the + historical full fan-out. A reduced plan is complete when its returned + expected set is recorded with this module's ``verdict`` function. + """ + if budget is None or session_tokens <= budget: + return { + "expected": list(ALL_LENSES), + "omitted": [], + "reduced": False, + } + return { + "expected": list(REQUIRED_LENSES), + "omitted": [lens for lens in ALL_LENSES if lens not in REQUIRED_LENSES], + "reduced": True, + } + + +def lens_plan_from_env(session_tokens, environ=None): + """Read the optional token budget and select the step-3 lens plan.""" + value = (environ or os.environ).get("CATSTACK_REFLECT_LENS_BUDGET") + if value is None: + return lens_plan(session_tokens) + try: + budget = int(value) + except ValueError as exc: + raise ValueError("CATSTACK_REFLECT_LENS_BUDGET must be an integer") from exc + if budget <= 0: + raise ValueError("CATSTACK_REFLECT_LENS_BUDGET must be positive") + return lens_plan(session_tokens, budget) + + +def reduced_status(session_tokens, budget, plan): + """Name every omitted lens for the step-4 synthesis prompt.""" + if not plan["reduced"]: + return "" + ran = ", ".join(plan["expected"]) + omitted = ", ".join(plan["omitted"]) + return f"reduced reflect: ran {ran}, omitted {omitted} (session {session_tokens} tokens over budget {budget})" + def verdict(expected, returned): """Outcome plus the lenses missing from this fan-out. diff --git a/engine/skills/reflect/scripts/tests/test_reflect_lens_budget.py b/engine/skills/reflect/scripts/tests/test_reflect_lens_budget.py new file mode 100644 index 000000000..ec32149c7 --- /dev/null +++ b/engine/skills/reflect/scripts/tests/test_reflect_lens_budget.py @@ -0,0 +1,55 @@ +import os +import sys +import unittest + +SCRIPTS = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +if SCRIPTS not in sys.path: + sys.path.insert(0, SCRIPTS) + +import fanout_complete as fc # noqa: E402 + + +class TestReflectLensBudget(unittest.TestCase): + def test_under_budget_runs_full_ordered_set(self): + plan = fc.lens_plan(99, 100) + self.assertEqual(plan["expected"], list(fc.ALL_LENSES)) + self.assertEqual(plan["omitted"], []) + self.assertFalse(plan["reduced"]) + + def test_over_budget_runs_required_set_and_names_omissions(self): + plan = fc.lens_plan(101, 100) + self.assertEqual(plan["expected"], list(fc.REQUIRED_LENSES)) + self.assertEqual(plan["omitted"], ["Tooling", "History", "Divergent"]) + self.assertTrue(plan["reduced"]) + + def test_unset_budget_preserves_unchanged_behavior(self): + plan = fc.lens_plan_from_env(80_000_000, {}) + self.assertEqual(plan["expected"], list(fc.ALL_LENSES)) + self.assertFalse(plan["reduced"]) + + def test_environment_budget_reduces_and_status_names_every_omission(self): + plan = fc.lens_plan_from_env(80_000_000, {"CATSTACK_REFLECT_LENS_BUDGET": "1000000"}) + self.assertEqual( + fc.reduced_status(80_000_000, 1_000_000, plan), + "reduced reflect: ran Judgment, Cost, Frustration, omitted Tooling, History, Divergent (session 80000000 tokens over budget 1000000)", + ) + + def test_incident_scale_session_reduces(self): + plan = fc.lens_plan(80_000_000, 1_000_000) + self.assertTrue(plan["reduced"]) + self.assertEqual(plan["expected"], ["Judgment", "Cost", "Frustration"]) + + def test_recorded_reduced_pass_is_complete(self): + plan = fc.lens_plan(101, 100) + outcome, missing, unexpected = fc.verdict(plan["expected"], plan["expected"]) + self.assertEqual((outcome, missing, unexpected), (fc.COMPLETE, [], [])) + + def test_recorded_reduced_pass_missing_required_lens_is_incomplete(self): + plan = fc.lens_plan(101, 100) + outcome, missing, _ = fc.verdict(plan["expected"], ["Judgment", "Cost"]) + self.assertEqual(outcome, fc.INCOMPLETE) + self.assertEqual(missing, ["Frustration"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/engine/skills/reflect/tests/__init__.py b/engine/skills/reflect/tests/__init__.py new file mode 100644 index 000000000..4bf2fd0b5 --- /dev/null +++ b/engine/skills/reflect/tests/__init__.py @@ -0,0 +1 @@ +"""Reflect skill acceptance-test entrypoint.""" diff --git a/engine/skills/reflect/tests/test_reflect_lens_budget.py b/engine/skills/reflect/tests/test_reflect_lens_budget.py new file mode 100644 index 000000000..9537e77bd --- /dev/null +++ b/engine/skills/reflect/tests/test_reflect_lens_budget.py @@ -0,0 +1,15 @@ +"""Run the lens-budget fixtures from the repository's script-test suite.""" + +import importlib.util +import os + +SOURCE = os.path.join( + os.path.dirname(os.path.dirname(os.path.abspath(__file__))), + "scripts", + "tests", + "test_reflect_lens_budget.py", +) +spec = importlib.util.spec_from_file_location("reflect_script_lens_budget", SOURCE) +module = importlib.util.module_from_spec(spec) +spec.loader.exec_module(module) +TestReflectLensBudget = module.TestReflectLensBudget