From 43a8b5496f6ca3da430f5804ede7ac7f5d8dd6c5 Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Tue, 29 Sep 2026 08:56:03 +0800 Subject: [PATCH 1/2] Add scalable spreadsheet authoring skill Change-Id: I4e2f319839830f71c7f873c0f4560bee194eaa86 --- product/skills/spreadsheet-authoring/SKILL.md | 49 ++++++++++++++ .../scripts/validate_schema.py | 66 +++++++++++++++++++ .../tests/fires_typed_temporal_schema.md | 9 +++ .../tests/stays_silent_unrelated_table.md | 5 ++ .../tests/test_validate_schema.py | 38 +++++++++++ 5 files changed, 167 insertions(+) create mode 100644 product/skills/spreadsheet-authoring/SKILL.md create mode 100644 product/skills/spreadsheet-authoring/scripts/validate_schema.py create mode 100644 product/skills/spreadsheet-authoring/tests/fires_typed_temporal_schema.md create mode 100644 product/skills/spreadsheet-authoring/tests/stays_silent_unrelated_table.md create mode 100644 product/skills/spreadsheet-authoring/tests/test_validate_schema.py diff --git a/product/skills/spreadsheet-authoring/SKILL.md b/product/skills/spreadsheet-authoring/SKILL.md new file mode 100644 index 00000000..896f0856 --- /dev/null +++ b/product/skills/spreadsheet-authoring/SKILL.md @@ -0,0 +1,49 @@ +--- +name: spreadsheet-authoring +description: >- + Design scalable spreadsheet models with category-level raw data, formula- + driven aggregation tables, provenance, and charts. Use when creating or + restructuring Excel or Google Sheets workbooks that must grow by adding rows. +--- + +# Spreadsheet authoring + +Build the workbook as a data pipeline, not as a decorated report: + +`raw observations → selector-driven aggregation table → chart` + +## Raw observation contract + +- Keep one raw sheet per category. +- Keep one numeric observation per row. +- Put dates in typed `Period Start` and `Period End` columns. +- Put `Average`, `Start`, `Peak`, `End`, `Low`, `High`, or similar qualifiers + in a separate categorical `Observation Point` column. +- Preserve the source wording in `Original Period Label` for audit, but never + use that display label as the aggregation key. +- Keep geography, scope, segment, entity, metric, value, unit, source, source + URL, and source location as separate fields. + +When parsing a range such as `Nov 2024–Sep 2025`, write the first and last +dates as typed values. When parsing `Q4 2023 end`, write the quarter dates and +`End` separately. Derive human-readable titles after the typed fields exist. + +## Aggregation and chart contract + +- The aggregation table must reference raw fields, not copied values. +- Include every dimension that distinguishes observations in its key, including + the observation point and cohort/segment. +- Let selectors choose the metric and the relevant subset of raw data. +- Reserve expandable formula ranges so adding a valid raw row updates the table. +- Charts must reference the aggregation table only, never the raw sheet. + +## Preflight and verification + +Before editing the live workbook, validate representative period labels and +the migration result. Reject any schema where a time column contains a +qualifier token or where two distinct observation points collapse to one key. +For CSV exports or fixtures, run `scripts/validate_schema.py `. + +After writing, reread raw headers, typed dates, observation points, table +formulas, selector behavior, and chart source ranges. Test both a new row and +an invalid or duplicate row when the workbook is intended to scale. diff --git a/product/skills/spreadsheet-authoring/scripts/validate_schema.py b/product/skills/spreadsheet-authoring/scripts/validate_schema.py new file mode 100644 index 00000000..fd493139 --- /dev/null +++ b/product/skills/spreadsheet-authoring/scripts/validate_schema.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +"""Validate the typed temporal columns used by spreadsheet authoring.""" + +from __future__ import annotations + +import csv +import re +import sys +from pathlib import Path + + +REQUIRED = { + "Original Period Label", + "Period Start", + "Period End", + "Observation Point", + "Geography", + "Scope", + "Segment", + "Entity", + "Metric", + "Value", +} +QUALIFIER = re.compile(r"\b(avg|average|start|peak|end|low|high)\b", re.I) +KEY_FIELDS = ["Period Start", "Period End", "Observation Point", "Geography", "Scope", "Segment", "Entity", "Metric"] + + +def validate_rows(rows: list[dict[str, str]]) -> list[str]: + errors: list[str] = [] + if not rows: + return ["no observations"] + missing = REQUIRED - set(rows[0]) + if missing: + errors.append("missing columns: " + ", ".join(sorted(missing))) + return errors + keys: dict[tuple[str, ...], int] = {} + for number, row in enumerate(rows, start=2): + if QUALIFIER.search(row["Period Start"]) or QUALIFIER.search(row["Period End"]): + errors.append(f"row {number}: qualifier embedded in period column") + key = tuple(row[field].strip() for field in KEY_FIELDS) + if key in keys: + errors.append(f"row {number}: duplicate observation key; first seen at row {keys[key]}") + else: + keys[key] = number + return errors + + +def validate_file(path: Path) -> list[str]: + with path.open(newline="", encoding="utf-8") as handle: + return validate_rows(list(csv.DictReader(handle))) + + +def main(argv: list[str]) -> int: + if len(argv) != 2: + print(f"usage: {argv[0]} FILE", file=sys.stderr) + return 2 + errors = validate_file(Path(argv[1])) + if errors: + print("\n".join(errors), file=sys.stderr) + return 1 + print("ok: typed temporal schema") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/product/skills/spreadsheet-authoring/tests/fires_typed_temporal_schema.md b/product/skills/spreadsheet-authoring/tests/fires_typed_temporal_schema.md new file mode 100644 index 00000000..e9d932d1 --- /dev/null +++ b/product/skills/spreadsheet-authoring/tests/fires_typed_temporal_schema.md @@ -0,0 +1,9 @@ +# Fires + +Invoke this skill when a workbook request includes raw observations, periods, +qualifiers such as peak/end/average, aggregation tables, or charts that must +update when new rows are added. + +Expected behavior: separate typed period start/end fields from the observation +point, preserve the original label for audit, and wire the table and chart to +formula-driven outputs. diff --git a/product/skills/spreadsheet-authoring/tests/stays_silent_unrelated_table.md b/product/skills/spreadsheet-authoring/tests/stays_silent_unrelated_table.md new file mode 100644 index 00000000..c9cedb54 --- /dev/null +++ b/product/skills/spreadsheet-authoring/tests/stays_silent_unrelated_table.md @@ -0,0 +1,5 @@ +# Stays silent + +Do not invoke this skill for a request to rename a single spreadsheet tab or +change a cell's color when no raw-data model, aggregation, provenance, or chart +relationship is involved. diff --git a/product/skills/spreadsheet-authoring/tests/test_validate_schema.py b/product/skills/spreadsheet-authoring/tests/test_validate_schema.py new file mode 100644 index 00000000..cf6e8ed1 --- /dev/null +++ b/product/skills/spreadsheet-authoring/tests/test_validate_schema.py @@ -0,0 +1,38 @@ +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parents[1] / "scripts")) +from validate_schema import validate_rows + + +def row(**overrides): + result = { + "Original Period Label": "Q4 2023 end", + "Period Start": "2023-10-01", + "Period End": "2023-12-31", + "Observation Point": "End", + "Geography": "US", + "Scope": "Unified", + "Segment": "", + "Entity": "Uber", + "Metric": "Weekly active users", + "Value": "11800000", + } + result.update(overrides) + return result + + +class SchemaTests(unittest.TestCase): + def test_accepts_typed_period_and_distinct_points(self): + errors = validate_rows([row(), row(**{"Original Period Label": "Q4 2023 peak", "Observation Point": "Peak"})]) + self.assertEqual(errors, []) + + def test_rejects_qualifier_in_period_start(self): + errors = validate_rows([row(**{"Period Start": "Q4 2023 end"})]) + self.assertEqual(len(errors), 1) + self.assertIn("qualifier embedded", errors[0]) + + +if __name__ == "__main__": + unittest.main() From ff6941577e13aa7d3c4736291063e874a8d26bfb Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Tue, 29 Sep 2026 11:22:53 +0800 Subject: [PATCH 2/2] Add spreadsheet completeness stop guard Change-Id: I913b91811835219351c9267075e078ccbe5795d1 --- engine/hooks/hooks.toml | 5 ++ .../spreadsheet-completeness-guard/README.md | 8 +++ .../claude.hook.json | 5 ++ .../claude_stop_check.py | 10 ++++ .../codex.hook.json | 5 ++ .../codex_stop_check.py | 10 ++++ .../cursor.hook.json | 5 ++ .../cursor_stop_check.py | 10 ++++ .../spreadsheet-completeness-guard/detect.py | 55 +++++++++++++++++++ .../tests/test_hooks.py | 38 +++++++++++++ 10 files changed, 151 insertions(+) create mode 100644 engine/hooks/spreadsheet-completeness-guard/README.md create mode 100644 engine/hooks/spreadsheet-completeness-guard/claude.hook.json create mode 100644 engine/hooks/spreadsheet-completeness-guard/claude_stop_check.py create mode 100644 engine/hooks/spreadsheet-completeness-guard/codex.hook.json create mode 100644 engine/hooks/spreadsheet-completeness-guard/codex_stop_check.py create mode 100644 engine/hooks/spreadsheet-completeness-guard/cursor.hook.json create mode 100644 engine/hooks/spreadsheet-completeness-guard/cursor_stop_check.py create mode 100644 engine/hooks/spreadsheet-completeness-guard/detect.py create mode 100644 engine/hooks/spreadsheet-completeness-guard/tests/test_hooks.py diff --git a/engine/hooks/hooks.toml b/engine/hooks/hooks.toml index 7c8f4268..82d9d6be 100644 --- a/engine/hooks/hooks.toml +++ b/engine/hooks/hooks.toml @@ -212,6 +212,11 @@ summary = "Stops drift after the user narrowed the task twice." enabled_by = "CATSTACK_REFLECT_ENFORCEMENT" rule_modes = { "scope-lock.prompt-instruction" = "warn", "scope-lock.reflection-acknowledged" = "warn" } +[hooks.spreadsheet-completeness-guard] +mode = "stop" +why_mode = "attention" +summary = "Blocks spreadsheet completion claims without source-backed entity coverage." + [hooks.scratchpad-collision] mode = "stop" why_mode = "outward" diff --git a/engine/hooks/spreadsheet-completeness-guard/README.md b/engine/hooks/spreadsheet-completeness-guard/README.md new file mode 100644 index 00000000..b8cd9959 --- /dev/null +++ b/engine/hooks/spreadsheet-completeness-guard/README.md @@ -0,0 +1,8 @@ +# spreadsheet-completeness-guard + +Stop hook for spreadsheet completion claims. A completion claim must carry a +source-backed `SPREADSHEET_COVERAGE: PASS` receipt from the expected-entity +validator. Missing observations must be explicitly classified; estimates and +zeros are not substitutes for absent competitor data. + +Tests: `python3 -m unittest discover -s engine/hooks/spreadsheet-completeness-guard/tests -v` diff --git a/engine/hooks/spreadsheet-completeness-guard/claude.hook.json b/engine/hooks/spreadsheet-completeness-guard/claude.hook.json new file mode 100644 index 00000000..3562a0c9 --- /dev/null +++ b/engine/hooks/spreadsheet-completeness-guard/claude.hook.json @@ -0,0 +1,5 @@ +{ + "hooks": { + "Stop": [{"matcher": "*", "hooks": [{"type": "command", "command": "python3 $HOME/.claude/hooks/spreadsheet-completeness-guard/claude_stop_check.py", "timeout": 10}]}] + } +} diff --git a/engine/hooks/spreadsheet-completeness-guard/claude_stop_check.py b/engine/hooks/spreadsheet-completeness-guard/claude_stop_check.py new file mode 100644 index 00000000..4a4d8806 --- /dev/null +++ b/engine/hooks/spreadsheet-completeness-guard/claude_stop_check.py @@ -0,0 +1,10 @@ +#!/usr/bin/env python3 +from __future__ import annotations +import os +import sys +sys.path.insert(0, os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "_sdk")) +from detect import detect # noqa: E402 +from runtime import run_hook # noqa: E402 + +if __name__ == "__main__": + run_hook("spreadsheet-completeness-guard", "claude", detect, "Stop", json_error_stderr=False) diff --git a/engine/hooks/spreadsheet-completeness-guard/codex.hook.json b/engine/hooks/spreadsheet-completeness-guard/codex.hook.json new file mode 100644 index 00000000..447e5ad2 --- /dev/null +++ b/engine/hooks/spreadsheet-completeness-guard/codex.hook.json @@ -0,0 +1,5 @@ +{ + "hooks": { + "Stop": [{"matcher": "*", "hooks": [{"type": "command", "command": "python3 $HOME/.codex/hooks/spreadsheet-completeness-guard/codex_stop_check.py", "timeout": 10}]}] + } +} diff --git a/engine/hooks/spreadsheet-completeness-guard/codex_stop_check.py b/engine/hooks/spreadsheet-completeness-guard/codex_stop_check.py new file mode 100644 index 00000000..dd667ce5 --- /dev/null +++ b/engine/hooks/spreadsheet-completeness-guard/codex_stop_check.py @@ -0,0 +1,10 @@ +#!/usr/bin/env python3 +from __future__ import annotations +import os +import sys +sys.path.insert(0, os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "_sdk")) +from detect import detect # noqa: E402 +from runtime import run_hook # noqa: E402 + +if __name__ == "__main__": + run_hook("spreadsheet-completeness-guard", "codex", detect, "Stop") diff --git a/engine/hooks/spreadsheet-completeness-guard/cursor.hook.json b/engine/hooks/spreadsheet-completeness-guard/cursor.hook.json new file mode 100644 index 00000000..8705fc89 --- /dev/null +++ b/engine/hooks/spreadsheet-completeness-guard/cursor.hook.json @@ -0,0 +1,5 @@ +{ + "hooks": { + "stop": [{"command": "python3 $HOME/.cursor/hooks/spreadsheet-completeness-guard/cursor_stop_check.py", "timeout": 10}] + } +} diff --git a/engine/hooks/spreadsheet-completeness-guard/cursor_stop_check.py b/engine/hooks/spreadsheet-completeness-guard/cursor_stop_check.py new file mode 100644 index 00000000..d586b3f5 --- /dev/null +++ b/engine/hooks/spreadsheet-completeness-guard/cursor_stop_check.py @@ -0,0 +1,10 @@ +#!/usr/bin/env python3 +from __future__ import annotations +import os +import sys +sys.path.insert(0, os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "_sdk")) +from detect import detect # noqa: E402 +from runtime import run_hook # noqa: E402 + +if __name__ == "__main__": + run_hook("spreadsheet-completeness-guard", "cursor", detect, "stop") diff --git a/engine/hooks/spreadsheet-completeness-guard/detect.py b/engine/hooks/spreadsheet-completeness-guard/detect.py new file mode 100644 index 00000000..62b6b32d --- /dev/null +++ b/engine/hooks/spreadsheet-completeness-guard/detect.py @@ -0,0 +1,55 @@ +"""Stop completion claims for spreadsheet work without a coverage receipt.""" +from __future__ import annotations + +import re +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "_sdk")) +from finding import Finding # noqa: E402 + +RULE_INCOMPLETE = "spreadsheet-completeness-guard.incomplete" +SPREADSHEET_RE = re.compile(r"\b(?:spreadsheet|google sheet|google sheets|workbook|raw tab|aggregation table|chart)\b", re.I) +COMPLETION_RE = re.compile(r"\b(?:complete|completed|done|filled|populated|regenerated|ready|all competitors)\b", re.I) +RECEIPT_RE = re.compile(r"SPREADSHEET_COVERAGE\s*:\s*PASS\b", re.I) +FAIL_RE = re.compile(r"SPREADSHEET_COVERAGE\s*:\s*FAIL\b", re.I) + + +def claims_completion(message: str) -> bool: + return bool(SPREADSHEET_RE.search(message or "") and COMPLETION_RE.search(message or "")) + + +def has_receipt(message: str) -> bool: + return bool(RECEIPT_RE.search(message or "")) + + +def decide(payload: dict) -> Finding | None: + message = str(payload.get("last_assistant_message") or payload.get("message") or "") + if not claims_completion(message) or has_receipt(message): + return None + try: + retry_count = int(payload.get("spreadsheet_retry_count") or 0) + except (TypeError, ValueError): + retry_count = 0 + retry_count = max(0, min(retry_count, 3)) + if FAIL_RE.search(message): + return Finding( + rule_id=RULE_INCOMPLETE, + subject="spreadsheet coverage failure", + message="Coverage failed after the bounded retry loop. Do not generate or declare the table/chart complete; list the unresolved entity/period keys.", + evidence=f"SPREADSHEET_COVERAGE: FAIL; retry {retry_count}/3", + ) + return Finding( + rule_id=RULE_INCOMPLETE, + subject="spreadsheet completion claim", + message=(f"Do not declare spreadsheet data complete until the expected-entity coverage " + f"validator has passed. Run source collection retry {min(retry_count + 1, 3)}/3; " + "add source-backed rows or state unresolved entity/period keys. " + "Only SPREADSHEET_COVERAGE: PASS permits completion."), + evidence=f"completion claim without a coverage receipt; retry {retry_count}/3", + ) + + +def detect(event: dict[str, object]) -> list[Finding]: + finding = decide(event) + return [] if finding is None else [finding] diff --git a/engine/hooks/spreadsheet-completeness-guard/tests/test_hooks.py b/engine/hooks/spreadsheet-completeness-guard/tests/test_hooks.py new file mode 100644 index 00000000..375deda3 --- /dev/null +++ b/engine/hooks/spreadsheet-completeness-guard/tests/test_hooks.py @@ -0,0 +1,38 @@ +import os +import sys +import unittest + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) +import detect # noqa: E402 + + +class SpreadsheetCompletenessTests(unittest.TestCase): + def test_blocks_completion_without_coverage_receipt(self): + finding = detect.decide({"last_assistant_message": "The spreadsheet is complete and all competitors are populated."}) + self.assertIsNotNone(finding) + self.assertEqual(finding.rule_id, "spreadsheet-completeness-guard.incomplete") + + def test_stays_silent_with_pass_receipt(self): + finding = detect.decide({"last_assistant_message": "SPREADSHEET_COVERAGE: PASS — the workbook is complete."}) + self.assertIsNone(finding) + + def test_stays_silent_for_non_spreadsheet_completion(self): + self.assertIsNone(detect.decide({"last_assistant_message": "The README edit is complete."})) + + def test_requests_bounded_retry_when_coverage_is_missing(self): + finding = detect.decide({ + "last_assistant_message": "The spreadsheet is complete.", + "spreadsheet_retry_count": 1, + }) + self.assertIn("retry 2/3", finding.message) + + def test_keeps_blocking_after_three_failed_retries(self): + finding = detect.decide({ + "last_assistant_message": "SPREADSHEET_COVERAGE: FAIL — the spreadsheet is complete.", + "spreadsheet_retry_count": 3, + }) + self.assertIn("after the bounded retry loop", finding.message) + + +if __name__ == "__main__": + unittest.main()