From 43a8b5496f6ca3da430f5804ede7ac7f5d8dd6c5 Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Tue, 29 Sep 2026 08:56:03 +0800 Subject: [PATCH 1/2] Add scalable spreadsheet authoring skill Change-Id: I4e2f319839830f71c7f873c0f4560bee194eaa86 --- product/skills/spreadsheet-authoring/SKILL.md | 49 ++++++++++++++ .../scripts/validate_schema.py | 66 +++++++++++++++++++ .../tests/fires_typed_temporal_schema.md | 9 +++ .../tests/stays_silent_unrelated_table.md | 5 ++ .../tests/test_validate_schema.py | 38 +++++++++++ 5 files changed, 167 insertions(+) create mode 100644 product/skills/spreadsheet-authoring/SKILL.md create mode 100644 product/skills/spreadsheet-authoring/scripts/validate_schema.py create mode 100644 product/skills/spreadsheet-authoring/tests/fires_typed_temporal_schema.md create mode 100644 product/skills/spreadsheet-authoring/tests/stays_silent_unrelated_table.md create mode 100644 product/skills/spreadsheet-authoring/tests/test_validate_schema.py diff --git a/product/skills/spreadsheet-authoring/SKILL.md b/product/skills/spreadsheet-authoring/SKILL.md new file mode 100644 index 00000000..896f0856 --- /dev/null +++ b/product/skills/spreadsheet-authoring/SKILL.md @@ -0,0 +1,49 @@ +--- +name: spreadsheet-authoring +description: >- + Design scalable spreadsheet models with category-level raw data, formula- + driven aggregation tables, provenance, and charts. Use when creating or + restructuring Excel or Google Sheets workbooks that must grow by adding rows. +--- + +# Spreadsheet authoring + +Build the workbook as a data pipeline, not as a decorated report: + +`raw observations → selector-driven aggregation table → chart` + +## Raw observation contract + +- Keep one raw sheet per category. +- Keep one numeric observation per row. +- Put dates in typed `Period Start` and `Period End` columns. +- Put `Average`, `Start`, `Peak`, `End`, `Low`, `High`, or similar qualifiers + in a separate categorical `Observation Point` column. +- Preserve the source wording in `Original Period Label` for audit, but never + use that display label as the aggregation key. +- Keep geography, scope, segment, entity, metric, value, unit, source, source + URL, and source location as separate fields. + +When parsing a range such as `Nov 2024–Sep 2025`, write the first and last +dates as typed values. When parsing `Q4 2023 end`, write the quarter dates and +`End` separately. Derive human-readable titles after the typed fields exist. + +## Aggregation and chart contract + +- The aggregation table must reference raw fields, not copied values. +- Include every dimension that distinguishes observations in its key, including + the observation point and cohort/segment. +- Let selectors choose the metric and the relevant subset of raw data. +- Reserve expandable formula ranges so adding a valid raw row updates the table. +- Charts must reference the aggregation table only, never the raw sheet. + +## Preflight and verification + +Before editing the live workbook, validate representative period labels and +the migration result. Reject any schema where a time column contains a +qualifier token or where two distinct observation points collapse to one key. +For CSV exports or fixtures, run `scripts/validate_schema.py `. + +After writing, reread raw headers, typed dates, observation points, table +formulas, selector behavior, and chart source ranges. Test both a new row and +an invalid or duplicate row when the workbook is intended to scale. diff --git a/product/skills/spreadsheet-authoring/scripts/validate_schema.py b/product/skills/spreadsheet-authoring/scripts/validate_schema.py new file mode 100644 index 00000000..fd493139 --- /dev/null +++ b/product/skills/spreadsheet-authoring/scripts/validate_schema.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +"""Validate the typed temporal columns used by spreadsheet authoring.""" + +from __future__ import annotations + +import csv +import re +import sys +from pathlib import Path + + +REQUIRED = { + "Original Period Label", + "Period Start", + "Period End", + "Observation Point", + "Geography", + "Scope", + "Segment", + "Entity", + "Metric", + "Value", +} +QUALIFIER = re.compile(r"\b(avg|average|start|peak|end|low|high)\b", re.I) +KEY_FIELDS = ["Period Start", "Period End", "Observation Point", "Geography", "Scope", "Segment", "Entity", "Metric"] + + +def validate_rows(rows: list[dict[str, str]]) -> list[str]: + errors: list[str] = [] + if not rows: + return ["no observations"] + missing = REQUIRED - set(rows[0]) + if missing: + errors.append("missing columns: " + ", ".join(sorted(missing))) + return errors + keys: dict[tuple[str, ...], int] = {} + for number, row in enumerate(rows, start=2): + if QUALIFIER.search(row["Period Start"]) or QUALIFIER.search(row["Period End"]): + errors.append(f"row {number}: qualifier embedded in period column") + key = tuple(row[field].strip() for field in KEY_FIELDS) + if key in keys: + errors.append(f"row {number}: duplicate observation key; first seen at row {keys[key]}") + else: + keys[key] = number + return errors + + +def validate_file(path: Path) -> list[str]: + with path.open(newline="", encoding="utf-8") as handle: + return validate_rows(list(csv.DictReader(handle))) + + +def main(argv: list[str]) -> int: + if len(argv) != 2: + print(f"usage: {argv[0]} FILE", file=sys.stderr) + return 2 + errors = validate_file(Path(argv[1])) + if errors: + print("\n".join(errors), file=sys.stderr) + return 1 + print("ok: typed temporal schema") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/product/skills/spreadsheet-authoring/tests/fires_typed_temporal_schema.md b/product/skills/spreadsheet-authoring/tests/fires_typed_temporal_schema.md new file mode 100644 index 00000000..e9d932d1 --- /dev/null +++ b/product/skills/spreadsheet-authoring/tests/fires_typed_temporal_schema.md @@ -0,0 +1,9 @@ +# Fires + +Invoke this skill when a workbook request includes raw observations, periods, +qualifiers such as peak/end/average, aggregation tables, or charts that must +update when new rows are added. + +Expected behavior: separate typed period start/end fields from the observation +point, preserve the original label for audit, and wire the table and chart to +formula-driven outputs. diff --git a/product/skills/spreadsheet-authoring/tests/stays_silent_unrelated_table.md b/product/skills/spreadsheet-authoring/tests/stays_silent_unrelated_table.md new file mode 100644 index 00000000..c9cedb54 --- /dev/null +++ b/product/skills/spreadsheet-authoring/tests/stays_silent_unrelated_table.md @@ -0,0 +1,5 @@ +# Stays silent + +Do not invoke this skill for a request to rename a single spreadsheet tab or +change a cell's color when no raw-data model, aggregation, provenance, or chart +relationship is involved. diff --git a/product/skills/spreadsheet-authoring/tests/test_validate_schema.py b/product/skills/spreadsheet-authoring/tests/test_validate_schema.py new file mode 100644 index 00000000..cf6e8ed1 --- /dev/null +++ b/product/skills/spreadsheet-authoring/tests/test_validate_schema.py @@ -0,0 +1,38 @@ +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parents[1] / "scripts")) +from validate_schema import validate_rows + + +def row(**overrides): + result = { + "Original Period Label": "Q4 2023 end", + "Period Start": "2023-10-01", + "Period End": "2023-12-31", + "Observation Point": "End", + "Geography": "US", + "Scope": "Unified", + "Segment": "", + "Entity": "Uber", + "Metric": "Weekly active users", + "Value": "11800000", + } + result.update(overrides) + return result + + +class SchemaTests(unittest.TestCase): + def test_accepts_typed_period_and_distinct_points(self): + errors = validate_rows([row(), row(**{"Original Period Label": "Q4 2023 peak", "Observation Point": "Peak"})]) + self.assertEqual(errors, []) + + def test_rejects_qualifier_in_period_start(self): + errors = validate_rows([row(**{"Period Start": "Q4 2023 end"})]) + self.assertEqual(len(errors), 1) + self.assertIn("qualifier embedded", errors[0]) + + +if __name__ == "__main__": + unittest.main() From c13bc1164d16d9630cfdf0c0321a5229ee95b797 Mon Sep 17 00:00:00 2001 From: Edbert Chan Date: Tue, 29 Sep 2026 11:22:52 +0800 Subject: [PATCH 2/2] Require provenance for spreadsheet derived estimates Change-Id: If2020ac1a6062fb2376a55589408419205058a18 --- product/skills/spreadsheet-authoring/SKILL.md | 48 +++++++++++--- .../scripts/validate_schema.py | 52 +++++++++++++-- .../tests/test_validate_schema.py | 66 +++++++++++++++++++ 3 files changed, 153 insertions(+), 13 deletions(-) diff --git a/product/skills/spreadsheet-authoring/SKILL.md b/product/skills/spreadsheet-authoring/SKILL.md index 896f0856..eb7a6395 100644 --- a/product/skills/spreadsheet-authoring/SKILL.md +++ b/product/skills/spreadsheet-authoring/SKILL.md @@ -21,8 +21,24 @@ Build the workbook as a data pipeline, not as a decorated report: in a separate categorical `Observation Point` column. - Preserve the source wording in `Original Period Label` for audit, but never use that display label as the aggregation key. +- When a source publishes a bounded start/end range but omits an average, a + midpoint may be added only as a derived estimate: retain the source URL, + label `Observation Type` as `derived estimate`, and record the formula in + `Notes`. Never derive a start or end value from a peak, low, or other + non-endpoint observation. - Keep geography, scope, segment, entity, metric, value, unit, source, source - URL, and source location as separate fields. + URL, source location, derivation method, and derived-from lineage as + separate fields. +- Synthetic observations may be added to the category raw sheet only when + they are explicitly marked `Observation Type = derived estimate`. They must + include `Derivation Method`, `Derived From`, a source URL for the underlying + observations, and a formula/method note. The aggregate must support a + published-only view and a published-plus-derived view; charts must make the + selected mode visible. +- The default permitted synthetic methods are `midpoint` from matching Start + and End observations and `linear interpolation` between bracketing periods. + Never synthesize a competitor's missing observation from another entity, or + infer Start/End/Peak/Low from an unrelated observation point. When parsing a range such as `Nov 2024–Sep 2025`, write the first and last dates as typed values. When parsing `Q4 2023 end`, write the quarter dates and @@ -36,14 +52,30 @@ dates as typed values. When parsing `Q4 2023 end`, write the quarter dates and - Let selectors choose the metric and the relevant subset of raw data. - Reserve expandable formula ranges so adding a valid raw row updates the table. - Charts must reference the aggregation table only, never the raw sheet. +- Declare the expected entity inventory for each category and measurement slice. + A table/chart is not complete until every expected entity has a valid, + source-backed observation for each aggregation key, or the raw data carries + an explicit `No observation published` status. Never convert an absent value + into zero or silently treat it as complete. Derived estimates do not satisfy + source-backed entity coverage. ## Preflight and verification -Before editing the live workbook, validate representative period labels and -the migration result. Reject any schema where a time column contains a -qualifier token or where two distinct observation points collapse to one key. -For CSV exports or fixtures, run `scripts/validate_schema.py `. +Before editing the live workbook, validate representative period labels, +source URLs, expected-entity coverage, and the migration result. Reject any +schema where a time column contains a qualifier token, where two distinct +observation points collapse to one key, or where an expected entity is absent +from a source-backed aggregation key. For CSV exports or fixtures, run +`scripts/validate_schema.py --expected-entities A,B,C`. -After writing, reread raw headers, typed dates, observation points, table -formulas, selector behavior, and chart source ranges. Test both a new row and -an invalid or duplicate row when the workbook is intended to scale. +If coverage fails, do not proceed to table or chart generation. Run a bounded +source-collection retry loop up to three times; each attempt must either add +new source-backed raw observations or record an explicit unresolved status. +After the third attempt, stop with `SPREADSHEET_COVERAGE: FAIL` and list the +missing entity/period keys. Never resolve the failure by copying, averaging, +zero-filling, or estimating an unreported competitor value. + +After writing, reread raw headers, typed dates, observation points, derivation +fields, table formulas, selector behavior, and chart source ranges. Test both +a new source-backed row and a derived row, plus invalid, duplicate, and +unprovenanceable rows when the workbook is intended to scale. diff --git a/product/skills/spreadsheet-authoring/scripts/validate_schema.py b/product/skills/spreadsheet-authoring/scripts/validate_schema.py index fd493139..a81cb4aa 100644 --- a/product/skills/spreadsheet-authoring/scripts/validate_schema.py +++ b/product/skills/spreadsheet-authoring/scripts/validate_schema.py @@ -20,12 +20,22 @@ "Entity", "Metric", "Value", + "Source URL", + "Observation Type", + "Derivation Method", + "Derived From", + "Notes", } QUALIFIER = re.compile(r"\b(avg|average|start|peak|end|low|high)\b", re.I) KEY_FIELDS = ["Period Start", "Period End", "Observation Point", "Geography", "Scope", "Segment", "Entity", "Metric"] +DEFAULT_COVERAGE_FIELDS = [field for field in KEY_FIELDS if field != "Entity"] -def validate_rows(rows: list[dict[str, str]]) -> list[str]: +def validate_rows( + rows: list[dict[str, str]], + expected_entities: set[str] | None = None, + coverage_fields: list[str] | None = None, +) -> list[str]: errors: list[str] = [] if not rows: return ["no observations"] @@ -34,6 +44,7 @@ def validate_rows(rows: list[dict[str, str]]) -> list[str]: errors.append("missing columns: " + ", ".join(sorted(missing))) return errors keys: dict[tuple[str, ...], int] = {} + coverage: dict[tuple[str, ...], set[str]] = {} for number, row in enumerate(rows, start=2): if QUALIFIER.search(row["Period Start"]) or QUALIFIER.search(row["Period End"]): errors.append(f"row {number}: qualifier embedded in period column") @@ -42,6 +53,26 @@ def validate_rows(rows: list[dict[str, str]]) -> list[str]: errors.append(f"row {number}: duplicate observation key; first seen at row {keys[key]}") else: keys[key] = number + if not row["Source URL"].strip() and row["Value"].strip(): + errors.append(f"row {number}: numeric observation missing Source URL") + if row["Observation Type"].strip().lower() == "derived estimate": + method = row["Derivation Method"].strip().lower() + if method not in {"midpoint", "linear interpolation"}: + errors.append(f"row {number}: derived estimate has unsupported or missing derivation method") + if not row["Derived From"].strip(): + errors.append(f"row {number}: derived estimate missing Derived From") + if not re.search(r"\b(midpoint|average|formula|interpolat|\+|/\s*2)\b", row["Notes"], re.I): + errors.append(f"row {number}: derived estimate missing derivation note") + if expected_entities and row["Value"].strip() and row["Source URL"].strip() and row["Observation Type"].strip().lower() != "derived estimate": + fields = coverage_fields or DEFAULT_COVERAGE_FIELDS + group = tuple(row[field].strip() for field in fields) + coverage.setdefault(group, set()).add(row["Entity"].strip()) + if expected_entities: + for group, entities in coverage.items(): + missing = sorted(expected_entities - entities) + if missing: + label = ", ".join(f"{field}={value or ''}" for field, value in zip(coverage_fields or DEFAULT_COVERAGE_FIELDS, group)) + errors.append(f"coverage {label}: missing expected entities: {', '.join(missing)}") return errors @@ -51,14 +82,25 @@ def validate_file(path: Path) -> list[str]: def main(argv: list[str]) -> int: - if len(argv) != 2: - print(f"usage: {argv[0]} FILE", file=sys.stderr) + if len(argv) not in (2, 4, 6): + print(f"usage: {argv[0]} FILE [--expected-entities A,B,C] [--coverage-fields A,B]", file=sys.stderr) return 2 - errors = validate_file(Path(argv[1])) + expected = None + fields = None + for index in range(2, len(argv), 2): + if argv[index] == "--expected-entities": + expected = {value.strip() for value in argv[index + 1].split(",") if value.strip()} + elif argv[index] == "--coverage-fields": + fields = [value.strip() for value in argv[index + 1].split(",") if value.strip()] + else: + print(f"unknown option: {argv[index]}", file=sys.stderr) + return 2 + with Path(argv[1]).open(newline="", encoding="utf-8") as handle: + errors = validate_rows(list(csv.DictReader(handle)), expected, fields) if errors: print("\n".join(errors), file=sys.stderr) return 1 - print("ok: typed temporal schema") + print("SPREADSHEET_COVERAGE: PASS" if expected else "ok: typed temporal schema") return 0 diff --git a/product/skills/spreadsheet-authoring/tests/test_validate_schema.py b/product/skills/spreadsheet-authoring/tests/test_validate_schema.py index cf6e8ed1..df0a9a69 100644 --- a/product/skills/spreadsheet-authoring/tests/test_validate_schema.py +++ b/product/skills/spreadsheet-authoring/tests/test_validate_schema.py @@ -18,6 +18,11 @@ def row(**overrides): "Entity": "Uber", "Metric": "Weekly active users", "Value": "11800000", + "Source URL": "https://example.com/source", + "Observation Type": "reported", + "Derivation Method": "", + "Derived From": "", + "Notes": "", } result.update(overrides) return result @@ -33,6 +38,67 @@ def test_rejects_qualifier_in_period_start(self): self.assertEqual(len(errors), 1) self.assertIn("qualifier embedded", errors[0]) + def test_rejects_missing_expected_entity_for_same_observation_key(self): + errors = validate_rows([row()], expected_entities={"Uber", "Lyft"}) + self.assertTrue(any("missing expected entities: Lyft" in error for error in errors)) + + def test_rejects_numeric_observation_without_source_url(self): + errors = validate_rows([row(**{"Source URL": ""})]) + self.assertTrue(any("missing Source URL" in error for error in errors)) + + def test_all_expected_entities_can_be_present_for_one_period(self): + errors = validate_rows( + [row(), row(**{"Entity": "Lyft", "Original Period Label": "Q4 2023 end"})], + expected_entities={"Uber", "Lyft"}, + ) + self.assertEqual(errors, []) + + def test_unreported_status_does_not_satisfy_entity_coverage(self): + errors = validate_rows( + [row(), row(**{"Entity": "Lyft", "Value": "", "Source URL": "https://example.com/source", "Data Status": "NO OBSERVATION PUBLISHED"})], + expected_entities={"Uber", "Lyft"}, + ) + self.assertTrue(any("missing expected entities: Lyft" in error for error in errors)) + + def test_derived_estimate_requires_derivation_note(self): + errors = validate_rows([row(**{"Observation Type": "derived estimate", "Notes": ""})]) + self.assertTrue(any("missing derivation method" in error for error in errors)) + self.assertTrue(any("missing Derived From" in error for error in errors)) + self.assertTrue(any("missing derivation note" in error for error in errors)) + + def test_derived_estimate_is_not_source_coverage(self): + errors = validate_rows( + [ + row(), + row( + **{ + "Entity": "Lyft", + "Observation Type": "derived estimate", + "Derivation Method": "midpoint", + "Derived From": "Start + End", + "Notes": "Midpoint formula: (Start + End) / 2", + } + ), + ], + expected_entities={"Uber", "Lyft"}, + ) + self.assertTrue(any("missing expected entities: Lyft" in error for error in errors)) + + def test_accepts_provenance_preserving_midpoint(self): + errors = validate_rows( + [ + row( + **{ + "Observation Type": "derived estimate", + "Derivation Method": "midpoint", + "Derived From": "Start + End", + "Notes": "Midpoint formula: (Start + End) / 2", + } + ) + ] + ) + self.assertEqual(errors, []) + if __name__ == "__main__": unittest.main()