Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions docs/ecosystem.md
Original file line number Diff line number Diff line change
Expand Up @@ -99,6 +99,7 @@ again.
| `handoff-needs-smoke-test` | hook |
| `hook-freshness` | hook (advisory) |
| `hook-health` | hook (advisory) |
| `skill-usage-log` | hook (metrics only; records each skill use in Claude, Cursor and Codex) |
| `llm-judge` | hook (shared background model judge; its inbox delivers finished verdicts on the next turn: Claude `UserPromptSubmit`, Cursor `stop`, Codex `notify`) |
| `engine/CLAUDE.core.md` | global hand-written Claude rules |
| `scripts/`, `always-on/`, `cursor/rules/` (repo root), root `install.sh` | runtime (engine-owned entrypoints at root for CI) |
Expand Down
69 changes: 69 additions & 0 deletions engine/hooks/_runner/report.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,10 @@

import registry

sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "skill-usage-log"))

import detect as skill_detect

FAILURE_OUTCOMES = {"crashed", "timed_out", "caught_error"}
OUTCOMES = ("spoke", "silent", "blocked", "crashed", "caught_error", "timed_out")

Expand Down Expand Up @@ -543,6 +547,58 @@ def format_judge_table(report: dict[str, Any]) -> str:
return "\n".join(lines) + "\n"


HARNESSES = ("claude", "cursor", "codex")


def build_skill_report(rows: list[dict[str, Any]], malformed: int, warnings: list[str], home: str) -> dict[str, Any]:
installed: dict[str, set[str]] = {}
for harness in HARNESSES:
names = skill_detect.installed_skills(harness, home)
if names is None:
warnings.append(f"unchecked: {harness} skill folders unreadable")
installed[harness] = names or set()
skills: dict[str, dict[str, Any]] = {}

def entry(name: str) -> dict[str, Any]:
return skills.setdefault(name, {"skill": name, **{h: 0 for h in HARNESSES}, "sources": {}, "installed": []})

for harness, names in installed.items():
for name in names:
entry(name)["installed"].append(harness)
unchecked = 0
for row in rows:
if row.get("hook") != "skill-usage-log":
continue
if row.get("action") == "skill_usage_unchecked":
unchecked += 1
continue
if row.get("action") != "skill_used" or not isinstance(row.get("skill"), str):
continue
item = entry(row["skill"])
harness = str(row.get("harness") or "")
if harness in HARNESSES:
item[harness] += 1
source = str(row.get("reason") or "")
item["sources"][source] = item["sources"].get(source, 0) + 1
ordered = sorted(skills.values(), key=lambda item: (-sum(item[h] for h in HARNESSES), item["skill"]))
return {"malformed_rows": malformed, "warnings": warnings, "unchecked_runs": unchecked, "skills": ordered}


def format_skill_table(report: dict[str, Any]) -> str:
lines = list(report["warnings"])
if report["malformed_rows"]:
lines.append(f"skipped {report['malformed_rows']} malformed event row(s)")
if report["unchecked_runs"]:
lines.append(f"unchecked: {report['unchecked_runs']} skill-usage-log run(s) could not read their input")
lines.append("skill " + " ".join(HARNESSES) + " sources")
for item in report["skills"]:
if not any(item[h] for h in HARNESSES):
lines.append(f"{item['skill']} no record")
continue
lines.append(f"{item['skill']} " + " ".join(str(item[h]) for h in HARNESSES) + f" {_counts(item['sources'])}")
return "\n".join(lines) + "\n"


def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--since", default="7d")
Expand All @@ -551,6 +607,7 @@ def main(argv: list[str] | None = None) -> int:
mode.add_argument("--events", action="store_true", default=True)
mode.add_argument("--runs", action="store_true")
mode.add_argument("--judge", action="store_true")
mode.add_argument("--skills", action="store_true")
parser.add_argument("--grace", default="1h")
parser.add_argument("--check", action="store_true")
args = parser.parse_args(argv)
Expand All @@ -560,6 +617,18 @@ def main(argv: list[str] | None = None) -> int:
except ValueError as exc:
print(str(exc), file=sys.stderr)
return 2
if args.skills:
rows, malformed, warnings = read_event_rows(metrics_dir(), datetime.now(timezone.utc) - since)
if rows is None:
for warning in warnings:
print(warning)
return 2
report = build_skill_report(rows, malformed, warnings, os.path.expanduser("~"))
if args.json:
print(json.dumps(report, sort_keys=True))
else:
print(format_skill_table(report), end="")
return 2 if report["warnings"] or report["unchecked_runs"] else 0
if args.judge:
now = datetime.now(timezone.utc)
rows, malformed, warnings = read_event_rows(metrics_dir(), now - since)
Expand Down
26 changes: 26 additions & 0 deletions engine/hooks/_runner/tests/test_report.py
Original file line number Diff line number Diff line change
Expand Up @@ -256,6 +256,32 @@ def test_codex_notify_scripts_count_as_registered_hooks(self) -> None:
self.assertIn("codex llm-judge/codex_notify.py 1 0 1", result.stdout)
self.assertIn("codex auto-pr/codex_notify.py no record", result.stdout)

def test_skills_report_counts_uses_per_harness_and_lists_unused_installed_skills(self) -> None:
for root, name in ((".claude/skills", "diu"), (".claude/skills", "idle"), (".codex/skills", "reflect")):
(self.home / root / name).mkdir(parents=True)
(self.home / root / name / "SKILL.md").write_text("x", encoding="utf-8")
now = datetime.now(timezone.utc).isoformat()

def used(harness: str, skill: str, source: str) -> dict[str, object]:
return {"ts": now, "hook": "skill-usage-log", "harness": harness, "action": "skill_used",
"reason": source, "skill": skill, "rule_id": "", "mode_source": "stage"}

self.write_events([used("claude", "diu", "skill_tool"), used("cursor", "diu", "read"), used("codex", "reflect", "mention")])
result = self.run_report("--skills", "--since", "1d")
self.assertEqual(result.returncode, 0, result.stdout + result.stderr)
lines = result.stdout.splitlines()
self.assertEqual(lines[0], "skill claude cursor codex sources")
self.assertIn("diu 1 1 0 read=1,skill_tool=1", lines)
self.assertIn("reflect 0 0 1 mention=1", lines)
self.assertIn("idle no record", lines)

def test_skills_report_exits_two_when_a_hook_run_could_not_read_its_input(self) -> None:
now = datetime.now(timezone.utc).isoformat()
self.write_events([{"ts": now, "hook": "skill-usage-log", "harness": "claude", "action": "skill_usage_unchecked", "reason": "bad_payload"}])
result = self.run_report("--skills", "--since", "1d")
self.assertEqual(result.returncode, 2)
self.assertIn("unchecked: 1 skill-usage-log run(s) could not read their input", result.stdout)

def test_seeded_rows_include_no_record_and_unregistered(self) -> None:
self.seed_configs()
self.write_rows(
Expand Down
2 changes: 2 additions & 0 deletions engine/hooks/_sdk/events.py
Original file line number Diff line number Diff line change
Expand Up @@ -73,9 +73,11 @@ def write_stage_event(
reason: str,
finding_id: str | None = None,
stderr: TextIO | None = None,
fields: Mapping[str, object] | None = None,
) -> bool:
err = stderr if stderr is not None else sys.stderr
row = _row(hook, harness, {"session_id": session_id}, None, "", "stage", action, 0, finding_id)
row.update(fields or {})
row["reason"] = reason
return _append_rows(hook, [row], err)

Expand Down
3 changes: 1 addition & 2 deletions engine/hooks/hooks.toml
Original file line number Diff line number Diff line change
Expand Up @@ -203,8 +203,7 @@ summary = "Stops two agents writing one temp file."
[hooks.skill-usage-log]
mode = "off"
why_mode = "habit"
summary = "Logs skill use when switched on."
enabled_by = "CATSTACK_SKILL_USAGE_LOG"
summary = "Records each skill use for metrics; never speaks."

[hooks.split-scope]
mode = "warn"
Expand Down
71 changes: 40 additions & 31 deletions engine/hooks/skill-usage-log/README.md
Original file line number Diff line number Diff line change
@@ -1,43 +1,52 @@
# skill-usage-log

PreToolUse hook (`Skill`): appends one JSON line per Skill tool call to a
local log file, so "which skills are actually used" can eventually be
answered with data instead of guessed from `git log` / last-modified dates.
Off by default -- catstack had no invocation-tracking mechanism at all
before this hook (confirmed by a repo-wide search: no hook matched the
`Skill` tool, no telemetry SDK, nothing in `scripts/` or the reflect
tooling counted per-skill firings).
Records one metrics row each time an agent uses a skill, in Claude, Cursor and
Codex. It never speaks to the agent.

Enable with:
## Fires on

| Harness | Event | Counted as |
| --- | --- | --- |
| Claude | `PreToolUse` (`Skill`, `Read`, `Bash`) | `skill_tool` for a Skill call, `read` for a Read of a `SKILL.md`, `shell_read` for `cat`/`sed`/`head`/`tail`/`nl`/`less`/`more`/`bat` on one |
| Claude | `UserPromptSubmit` | `slash` for a prompt that starts with `/<installed skill>` |
| Cursor | `preToolUse` | `read` / `shell_read` as above, including `skills-cursor/` |
| Cursor | `beforeSubmitPrompt` | `slash` |
| Codex | `PreToolUse` | `shell_read`, including a path inside an `exec` code string |
| Codex | `UserPromptSubmit` | `slash`, and `mention` for each `$<installed skill>` |

## Silent on

A `SKILL.md` path that only appears inside a Write, Edit, StrReplace, Task,
Agent, Glob, Grep, search or fetch call; a URL; `rg`, `grep` or `wc` over
skill files; a `/path` or `/word` that is not an installed skill; a skill
named mid-sentence.

## Where rows go

`~/.cache/catstack-hook-metrics/events-<date>.jsonl` (or
`$CATSTACK_HOOK_METRICS_DIR`), as `catstack.hook_event.v1` rows with
`action: skill_used`, `reason: <source>` and `skill: <name>`. Read them with:

```sh
export CATSTACK_SKILL_USAGE_LOG=1
python3 engine/hooks/_runner/report.py --skills --since 7d
```

Log location: `~/.cache/catstack-skill-usage-log/skill-usage.jsonl`
(override the directory with `CATSTACK_SKILL_USAGE_LOG_STATE_DIR`). Each
line: `{"ts": <unix-epoch-seconds>, "skill": "<name>", "args": "<string or
null>", "session_id": "<string or null>", "cwd": "<string or null>"}`.
Every installed skill with no use in the window prints `no record`.

This is intentionally crude: a local append-only file, no rotation, no
aggregation, no query tool. It exists to draw the architectural boundary
first -- one write function, `record_skill_usage()` in
`claude_pretooluse_log.py` -- so a later swap to a real metrics ingester
(PostHog or otherwise) is a change to that one function, not a rewrite of
the hook.
## Fail direction

Claude-only: the `Skill` tool is a Claude Code concept, so this hook is not
linked into Cursor or Codex hook directories.
Open: the hook never blocks or changes a tool call or prompt. Input it cannot
read writes a `skill_usage_unchecked` row and a `catstack-hook-error` line,
so the run counts as `caught_error` and `report.py --skills` exits 2. A skill
folder that exists but cannot be listed marks typed commands unchecked; a
folder that does not exist means no skills are installed there.

Register in `~/.claude/settings.json`:
## Escape hatch

```json
"PreToolUse": [
{ "matcher": "Skill",
"hooks": [ { "type": "command", "command": "python3 $HOME/.claude/hooks/skill-usage-log/claude_pretooluse_log.py", "timeout": 5 } ] }
]
```
`CATSTACK_SKILL_USAGE_LOG=0` turns recording off.

## Tests

Tests: `python3 -m unittest discover -s engine/hooks/skill-usage-log/tests -v`
Fail-open: env var unset, malformed stdin, non-dict payload, missing skill
name, or an unwritable log directory all leave the Skill call unaffected.
```sh
python3 engine/hooks/skill-usage-log/tests/test_hooks.py
```
13 changes: 12 additions & 1 deletion engine/hooks/skill-usage-log/claude.hook.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
"hooks": {
"PreToolUse": [
{
"matcher": "Skill",
"matcher": "Skill|Read|Bash",
"hooks": [
{
"type": "command",
Expand All @@ -11,6 +11,17 @@
}
]
}
],
"UserPromptSubmit": [
{
"hooks": [
{
"type": "command",
"command": "python3 $HOME/.claude/hooks/skill-usage-log/claude_prompt_submit.py",
"timeout": 5
}
]
}
]
}
}
82 changes: 2 additions & 80 deletions engine/hooks/skill-usage-log/claude_pretooluse_log.py
Original file line number Diff line number Diff line change
@@ -1,85 +1,7 @@
#!/usr/bin/env python3
"""Claude Code PreToolUse hook (Skill): crude local usage logging, off by default.

Catstack has no mechanism anywhere that records which skill fires, when, or
how often -- "which skills are stale" can currently only be guessed from
`git log` / last-modified dates on skill files. This hook is the first
slice toward real data: append one JSON line per Skill invocation to a
local log file, gated behind an env var so it costs nothing when unset.

The write path is deliberately a single function, record_skill_usage().
Swapping local-file logging for a real metrics ingester later is a change
to that one function, not to the hook's read/gate/build logic around it.

Never blocks the Skill call and fails open on any error (env var unset,
malformed stdin, unwritable log dir): logging must never break the call.
"""
from __future__ import annotations

import json
import os
import sys
import time

ENABLED_ENV = "CATSTACK_SKILL_USAGE_LOG"
STATE_DIR = os.environ.get(
"CATSTACK_SKILL_USAGE_LOG_STATE_DIR",
os.path.join(os.path.expanduser("~"), ".cache", "catstack-skill-usage-log"),
)
LOG_FILE_NAME = "skill-usage.jsonl"


def enabled() -> bool:
return os.environ.get(ENABLED_ENV, "") == "1"


def _skill_name(tool_input: dict) -> str:
value = tool_input.get("skill")
return value.strip() if isinstance(value, str) and value.strip() else ""


def build_event(payload: dict) -> dict:
tool_input = payload.get("tool_input") or payload.get("toolInput") or {}
if not isinstance(tool_input, dict):
tool_input = {}
args = tool_input.get("args")
return {
"ts": time.time(),
"skill": _skill_name(tool_input),
"args": args if isinstance(args, str) and args.strip() else None,
"session_id": payload.get("session_id") or payload.get("sessionId"),
"cwd": payload.get("cwd"),
}


def log_path() -> str:
return os.path.join(STATE_DIR, LOG_FILE_NAME)


def record_skill_usage(event: dict) -> None:
"""The single write path -- swap this for a real ingester later."""
os.makedirs(STATE_DIR, exist_ok=True)
with open(log_path(), "a") as handle:
handle.write(json.dumps(event, sort_keys=True) + "\n")


def main() -> None:
if not enabled():
return
try:
payload = json.load(sys.stdin)
except (json.JSONDecodeError, ValueError):
return
if not isinstance(payload, dict):
return
event = build_event(payload)
if not event["skill"]:
return
try:
record_skill_usage(event)
except OSError:
return

from log import main

if __name__ == "__main__":
main()
main("claude", "tool", "")
7 changes: 7 additions & 0 deletions engine/hooks/skill-usage-log/claude_prompt_submit.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
#!/usr/bin/env python3
from __future__ import annotations

from log import main

if __name__ == "__main__":
main("claude", "prompt", "")
26 changes: 26 additions & 0 deletions engine/hooks/skill-usage-log/codex.hook.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,26 @@
{
"hooks": {
"PreToolUse": [
{
"hooks": [
{
"type": "command",
"command": "python3 $HOME/.codex/hooks/skill-usage-log/codex_pretooluse.py",
"timeout": 5
}
]
}
],
"UserPromptSubmit": [
{
"hooks": [
{
"type": "command",
"command": "python3 $HOME/.codex/hooks/skill-usage-log/codex_prompt_submit.py",
"timeout": 5
}
]
}
]
}
}
Loading
Loading