diff --git a/docs/ecosystem.md b/docs/ecosystem.md index 88087bac7..845a4f8d7 100644 --- a/docs/ecosystem.md +++ b/docs/ecosystem.md @@ -99,6 +99,7 @@ again. | `handoff-needs-smoke-test` | hook | | `hook-freshness` | hook (advisory) | | `hook-health` | hook (advisory) | +| `skill-usage-log` | hook (metrics only; records each skill use in Claude, Cursor and Codex) | | `llm-judge` | hook (shared background model judge; its inbox delivers finished verdicts on the next turn: Claude `UserPromptSubmit`, Cursor `stop`, Codex `notify`) | | `engine/CLAUDE.core.md` | global hand-written Claude rules | | `scripts/`, `always-on/`, `cursor/rules/` (repo root), root `install.sh` | runtime (engine-owned entrypoints at root for CI) | diff --git a/engine/hooks/_runner/report.py b/engine/hooks/_runner/report.py index 1928df6aa..2f8d01948 100644 --- a/engine/hooks/_runner/report.py +++ b/engine/hooks/_runner/report.py @@ -18,6 +18,10 @@ import registry +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "skill-usage-log")) + +import detect as skill_detect + FAILURE_OUTCOMES = {"crashed", "timed_out", "caught_error"} OUTCOMES = ("spoke", "silent", "blocked", "crashed", "caught_error", "timed_out") @@ -543,6 +547,58 @@ def format_judge_table(report: dict[str, Any]) -> str: return "\n".join(lines) + "\n" +HARNESSES = ("claude", "cursor", "codex") + + +def build_skill_report(rows: list[dict[str, Any]], malformed: int, warnings: list[str], home: str) -> dict[str, Any]: + installed: dict[str, set[str]] = {} + for harness in HARNESSES: + names = skill_detect.installed_skills(harness, home) + if names is None: + warnings.append(f"unchecked: {harness} skill folders unreadable") + installed[harness] = names or set() + skills: dict[str, dict[str, Any]] = {} + + def entry(name: str) -> dict[str, Any]: + return skills.setdefault(name, {"skill": name, **{h: 0 for h in HARNESSES}, "sources": {}, "installed": []}) + + for harness, names in installed.items(): + for name in names: + entry(name)["installed"].append(harness) + unchecked = 0 + for row in rows: + if row.get("hook") != "skill-usage-log": + continue + if row.get("action") == "skill_usage_unchecked": + unchecked += 1 + continue + if row.get("action") != "skill_used" or not isinstance(row.get("skill"), str): + continue + item = entry(row["skill"]) + harness = str(row.get("harness") or "") + if harness in HARNESSES: + item[harness] += 1 + source = str(row.get("reason") or "") + item["sources"][source] = item["sources"].get(source, 0) + 1 + ordered = sorted(skills.values(), key=lambda item: (-sum(item[h] for h in HARNESSES), item["skill"])) + return {"malformed_rows": malformed, "warnings": warnings, "unchecked_runs": unchecked, "skills": ordered} + + +def format_skill_table(report: dict[str, Any]) -> str: + lines = list(report["warnings"]) + if report["malformed_rows"]: + lines.append(f"skipped {report['malformed_rows']} malformed event row(s)") + if report["unchecked_runs"]: + lines.append(f"unchecked: {report['unchecked_runs']} skill-usage-log run(s) could not read their input") + lines.append("skill " + " ".join(HARNESSES) + " sources") + for item in report["skills"]: + if not any(item[h] for h in HARNESSES): + lines.append(f"{item['skill']} no record") + continue + lines.append(f"{item['skill']} " + " ".join(str(item[h]) for h in HARNESSES) + f" {_counts(item['sources'])}") + return "\n".join(lines) + "\n" + + def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser() parser.add_argument("--since", default="7d") @@ -551,6 +607,7 @@ def main(argv: list[str] | None = None) -> int: mode.add_argument("--events", action="store_true", default=True) mode.add_argument("--runs", action="store_true") mode.add_argument("--judge", action="store_true") + mode.add_argument("--skills", action="store_true") parser.add_argument("--grace", default="1h") parser.add_argument("--check", action="store_true") args = parser.parse_args(argv) @@ -560,6 +617,18 @@ def main(argv: list[str] | None = None) -> int: except ValueError as exc: print(str(exc), file=sys.stderr) return 2 + if args.skills: + rows, malformed, warnings = read_event_rows(metrics_dir(), datetime.now(timezone.utc) - since) + if rows is None: + for warning in warnings: + print(warning) + return 2 + report = build_skill_report(rows, malformed, warnings, os.path.expanduser("~")) + if args.json: + print(json.dumps(report, sort_keys=True)) + else: + print(format_skill_table(report), end="") + return 2 if report["warnings"] or report["unchecked_runs"] else 0 if args.judge: now = datetime.now(timezone.utc) rows, malformed, warnings = read_event_rows(metrics_dir(), now - since) diff --git a/engine/hooks/_runner/tests/test_report.py b/engine/hooks/_runner/tests/test_report.py index caf66acdf..035a5866a 100644 --- a/engine/hooks/_runner/tests/test_report.py +++ b/engine/hooks/_runner/tests/test_report.py @@ -256,6 +256,32 @@ def test_codex_notify_scripts_count_as_registered_hooks(self) -> None: self.assertIn("codex llm-judge/codex_notify.py 1 0 1", result.stdout) self.assertIn("codex auto-pr/codex_notify.py no record", result.stdout) + def test_skills_report_counts_uses_per_harness_and_lists_unused_installed_skills(self) -> None: + for root, name in ((".claude/skills", "diu"), (".claude/skills", "idle"), (".codex/skills", "reflect")): + (self.home / root / name).mkdir(parents=True) + (self.home / root / name / "SKILL.md").write_text("x", encoding="utf-8") + now = datetime.now(timezone.utc).isoformat() + + def used(harness: str, skill: str, source: str) -> dict[str, object]: + return {"ts": now, "hook": "skill-usage-log", "harness": harness, "action": "skill_used", + "reason": source, "skill": skill, "rule_id": "", "mode_source": "stage"} + + self.write_events([used("claude", "diu", "skill_tool"), used("cursor", "diu", "read"), used("codex", "reflect", "mention")]) + result = self.run_report("--skills", "--since", "1d") + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + lines = result.stdout.splitlines() + self.assertEqual(lines[0], "skill claude cursor codex sources") + self.assertIn("diu 1 1 0 read=1,skill_tool=1", lines) + self.assertIn("reflect 0 0 1 mention=1", lines) + self.assertIn("idle no record", lines) + + def test_skills_report_exits_two_when_a_hook_run_could_not_read_its_input(self) -> None: + now = datetime.now(timezone.utc).isoformat() + self.write_events([{"ts": now, "hook": "skill-usage-log", "harness": "claude", "action": "skill_usage_unchecked", "reason": "bad_payload"}]) + result = self.run_report("--skills", "--since", "1d") + self.assertEqual(result.returncode, 2) + self.assertIn("unchecked: 1 skill-usage-log run(s) could not read their input", result.stdout) + def test_seeded_rows_include_no_record_and_unregistered(self) -> None: self.seed_configs() self.write_rows( diff --git a/engine/hooks/_sdk/events.py b/engine/hooks/_sdk/events.py index 8ee1a7179..6af0b8490 100644 --- a/engine/hooks/_sdk/events.py +++ b/engine/hooks/_sdk/events.py @@ -73,9 +73,11 @@ def write_stage_event( reason: str, finding_id: str | None = None, stderr: TextIO | None = None, + fields: Mapping[str, object] | None = None, ) -> bool: err = stderr if stderr is not None else sys.stderr row = _row(hook, harness, {"session_id": session_id}, None, "", "stage", action, 0, finding_id) + row.update(fields or {}) row["reason"] = reason return _append_rows(hook, [row], err) diff --git a/engine/hooks/hooks.toml b/engine/hooks/hooks.toml index bc713bd13..ef9611585 100644 --- a/engine/hooks/hooks.toml +++ b/engine/hooks/hooks.toml @@ -203,8 +203,7 @@ summary = "Stops two agents writing one temp file." [hooks.skill-usage-log] mode = "off" why_mode = "habit" -summary = "Logs skill use when switched on." -enabled_by = "CATSTACK_SKILL_USAGE_LOG" +summary = "Records each skill use for metrics; never speaks." [hooks.split-scope] mode = "warn" diff --git a/engine/hooks/skill-usage-log/README.md b/engine/hooks/skill-usage-log/README.md index 0c1116ab1..b07b11240 100644 --- a/engine/hooks/skill-usage-log/README.md +++ b/engine/hooks/skill-usage-log/README.md @@ -1,43 +1,52 @@ # skill-usage-log -PreToolUse hook (`Skill`): appends one JSON line per Skill tool call to a -local log file, so "which skills are actually used" can eventually be -answered with data instead of guessed from `git log` / last-modified dates. -Off by default -- catstack had no invocation-tracking mechanism at all -before this hook (confirmed by a repo-wide search: no hook matched the -`Skill` tool, no telemetry SDK, nothing in `scripts/` or the reflect -tooling counted per-skill firings). +Records one metrics row each time an agent uses a skill, in Claude, Cursor and +Codex. It never speaks to the agent. -Enable with: +## Fires on + +| Harness | Event | Counted as | +| --- | --- | --- | +| Claude | `PreToolUse` (`Skill`, `Read`, `Bash`) | `skill_tool` for a Skill call, `read` for a Read of a `SKILL.md`, `shell_read` for `cat`/`sed`/`head`/`tail`/`nl`/`less`/`more`/`bat` on one | +| Claude | `UserPromptSubmit` | `slash` for a prompt that starts with `/` | +| Cursor | `preToolUse` | `read` / `shell_read` as above, including `skills-cursor/` | +| Cursor | `beforeSubmitPrompt` | `slash` | +| Codex | `PreToolUse` | `shell_read`, including a path inside an `exec` code string | +| Codex | `UserPromptSubmit` | `slash`, and `mention` for each `$` | + +## Silent on + +A `SKILL.md` path that only appears inside a Write, Edit, StrReplace, Task, +Agent, Glob, Grep, search or fetch call; a URL; `rg`, `grep` or `wc` over +skill files; a `/path` or `/word` that is not an installed skill; a skill +named mid-sentence. + +## Where rows go + +`~/.cache/catstack-hook-metrics/events-.jsonl` (or +`$CATSTACK_HOOK_METRICS_DIR`), as `catstack.hook_event.v1` rows with +`action: skill_used`, `reason: ` and `skill: `. Read them with: ```sh -export CATSTACK_SKILL_USAGE_LOG=1 +python3 engine/hooks/_runner/report.py --skills --since 7d ``` -Log location: `~/.cache/catstack-skill-usage-log/skill-usage.jsonl` -(override the directory with `CATSTACK_SKILL_USAGE_LOG_STATE_DIR`). Each -line: `{"ts": , "skill": "", "args": "", "session_id": "", "cwd": ""}`. +Every installed skill with no use in the window prints `no record`. -This is intentionally crude: a local append-only file, no rotation, no -aggregation, no query tool. It exists to draw the architectural boundary -first -- one write function, `record_skill_usage()` in -`claude_pretooluse_log.py` -- so a later swap to a real metrics ingester -(PostHog or otherwise) is a change to that one function, not a rewrite of -the hook. +## Fail direction -Claude-only: the `Skill` tool is a Claude Code concept, so this hook is not -linked into Cursor or Codex hook directories. +Open: the hook never blocks or changes a tool call or prompt. Input it cannot +read writes a `skill_usage_unchecked` row and a `catstack-hook-error` line, +so the run counts as `caught_error` and `report.py --skills` exits 2. A skill +folder that exists but cannot be listed marks typed commands unchecked; a +folder that does not exist means no skills are installed there. -Register in `~/.claude/settings.json`: +## Escape hatch -```json -"PreToolUse": [ - { "matcher": "Skill", - "hooks": [ { "type": "command", "command": "python3 $HOME/.claude/hooks/skill-usage-log/claude_pretooluse_log.py", "timeout": 5 } ] } -] -``` +`CATSTACK_SKILL_USAGE_LOG=0` turns recording off. + +## Tests -Tests: `python3 -m unittest discover -s engine/hooks/skill-usage-log/tests -v` -Fail-open: env var unset, malformed stdin, non-dict payload, missing skill -name, or an unwritable log directory all leave the Skill call unaffected. +```sh +python3 engine/hooks/skill-usage-log/tests/test_hooks.py +``` diff --git a/engine/hooks/skill-usage-log/claude.hook.json b/engine/hooks/skill-usage-log/claude.hook.json index e61e4ffa1..31d5e9704 100644 --- a/engine/hooks/skill-usage-log/claude.hook.json +++ b/engine/hooks/skill-usage-log/claude.hook.json @@ -2,7 +2,7 @@ "hooks": { "PreToolUse": [ { - "matcher": "Skill", + "matcher": "Skill|Read|Bash", "hooks": [ { "type": "command", @@ -11,6 +11,17 @@ } ] } + ], + "UserPromptSubmit": [ + { + "hooks": [ + { + "type": "command", + "command": "python3 $HOME/.claude/hooks/skill-usage-log/claude_prompt_submit.py", + "timeout": 5 + } + ] + } ] } } diff --git a/engine/hooks/skill-usage-log/claude_pretooluse_log.py b/engine/hooks/skill-usage-log/claude_pretooluse_log.py index c851f94eb..41d99653e 100644 --- a/engine/hooks/skill-usage-log/claude_pretooluse_log.py +++ b/engine/hooks/skill-usage-log/claude_pretooluse_log.py @@ -1,85 +1,7 @@ #!/usr/bin/env python3 -"""Claude Code PreToolUse hook (Skill): crude local usage logging, off by default. - -Catstack has no mechanism anywhere that records which skill fires, when, or -how often -- "which skills are stale" can currently only be guessed from -`git log` / last-modified dates on skill files. This hook is the first -slice toward real data: append one JSON line per Skill invocation to a -local log file, gated behind an env var so it costs nothing when unset. - -The write path is deliberately a single function, record_skill_usage(). -Swapping local-file logging for a real metrics ingester later is a change -to that one function, not to the hook's read/gate/build logic around it. - -Never blocks the Skill call and fails open on any error (env var unset, -malformed stdin, unwritable log dir): logging must never break the call. -""" from __future__ import annotations -import json -import os -import sys -import time - -ENABLED_ENV = "CATSTACK_SKILL_USAGE_LOG" -STATE_DIR = os.environ.get( - "CATSTACK_SKILL_USAGE_LOG_STATE_DIR", - os.path.join(os.path.expanduser("~"), ".cache", "catstack-skill-usage-log"), -) -LOG_FILE_NAME = "skill-usage.jsonl" - - -def enabled() -> bool: - return os.environ.get(ENABLED_ENV, "") == "1" - - -def _skill_name(tool_input: dict) -> str: - value = tool_input.get("skill") - return value.strip() if isinstance(value, str) and value.strip() else "" - - -def build_event(payload: dict) -> dict: - tool_input = payload.get("tool_input") or payload.get("toolInput") or {} - if not isinstance(tool_input, dict): - tool_input = {} - args = tool_input.get("args") - return { - "ts": time.time(), - "skill": _skill_name(tool_input), - "args": args if isinstance(args, str) and args.strip() else None, - "session_id": payload.get("session_id") or payload.get("sessionId"), - "cwd": payload.get("cwd"), - } - - -def log_path() -> str: - return os.path.join(STATE_DIR, LOG_FILE_NAME) - - -def record_skill_usage(event: dict) -> None: - """The single write path -- swap this for a real ingester later.""" - os.makedirs(STATE_DIR, exist_ok=True) - with open(log_path(), "a") as handle: - handle.write(json.dumps(event, sort_keys=True) + "\n") - - -def main() -> None: - if not enabled(): - return - try: - payload = json.load(sys.stdin) - except (json.JSONDecodeError, ValueError): - return - if not isinstance(payload, dict): - return - event = build_event(payload) - if not event["skill"]: - return - try: - record_skill_usage(event) - except OSError: - return - +from log import main if __name__ == "__main__": - main() + main("claude", "tool", "") diff --git a/engine/hooks/skill-usage-log/claude_prompt_submit.py b/engine/hooks/skill-usage-log/claude_prompt_submit.py new file mode 100755 index 000000000..8e45141f9 --- /dev/null +++ b/engine/hooks/skill-usage-log/claude_prompt_submit.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +from log import main + +if __name__ == "__main__": + main("claude", "prompt", "") diff --git a/engine/hooks/skill-usage-log/codex.hook.json b/engine/hooks/skill-usage-log/codex.hook.json new file mode 100644 index 000000000..c812b739d --- /dev/null +++ b/engine/hooks/skill-usage-log/codex.hook.json @@ -0,0 +1,26 @@ +{ + "hooks": { + "PreToolUse": [ + { + "hooks": [ + { + "type": "command", + "command": "python3 $HOME/.codex/hooks/skill-usage-log/codex_pretooluse.py", + "timeout": 5 + } + ] + } + ], + "UserPromptSubmit": [ + { + "hooks": [ + { + "type": "command", + "command": "python3 $HOME/.codex/hooks/skill-usage-log/codex_prompt_submit.py", + "timeout": 5 + } + ] + } + ] + } +} diff --git a/engine/hooks/skill-usage-log/codex_pretooluse.py b/engine/hooks/skill-usage-log/codex_pretooluse.py new file mode 100755 index 000000000..58f36be00 --- /dev/null +++ b/engine/hooks/skill-usage-log/codex_pretooluse.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +from log import main + +if __name__ == "__main__": + main("codex", "tool", "") diff --git a/engine/hooks/skill-usage-log/codex_prompt_submit.py b/engine/hooks/skill-usage-log/codex_prompt_submit.py new file mode 100755 index 000000000..6ec58fb96 --- /dev/null +++ b/engine/hooks/skill-usage-log/codex_prompt_submit.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +from log import main + +if __name__ == "__main__": + main("codex", "prompt", "") diff --git a/engine/hooks/skill-usage-log/cursor.hook.json b/engine/hooks/skill-usage-log/cursor.hook.json new file mode 100644 index 000000000..d2d5e249d --- /dev/null +++ b/engine/hooks/skill-usage-log/cursor.hook.json @@ -0,0 +1,16 @@ +{ + "hooks": { + "preToolUse": [ + { + "command": "python3 $HOME/.cursor/hooks/skill-usage-log/cursor_pretooluse.py", + "timeout": 5 + } + ], + "beforeSubmitPrompt": [ + { + "command": "python3 $HOME/.cursor/hooks/skill-usage-log/cursor_before_submit.py", + "timeout": 5 + } + ] + } +} diff --git a/engine/hooks/skill-usage-log/cursor_before_submit.py b/engine/hooks/skill-usage-log/cursor_before_submit.py new file mode 100755 index 000000000..499a6084e --- /dev/null +++ b/engine/hooks/skill-usage-log/cursor_before_submit.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +from log import main + +if __name__ == "__main__": + main("cursor", "prompt", '{"continue": true}\n') diff --git a/engine/hooks/skill-usage-log/cursor_pretooluse.py b/engine/hooks/skill-usage-log/cursor_pretooluse.py new file mode 100755 index 000000000..98cfde75e --- /dev/null +++ b/engine/hooks/skill-usage-log/cursor_pretooluse.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +from log import main + +if __name__ == "__main__": + main("cursor", "tool", '{"continue": true}\n') diff --git a/engine/hooks/skill-usage-log/detect.py b/engine/hooks/skill-usage-log/detect.py new file mode 100644 index 000000000..d9b71e8a4 --- /dev/null +++ b/engine/hooks/skill-usage-log/detect.py @@ -0,0 +1,80 @@ +from __future__ import annotations + +import os +import re + +SKILL_PATH = r"skills(?:-cursor)?/([A-Za-z0-9][A-Za-z0-9_.-]*)/SKILL\.md" +WHOLE_PATH_RE = re.compile(r"^[^\s:]*" + SKILL_PATH + r"$") +SHELL_READ_RE = re.compile(r"\b(?:cat|sed|head|tail|nl|less|more|bat)\b[^;|&\n]*?" + SKILL_PATH) +NOT_A_USE_TOOL_RE = re.compile(r"write|edit|replace|task|agent|todo|plan|glob|grep|fetch|search", re.IGNORECASE) +SLASH_RE = re.compile(r"^\s*/([A-Za-z0-9][A-Za-z0-9_.:-]*)") +MENTION_RE = re.compile(r"(?:^|\s)\$([A-Za-z0-9][A-Za-z0-9_.:-]*)") +PROMPT_KEYS = ("prompt", "user_prompt", "userPrompt", "message") +TOOL_INPUT_KEYS = ("tool_input", "toolInput", "arguments") +SKILL_ROOTS = { + "claude": (".claude/skills",), + "cursor": (".cursor/skills", ".cursor/skills-cursor"), + "codex": (".codex/skills", ".agents/skills"), +} + + +def installed_skills(harness: str, home: str | None = None) -> set[str] | None: + base = home or os.path.expanduser("~") + names: set[str] = set() + for relative in SKILL_ROOTS.get(harness, ()): + root = os.path.join(base, relative) + try: + entries = os.listdir(root) + except FileNotFoundError: + continue + except OSError: + return None + names.update(name for name in entries if os.path.isfile(os.path.join(root, name, "SKILL.md"))) + return names + + +def _strings(node: object) -> list[str]: + if isinstance(node, str): + return [node] + if isinstance(node, dict): + return [text for value in node.values() for text in _strings(value)] + if isinstance(node, list): + return [text for value in node for text in _strings(value)] + return [] + + +def tool_uses(payload: dict) -> list[tuple[str, str]]: + name = str(payload.get("tool_name") or payload.get("toolName") or payload.get("tool") or "") + tool_input = next((payload[key] for key in TOOL_INPUT_KEYS if payload.get(key) is not None), {}) + if name == "Skill" and isinstance(tool_input, dict): + skill = tool_input.get("skill") + return [(skill.strip(), "skill_tool")] if isinstance(skill, str) and skill.strip() else [] + if NOT_A_USE_TOOL_RE.search(name): + return [] + found: list[tuple[str, str]] = [] + for text in _strings(tool_input): + whole = WHOLE_PATH_RE.match(text.strip()) + if whole: + found.append((whole.group(1), "read")) + continue + found.extend((skill, "shell_read") for skill in SHELL_READ_RE.findall(text)) + return list(dict.fromkeys(found)) + + +def prompt_text(payload: dict) -> str: + for key in PROMPT_KEYS: + value = payload.get(key) + if isinstance(value, str): + return value + return "" + + +def prompt_uses(payload: dict, harness: str, installed: set[str]) -> list[tuple[str, str]]: + text = prompt_text(payload) + found: list[tuple[str, str]] = [] + slash = SLASH_RE.match(text) + if slash and slash.group(1) in installed: + found.append((slash.group(1), "slash")) + if harness == "codex": + found.extend((name, "mention") for name in MENTION_RE.findall(text) if name in installed) + return list(dict.fromkeys(found)) diff --git a/engine/hooks/skill-usage-log/install_claude_hook.py b/engine/hooks/skill-usage-log/install_claude_hook.py index 8a505f39d..ba25e9025 100644 --- a/engine/hooks/skill-usage-log/install_claude_hook.py +++ b/engine/hooks/skill-usage-log/install_claude_hook.py @@ -1,46 +1,7 @@ #!/usr/bin/env python3 -"""Merge skill-usage-log into ~/.claude/settings.json PreToolUse hooks. Idempotent.""" from __future__ import annotations -import json -import os - -HERE = os.path.dirname(os.path.abspath(__file__)) -SETTINGS_PATH = os.path.expanduser("~/.claude/settings.json") -FRAGMENT_PATH = os.path.join(HERE, "claude.hook.json") -MARKER = "skill-usage-log/claude_pretooluse_log.py" -EVENT = "PreToolUse" - - -def _is_ours(entry: dict) -> bool: - return any(MARKER in h.get("command", "") for h in entry.get("hooks", [])) - - -def merge_hook(settings: dict, fragment: dict) -> bool: - entry_list = settings.setdefault("hooks", {}).setdefault(EVENT, []) - new_entries = fragment.get("hooks", {}).get(EVENT, []) - before = json.dumps(entry_list, sort_keys=True) - kept = [e for e in entry_list if not _is_ours(e)] - entry_list[:] = kept + new_entries - return json.dumps(entry_list, sort_keys=True) != before - - -def main() -> None: - settings: dict = {} - if os.path.exists(SETTINGS_PATH): - with open(SETTINGS_PATH) as handle: - settings = json.load(handle) - with open(FRAGMENT_PATH) as handle: - fragment = json.load(handle) - if not merge_hook(settings, fragment): - print("ok claude PreToolUse skill-usage-log already up to date") - return - os.makedirs(os.path.dirname(SETTINGS_PATH), exist_ok=True) - with open(SETTINGS_PATH, "w") as handle: - json.dump(settings, handle, indent=2) - handle.write("\n") - print("added claude PreToolUse skill-usage-log") - +from install_common import install if __name__ == "__main__": - main() + install("claude", "~/.claude/settings.json", {}) diff --git a/engine/hooks/skill-usage-log/install_codex_hook.py b/engine/hooks/skill-usage-log/install_codex_hook.py new file mode 100644 index 000000000..6527f4369 --- /dev/null +++ b/engine/hooks/skill-usage-log/install_codex_hook.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +from install_common import install + +if __name__ == "__main__": + install("codex", "~/.codex/hooks.json", {}) diff --git a/engine/hooks/skill-usage-log/install_common.py b/engine/hooks/skill-usage-log/install_common.py new file mode 100644 index 000000000..a63a49b68 --- /dev/null +++ b/engine/hooks/skill-usage-log/install_common.py @@ -0,0 +1,41 @@ +from __future__ import annotations + +import copy +import json +import os + +HERE = os.path.dirname(os.path.abspath(__file__)) +MARKER = "skill-usage-log/" + + +def merge(config: dict, fragment: dict) -> dict: + result = copy.deepcopy(config) + hooks = result.get("hooks") + if not isinstance(hooks, dict): + hooks = {} + result["hooks"] = hooks + for event, incoming in fragment["hooks"].items(): + existing = hooks.get(event, []) + hooks[event] = [entry for entry in existing if MARKER not in json.dumps(entry)] + incoming + return result + + +def install(harness: str, config_path: str, default: dict) -> None: + path = os.path.expanduser(config_path) + config = copy.deepcopy(default) + if os.path.exists(path): + with open(path, encoding="utf-8") as handle: + config = json.load(handle) + with open(os.path.join(HERE, f"{harness}.hook.json"), encoding="utf-8") as handle: + fragment = json.load(handle) + merged = merge(config, fragment) + if merged == config and not os.path.islink(path): + print(f"ok {harness} skill-usage-log already up to date") + return + if os.path.islink(path): + os.unlink(path) + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "w", encoding="utf-8") as handle: + json.dump(merged, handle, indent=2) + handle.write("\n") + print(f"link {harness} skill-usage-log merged into {path}") diff --git a/engine/hooks/skill-usage-log/install_cursor_hook.py b/engine/hooks/skill-usage-log/install_cursor_hook.py new file mode 100644 index 000000000..ced5e8e3f --- /dev/null +++ b/engine/hooks/skill-usage-log/install_cursor_hook.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +from install_common import install + +if __name__ == "__main__": + install("cursor", "~/.cursor/hooks.json", {"version": 1, "hooks": {}}) diff --git a/engine/hooks/skill-usage-log/log.py b/engine/hooks/skill-usage-log/log.py new file mode 100644 index 000000000..a711115a1 --- /dev/null +++ b/engine/hooks/skill-usage-log/log.py @@ -0,0 +1,51 @@ +from __future__ import annotations + +import json +import os +import sys + +sys.path.insert(0, os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "_sdk")) + +from detect import installed_skills, prompt_uses, tool_uses # noqa: E402 +from events import write_stage_event # noqa: E402 + +HOOK = "skill-usage-log" +OPT_OUT_ENV = "CATSTACK_SKILL_USAGE_LOG" + + +def session_id(payload: dict) -> str: + for key in ("session_id", "sessionId", "conversation_id", "conversationId"): + value = payload.get(key) + if isinstance(value, str) and value: + return value + return "" + + +def record(harness: str, kind: str, payload: dict) -> list[tuple[str, str]]: + if kind == "prompt": + installed = installed_skills(harness) + if installed is None: + print(f"catstack-hook-error {HOOK}: {harness} skill folders unreadable; typed skill commands unchecked", file=sys.stderr) + write_stage_event(HOOK, harness, session_id(payload), "skill_usage_unchecked", "skills_unreadable") + return [] + uses = prompt_uses(payload, harness, installed) + else: + uses = tool_uses(payload) + for skill, source in uses: + write_stage_event(HOOK, harness, session_id(payload), "skill_used", source, fields={"skill": skill}) + return uses + + +def main(harness: str, kind: str, allow_output: str = "") -> None: + if os.environ.get(OPT_OUT_ENV) == "0": + print(allow_output, end="") + return + try: + payload = json.load(sys.stdin) + if not isinstance(payload, dict): + raise ValueError(f"payload is a JSON {type(payload).__name__}, not an object") + record(harness, kind, payload) + except Exception as exc: + print(f"catstack-hook-error {HOOK}: {type(exc).__name__}: {exc}", file=sys.stderr) + write_stage_event(HOOK, harness, "", "skill_usage_unchecked", "bad_payload") + print(allow_output, end="") diff --git a/engine/hooks/skill-usage-log/tests/test_hooks.py b/engine/hooks/skill-usage-log/tests/test_hooks.py index 27cd4c2dc..c5bf6066c 100644 --- a/engine/hooks/skill-usage-log/tests/test_hooks.py +++ b/engine/hooks/skill-usage-log/tests/test_hooks.py @@ -1,110 +1,177 @@ #!/usr/bin/env python3 -"""Unit tests for the skill-usage-log PreToolUse hook. +"""Tests for skill-usage-log. Run: python3 -m unittest discover -s engine/hooks/skill-usage-log/tests -v """ import io import json import os +import subprocess import sys import tempfile import unittest +from contextlib import redirect_stderr, redirect_stdout from unittest.mock import patch HOOKS_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) sys.path.insert(0, HOOKS_DIR) -import claude_pretooluse_log # noqa: E402 - - -def run_hook(payload, enabled=True, state_dir=None): - with patch.object(claude_pretooluse_log, "STATE_DIR", state_dir or ""): - with patch.dict( - os.environ, - {claude_pretooluse_log.ENABLED_ENV: "1" if enabled else "0"}, - ): - with patch.object(sys, "stdin", io.StringIO(json.dumps(payload))): - claude_pretooluse_log.main() - - -def read_lines(state_dir): - path = os.path.join(state_dir, claude_pretooluse_log.LOG_FILE_NAME) - if not os.path.exists(path): - return [] - with open(path) as handle: - return [json.loads(line) for line in handle if line.strip()] - - -class TestSkillUsageLog(unittest.TestCase): - def test_enabled_skill_call_is_logged(self): - with tempfile.TemporaryDirectory() as tmp: - run_hook( - { - "tool_name": "Skill", - "tool_input": {"skill": "draft-pr", "args": "foo"}, - "session_id": "sess-1", - "cwd": "/repo", - }, - enabled=True, - state_dir=tmp, - ) - lines = read_lines(tmp) - self.assertEqual(len(lines), 1) - self.assertEqual(lines[0]["skill"], "draft-pr") - self.assertEqual(lines[0]["args"], "foo") - self.assertEqual(lines[0]["session_id"], "sess-1") - self.assertEqual(lines[0]["cwd"], "/repo") - self.assertIn("ts", lines[0]) - - def test_disabled_flag_writes_nothing(self): - with tempfile.TemporaryDirectory() as tmp: - run_hook( - {"tool_name": "Skill", "tool_input": {"skill": "draft-pr"}}, - enabled=False, - state_dir=tmp, - ) - self.assertEqual(read_lines(tmp), []) - self.assertFalse( - os.path.exists(os.path.join(tmp, claude_pretooluse_log.LOG_FILE_NAME)) - ) - - def test_missing_skill_name_writes_nothing(self): - with tempfile.TemporaryDirectory() as tmp: - run_hook( - {"tool_name": "Skill", "tool_input": {}}, - enabled=True, - state_dir=tmp, - ) - self.assertEqual(read_lines(tmp), []) - - def test_malformed_stdin_json_does_not_crash(self): - with tempfile.TemporaryDirectory() as tmp: - with patch.object(claude_pretooluse_log, "STATE_DIR", tmp): - with patch.dict(os.environ, {claude_pretooluse_log.ENABLED_ENV: "1"}): - with patch.object(sys, "stdin", io.StringIO("not json")): - claude_pretooluse_log.main() - self.assertEqual(read_lines(tmp), []) - - def test_non_dict_payload_is_ignored(self): - with tempfile.TemporaryDirectory() as tmp: - run_hook(["not", "a", "dict"], enabled=True, state_dir=tmp) - self.assertEqual(read_lines(tmp), []) - - def test_two_calls_append_two_lines(self): - with tempfile.TemporaryDirectory() as tmp: - run_hook({"tool_input": {"skill": "a"}}, enabled=True, state_dir=tmp) - run_hook({"tool_input": {"skill": "b"}}, enabled=True, state_dir=tmp) - lines = read_lines(tmp) - self.assertEqual([entry["skill"] for entry in lines], ["a", "b"]) - - def test_unwritable_state_dir_fails_open(self): - with patch.object(claude_pretooluse_log, "STATE_DIR", "/nonexistent-root/nope"): - with patch("os.makedirs", side_effect=OSError("no permission")): - with patch.object( - sys, "stdin", io.StringIO(json.dumps({"tool_input": {"skill": "x"}})) - ): - with patch.dict(os.environ, {claude_pretooluse_log.ENABLED_ENV: "1"}): - claude_pretooluse_log.main() +import detect # noqa: E402 +import log # noqa: E402 + +HOME = "/Users/someone" +CODEX_EXEC = ( + "const r = await tools.exec_command({cmd:\"sed -n '1,240p' " + f"{HOME}/.codex/skills/invoker-make-pr/SKILL.md && pwd\"}});" +) + + +def tool(name, tool_input): + return {"hook_event_name": "PreToolUse", "session_id": "s-1", "tool_name": name, "tool_input": tool_input} + + +class TestToolUses(unittest.TestCase): + def test_detects_real_shapes_from_each_harness_as_uses(self): + cases = [ + (tool("Skill", {"skill": "invoker-chat-submit"}), [("invoker-chat-submit", "skill_tool")]), + (tool("Bash", {"command": "cat engine/skills/make-pr/SKILL.md 2>/dev/null | head -150"}), [("make-pr", "shell_read")]), + (tool("Read", {"path": f"{HOME}/.claude/skills/reflect/SKILL.md"}), [("reflect", "read")]), + (tool("ReadFile", {"path": f"{HOME}/.cursor/skills-cursor/canvas/SKILL.md"}), [("canvas", "read")]), + (tool("Read", {"file_path": f"{HOME}/.claude/skills/diu/SKILL.md"}), [("diu", "read")]), + (tool("exec", CODEX_EXEC), [("invoker-make-pr", "shell_read")]), + ] + for payload, expected in cases: + with self.subTest(tool=payload["tool_name"]): + self.assertEqual(detect.tool_uses(payload), expected) + + def test_mentions_that_are_not_uses_are_ignored(self): + cases = [ + tool("Glob", {"glob_pattern": "**/cat-mode/SKILL.md", "target_directory": HOME}), + tool("Grep", {"path": f"{HOME}/.claude/skills", "pattern": "x", "glob": "**/SKILL.md"}), + tool("Write", {"file_path": "/tmp/a.py", "content": f"open('{HOME}/.claude/skills/diu/SKILL.md')"}), + tool("StrReplace", {"path": f"{HOME}/.claude/skills/diu/SKILL.md", "new_string": "x"}), + tool("Task", {"prompt": f"cat {HOME}/.claude/skills/reflect/SKILL.md and follow it"}), + tool("Agent", {"prompt": f"read {HOME}/.claude/skills/reflect/SKILL.md"}), + tool("WebFetch", {"url": "https://raw.githubusercontent.com/o/r/main/.cursor/skills/verify-atlas/SKILL.md"}), + tool("Bash", {"command": "rg -n 'skill' skills/ -g '*.md' | head"}), + tool("Bash", {"command": "wc -l engine/skills/draft-pr/SKILL.md"}), + tool("Read", {"file_path": f"{HOME}/notes/skills.md"}), + ] + for payload in cases: + with self.subTest(tool=payload["tool_name"], data=json.dumps(payload["tool_input"])[:60]): + self.assertEqual(detect.tool_uses(payload), []) + + +class TestPromptUses(unittest.TestCase): + INSTALLED = {"cat-mode", "diu", "reflect"} + + def test_a_typed_command_at_the_start_counts(self): + self.assertEqual(detect.prompt_uses({"prompt": "/cat-mode and add metrics"}, "claude", self.INSTALLED), [("cat-mode", "slash")]) + self.assertEqual(detect.prompt_uses({"prompt": "/diu"}, "cursor", self.INSTALLED), [("diu", "slash")]) + + def test_codex_dollar_mentions_count_only_on_codex(self): + self.assertEqual(detect.prompt_uses({"prompt": "fix it $reflect"}, "codex", self.INSTALLED), [("reflect", "mention")]) + self.assertEqual(detect.prompt_uses({"prompt": "fix it $reflect"}, "claude", self.INSTALLED), []) + + def test_paths_unknown_names_and_mid_sentence_mentions_do_not_count(self): + for prompt in ("/tmp/x.log is empty", "/nosuchskill go", "the /cat-mode skill says so", "costs $5"): + with self.subTest(prompt=prompt): + self.assertEqual(detect.prompt_uses({"prompt": prompt}, "codex", self.INSTALLED), []) + + def test_installed_skills_reads_every_root_for_the_harness(self): + with tempfile.TemporaryDirectory() as home: + for root, name in ((".cursor/skills", "a"), (".cursor/skills-cursor", "b"), (".cursor/skills", "no-skill-md")): + os.makedirs(os.path.join(home, root, name)) + for root, name in ((".cursor/skills", "a"), (".cursor/skills-cursor", "b")): + open(os.path.join(home, root, name, "SKILL.md"), "w").close() + self.assertEqual(detect.installed_skills("cursor", home), {"a", "b"}) + self.assertEqual(detect.installed_skills("claude", home), set()) + + def test_an_unreadable_skill_folder_is_unchecked_not_empty(self): + with patch.object(detect.os, "listdir", side_effect=PermissionError("denied")): + self.assertIsNone(detect.installed_skills("claude", "/anywhere")) + + +class TestLog(unittest.TestCase): + def setUp(self): + self.metrics = tempfile.TemporaryDirectory() + self.addCleanup(self.metrics.cleanup) + env = patch.dict(os.environ, {"CATSTACK_HOOK_METRICS_DIR": self.metrics.name}) + env.start() + self.addCleanup(env.stop) + os.environ.pop(log.OPT_OUT_ENV, None) + + def rows(self): + rows = [] + for name in sorted(os.listdir(self.metrics.name)): + if name.startswith("events-"): + with open(os.path.join(self.metrics.name, name), encoding="utf-8") as handle: + rows.extend(json.loads(line) for line in handle) + return rows + + def run_main(self, harness, kind, stdin, allow=""): + out, err = io.StringIO(), io.StringIO() + with patch.object(sys, "stdin", io.StringIO(stdin)), redirect_stdout(out), redirect_stderr(err): + log.main(harness, kind, allow) + return out.getvalue(), err.getvalue() + + def test_a_use_writes_one_row_with_skill_source_and_harness(self): + out, err = self.run_main("claude", "tool", json.dumps(tool("Skill", {"skill": "diu"}))) + self.assertEqual((out, err), ("", "")) + [row] = self.rows() + self.assertEqual( + {k: row[k] for k in ("hook", "harness", "session_id", "action", "reason", "skill")}, + {"hook": "skill-usage-log", "harness": "claude", "session_id": "s-1", "action": "skill_used", "reason": "skill_tool", "skill": "diu"}, + ) + + def test_a_tool_call_that_uses_no_skill_writes_nothing(self): + self.run_main("claude", "tool", json.dumps(tool("Bash", {"command": "ls"}))) + self.assertEqual(self.rows(), []) + + def test_opt_out_writes_nothing_and_still_allows_cursor(self): + with patch.dict(os.environ, {log.OPT_OUT_ENV: "0"}): + out, _ = self.run_main("cursor", "tool", json.dumps(tool("Skill", {"skill": "diu"})), '{"continue": true}\n') + self.assertEqual(out, '{"continue": true}\n') + self.assertEqual(self.rows(), []) + + def test_unreadable_payload_is_recorded_as_unchecked_and_reported(self): + for stdin in ("{not json", "[1, 2]"): + with self.subTest(stdin=stdin): + out, err = self.run_main("cursor", "tool", stdin, '{"continue": true}\n') + self.assertEqual(out, '{"continue": true}\n') + self.assertTrue(err.startswith("catstack-hook-error skill-usage-log: ")) + self.assertEqual([(r["action"], r["reason"]) for r in self.rows()], [("skill_usage_unchecked", "bad_payload")] * 2) + + def test_unreadable_skill_folders_leave_typed_commands_unchecked(self): + with patch.object(log, "installed_skills", return_value=None): + _, err = self.run_main("claude", "prompt", json.dumps({"prompt": "/diu", "session_id": "s-2"})) + self.assertIn("typed skill commands unchecked", err) + self.assertEqual([(r["action"], r["reason"]) for r in self.rows()], [("skill_usage_unchecked", "skills_unreadable")]) + + def test_every_entry_script_runs_as_a_real_process(self): + home = tempfile.TemporaryDirectory() + self.addCleanup(home.cleanup) + os.makedirs(os.path.join(home.name, ".codex", "skills", "reflect")) + open(os.path.join(home.name, ".codex", "skills", "reflect", "SKILL.md"), "w").close() + cases = [ + ("claude_pretooluse_log.py", tool("Skill", {"skill": "diu"}), ""), + ("codex_pretooluse.py", tool("exec", CODEX_EXEC), ""), + ("codex_prompt_submit.py", {"prompt": "go $reflect", "session_id": "s-3"}, ""), + ("cursor_pretooluse.py", tool("Read", {"path": f"{HOME}/.claude/skills/reflect/SKILL.md"}), '{"continue": true}\n'), + ("cursor_before_submit.py", {"prompt": "hi"}, '{"continue": true}\n'), + ("claude_prompt_submit.py", {"prompt": "hi"}, ""), + ] + env = {**os.environ, "HOME": home.name, "CATSTACK_HOOK_METRICS_DIR": self.metrics.name} + for script, payload, expected in cases: + with self.subTest(script=script): + proc = subprocess.run([sys.executable, os.path.join(HOOKS_DIR, script)], input=json.dumps(payload), + capture_output=True, text=True, env=env, timeout=10) + self.assertEqual((proc.returncode, proc.stdout, proc.stderr), (0, expected, "")) + self.assertEqual( + [(r["harness"], r["skill"], r["reason"]) for r in self.rows()], + [("claude", "diu", "skill_tool"), ("codex", "invoker-make-pr", "shell_read"), + ("codex", "reflect", "mention"), ("cursor", "reflect", "read")], + ) if __name__ == "__main__": diff --git a/install.sh b/install.sh index cc4195ba2..6f8c0770f 100755 --- a/install.sh +++ b/install.sh @@ -317,6 +317,7 @@ link_item "pr-schema-gate" "$REPO_DIR/engine/hooks/pr-schema-gate" "$HOME/.curso link_item "wrong-check-reflect" "$REPO_DIR/engine/hooks/wrong-check-reflect" "$HOME/.cursor/hooks/wrong-check-reflect" link_item "llm-judge" "$REPO_DIR/engine/hooks/llm-judge" "$HOME/.cursor/hooks/llm-judge" link_item "hook-health" "$REPO_DIR/engine/hooks/hook-health" "$HOME/.cursor/hooks/hook-health" +link_item "skill-usage-log" "$REPO_DIR/engine/hooks/skill-usage-log" "$HOME/.cursor/hooks/skill-usage-log" link_item "build-the-lever" "$REPO_DIR/engine/hooks/build-the-lever" "$HOME/.cursor/hooks/build-the-lever" link_item "split-scope" "$REPO_DIR/engine/hooks/split-scope" "$HOME/.cursor/hooks/split-scope" link_item "repeat-error-stop" "$REPO_DIR/engine/hooks/repeat-error-stop" "$HOME/.cursor/hooks/repeat-error-stop" @@ -337,6 +338,7 @@ link_item "pr-schema-gate" "$REPO_DIR/engine/hooks/pr-schema-gate" "$HOME/.codex link_item "wrong-check-reflect" "$REPO_DIR/engine/hooks/wrong-check-reflect" "$HOME/.codex/hooks/wrong-check-reflect" link_item "llm-judge" "$REPO_DIR/engine/hooks/llm-judge" "$HOME/.codex/hooks/llm-judge" link_item "hook-health" "$REPO_DIR/engine/hooks/hook-health" "$HOME/.codex/hooks/hook-health" +link_item "skill-usage-log" "$REPO_DIR/engine/hooks/skill-usage-log" "$HOME/.codex/hooks/skill-usage-log" link_item "build-the-lever" "$REPO_DIR/engine/hooks/build-the-lever" "$HOME/.codex/hooks/build-the-lever" link_item "split-scope" "$REPO_DIR/engine/hooks/split-scope" "$HOME/.codex/hooks/split-scope" link_item "repeat-error-stop" "$REPO_DIR/engine/hooks/repeat-error-stop" "$HOME/.codex/hooks/repeat-error-stop" @@ -425,6 +427,7 @@ python3 "$REPO_DIR/engine/hooks/pr-schema-gate/install_cursor_hook.py" python3 "$REPO_DIR/engine/hooks/wrong-check-reflect/install_cursor_hook.py" python3 "$REPO_DIR/engine/hooks/llm-judge/install_cursor_hook.py" python3 "$REPO_DIR/engine/hooks/hook-health/install_cursor_hook.py" +python3 "$REPO_DIR/engine/hooks/skill-usage-log/install_cursor_hook.py" python3 "$REPO_DIR/engine/hooks/build-the-lever/install_cursor_hook.py" python3 "$REPO_DIR/engine/hooks/split-scope/install_cursor_hook.py" python3 "$REPO_DIR/engine/hooks/repeat-error-stop/install_cursor_hook.py" @@ -444,6 +447,7 @@ python3 "$REPO_DIR/engine/hooks/pr-schema-gate/install_codex_hook.py" echo "--- codex native scope-lock hooks (\$HOME/.codex/hooks.json) ---" python3 "$REPO_DIR/engine/hooks/scope-lock/install_codex_hook.py" python3 "$REPO_DIR/engine/hooks/hook-health/install_codex_hook.py" +python3 "$REPO_DIR/engine/hooks/skill-usage-log/install_codex_hook.py" python3 "$REPO_DIR/engine/hooks/build-the-lever/install_codex_hook.py" python3 "$REPO_DIR/engine/hooks/split-scope/install_codex_hook.py" python3 "$REPO_DIR/engine/hooks/repeat-error-stop/install_codex_hook.py" diff --git a/tests/test_install.py b/tests/test_install.py index f75607d43..807a3e183 100644 --- a/tests/test_install.py +++ b/tests/test_install.py @@ -609,6 +609,34 @@ def test_hook_health_wired_for_claude_cursor_and_codex(self): ] self.assertTrue(any("hook-health/codex_prompt_submit.py" in command for command in codex_prompt_commands)) + def test_skill_usage_log_wired_for_claude_cursor_and_codex(self): + for agent_dir in (".claude", ".cursor", ".codex"): + target = os.path.join(self.fake_home, agent_dir, "hooks", "skill-usage-log") + self.assertTrue(os.path.islink(target), target) + self.assertEqual(os.readlink(target), hook_src("skill-usage-log")) + + with open(os.path.join(self.fake_home, ".claude", "settings.json")) as handle: + claude_hooks = json.load(handle)["hooks"] + with open(os.path.join(self.fake_home, ".cursor", "hooks.json")) as handle: + cursor_hooks = json.load(handle)["hooks"] + with open(os.path.join(self.fake_home, ".codex", "hooks.json")) as handle: + codex_hooks = json.load(handle)["hooks"] + expected = ( + (claude_hooks, "PreToolUse", "skill-usage-log/claude_pretooluse_log.py"), + (claude_hooks, "UserPromptSubmit", "skill-usage-log/claude_prompt_submit.py"), + (cursor_hooks, "preToolUse", "skill-usage-log/cursor_pretooluse.py"), + (cursor_hooks, "beforeSubmitPrompt", "skill-usage-log/cursor_before_submit.py"), + (codex_hooks, "PreToolUse", "skill-usage-log/codex_pretooluse.py"), + (codex_hooks, "UserPromptSubmit", "skill-usage-log/codex_prompt_submit.py"), + ) + for hooks, event, marker in expected: + with self.subTest(event=event, marker=marker): + matching = [entry for entry in hooks[event] if marker in json.dumps(entry)] + self.assertEqual(len(matching), 1, matching) + self.assertIn("_runner/run.py", json.dumps(matching[0])) + claude_pre = [entry for entry in claude_hooks["PreToolUse"] if "skill-usage-log/" in json.dumps(entry)] + self.assertEqual(claude_pre[0]["matcher"], "Skill|Read|Bash") + def test_llm_judge_inbox_wired_for_claude_cursor_and_codex(self): for agent_dir in (".claude", ".cursor", ".codex"): target = os.path.join(self.fake_home, agent_dir, "hooks", "llm-judge")