Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion engine/hooks/llm-judge/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -76,7 +76,7 @@ the dictionary's `on_hit` text.

`ask(prompt)` tries these in order and stops at the first one that answers:

1. **codex**: `codex exec --skip-git-repo-check -m gpt-5.3-codex-spark --sandbox read-only -c notify=[] PROMPT`
1. **codex**: `codex exec --skip-git-repo-check -m <model> --sandbox read-only -c notify=[] PROMPT`, where `<model>` is the `model` in `~/.codex/config.toml` when `codex debug models` lists it, else the first listed model; if the catalog cannot be read, `-m` is left out and logged to `judge.log`
2. **claude**: `claude -p --model haiku --settings '{"disableAllHooks": true}' PROMPT`
3. **cursor**: `cursor-agent -p --output-format text PROMPT`

Expand Down
83 changes: 76 additions & 7 deletions engine/hooks/llm-judge/judge.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
import sys
import tempfile
import time
import tomllib
import traceback
import uuid

Expand All @@ -32,7 +33,7 @@
RUNNERS_ENV = "CATSTACK_LLM_JUDGE_RUNNERS"
STATE_ENV = "CATSTACK_LLM_JUDGE_STATE_DIR"
DEFAULT_RUNNERS = (
("codex", ["codex", "exec", "--skip-git-repo-check", "-m", "gpt-5.3-codex-spark", "--sandbox", "read-only", "-c", "notify=[]", PROMPT_SLOT]),
("codex", ["codex", "exec", "--skip-git-repo-check", "--sandbox", "read-only", "-c", "notify=[]", PROMPT_SLOT]),
("claude", ["claude", "-p", "--model", "haiku", "--settings", '{"disableAllHooks": true}', PROMPT_SLOT]),
("cursor", ["cursor-agent", "-p", "--output-format", "text", PROMPT_SLOT]),
)
Expand All @@ -42,6 +43,64 @@
)


CODEX_CATALOG_ARGV = ("codex", "debug", "models")
CODEX_CATALOG_TIMEOUT = 15


def codex_config_path() -> str:
return os.path.join(os.path.expanduser("~"), ".codex", "config.toml")


def codex_listed_models() -> list[str]:
"""Slugs this Codex login offers, most preferred first, from its own catalog."""
proc = subprocess.run(
list(CODEX_CATALOG_ARGV), capture_output=True, text=True, timeout=CODEX_CATALOG_TIMEOUT,
)
if proc.returncode != 0:
raise RuntimeError(f"codex debug models exited {proc.returncode}: {proc.stderr.strip()[-200:]}")
models = json.loads(proc.stdout)["models"]
listed = [m for m in models if isinstance(m, dict) and m.get("visibility") == "list" and m.get("slug")]
listed.sort(key=lambda m: m.get("priority") if isinstance(m.get("priority"), int) else sys.maxsize)
return [m["slug"] for m in listed]


def codex_configured_model() -> str | None:
path = codex_config_path()
try:
with open(path, "rb") as handle:
model = tomllib.load(handle).get("model")
except FileNotFoundError:
return None
except (OSError, tomllib.TOMLDecodeError) as exc:
log(f"codex model: could not read {path}: {exc}")
return None
return model if isinstance(model, str) else None


def codex_model() -> str | None:
"""The user's configured Codex model when the catalog lists it, else the catalog's first.

None means the catalog could not be read; the runner then omits -m and Codex
uses its own default, and the reason is logged."""
try:
listed = codex_listed_models()
except (OSError, subprocess.SubprocessError, RuntimeError, ValueError, KeyError, TypeError) as exc:
log(f"codex model: catalog unreadable, using Codex default: {type(exc).__name__}: {exc}")
return None
if not listed:
log("codex model: catalog lists no models, using Codex default")
return None
configured = codex_configured_model()
return configured if configured in listed else listed[0]


def with_codex_model(argv: list[str]) -> list[str]:
model = codex_model()
if model is None:
return argv
return argv[:3] + ["-m", model] + argv[3:]


def state_root() -> str:
return os.environ.get(STATE_ENV) or os.path.join(os.path.expanduser("~"), ".cache", "catstack-llm-judge")

Expand All @@ -68,7 +127,10 @@ def runners(mode: object = None) -> list[tuple[str, list[str]]]:
default = INVESTIGATE_RUNNERS if mode == "investigate" else DEFAULT_RUNNERS
raw = os.environ.get(RUNNERS_ENV)
if not raw:
return [(name, list(argv)) for name, argv in default]
return [
(name, with_codex_model(list(argv)) if name == "codex" and default is DEFAULT_RUNNERS else list(argv))
for name, argv in default
]
try:
parsed = json.loads(raw)
except ValueError as exc:
Expand Down Expand Up @@ -214,7 +276,9 @@ def unavailable_path(name: str) -> str:
return os.path.join(state_root(), "unavailable", f"{digest}.json")


def unavailable_until(name: str) -> float:
def unavailable_until(name: str, argv: list[str]) -> float:
"""When a benched runner may be tried again. A bench earned by a different
command (another model, another flag) does not apply to this one."""
path = unavailable_path(name)
try:
with open(path, encoding="utf-8") as handle:
Expand All @@ -227,6 +291,9 @@ def unavailable_until(name: str) -> float:
until = data.get("until") if isinstance(data, dict) else None
if isinstance(until, bool) or not isinstance(until, (int, float)):
return 0.0
if data.get("argv") != argv:
log(f"runner {name}: unavailable marker was for a different command, trying the current one")
return 0.0
return float(until)


Expand All @@ -235,9 +302,11 @@ def shows_unavailable(attempt: dict) -> bool:
return not attempt.get("ok") and (reason == NOT_INSTALLED or reason.startswith("exit "))


def mark_unavailable(name: str, reason: str) -> None:
def mark_unavailable(name: str, reason: str, argv: list[str]) -> None:
try:
write_json_atomic(unavailable_path(name), {"runner": name, "until": time.time() + UNAVAILABLE_SECONDS, "reason": reason})
write_json_atomic(unavailable_path(name), {
"runner": name, "argv": argv, "until": time.time() + UNAVAILABLE_SECONDS, "reason": reason,
})
log(f"runner {name}: left out of the judge table for {UNAVAILABLE_SECONDS}s after: {reason}")
except OSError as exc:
print(f"catstack-hook-error llm-judge: could not mark runner {name} unavailable: {exc}", file=sys.stderr)
Expand All @@ -256,7 +325,7 @@ def mark_available(name: str) -> None:
def available_runners(mode: object = None) -> list[tuple[str, list[str]]]:
table = runners(mode)
now = time.time()
kept = [(name, argv) for name, argv in table if unavailable_until(name) <= now]
kept = [(name, argv) for name, argv in table if unavailable_until(name, argv) <= now]
return kept or table


Expand All @@ -271,7 +340,7 @@ def ask(prompt: str, mode: object = None, timeout_seconds: object = None, cwd: o
mark_available(name)
return {"outcome": "answered", "runner": name, "answer": answer, "attempts": attempts}
if shows_unavailable(attempt):
mark_unavailable(name, attempt["reason"])
mark_unavailable(name, attempt["reason"], argv)
return {"outcome": "unchecked", "runner": None, "answer": None, "attempts": attempts}


Expand Down
128 changes: 128 additions & 0 deletions engine/hooks/llm-judge/tests/test_codex_model.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,128 @@
#!/usr/bin/env python3
"""The codex runner's model comes from Codex's own catalog, never a fixed slug.

Run: python3 -m unittest discover -s engine/hooks/llm-judge/tests -v

A fixed slug stops answering the day the login stops offering it, and every
judge call then reports "could not judge". These tests stand in a fake
`codex debug models` and a fake config, and pin which `-m` the runner gets.
"""
import json
import os
import sys
import tempfile
import unittest
from unittest.mock import patch

LIB_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, LIB_DIR)

import judge # noqa: E402

PY = sys.executable
CATALOG = {"models": [
{"slug": "big-model", "visibility": "list", "priority": 1},
{"slug": "hidden-model", "visibility": "hide", "priority": 0},
{"slug": "fast-model", "visibility": "list", "priority": 8},
]}


def catalog_argv(payload, code=0):
return (PY, "-c", f"import sys; print({json.dumps(payload)!r}); sys.exit({code})")


class CodexModelTestCase(unittest.TestCase):
def setUp(self):
self.work = tempfile.TemporaryDirectory()
self.config = os.path.join(self.work.name, "config.toml")
self.env = patch.dict(os.environ, {judge.STATE_ENV: self.work.name})
self.env.start()
os.environ.pop(judge.RUNNERS_ENV, None)
self.config_path = patch.object(judge, "codex_config_path", return_value=self.config)
self.config_path.start()

def tearDown(self):
self.config_path.stop()
self.env.stop()
self.work.cleanup()

def configure(self, model):
with open(self.config, "w", encoding="utf-8") as handle:
handle.write(f'model = "{model}"\n')

def codex_argv(self):
return dict(judge.runners())["codex"]

def judge_log(self):
path = os.path.join(self.work.name, "judge.log")
return open(path, encoding="utf-8").read() if os.path.exists(path) else ""

def test_configured_model_is_used_when_the_catalog_lists_it(self):
self.configure("fast-model")
with patch.object(judge, "CODEX_CATALOG_ARGV", catalog_argv(CATALOG)):
argv = self.codex_argv()
self.assertEqual(argv[:5], ["codex", "exec", "--skip-git-repo-check", "-m", "fast-model"])

def test_unlisted_configured_model_falls_back_to_first_listed(self):
self.configure("gpt-5.3-codex-spark")
with patch.object(judge, "CODEX_CATALOG_ARGV", catalog_argv(CATALOG)):
argv = self.codex_argv()
self.assertIn("big-model", argv)
self.assertNotIn("gpt-5.3-codex-spark", argv)

def test_hidden_models_are_never_picked(self):
with patch.object(judge, "CODEX_CATALOG_ARGV", catalog_argv(CATALOG)):
self.assertNotIn("hidden-model", self.codex_argv())

def test_default_runner_carries_no_fixed_model(self):
codex = dict(judge.DEFAULT_RUNNERS)["codex"]
self.assertNotIn("-m", codex)

def test_unreadable_catalog_omits_model_and_logs_it(self):
with patch.object(judge, "CODEX_CATALOG_ARGV", catalog_argv({"oops": 1}, code=2)):
argv = self.codex_argv()
self.assertNotIn("-m", argv)
self.assertIn("catalog unreadable", self.judge_log())

def test_malformed_catalog_json_omits_model_and_logs_it(self):
with patch.object(judge, "CODEX_CATALOG_ARGV", (PY, "-c", "print('not json')")):
argv = self.codex_argv()
self.assertNotIn("-m", argv)
self.assertIn("catalog unreadable", self.judge_log())

def test_runner_override_env_is_left_alone(self):
with patch.dict(os.environ, {judge.RUNNERS_ENV: json.dumps([["codex", ["codex", "{prompt}"]]])}):
self.assertEqual(self.codex_argv(), ["codex", "{prompt}"])

def test_investigate_runners_are_left_alone(self):
with patch.object(judge, "CODEX_CATALOG_ARGV", catalog_argv(CATALOG)):
codex = dict(judge.runners("investigate"))["codex"]
self.assertNotIn("-m", codex)


if __name__ == "__main__":
unittest.main()


class BenchFollowsTheCommandTestCase(unittest.TestCase):
def setUp(self):
self.work = tempfile.TemporaryDirectory()
self.env = patch.dict(os.environ, {judge.STATE_ENV: self.work.name})
self.env.start()

def tearDown(self):
self.env.stop()
self.work.cleanup()

def test_bench_for_the_same_command_still_applies(self):
argv = ["codex", "exec", "-m", "fast-model", "{prompt}"]
judge.mark_unavailable("codex", "exit 1", argv)
self.assertGreater(judge.unavailable_until("codex", argv), 0)

def test_bench_earned_by_an_old_model_does_not_block_a_new_one(self):
judge.mark_unavailable("codex", "exit 1: model not supported", ["codex", "exec", "-m", "gpt-5.3-codex-spark", "{prompt}"])
self.assertEqual(judge.unavailable_until("codex", ["codex", "exec", "-m", "fast-model", "{prompt}"]), 0.0)

def test_marker_with_no_recorded_command_does_not_block(self):
judge.write_json_atomic(judge.unavailable_path("codex"), {"runner": "codex", "until": 9e12, "reason": "exit 1"})
self.assertEqual(judge.unavailable_until("codex", ["codex", "{prompt}"]), 0.0)
Loading