Files
bench/tests/test_catalog.py
Karti Tripathi 006feee0f7
CI / verify (push) Successful in 24s
CI / deploy (push) Failing after 1m14s
Lumbridge Bench
2026-08-04 00:44:07 -07:00

199 lines
8.6 KiB
Python

"""The catalog must describe things that exist.
This guards one specific failure that has now happened twice in this repo, in two different
shapes: a control that cannot fire. First the canary GUID that was never embedded in a prompt,
then the canary task that was never registered — in both cases every surface reported success
because nothing raised, and the thing simply never ran.
The pattern is the same each time: a component is *described* somewhere and *wired* nowhere,
and no test notices because nothing errors. So these tests check the wiring itself — that every
catalog entry resolves to a real task in a real module over a real data file.
Deliberately parses source rather than importing: `kbench.tasks.signal` imports `inspect_ai` at
module scope, and a structural check that only runs when a heavy optional dependency is
installed is a check that does not run.
python3 -m unittest discover -s tests -v
"""
from __future__ import annotations
import ast
import re
import sys
import unittest
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(ROOT))
CATALOG_SRC = (ROOT / "kbench" / "tasks" / "__init__.py").read_text()
def catalog_entries() -> dict[str, dict[str, str]]:
"""Pull (name -> {field: literal}) out of the CATALOG without importing it."""
tree = ast.parse(CATALOG_SRC)
entries: dict[str, dict[str, str]] = {}
for node in ast.walk(tree):
if not isinstance(node, ast.AnnAssign) or getattr(node.target, "id", None) != "CATALOG":
continue
assert isinstance(node.value, ast.Dict)
for key, value in zip(node.value.keys, node.value.values):
assert isinstance(key, ast.Constant) and isinstance(value, ast.Call)
fields = {
kw.arg: kw.value.value
for kw in value.keywords
if isinstance(kw.value, ast.Constant)
}
entries[key.value] = fields
return entries
ENTRIES = catalog_entries()
class CatalogWiring(unittest.TestCase):
def test_the_catalog_was_parsed_at_all(self):
self.assertTrue(ENTRIES, "could not read CATALOG — this test is not testing anything")
def test_the_key_matches_the_spec_name(self):
for key, fields in ENTRIES.items():
with self.subTest(task=key):
self.assertEqual(fields.get("name"), key)
def test_every_local_task_resolves_to_a_defined_function(self):
# `module.py@func` must name a function that actually exists and is decorated @task.
for key, fields in ENTRIES.items():
ref = fields.get("inspect_task", "")
if "@" not in ref or ref.startswith("inspect_evals/"):
continue
module_path, func = ref.split("@", 1)
with self.subTest(task=key):
path = ROOT / module_path
self.assertTrue(path.exists(), f"{ref}: {module_path} does not exist")
src = path.read_text()
self.assertRegex(
src,
rf"@task\s*\ndef {re.escape(func)}\s*\(",
f"{ref}: no @task-decorated def {func}() in {module_path}",
)
def test_every_local_task_has_a_data_file_with_samples(self):
# A registered task over an empty family raises only at run time, after a target has
# been stood up and paid for.
from kbench.schema import read_records # local import: stdlib-only module
tier_for = {"signal": "private", "canary": "canary", "example": "public"}
for key, fields in ENTRIES.items():
ref = fields.get("inspect_task", "")
if "@" not in ref or ref.startswith("inspect_evals/"):
continue
tier = tier_for.get(fields.get("tier", ""))
if tier is None:
continue
family = ref.split("@", 1)[1]
path = ROOT / "data" / tier / f"{family}.jsonl"
with self.subTest(task=key):
# A missing tier DIRECTORY means the published tree, which ships neither
# the signal set nor the canary — expected, and not a wiring bug. A missing
# FILE inside a directory that exists is the real defect this guards: a task
# registered in the catalog with nothing behind it.
if not path.parent.exists():
self.skipTest(f"{tier}/ not present (expected in the published tree)")
self.assertTrue(path.exists(), f"{key}: no data file at {path}")
self.assertTrue(read_records(path), f"{key}: {path} has no samples")
def test_declared_sample_count_matches_the_file(self):
# A drifting count is how "we ran the whole set" quietly becomes false.
from kbench.schema import read_records
tier_for = {"signal": "private", "canary": "canary", "example": "public"}
for key, fields in ENTRIES.items():
ref = fields.get("inspect_task", "")
declared = fields.get("dataset_samples")
tier = tier_for.get(fields.get("tier", ""))
if "@" not in ref or declared is None or tier is None:
continue
family = ref.split("@", 1)[1]
path = ROOT / "data" / tier / f"{family}.jsonl"
if not path.exists():
continue
actual = len([r for _, r in read_records(path) if r["split"] == "test"])
with self.subTest(task=key):
self.assertEqual(
actual, declared, f"{key}: catalog says {declared} samples, file has {actual}"
)
class ContaminationIsWired(unittest.TestCase):
"""The gate is only real if the probe runs and something reads the result."""
def test_a_canary_tier_task_is_registered(self):
canary = [k for k, f in ENTRIES.items() if f.get("tier") == "canary"]
self.assertTrue(
canary,
"no canary-tier task in the CATALOG — the contamination gate would report "
"'unverified' forever, which is indistinguishable from having no probe at all",
)
def test_the_verdict_reads_the_canary_tier(self):
src = (ROOT / "kbench" / "results.py").read_text()
self.assertIn('tier == "canary"', src, "compute_verdict ignores the canary tier")
self.assertIn(
'verdict["signal_score"] = None',
src,
"a fired probe must remove the signal score, not merely annotate it",
)
def test_at_least_one_signal_family_exists(self):
signal = [k for k, f in ENTRIES.items() if f.get("tier") == "signal"]
self.assertTrue(signal, "no signal-tier family — the bench measures nothing of its own")
class SamplingResolution(unittest.TestCase):
"""Registry sampling overrides the default; the card and the run share one source."""
def setUp(self):
try:
from kbench.run import DEFAULT_SAMPLING, resolve_sampling
except ImportError as exc:
self.skipTest(f"serving deps unavailable: {exc}")
self.resolve, self.default = resolve_sampling, DEFAULT_SAMPLING
def test_default_is_not_greedy(self):
# Greedy decoding is the intuitive choice for reproducibility and makes reasoning
# models repeat forever -- a 90s probe ran past 12 minutes at temperature 0.
# Reproducibility comes from the seed instead.
self.assertGreater(self.default["temperature"], 0.0)
self.assertIn("seed", self.default)
self.assertIn("max_tokens", self.default)
def test_registry_values_win_over_defaults(self):
from kbench.registry import load_registry
target = load_registry().target("brain")
resolved = self.resolve(target)
for key, value in (target.serving.sampling or {}).items():
with self.subTest(key=key):
self.assertEqual(resolved[key], value)
def test_every_key_is_a_real_generate_config_field(self):
# A sampling key inspect does not know is not ignored -- it aborts the run. Worse,
# a key that IS accepted but silently dropped would look pinned and still sample.
try:
from inspect_ai.model import GenerateConfig
except ImportError as exc:
self.skipTest(f"inspect-ai unavailable: {exc}")
from kbench.registry import load_registry
fields = set(GenerateConfig.model_fields)
for target in load_registry().targets.values():
resolved = self.resolve(target)
for key in resolved:
with self.subTest(target=target.id, key=key):
self.assertIn(key, fields, f"{key} is not a GenerateConfig field")
if __name__ == "__main__":
unittest.main()