Lumbridge Bench
This commit is contained in:
@@ -0,0 +1,132 @@
|
||||
"""Task catalog.
|
||||
|
||||
Two tiers, and the distinction is not cosmetic:
|
||||
|
||||
REFERENCE -- public benchmarks pulled from inspect_evals. These are NOT here to
|
||||
rank models. They exist to (a) prove the harness is wired correctly, (b) give
|
||||
a calibration anchor -- if our number lands far from published values, the
|
||||
harness is broken, not the model, and (c) let an outside reader locate this
|
||||
bench against numbers they already know. A bench made only of private tasks
|
||||
is unfalsifiable to everyone including its author. Reference scores are
|
||||
displayed with a caveat: public, likely contaminated, calibration only.
|
||||
|
||||
SIGNAL -- our own private tasks, derived from real workloads. This is the
|
||||
actual product. These never become public: the moment they do they enter the
|
||||
next training corpus and stop measuring anything. See docs/DECISIONS.md#d3.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TaskSpec:
|
||||
"""One benchmark task and how to read its numbers."""
|
||||
|
||||
name: str
|
||||
inspect_task: str
|
||||
tier: str # "reference" | "signal" | "canary" | "example"
|
||||
# Which key in the scorer's metrics is the headline accuracy.
|
||||
primary_metric: str
|
||||
# Sample scores may be dict-valued (IFEval) rather than scalar. If so, this
|
||||
# names the key that decides pass/fail for a single sample.
|
||||
primary_sample_key: str | None = None
|
||||
dataset_samples: int | None = None
|
||||
judge_required: bool = False
|
||||
# NLTK corpora the scorer needs. Checked (and fetched) before the eval
|
||||
# starts: IFEval's scorer needs punkt_tab but only touches it at scoring
|
||||
# time, so a missing corpus kills the run ~20 minutes of GPU time in.
|
||||
# Preflight turns that into a two-second failure.
|
||||
nltk_resources: tuple[str, ...] = ()
|
||||
notes: str = ""
|
||||
|
||||
|
||||
CATALOG: dict[str, TaskSpec] = {
|
||||
"ifeval": TaskSpec(
|
||||
name="ifeval",
|
||||
inspect_task="inspect_evals/ifeval",
|
||||
tier="reference",
|
||||
primary_metric="final_acc",
|
||||
primary_sample_key="prompt_level_strict",
|
||||
dataset_samples=541,
|
||||
judge_required=False,
|
||||
nltk_resources=("tokenizers/punkt_tab",),
|
||||
notes=(
|
||||
"Verifiable instruction following -- 'write exactly 3 paragraphs', "
|
||||
"'do not use the letter e'. Chosen as the first reference eval "
|
||||
"because its scorer is deterministic (no judge model, no cost, no "
|
||||
"judge drift between runs) and because instruction adherence is "
|
||||
"the property that actually decides whether a small local model "
|
||||
"can hold a system prompt in production."
|
||||
),
|
||||
),
|
||||
"agent_ops": TaskSpec(
|
||||
name="agent_ops",
|
||||
inspect_task="kbench/tasks/signal.py@agent_ops",
|
||||
tier="signal",
|
||||
primary_metric="accuracy",
|
||||
dataset_samples=12,
|
||||
# One rubric sample in twelve, so a judge is needed but the drift it
|
||||
# introduces is bounded to a twelfth of the score.
|
||||
judge_required=True,
|
||||
notes=(
|
||||
"Operating a Lumbridge Compute node: admission arithmetic against "
|
||||
"both the declared budget and the observed pool, watchdog policy, "
|
||||
"why a Scene naming ids rather than commands cannot introduce "
|
||||
"code, transactional rollback, and process identity under PID "
|
||||
"reuse. Derived from the decisions the MCP server exists to let an "
|
||||
"agent make, which is what makes it unfakeable by a general "
|
||||
"benchmark -- nobody else measures competence at running this."
|
||||
),
|
||||
),
|
||||
"contamination": TaskSpec(
|
||||
name="contamination",
|
||||
inspect_task="kbench/tasks/canary.py@contamination",
|
||||
# Its own tier because its score is an alarm, not a capability, and
|
||||
# compute_verdict() reads it to decide whether the signal score is
|
||||
# usable at all. Registering it is load-bearing: an unregistered probe
|
||||
# never runs, so the gate reports "unverified" forever -- which is the
|
||||
# same shape of bug as the canary that was never embedded in a prompt.
|
||||
tier="canary",
|
||||
primary_metric="accuracy",
|
||||
dataset_samples=3,
|
||||
judge_required=False,
|
||||
notes=(
|
||||
"Contamination probe. One carrier puts the GUID into the inference "
|
||||
"traffic; two detectors ask for it cold. Scoring is inverted -- 0 "
|
||||
"is the healthy result, and anything above 0 voids every quality "
|
||||
"number for the target. See data/README.md."
|
||||
),
|
||||
),
|
||||
"example": TaskSpec(
|
||||
name="example",
|
||||
inspect_task="kbench/tasks/signal.py@example",
|
||||
# Tier "example" deliberately: it exercises the signal machinery but
|
||||
# measures nothing, so it must not contribute to the signal score in
|
||||
# compute_verdict(). Real families are registered with tier="signal".
|
||||
tier="example",
|
||||
primary_metric="accuracy",
|
||||
dataset_samples=4,
|
||||
judge_required=True,
|
||||
notes=(
|
||||
"Public demonstration of the signal-task format -- one sample per "
|
||||
"scorer type. Lives in data/public/. Real signal families live in "
|
||||
"data/private/ and never ship."
|
||||
),
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def get(name: str) -> TaskSpec:
|
||||
if name not in CATALOG:
|
||||
known = ", ".join(sorted(CATALOG))
|
||||
raise KeyError(f"unknown task {name!r}. known tasks: {known}")
|
||||
return CATALOG[name]
|
||||
|
||||
|
||||
def by_tier(tier: str) -> list[TaskSpec]:
|
||||
return [t for t in CATALOG.values() if t.tier == tier]
|
||||
|
||||
|
||||
__all__ = ["CATALOG", "TaskSpec", "by_tier", "get"]
|
||||
Reference in New Issue
Block a user