Files
Karti Tripathi 006feee0f7
CI / verify (push) Successful in 24s
CI / deploy (push) Failing after 1m14s
Lumbridge Bench
2026-08-04 00:44:07 -07:00

150 lines
6.0 KiB
Python

"""Sweep a matrix of targets and reduce it to a decision.
`run` answers "what did this target do". `compare` answers "what changed between two". Neither
answers the question people actually start with, which is **which of these should I run**, and
that one is a matrix: the same model at three quantisations, or one quantisation under four sets
of serving flags, all on the box you actually own.
Three things make this more than a shell loop:
**It holds the box constant.** A sweep varies configuration and nothing else. Varying the host
too produces numbers with two causes, and the tempting reading — "fp8 is faster" — is then
unsupported by the data that appears to show it.
**It runs sequentially, and that is not a simplification.** On a unified-memory machine two
targets cannot both be resident; that constraint is the entire premise of the stack this bench
was built for. Running them concurrently would measure contention between them rather than
either one.
**It reduces to a frontier, not a leaderboard.** Self-hosting is a trade: quality against
throughput, under a memory ceiling. A single ranking hides that, so the output is the set of
targets that are not beaten on both axes at once — plus, explicitly, the ones that are, because
a target that is worse at everything is the one useful thing a matrix can tell you.
Every point is an ordinary score card saved to `results/`. There is no sweep-shaped result
format, so `compare`, `card` and the git history all keep working unchanged.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from typing import Any
@dataclass
class SweepPoint:
"""One target's outcome within a sweep."""
target_id: str
quality: float | None
throughput_tps: float | None
single_stream_tps: float | None = None
contamination: str | None = None
error: str | None = None
run_path: str | None = None
@property
def usable(self) -> bool:
"""Whether this point can take part in the trade-off at all."""
return self.error is None and self.throughput_tps is not None
@property
def quality_comparable(self) -> bool:
"""A contaminated or unmeasured quality score cannot be traded against anything."""
return self.quality is not None and self.contamination != "detected"
@dataclass
class SweepResult:
points: list[SweepPoint] = field(default_factory=list)
best: list[str] = field(default_factory=list)
dominated: dict[str, str] = field(default_factory=dict) # target -> what beats it
warnings: list[str] = field(default_factory=list)
@property
def failed(self) -> list[SweepPoint]:
return [p for p in self.points if p.error]
def _dominates(a: SweepPoint, b: SweepPoint, use_quality: bool) -> bool:
"""True when `a` is at least as good as `b` everywhere and better somewhere.
Strict domination is the only claim worth making from a small matrix. "Higher throughput
and slightly lower quality" is a trade the operator has to make with knowledge this tool
does not have — how much quality their workload can spare — so it is deliberately left to
them rather than collapsed into a score.
"""
if not (a.usable and b.usable):
return False
better_anywhere = a.throughput_tps > b.throughput_tps
at_least_equal = a.throughput_tps >= b.throughput_tps
if use_quality and a.quality_comparable and b.quality_comparable:
at_least_equal = at_least_equal and a.quality >= b.quality
better_anywhere = better_anywhere or a.quality > b.quality
return at_least_equal and better_anywhere
def analyse(points: list[SweepPoint], hosts: list[str] | None = None) -> SweepResult:
"""Reduce a set of measured targets to a frontier and a set of dominated options."""
result = SweepResult(points=list(points))
if hosts and len(set(hosts)) > 1:
result.warnings.append(
f"Targets ran on more than one host ({', '.join(sorted(set(hosts)))}). A sweep is "
"meant to hold the machine constant — these differences have two causes and "
"neither can be isolated."
)
usable = [p for p in points if p.usable]
if not usable:
result.warnings.append("No target produced a usable measurement.")
return result
with_quality = [p for p in usable if p.quality_comparable]
use_quality = len(with_quality) == len(usable)
if not use_quality:
missing = sorted(p.target_id for p in usable if not p.quality_comparable)
result.warnings.append(
f"Ranking on throughput alone: {', '.join(missing)} has no comparable quality score "
"(never measured, or voided by a contamination probe)."
)
for candidate in usable:
beaten_by = next(
(other.target_id for other in usable
if other.target_id != candidate.target_id and _dominates(other, candidate, use_quality)),
None,
)
if beaten_by:
result.dominated[candidate.target_id] = beaten_by
else:
result.best.append(candidate.target_id)
# Present the frontier the way the decision is made: fastest first, quality breaking ties.
order = {p.target_id: p for p in usable}
result.best.sort(
key=lambda t: (
-(order[t].throughput_tps or 0),
-(order[t].quality if order[t].quality_comparable else 0),
)
)
return result
def point_from_run(run: dict[str, Any], path: str | None = None) -> SweepPoint:
"""Read a saved score card into a sweep point.
Reads `signal_score` rather than any raw value: a contaminated run has had that nulled by
compute_verdict, which is exactly the behaviour a sweep should inherit — a memorised score
must not win a frontier.
"""
verdict = run.get("verdict") or {}
return SweepPoint(
target_id=run.get("target_id", "?"),
quality=verdict.get("signal_score"),
throughput_tps=verdict.get("peak_throughput_tps"),
single_stream_tps=verdict.get("single_stream_tps"),
contamination=verdict.get("contamination"),
run_path=path,
)