Files
Karti Tripathi 006feee0f7
CI / verify (push) Successful in 24s
CI / deploy (push) Failing after 1m14s
Lumbridge Bench
2026-08-04 00:44:07 -07:00

136 lines
5.5 KiB
Python

"""Frontier reduction.
The claim a sweep makes is "you can stop considering this one", so the tests are mostly about
when that claim is NOT safe to make.
python3 -m unittest discover -s tests -v
"""
from __future__ import annotations
import sys
import unittest
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(ROOT))
from kbench.sweep import SweepPoint, analyse, point_from_run # noqa: E402
def p(tid, quality=None, tps=None, contamination="clean", error=None):
return SweepPoint(
target_id=tid, quality=quality, throughput_tps=tps, contamination=contamination, error=error
)
class Domination(unittest.TestCase):
def test_worse_on_both_axes_is_dominated(self):
r = analyse([p("fast-good", 0.80, 200.0), p("slow-bad", 0.70, 100.0)])
self.assertEqual(r.best, ["fast-good"])
self.assertEqual(r.dominated["slow-bad"], "fast-good")
def test_a_genuine_trade_off_keeps_both_on_the_frontier(self):
# Higher quality at lower throughput is a decision the operator has to make with
# knowledge this tool does not have. Collapsing it to one winner would be inventing
# a preference.
r = analyse([p("accurate", 0.90, 100.0), p("quick", 0.70, 200.0)])
self.assertEqual(sorted(r.best), ["accurate", "quick"])
self.assertEqual(r.dominated, {})
def test_equal_quality_and_more_throughput_dominates(self):
r = analyse([p("a", 0.80, 200.0), p("b", 0.80, 150.0)])
self.assertEqual(r.best, ["a"])
self.assertIn("b", r.dominated)
def test_identical_points_do_not_dominate_each_other(self):
# Nothing is better anywhere, so neither may be dismissed — otherwise a tie would
# silently eliminate an option.
r = analyse([p("a", 0.80, 200.0), p("b", 0.80, 200.0)])
self.assertEqual(sorted(r.best), ["a", "b"])
self.assertEqual(r.dominated, {})
def test_the_frontier_is_ordered_fastest_first(self):
r = analyse([p("mid", 0.85, 150.0), p("fast", 0.70, 200.0), p("slow", 0.95, 100.0)])
self.assertEqual(r.best, ["fast", "mid", "slow"])
class WhenQualityCannotBeTraded(unittest.TestCase):
def test_a_contaminated_point_is_ranked_on_throughput_only(self):
# Its quality number is memorisation. Letting it win a quality comparison would launder
# a void score into a recommendation.
r = analyse([p("clean", 0.80, 100.0), p("dirty", 0.99, 90.0, contamination="detected")])
self.assertTrue(any("throughput alone" in w for w in r.warnings))
self.assertEqual(r.best, ["clean"])
self.assertIn("dirty", r.dominated)
def test_a_missing_quality_score_also_falls_back_to_throughput(self):
r = analyse([p("measured", 0.80, 100.0), p("unmeasured", None, 150.0)])
self.assertTrue(any("throughput alone" in w for w in r.warnings))
self.assertEqual(r.best, ["unmeasured"])
def test_quality_is_used_when_every_point_has_a_comparable_one(self):
r = analyse([p("a", 0.90, 100.0), p("b", 0.70, 120.0)])
self.assertFalse(any("throughput alone" in w for w in r.warnings))
self.assertEqual(sorted(r.best), ["a", "b"])
class Robustness(unittest.TestCase):
def test_a_failed_target_never_reaches_the_frontier(self):
r = analyse([p("ok", 0.80, 100.0), p("broken", error="server never came up")])
self.assertEqual(r.best, ["ok"])
self.assertEqual([f.target_id for f in r.failed], ["broken"])
self.assertNotIn("broken", r.dominated)
def test_all_failed_is_reported_rather_than_returning_an_empty_winner(self):
r = analyse([p("a", error="boom"), p("b", error="boom")])
self.assertEqual(r.best, [])
self.assertTrue(any("No target produced" in w for w in r.warnings))
def test_mixed_hosts_are_warned_about(self):
# A sweep is meant to hold the machine constant; two hosts means two causes.
r = analyse([p("a", 0.8, 100.0), p("b", 0.8, 200.0)], hosts=["your-node", "metal"])
self.assertTrue(any("more than one host" in w for w in r.warnings))
def test_one_host_is_not_warned_about(self):
r = analyse([p("a", 0.8, 100.0)], hosts=["your-node", "your-node"])
self.assertFalse(any("more than one host" in w for w in r.warnings))
class ReadingSavedRuns(unittest.TestCase):
def test_a_contaminated_run_contributes_no_quality(self):
# compute_verdict nulls signal_score when a probe fires; the sweep must inherit that
# rather than reaching for the withheld value.
run = {
"target_id": "t",
"verdict": {
"signal_score": None,
"signal_score_unverified": 0.99,
"peak_throughput_tps": 180.0,
"contamination": "detected",
},
}
point = point_from_run(run)
self.assertIsNone(point.quality)
self.assertFalse(point.quality_comparable)
self.assertTrue(point.usable, "throughput is still valid evidence")
def test_a_clean_run_is_fully_comparable(self):
run = {
"target_id": "t",
"verdict": {
"signal_score": 0.81,
"peak_throughput_tps": 180.0,
"single_stream_tps": 30.0,
"contamination": "clean",
},
}
point = point_from_run(run)
self.assertTrue(point.quality_comparable)
self.assertEqual(point.quality, 0.81)
self.assertEqual(point.single_stream_tps, 30.0)
if __name__ == "__main__":
unittest.main()