b601511e7f
The browser engine is a port of the Python one and CI proves it: all 21.2M (guess, answer) pairs hashed on both sides to the same SHA-256. Six TS tests, including the duplicate-letter table and the twelve pinned seed vectors that keep ?seed= permalinks pointing at the same word the recording used. Word lists are split by how they are used. answers.json is inlined because the board needs it before first paint to turn a seed into a word, and a fetch there means a visibly empty board on a cold cache. guesses.json is fetched, because it is three times larger and only needed the first time somebody presses Enter; until it lands, validation falls back to the answer list, which accepts strictly fewer words. The failure mode is 'your real word was briefly rejected', not 'a non-word was accepted' — the right way round. The solver runs in a worker constructed from a same-origin module URL, never Vite's ?worker&inline: that yields a blob:, and production CSP has no worker-src, so it falls back to default-src 'self' and the worker is blocked with no console error. It would fail in production only. deploy.sh smoke-tests the real public hostname from the deploying machine and fails on a body under 1 kB, because the bind bug's signature is a valid certificate over an empty 200 and a local --resolve check passes anyway. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019mt6sHQHEnEYrJZvoMCJSB
79 lines
2.7 KiB
Python
79 lines
2.7 KiB
Python
#!/usr/bin/env python3
|
|
"""Build public/traces/manifest.json from the captured fixtures.
|
|
|
|
The manifest is what the browser reads to know which runs exist. It is
|
|
generated rather than hand-written so a fixture can never be referenced without
|
|
existing, or exist without being referenced.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from pathlib import Path
|
|
|
|
TRACES = Path(__file__).parent.parent / "public" / "traces"
|
|
|
|
ARMS = {
|
|
"base-off": ("Out of the box", "recorded"),
|
|
"base-on": ("Allowed to think", "intervened"),
|
|
"solver": ("Best-known play", "generated"),
|
|
}
|
|
|
|
# What was done to the run, for arms that had something done to them. Required
|
|
# by the contract on any `intervened` run so that a sampling change can never be
|
|
# presented as a training result by omitting to mention it.
|
|
INTERVENTIONS = {
|
|
"base-on": "Same model, same seeds, sampled with thinking enabled. No training, no fine-tuning.",
|
|
}
|
|
|
|
ORDER = ["base-off", "base-on", "solver"]
|
|
|
|
|
|
def main() -> int:
|
|
manifest: dict[str, list[dict]] = {}
|
|
|
|
for demo_dir in sorted(p for p in TRACES.iterdir() if p.is_dir()):
|
|
runs = []
|
|
for path in sorted(demo_dir.glob("*.json")):
|
|
data = json.loads(path.read_text())
|
|
arm = data["runId"].rsplit("-s", 1)[0]
|
|
label, kind = ARMS.get(arm, (arm, "recorded"))
|
|
run = {
|
|
"id": data["runId"],
|
|
"label": label,
|
|
"path": f"/traces/{demo_dir.name}/{path.name}",
|
|
"kind": kind,
|
|
"model": data["model"],
|
|
"capturedAt": data["capturedAt"],
|
|
"seed": data["seed"],
|
|
}
|
|
if arm in INTERVENTIONS:
|
|
run["intervention"] = INTERVENTIONS[arm]
|
|
runs.append(run)
|
|
|
|
runs.sort(key=lambda r: (ORDER.index(r["id"].rsplit("-s", 1)[0])
|
|
if r["id"].rsplit("-s", 1)[0] in ORDER else 99,
|
|
r["seed"]))
|
|
if runs:
|
|
manifest[demo_dir.name] = runs
|
|
|
|
out = TRACES / "manifest.json"
|
|
out.write_text(json.dumps(manifest, indent=2) + "\n")
|
|
|
|
for slug, runs in manifest.items():
|
|
by_arm: dict[str, list[dict]] = {}
|
|
for r in runs:
|
|
by_arm.setdefault(r["id"].rsplit("-s", 1)[0], []).append(r)
|
|
print(f"{slug}: {len(runs)} runs")
|
|
for arm, group in by_arm.items():
|
|
solved = 0
|
|
for r in group:
|
|
data = json.loads((TRACES.parent / r["path"].lstrip("/")).read_text())
|
|
solved += 1 if data["outcome"] == "solved" else 0
|
|
print(f" {arm:<10} {len(group)} runs, solved {solved}/{len(group)}")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|