diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..bdd9731 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,86 @@ +name: ci + +on: + push: + branches: [main] + pull_request: + +concurrency: + group: ci-${{ github.ref }} + cancel-in-progress: true + +jobs: + web: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v5 + - uses: pnpm/action-setup@v4 + with: { version: 11.21.0 } + - uses: actions/setup-node@v5 + with: { node-version: 22, cache: pnpm } + + # Asserts the toolchain matches what package.json pins, rather than + # discovering a mismatch three steps later as an unrelated build error. + - name: preflight + run: | + test "$(pnpm -v)" = "11.21.0" || { echo "pnpm $(pnpm -v) != 11.21.0"; exit 1; } + + - run: CI=true pnpm install --frozen-lockfile + - run: pnpm typecheck + - run: pnpm check + - run: pnpm test + - run: pnpm build + - run: node scripts/bundle-budget.mjs + + # The prerender pass writes a real HTML file per route. Crawlers do not + # run JavaScript, so without these every shared link previews as the + # homepage — assert the baked tags actually landed. + - name: prerendered head is real + run: | + test -f dist/demos/wordle/index.html || { echo "no prerendered demo route"; exit 1; } + grep -q 'og:title' dist/demos/wordle/index.html || { echo "og tags missing"; exit 1; } + grep -qv 'PIG Demo — RL environments you can play' dist/demos/wordle/index.html \ + || { echo "demo route kept the homepage title"; exit 1; } + test -f dist/404.html || { echo "no 404.html"; exit 1; } + + # A blob-backed worker is blocked in production and nowhere else: the + # site's CSP has no worker-src, so it falls back to default-src 'self'. + # The failure is silent — the solver simply never boots. + - name: no inline workers + run: | + ! grep -rqE "createObjectURL|blob:" dist/assets/*.js \ + || { echo "a blob: URL reached the bundle; CSP will block it in prod"; exit 1; } + + python: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v5 + - uses: astral-sh/setup-uv@v6 + with: { enable-cache: true } + + - run: uv sync --all-packages + + # Rebuild the word lists from their committed sources and assert the + # output is byte-identical. If it is not, every downstream number — + # the conformance digest included — is describing a different game. + - name: word lists rebuild identically + run: | + uv run python envs/wordle_five/words/build_words.py + git diff --exit-code envs/wordle_five/words/*.json + + - run: uv run pytest envs/wordle_five/tests -q + - run: uv run python envs/probe.py + + # The cross-language gate. Both halves score all 21.2M (guess, answer) + # pairs; the digests must match each other and the committed value. + - uses: pnpm/action-setup@v4 + with: { version: 11.21.0 } + - uses: actions/setup-node@v5 + with: { node-version: 22, cache: pnpm } + - run: CI=true pnpm install --frozen-lockfile + - run: pnpm conformance + + # Every committed fixture must replay through the Python engine and + # reproduce its own recorded rewards. A fixture that cannot be + # regenerated is a claim with no receipt behind it. + - run: uv run python envs/verify_fixtures.py diff --git a/README.md b/README.md new file mode 100644 index 0000000..69ccbdb --- /dev/null +++ b/README.md @@ -0,0 +1,144 @@ +# PIG-Demo + +Interactive demos of reinforcement-learning **environments**, built for people +who sign the budget rather than write the training loop. + +**https://demo.primeintellectgrowth.com** + +An environment is four things: a task, a set of legal moves, a grader that +cannot be argued with, and a score that moves. Every demo here renders those +four boxes, plays a real recorded rollout against them, and then lets you +change what "good" means and watch the ranking flip. + +Sibling project to [PIG](https://primeintellectgrowth.com). + +--- + +## Why a word game first + +Because it is **Prime Intellect's own hello-world**, and that is checkable in +three public repositories in about thirty seconds: + +- one of five `basic` examples in [prime-rl](https://github.com/PrimeIntellect-ai/prime-rl/tree/main/examples/basic/wordle) +- an environment in [verifiers](https://github.com/PrimeIntellect-ai/verifiers/tree/main/environments/wordle) +- the environment used in the official [lab-cookbook](https://github.com/PrimeIntellect-ai/lab-cookbook) GEPA prompt-optimisation tutorial + +We didn't pick a game. We picked theirs. + +It also needs zero domain knowledge, which means an executive learns the +*machine* — task, moves, grader, score — before any vertical vocabulary gets in +the way. Every demo after it renders the same four boxes with a different +grader. + +## What is actually in here + +``` +envs/wordle_five/ A real environment: engine, reward, reference solver, tests +envs/probe.py The ladder that proves the reward measures something +src/demos/wordle/ The browser half — a port of the same engine +public/traces/wordle/ Recorded rollouts, committed. The evidence. +``` + +The site does **no live inference**. Rollouts are captured once against a real +model and replayed at their recorded wall-clock. That is a deliberate choice: a +public demo with no auth cannot hold an API key, a live call is slow and flaky +on conference wifi, and a recorded run can be scrubbed, permalinked, +blind-compared and *verified* in ways a live one cannot. + +What keeps it from being a video is that the browser re-derives every number it +shows. It re-runs the recorded moves through its own copy of the engine, +rescores them, and prints the delta against what the environment recorded. + +## The three commands that check the claims + +```bash +uv sync --all-packages +uv run pytest envs/wordle_five/tests # 25 tests, incl. the duplicate-letter cases +uv run python envs/probe.py # the reward ladder, 7 hard assertions +pnpm install && pnpm check && pnpm build +``` + +### The cross-language gate + +`engine.py` and `engine.ts` must agree on every tile. CI scores **all 4,603² = +21.2 million** (guess, answer) pairs through both and compares a SHA-256: + +``` +69f4e8dfbdc492176d7b7f04b080d8d3cccb5f18aca2d592e622938a5aeec8e2 +``` + +A hand-picked vector file only ever catches the cases somebody thought of. This +catches all of them, and it is what lets the page claim it *verified* a run. + +The subtlety it protects is the duplicate-letter rule. Scoring runs in two +passes — every green in the word is resolved before any yellow is assigned — +because a letter may be marked non-grey at most as many times as it occurs in +the answer, and greens have first claim. `SASSY` against `BASIS` is `YGGXX`: +three S's guessed, two available, so one yellow and one grey. Getting this +wrong is the most common bug in implementations of this game, and it is the bug +that put a correction video on the most-watched explanation of it ever made. + +## The reward, and why the third component is the point + +| component | weight | role | measures | +|---|---|---|---| +| `solved` | 0.50 | objective | did it win | +| `economy` | 0.30 | objective | turns used, as a ratio against the reference player on the same word | +| `consistency` | 0.20 | **counterweight** | share of turns spent on a move that could still have won | + +A counterweight has to be in genuine tension with the objective, or it is a +gate wearing a counterweight's name. This one is: a player maximising +information deliberately guesses words that **cannot** win, because a word +splitting the candidates evenly teaches more than a word that might happen to +be right. That is good play, and it costs consistency. + +The probe ladder measures the trade rather than asserting it: + +``` +policy solved economy consistency TOTAL +inaction 0.0000 0.0000 0.0000 0.0000 +crude 0.0000 0.0000 0.0556 0.0111 +plausible 0.1250 0.1000 0.1493 0.1224 +candidate_only 0.9375 0.7458 1.0000 0.8925 +exhaustive 1.0000 0.9375 0.6094 0.9031 +oracle 1.0000 1.0000 0.7292 0.9458 +``` + +The two good policies are 0.05 apart and **neither dominates the other**. Which +one wins is a decision about what you actually want — which is the whole +argument this site exists to make, and it is why the reward editor on the page +can flip the ranking with one slider. `probe.py` fails CI if either policy +starts dominating, if any weighted component goes flat, or if doing nothing +scores above zero. + +## Honesty + +- Every rollout is **recorded, not live**, and labelled as such on the page. +- The reward editor **re-scores recorded attempts**. It does not retrain + anything, and it says so where you use it. +- The word list is [our own construction](envs/wordle_five/words/PROVENANCE.md) + from Wordnik (MIT) and SCOWL. Our answer pool is 4,603 — roughly twice the + original game's — so this is **harder** than the original and our numbers are + not comparable to published figures for it. The famous SALET / 3.4212-guess + optimum belongs to that list, not ours, and is cited as such. +- The verticals listed on the site are **our proposals**. They are not Prime + Intellect's roadmap and not a customer list. No customer logos, ever. +- Not affiliated with The New York Times. Independent implementation, own word + lists, own palette, own reward. + +Full accounting: [demo.primeintellectgrowth.com/honesty](https://demo.primeintellectgrowth.com/honesty) + +## Adding a demo + +```bash +pnpm demo:new +``` + +Copies the templates and wires nothing — the registry finds demos by existence, +so the header, the gallery, the router and the sitemap all pick it up with zero +edits to shared files. `pnpm check` enforces the contract; see +[CONTRACT.md](CONTRACT.md). + +## Licence + +Apache-2.0. Third-party attributions in [NOTICE](NOTICE). diff --git a/envs/capture.py b/envs/capture.py new file mode 100644 index 0000000..94070ef --- /dev/null +++ b/envs/capture.py @@ -0,0 +1,228 @@ +#!/usr/bin/env python3 +"""Capture real rollouts into the fixtures the site replays. + +The site does no live inference. That is a deliberate architecture choice, not +a limitation: a public demo with no auth cannot hold an API key, a live call is +slow and flaky on conference wifi, and a recorded run can be scrubbed, +verified, permalinked and blind-compared in ways a live one cannot. + +What makes it honest rather than a video is that the browser re-derives every +number it shows from the recorded moves, and says so. See verify.ts. + +Usage: + uv run python envs/capture.py --arm base-off --seeds 0-7 + uv run python envs/capture.py --arm solver --seeds 0-7 # no model needed +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +import time +import urllib.error +import urllib.request +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parent / "wordle_five")) + +import numpy as np # noqa: E402 + +from wordle_five import solver as S # noqa: E402 +from wordle_five.engine import MAX_GUESSES, Game, answers # noqa: E402 +from wordle_five.protocol import ( # noqa: E402 + parse_guess, + render_feedback, + render_rejection, + system_prompt, +) +from wordle_five.reward import Episode, metrics, score # noqa: E402 + +OUT = Path(__file__).parent.parent / "public" / "traces" / "wordle" +ENDPOINT = os.environ.get("PIG_DEMO_INFERENCE", "http://100.127.247.67:8001/v1/chat/completions") +MODEL = os.environ.get("PIG_DEMO_MODEL", "brain-qwen38-dspark") + +ARMS = { + "base-off": {"label": "Out of the box", "thinking": False}, + "base-on": {"label": "Allowed to think", "thinking": True}, + "solver": {"label": "Best-known play", "thinking": None}, +} + + +def call_model(messages: list[dict], thinking: bool) -> dict: + """One completion. Returns reply, reasoning and the real call metrics. + + Never raises on an upstream failure — a dropped call becomes a turn with a + null reply, which the game scores as a rejected guess. A capture that + silently discarded failed turns would be reporting a better model than the + one that ran. + """ + body = { + "model": MODEL, + "messages": messages, + "temperature": 0.7, + "max_tokens": 2048 if thinking else 512, + "chat_template_kwargs": {"enable_thinking": bool(thinking)}, + } + request = urllib.request.Request( + ENDPOINT, + data=json.dumps(body).encode(), + headers={"Content-Type": "application/json"}, + ) + started = time.time() + try: + with urllib.request.urlopen(request, timeout=300) as response: + payload = json.loads(response.read()) + except (urllib.error.URLError, TimeoutError, OSError) as exc: + return { + "reply": None, + "reasoning": None, + "call": { + "promptTokens": None, + "completionTokens": None, + "reasoningTokens": None, + "durationMs": round((time.time() - started) * 1000), + "finishReason": f"error: {type(exc).__name__}", + }, + } + + elapsed = round((time.time() - started) * 1000) + choice = payload["choices"][0] + message = choice.get("message", {}) + usage = payload.get("usage", {}) or {} + details = usage.get("completion_tokens_details") or {} + return { + "reply": message.get("content"), + "reasoning": message.get("reasoning_content"), + "call": { + "promptTokens": usage.get("prompt_tokens"), + "completionTokens": usage.get("completion_tokens"), + "reasoningTokens": details.get("reasoning_tokens"), + "durationMs": elapsed, + "finishReason": choice.get("finish_reason"), + }, + } + + +def solver_turn(game: Game) -> dict: + """The reference player, recorded in the same shape as a model turn. + + durationMs is null rather than invented: nothing waited for this, and the + player must not pretend otherwise. The UI renders a null duration as an + instant step and labels the run as generated. + """ + pool = answers() + if not game.history: + guess = pool[S._best_opener()] + else: + alive = S.consistent_candidates(game.history) + guess = ( + pool[S.best_guess(np.array([pool.index(w) for w in alive]))] if alive else "tares" + ) + return { + "reply": f"[{guess}]", + "reasoning": None, + "call": { + "promptTokens": None, + "completionTokens": None, + "reasoningTokens": None, + "durationMs": None, + "finishReason": "generated", + }, + } + + +def capture(arm: str, seed: int) -> dict: + config = ARMS[arm] + game = Game(seed=seed) + messages = [ + {"role": "system", "content": system_prompt()}, + {"role": "user", "content": "Enter your guess to begin."}, + ] + turns: list[dict] = [] + + while not game.over and len(turns) < MAX_GUESSES * 2: + if arm == "solver": + result = solver_turn(game) + else: + result = call_model(messages, bool(config["thinking"])) + + guess = parse_guess(result["reply"]) + left = MAX_GUESSES - len(game.history) + + if guess is None: + game.rejected += 1 + observation = render_rejection("no bracketed guess found", left) + else: + pattern, rejection = game.play(guess) + observation = ( + render_rejection(rejection, left) + if rejection is not None + else render_feedback(guess, pattern or "", MAX_GUESSES - len(game.history)) + ) + + turns.append({**result, "info": {"guess": guess, "observation": observation}}) + messages.append({"role": "assistant", "content": result["reply"] or ""}) + messages.append({"role": "user", "content": observation}) + + if game.rejected >= MAX_GUESSES: + break + + episode = Episode( + answer=game.answer, + guesses=[g for g, _ in game.history], + patterns=[p for _, p in game.history], + rejected=game.rejected, + reference_depth=S.reference_depth(game.answer), + ) + # A run that never reached a terminal state is marked truncated, and the + # browser renders its verification as "unverifiable" rather than as a zero. + truncated = not game.over and game.rejected < MAX_GUESSES + + return { + "runId": f"{arm}-s{seed}", + "seed": seed, + "model": MODEL if arm != "solver" else "entropy-solver", + "capturedAt": time.strftime("%Y-%m-%d"), + "rewards": score(episode), + "metrics": metrics(episode), + "truncated": truncated, + "outcome": "solved" if game.solved else "failed", + "answer": game.answer, + "turns": turns, + } + + +def parse_seeds(spec: str) -> list[int]: + if "-" in spec: + lo, hi = spec.split("-") + return list(range(int(lo), int(hi) + 1)) + return [int(s) for s in spec.split(",")] + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--arm", required=True, choices=sorted(ARMS)) + parser.add_argument("--seeds", default="0-7") + args = parser.parse_args() + + OUT.mkdir(parents=True, exist_ok=True) + for seed in parse_seeds(args.seeds): + started = time.time() + episode = capture(args.arm, seed) + path = OUT / f"{episode['runId']}.json" + path.write_text(json.dumps(episode, indent=2) + "\n") + guesses = [t["info"]["guess"] for t in episode["turns"]] + print( + f"{episode['runId']:>16} {episode['answer']} {episode['outcome']:<7}" + f" solved={episode['rewards']['solved']:.0f}" + f" econ={episode['rewards']['economy']:.2f}" + f" cons={episode['rewards']['consistency']:.2f}" + f" {time.time()-started:5.1f}s {guesses}" + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/envs/verify_fixtures.py b/envs/verify_fixtures.py new file mode 100644 index 0000000..84e71ad --- /dev/null +++ b/envs/verify_fixtures.py @@ -0,0 +1,79 @@ +#!/usr/bin/env python3 +"""Replay every committed fixture and reproduce its recorded rewards. + +A fixture is a claim: "this model, on this seed, scored this." The claim is +only worth anything if it can be re-derived from the moves it records. This +does that in Python; `verify.ts` does the same thing in the browser, live, in +front of the visitor. + +Exits non-zero if any fixture's rewards cannot be reproduced from its own turns. +""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parent / "wordle_five")) + +from wordle_five import solver as S # noqa: E402 +from wordle_five.engine import Game # noqa: E402 +from wordle_five.protocol import parse_guess # noqa: E402 +from wordle_five.reward import Episode, score # noqa: E402 + +TRACES = Path(__file__).parent.parent / "public" / "traces" / "wordle" +TOLERANCE = 1e-9 + + +def replay(fixture: dict) -> dict[str, float]: + game = Game(seed=fixture["seed"]) + if game.answer != fixture["answer"]: + raise AssertionError( + f"seed {fixture['seed']} gives {game.answer!r}, fixture claims {fixture['answer']!r}" + ) + for turn in fixture["turns"]: + guess = parse_guess(turn["reply"]) + if guess is None: + game.rejected += 1 + continue + game.play(guess) + + episode = Episode( + answer=game.answer, + guesses=[g for g, _ in game.history], + patterns=[p for _, p in game.history], + rejected=game.rejected, + reference_depth=S.reference_depth(game.answer), + ) + return score(episode) + + +def main() -> int: + fixtures = sorted(TRACES.glob("*.json")) + fixtures = [f for f in fixtures if f.name != "manifest.json"] + if not fixtures: + print("no fixtures found — nothing to verify") + return 0 + + failures = 0 + for path in fixtures: + data = json.loads(path.read_text()) + recomputed = replay(data) + recorded = data["rewards"] + deltas = {k: abs(recomputed[k] - recorded[k]) for k in recorded} + worst = max(deltas.values()) + status = "ok" if worst <= TOLERANCE else "MISMATCH" + if worst > TOLERANCE: + failures += 1 + culprit = max(deltas, key=deltas.get) + print(f"{path.name:>20} {status} worst delta {worst:.3g} on '{culprit}'") + else: + print(f"{path.name:>20} {status} delta {worst:.1e}") + + print(f"\n{len(fixtures)} fixtures, {failures} mismatched") + return 1 if failures else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/public/traces/wordle/base-off-s0.json b/public/traces/wordle/base-off-s0.json new file mode 100644 index 0000000..a9ad3c9 --- /dev/null +++ b/public/traces/wordle/base-off-s0.json @@ -0,0 +1,143 @@ +{ + "runId": "base-off-s0", + "seed": 0, + "model": "brain-qwen38-dspark", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 0.0, + "economy": 0.0, + "consistency": 0.125 + }, + "metrics": { + "guesses_used": 6.0, + "rejected_replies": 2.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 5.0 + }, + "truncated": false, + "outcome": "failed", + "answer": "wants", + "turns": [ + { + "reply": "[crane]", + "reasoning": null, + "call": { + "promptTokens": 147, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 433, + "finishReason": "stop" + }, + "info": { + "guess": "crane", + "observation": "C R A N E\nX X Y Y X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[slate]", + "reasoning": null, + "call": { + "promptTokens": 184, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 442, + "finishReason": "stop" + }, + "info": { + "guess": "slate", + "observation": "S L A T E\nY X Y G X\n\nYou have 4 guesses left." + } + }, + { + "reply": "[stale]", + "reasoning": null, + "call": { + "promptTokens": 221, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 496, + "finishReason": "stop" + }, + "info": { + "guess": "stale", + "observation": "S T A L E\nY Y Y X X\n\nYou have 3 guesses left." + } + }, + { + "reply": "[prate]", + "reasoning": null, + "call": { + "promptTokens": 258, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 488, + "finishReason": "stop" + }, + "info": { + "guess": "prate", + "observation": "P R A T E\nX X Y G X\n\nYou have 2 guesses left." + } + }, + { + "reply": "[prate]", + "reasoning": null, + "call": { + "promptTokens": 295, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 443, + "finishReason": "stop" + }, + "info": { + "guess": "prate", + "observation": "That guess was not accepted: 'prate' has already been guessed.\nIt did not use up a turn. You have 2 guesses left." + } + }, + { + "reply": "[spate]", + "reasoning": null, + "call": { + "promptTokens": 344, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 495, + "finishReason": "stop" + }, + "info": { + "guess": "spate", + "observation": "S P A T E\nY X Y G X\n\nYou have 1 guess left." + } + }, + { + "reply": "[prate]", + "reasoning": null, + "call": { + "promptTokens": 381, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 445, + "finishReason": "stop" + }, + "info": { + "guess": "prate", + "observation": "That guess was not accepted: 'prate' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[stare]", + "reasoning": null, + "call": { + "promptTokens": 430, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 502, + "finishReason": "stop" + }, + "info": { + "guess": "stare", + "observation": "S T A R E\nY Y Y X X\n\nYou have 0 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/base-off-s1.json b/public/traces/wordle/base-off-s1.json new file mode 100644 index 0000000..54b27db --- /dev/null +++ b/public/traces/wordle/base-off-s1.json @@ -0,0 +1,113 @@ +{ + "runId": "base-off-s1", + "seed": 1, + "model": "brain-qwen38-dspark", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 0.0, + "economy": 0.0, + "consistency": 0.16666666666666666 + }, + "metrics": { + "guesses_used": 6.0, + "rejected_replies": 0.0, + "reference_depth": 3.0, + "turns_granted": 6.0, + "inconsistent_guesses": 5.0 + }, + "truncated": false, + "outcome": "failed", + "answer": "amber", + "turns": [ + { + "reply": "[crane]", + "reasoning": null, + "call": { + "promptTokens": 147, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 385, + "finishReason": "stop" + }, + "info": { + "guess": "crane", + "observation": "C R A N E\nX Y Y X Y\n\nYou have 5 guesses left." + } + }, + { + "reply": "[slate]", + "reasoning": null, + "call": { + "promptTokens": 184, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 443, + "finishReason": "stop" + }, + "info": { + "guess": "slate", + "observation": "S L A T E\nX X Y X Y\n\nYou have 4 guesses left." + } + }, + { + "reply": "[plume]", + "reasoning": null, + "call": { + "promptTokens": 221, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 499, + "finishReason": "stop" + }, + "info": { + "guess": "plume", + "observation": "P L U M E\nX X X Y Y\n\nYou have 3 guesses left." + } + }, + { + "reply": "[grape]", + "reasoning": null, + "call": { + "promptTokens": 258, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 488, + "finishReason": "stop" + }, + "info": { + "guess": "grape", + "observation": "G R A P E\nX Y Y X Y\n\nYou have 2 guesses left." + } + }, + { + "reply": "[stare]", + "reasoning": null, + "call": { + "promptTokens": 295, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 439, + "finishReason": "stop" + }, + "info": { + "guess": "stare", + "observation": "S T A R E\nX X Y Y Y\n\nYou have 1 guess left." + } + }, + { + "reply": "[crate]", + "reasoning": null, + "call": { + "promptTokens": 332, + "completionTokens": 4, + "reasoningTokens": null, + "durationMs": 493, + "finishReason": "stop" + }, + "info": { + "guess": "crate", + "observation": "C R A T E\nX Y Y X Y\n\nYou have 0 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/base-off-s2.json b/public/traces/wordle/base-off-s2.json new file mode 100644 index 0000000..35a8f80 --- /dev/null +++ b/public/traces/wordle/base-off-s2.json @@ -0,0 +1,188 @@ +{ + "runId": "base-off-s2", + "seed": 2, + "model": "brain-qwen38-dspark", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 0.0, + "economy": 0.0, + "consistency": 0.09090909090909091 + }, + "metrics": { + "guesses_used": 5.0, + "rejected_replies": 6.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 5.0 + }, + "truncated": false, + "outcome": "failed", + "answer": "spume", + "turns": [ + { + "reply": "[crane]", + "reasoning": null, + "call": { + "promptTokens": 147, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 301, + "finishReason": "stop" + }, + "info": { + "guess": "crane", + "observation": "C R A N E\nX X X X G\n\nYou have 5 guesses left." + } + }, + { + "reply": "[shine]", + "reasoning": null, + "call": { + "promptTokens": 184, + "completionTokens": 4, + "reasoningTokens": null, + "durationMs": 444, + "finishReason": "stop" + }, + "info": { + "guess": "shine", + "observation": "S H I N E\nG X X X G\n\nYou have 4 guesses left." + } + }, + { + "reply": "[shone]", + "reasoning": null, + "call": { + "promptTokens": 220, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 496, + "finishReason": "stop" + }, + "info": { + "guess": "shone", + "observation": "S H O N E\nG X X X G\n\nYou have 3 guesses left." + } + }, + { + "reply": "[swine]", + "reasoning": null, + "call": { + "promptTokens": 257, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 606, + "finishReason": "stop" + }, + "info": { + "guess": "swine", + "observation": "S W I N E\nG X X X G\n\nYou have 2 guesses left." + } + }, + { + "reply": "[shine]", + "reasoning": null, + "call": { + "promptTokens": 294, + "completionTokens": 4, + "reasoningTokens": null, + "durationMs": 439, + "finishReason": "stop" + }, + "info": { + "guess": "shine", + "observation": "That guess was not accepted: 'shine' has already been guessed.\nIt did not use up a turn. You have 2 guesses left." + } + }, + { + "reply": "[spine]", + "reasoning": null, + "call": { + "promptTokens": 341, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 384, + "finishReason": "stop" + }, + "info": { + "guess": "spine", + "observation": "S P I N E\nG G X X G\n\nYou have 1 guess left." + } + }, + { + "reply": "[spine]", + "reasoning": null, + "call": { + "promptTokens": 378, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 331, + "finishReason": "stop" + }, + "info": { + "guess": "spine", + "observation": "That guess was not accepted: 'spine' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[spine]", + "reasoning": null, + "call": { + "promptTokens": 427, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 389, + "finishReason": "stop" + }, + "info": { + "guess": "spine", + "observation": "That guess was not accepted: 'spine' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[spine]", + "reasoning": null, + "call": { + "promptTokens": 476, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 386, + "finishReason": "stop" + }, + "info": { + "guess": "spine", + "observation": "That guess was not accepted: 'spine' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[spine]", + "reasoning": null, + "call": { + "promptTokens": 525, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 379, + "finishReason": "stop" + }, + "info": { + "guess": "spine", + "observation": "That guess was not accepted: 'spine' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[spine]", + "reasoning": null, + "call": { + "promptTokens": 574, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 334, + "finishReason": "stop" + }, + "info": { + "guess": "spine", + "observation": "That guess was not accepted: 'spine' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + } + ] +} diff --git a/public/traces/wordle/base-off-s3.json b/public/traces/wordle/base-off-s3.json new file mode 100644 index 0000000..6f20880 --- /dev/null +++ b/public/traces/wordle/base-off-s3.json @@ -0,0 +1,188 @@ +{ + "runId": "base-off-s3", + "seed": 3, + "model": "brain-qwen38-dspark", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 0.0, + "economy": 0.0, + "consistency": 0.09090909090909091 + }, + "metrics": { + "guesses_used": 5.0, + "rejected_replies": 6.0, + "reference_depth": 3.0, + "turns_granted": 6.0, + "inconsistent_guesses": 5.0 + }, + "truncated": false, + "outcome": "failed", + "answer": "toady", + "turns": [ + { + "reply": "[crane]", + "reasoning": null, + "call": { + "promptTokens": 147, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 393, + "finishReason": "stop" + }, + "info": { + "guess": "crane", + "observation": "C R A N E\nX X G X X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[slate]", + "reasoning": null, + "call": { + "promptTokens": 184, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 557, + "finishReason": "stop" + }, + "info": { + "guess": "slate", + "observation": "S L A T E\nX X G Y X\n\nYou have 4 guesses left." + } + }, + { + "reply": "[plate]", + "reasoning": null, + "call": { + "promptTokens": 221, + "completionTokens": 4, + "reasoningTokens": null, + "durationMs": 620, + "finishReason": "stop" + }, + "info": { + "guess": "plate", + "observation": "P L A T E\nX X G Y X\n\nYou have 3 guesses left." + } + }, + { + "reply": "[crate]", + "reasoning": null, + "call": { + "promptTokens": 257, + "completionTokens": 4, + "reasoningTokens": null, + "durationMs": 9070, + "finishReason": "stop" + }, + "info": { + "guess": "crate", + "observation": "C R A T E\nX X G Y X\n\nYou have 2 guesses left." + } + }, + { + "reply": "[plate]", + "reasoning": null, + "call": { + "promptTokens": 293, + "completionTokens": 4, + "reasoningTokens": null, + "durationMs": 1037, + "finishReason": "stop" + }, + "info": { + "guess": "plate", + "observation": "That guess was not accepted: 'plate' has already been guessed.\nIt did not use up a turn. You have 2 guesses left." + } + }, + { + "reply": "[trape]", + "reasoning": null, + "call": { + "promptTokens": 340, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 1620, + "finishReason": "stop" + }, + "info": { + "guess": "trape", + "observation": "T R A P E\nG X G X X\n\nYou have 1 guess left." + } + }, + { + "reply": "[trape]", + "reasoning": null, + "call": { + "promptTokens": 377, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 985, + "finishReason": "stop" + }, + "info": { + "guess": "trape", + "observation": "That guess was not accepted: 'trape' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[trape]", + "reasoning": null, + "call": { + "promptTokens": 426, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 1105, + "finishReason": "stop" + }, + "info": { + "guess": "trape", + "observation": "That guess was not accepted: 'trape' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[trape]", + "reasoning": null, + "call": { + "promptTokens": 475, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 1094, + "finishReason": "stop" + }, + "info": { + "guess": "trape", + "observation": "That guess was not accepted: 'trape' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[trape]", + "reasoning": null, + "call": { + "promptTokens": 524, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 1349, + "finishReason": "stop" + }, + "info": { + "guess": "trape", + "observation": "That guess was not accepted: 'trape' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[trape]", + "reasoning": null, + "call": { + "promptTokens": 573, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 983, + "finishReason": "stop" + }, + "info": { + "guess": "trape", + "observation": "That guess was not accepted: 'trape' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + } + ] +} diff --git a/public/traces/wordle/base-off-s4.json b/public/traces/wordle/base-off-s4.json new file mode 100644 index 0000000..e5cc6d1 --- /dev/null +++ b/public/traces/wordle/base-off-s4.json @@ -0,0 +1,188 @@ +{ + "runId": "base-off-s4", + "seed": 4, + "model": "brain-qwen38-dspark", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 0.0, + "economy": 0.0, + "consistency": 0.09090909090909091 + }, + "metrics": { + "guesses_used": 5.0, + "rejected_replies": 6.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 5.0 + }, + "truncated": false, + "outcome": "failed", + "answer": "divot", + "turns": [ + { + "reply": "[crane]", + "reasoning": null, + "call": { + "promptTokens": 147, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 952, + "finishReason": "stop" + }, + "info": { + "guess": "crane", + "observation": "C R A N E\nX X X X X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[slith]", + "reasoning": null, + "call": { + "promptTokens": 184, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 1501, + "finishReason": "stop" + }, + "info": { + "guess": "slith", + "observation": "That guess was not accepted: 'slith' is not in the word list.\nIt did not use up a turn. You have 5 guesses left." + } + }, + { + "reply": "[slant]", + "reasoning": null, + "call": { + "promptTokens": 235, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 1363, + "finishReason": "stop" + }, + "info": { + "guess": "slant", + "observation": "S L A N T\nX X X X G\n\nYou have 4 guesses left." + } + }, + { + "reply": "[roast]", + "reasoning": null, + "call": { + "promptTokens": 272, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 1609, + "finishReason": "stop" + }, + "info": { + "guess": "roast", + "observation": "R O A S T\nX Y X X G\n\nYou have 3 guesses left." + } + }, + { + "reply": "[toast]", + "reasoning": null, + "call": { + "promptTokens": 309, + "completionTokens": 4, + "reasoningTokens": null, + "durationMs": 1240, + "finishReason": "stop" + }, + "info": { + "guess": "toast", + "observation": "T O A S T\nX Y X X G\n\nYou have 2 guesses left." + } + }, + { + "reply": "[plout]", + "reasoning": null, + "call": { + "promptTokens": 345, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 1613, + "finishReason": "stop" + }, + "info": { + "guess": "plout", + "observation": "That guess was not accepted: 'plout' is not in the word list.\nIt did not use up a turn. You have 2 guesses left." + } + }, + { + "reply": "[postt]", + "reasoning": null, + "call": { + "promptTokens": 396, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 1349, + "finishReason": "stop" + }, + "info": { + "guess": "postt", + "observation": "That guess was not accepted: 'postt' is not in the word list.\nIt did not use up a turn. You have 2 guesses left." + } + }, + { + "reply": "[roost]", + "reasoning": null, + "call": { + "promptTokens": 447, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 868, + "finishReason": "stop" + }, + "info": { + "guess": "roost", + "observation": "R O O S T\nX Y X X G\n\nYou have 1 guess left." + } + }, + { + "reply": "[plout]", + "reasoning": null, + "call": { + "promptTokens": 484, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 946, + "finishReason": "stop" + }, + "info": { + "guess": "plout", + "observation": "That guess was not accepted: 'plout' is not in the word list.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[toast]", + "reasoning": null, + "call": { + "promptTokens": 535, + "completionTokens": 4, + "reasoningTokens": null, + "durationMs": 498, + "finishReason": "stop" + }, + "info": { + "guess": "toast", + "observation": "That guess was not accepted: 'toast' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[most]", + "reasoning": null, + "call": { + "promptTokens": 582, + "completionTokens": 4, + "reasoningTokens": null, + "durationMs": 490, + "finishReason": "stop" + }, + "info": { + "guess": "most", + "observation": "That guess was not accepted: 'most' is not 5 letters.\nIt did not use up a turn. You have 1 guess left." + } + } + ] +} diff --git a/public/traces/wordle/base-off-s5.json b/public/traces/wordle/base-off-s5.json new file mode 100644 index 0000000..9659f60 --- /dev/null +++ b/public/traces/wordle/base-off-s5.json @@ -0,0 +1,143 @@ +{ + "runId": "base-off-s5", + "seed": 5, + "model": "brain-qwen38-dspark", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 0.0, + "economy": 0.0, + "consistency": 0.125 + }, + "metrics": { + "guesses_used": 6.0, + "rejected_replies": 2.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 5.0 + }, + "truncated": false, + "outcome": "failed", + "answer": "filly", + "turns": [ + { + "reply": "[slate]", + "reasoning": null, + "call": { + "promptTokens": 147, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 392, + "finishReason": "stop" + }, + "info": { + "guess": "slate", + "observation": "S L A T E\nX Y X X X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[crane]", + "reasoning": null, + "call": { + "promptTokens": 184, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 555, + "finishReason": "stop" + }, + "info": { + "guess": "crane", + "observation": "C R A N E\nX X X X X\n\nYou have 4 guesses left." + } + }, + { + "reply": "[plaid]", + "reasoning": null, + "call": { + "promptTokens": 221, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 496, + "finishReason": "stop" + }, + "info": { + "guess": "plaid", + "observation": "P L A I D\nX Y X Y X\n\nYou have 3 guesses left." + } + }, + { + "reply": "[blint]", + "reasoning": null, + "call": { + "promptTokens": 258, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 491, + "finishReason": "stop" + }, + "info": { + "guess": "blint", + "observation": "That guess was not accepted: 'blint' is not in the word list.\nIt did not use up a turn. You have 3 guesses left." + } + }, + { + "reply": "[plaid]", + "reasoning": null, + "call": { + "promptTokens": 309, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 443, + "finishReason": "stop" + }, + "info": { + "guess": "plaid", + "observation": "That guess was not accepted: 'plaid' has already been guessed.\nIt did not use up a turn. You have 3 guesses left." + } + }, + { + "reply": "[plume]", + "reasoning": null, + "call": { + "promptTokens": 358, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 499, + "finishReason": "stop" + }, + "info": { + "guess": "plume", + "observation": "P L U M E\nX Y X X X\n\nYou have 2 guesses left." + } + }, + { + "reply": "[climb]", + "reasoning": null, + "call": { + "promptTokens": 395, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 603, + "finishReason": "stop" + }, + "info": { + "guess": "climb", + "observation": "C L I M B\nX Y Y X X\n\nYou have 1 guess left." + } + }, + { + "reply": "[slimy]", + "reasoning": null, + "call": { + "promptTokens": 432, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 555, + "finishReason": "stop" + }, + "info": { + "guess": "slimy", + "observation": "S L I M Y\nX Y Y X G\n\nYou have 0 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/base-off-s6.json b/public/traces/wordle/base-off-s6.json new file mode 100644 index 0000000..ae41cbf --- /dev/null +++ b/public/traces/wordle/base-off-s6.json @@ -0,0 +1,128 @@ +{ + "runId": "base-off-s6", + "seed": 6, + "model": "brain-qwen38-dspark", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 0.0, + "economy": 0.0, + "consistency": 0.14285714285714285 + }, + "metrics": { + "guesses_used": 6.0, + "rejected_replies": 1.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 5.0 + }, + "truncated": false, + "outcome": "failed", + "answer": "bobby", + "turns": [ + { + "reply": "[crane]", + "reasoning": null, + "call": { + "promptTokens": 147, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 381, + "finishReason": "stop" + }, + "info": { + "guess": "crane", + "observation": "C R A N E\nX X X X X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[slite]", + "reasoning": null, + "call": { + "promptTokens": 184, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 555, + "finishReason": "stop" + }, + "info": { + "guess": "slite", + "observation": "That guess was not accepted: 'slite' is not in the word list.\nIt did not use up a turn. You have 5 guesses left." + } + }, + { + "reply": "[slate]", + "reasoning": null, + "call": { + "promptTokens": 235, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 503, + "finishReason": "stop" + }, + "info": { + "guess": "slate", + "observation": "S L A T E\nX X X X X\n\nYou have 4 guesses left." + } + }, + { + "reply": "[robin]", + "reasoning": null, + "call": { + "promptTokens": 272, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 607, + "finishReason": "stop" + }, + "info": { + "guess": "robin", + "observation": "R O B I N\nX G G X X\n\nYou have 3 guesses left." + } + }, + { + "reply": "[boomy]", + "reasoning": null, + "call": { + "promptTokens": 309, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 555, + "finishReason": "stop" + }, + "info": { + "guess": "boomy", + "observation": "B O O M Y\nG G X X G\n\nYou have 2 guesses left." + } + }, + { + "reply": "[booth]", + "reasoning": null, + "call": { + "promptTokens": 346, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 385, + "finishReason": "stop" + }, + "info": { + "guess": "booth", + "observation": "B O O T H\nG G X X X\n\nYou have 1 guess left." + } + }, + { + "reply": "[booby]", + "reasoning": null, + "call": { + "promptTokens": 383, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 446, + "finishReason": "stop" + }, + "info": { + "guess": "booby", + "observation": "B O O B Y\nG G X G G\n\nYou have 0 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/base-off-s7.json b/public/traces/wordle/base-off-s7.json new file mode 100644 index 0000000..25fa596 --- /dev/null +++ b/public/traces/wordle/base-off-s7.json @@ -0,0 +1,128 @@ +{ + "runId": "base-off-s7", + "seed": 7, + "model": "brain-qwen38-dspark", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 0.0, + "economy": 0.0, + "consistency": 0.14285714285714285 + }, + "metrics": { + "guesses_used": 6.0, + "rejected_replies": 1.0, + "reference_depth": 3.0, + "turns_granted": 6.0, + "inconsistent_guesses": 5.0 + }, + "truncated": false, + "outcome": "failed", + "answer": "clews", + "turns": [ + { + "reply": "[crane]", + "reasoning": null, + "call": { + "promptTokens": 147, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 254, + "finishReason": "stop" + }, + "info": { + "guess": "crane", + "observation": "C R A N E\nG X X X Y\n\nYou have 5 guesses left." + } + }, + { + "reply": "[prize]", + "reasoning": null, + "call": { + "promptTokens": 184, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 557, + "finishReason": "stop" + }, + "info": { + "guess": "prize", + "observation": "P R I Z E\nX X X X Y\n\nYou have 4 guesses left." + } + }, + { + "reply": "[crepe]", + "reasoning": null, + "call": { + "promptTokens": 221, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 607, + "finishReason": "stop" + }, + "info": { + "guess": "crepe", + "observation": "C R E P E\nG X G X X\n\nYou have 3 guesses left." + } + }, + { + "reply": "[crave]", + "reasoning": null, + "call": { + "promptTokens": 258, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 488, + "finishReason": "stop" + }, + "info": { + "guess": "crave", + "observation": "C R A V E\nG X X X Y\n\nYou have 2 guesses left." + } + }, + { + "reply": "[credo]", + "reasoning": null, + "call": { + "promptTokens": 295, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 553, + "finishReason": "stop" + }, + "info": { + "guess": "credo", + "observation": "C R E D O\nG X G X X\n\nYou have 1 guess left." + } + }, + { + "reply": "[crepe]", + "reasoning": null, + "call": { + "promptTokens": 332, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 492, + "finishReason": "stop" + }, + "info": { + "guess": "crepe", + "observation": "That guess was not accepted: 'crepe' has already been guessed.\nIt did not use up a turn. You have 1 guess left." + } + }, + { + "reply": "[creed]", + "reasoning": null, + "call": { + "promptTokens": 381, + "completionTokens": 5, + "reasoningTokens": null, + "durationMs": 444, + "finishReason": "stop" + }, + "info": { + "guess": "creed", + "observation": "C R E E D\nG X G X X\n\nYou have 0 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/solver-s0.json b/public/traces/wordle/solver-s0.json new file mode 100644 index 0000000..e12f078 --- /dev/null +++ b/public/traces/wordle/solver-s0.json @@ -0,0 +1,83 @@ +{ + "runId": "solver-s0", + "seed": 0, + "model": "entropy-solver", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 1.0, + "economy": 1.0, + "consistency": 0.5 + }, + "metrics": { + "guesses_used": 4.0, + "rejected_replies": 0.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 2.0 + }, + "truncated": false, + "outcome": "solved", + "answer": "wants", + "turns": [ + { + "reply": "[tares]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "tares", + "observation": "T A R E S\nY G X X G\n\nYou have 5 guesses left." + } + }, + { + "reply": "[filch]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "filch", + "observation": "F I L C H\nX X X X X\n\nYou have 4 guesses left." + } + }, + { + "reply": "[spams]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "spams", + "observation": "S P A M S\nX X Y X G\n\nYou have 3 guesses left." + } + }, + { + "reply": "[wants]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "wants", + "observation": "W A N T S\nG G G G G\n\nYou have 2 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/solver-s1.json b/public/traces/wordle/solver-s1.json new file mode 100644 index 0000000..a5bafd6 --- /dev/null +++ b/public/traces/wordle/solver-s1.json @@ -0,0 +1,68 @@ +{ + "runId": "solver-s1", + "seed": 1, + "model": "entropy-solver", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 1.0, + "economy": 1.0, + "consistency": 0.6666666666666666 + }, + "metrics": { + "guesses_used": 3.0, + "rejected_replies": 0.0, + "reference_depth": 3.0, + "turns_granted": 6.0, + "inconsistent_guesses": 1.0 + }, + "truncated": false, + "outcome": "solved", + "answer": "amber", + "turns": [ + { + "reply": "[tares]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "tares", + "observation": "T A R E S\nX Y Y G X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[blend]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "blend", + "observation": "B L E N D\nY X Y X X\n\nYou have 4 guesses left." + } + }, + { + "reply": "[amber]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "amber", + "observation": "A M B E R\nG G G G G\n\nYou have 3 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/solver-s2.json b/public/traces/wordle/solver-s2.json new file mode 100644 index 0000000..34052eb --- /dev/null +++ b/public/traces/wordle/solver-s2.json @@ -0,0 +1,83 @@ +{ + "runId": "solver-s2", + "seed": 2, + "model": "entropy-solver", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 1.0, + "economy": 1.0, + "consistency": 0.75 + }, + "metrics": { + "guesses_used": 4.0, + "rejected_replies": 0.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 1.0 + }, + "truncated": false, + "outcome": "solved", + "answer": "spume", + "turns": [ + { + "reply": "[tares]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "tares", + "observation": "T A R E S\nX X X Y Y\n\nYou have 5 guesses left." + } + }, + { + "reply": "[spoil]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "spoil", + "observation": "S P O I L\nG G X X X\n\nYou have 4 guesses left." + } + }, + { + "reply": "[speck]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "speck", + "observation": "S P E C K\nG G Y X X\n\nYou have 3 guesses left." + } + }, + { + "reply": "[spume]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "spume", + "observation": "S P U M E\nG G G G G\n\nYou have 2 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/solver-s3.json b/public/traces/wordle/solver-s3.json new file mode 100644 index 0000000..8e7deee --- /dev/null +++ b/public/traces/wordle/solver-s3.json @@ -0,0 +1,68 @@ +{ + "runId": "solver-s3", + "seed": 3, + "model": "entropy-solver", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 1.0, + "economy": 1.0, + "consistency": 0.6666666666666666 + }, + "metrics": { + "guesses_used": 3.0, + "rejected_replies": 0.0, + "reference_depth": 3.0, + "turns_granted": 6.0, + "inconsistent_guesses": 1.0 + }, + "truncated": false, + "outcome": "solved", + "answer": "toady", + "turns": [ + { + "reply": "[tares]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "tares", + "observation": "T A R E S\nG Y X X X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[bland]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "bland", + "observation": "B L A N D\nX X G X Y\n\nYou have 4 guesses left." + } + }, + { + "reply": "[toady]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "toady", + "observation": "T O A D Y\nG G G G G\n\nYou have 3 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/solver-s4.json b/public/traces/wordle/solver-s4.json new file mode 100644 index 0000000..efd0779 --- /dev/null +++ b/public/traces/wordle/solver-s4.json @@ -0,0 +1,83 @@ +{ + "runId": "solver-s4", + "seed": 4, + "model": "entropy-solver", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 1.0, + "economy": 1.0, + "consistency": 1.0 + }, + "metrics": { + "guesses_used": 4.0, + "rejected_replies": 0.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 0.0 + }, + "truncated": false, + "outcome": "solved", + "answer": "divot", + "turns": [ + { + "reply": "[tares]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "tares", + "observation": "T A R E S\nY X X X X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[count]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "count", + "observation": "C O U N T\nX Y X X G\n\nYou have 4 guesses left." + } + }, + { + "reply": "[pivot]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "pivot", + "observation": "P I V O T\nX G G G G\n\nYou have 3 guesses left." + } + }, + { + "reply": "[divot]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "divot", + "observation": "D I V O T\nG G G G G\n\nYou have 2 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/solver-s5.json b/public/traces/wordle/solver-s5.json new file mode 100644 index 0000000..28b8efe --- /dev/null +++ b/public/traces/wordle/solver-s5.json @@ -0,0 +1,83 @@ +{ + "runId": "solver-s5", + "seed": 5, + "model": "entropy-solver", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 1.0, + "economy": 1.0, + "consistency": 0.75 + }, + "metrics": { + "guesses_used": 4.0, + "rejected_replies": 0.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 1.0 + }, + "truncated": false, + "outcome": "solved", + "answer": "filly", + "turns": [ + { + "reply": "[tares]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "tares", + "observation": "T A R E S\nX X X X X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[doily]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "doily", + "observation": "D O I L Y\nX X Y G G\n\nYou have 4 guesses left." + } + }, + { + "reply": "[belch]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "belch", + "observation": "B E L C H\nX X G X X\n\nYou have 3 guesses left." + } + }, + { + "reply": "[filly]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "filly", + "observation": "F I L L Y\nG G G G G\n\nYou have 2 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/solver-s6.json b/public/traces/wordle/solver-s6.json new file mode 100644 index 0000000..6331832 --- /dev/null +++ b/public/traces/wordle/solver-s6.json @@ -0,0 +1,83 @@ +{ + "runId": "solver-s6", + "seed": 6, + "model": "entropy-solver", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 1.0, + "economy": 1.0, + "consistency": 0.75 + }, + "metrics": { + "guesses_used": 4.0, + "rejected_replies": 0.0, + "reference_depth": 4.0, + "turns_granted": 6.0, + "inconsistent_guesses": 1.0 + }, + "truncated": false, + "outcome": "solved", + "answer": "bobby", + "turns": [ + { + "reply": "[tares]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "tares", + "observation": "T A R E S\nX X X X X\n\nYou have 5 guesses left." + } + }, + { + "reply": "[doily]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "doily", + "observation": "D O I L Y\nX G X X G\n\nYou have 4 guesses left." + } + }, + { + "reply": "[knobs]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "knobs", + "observation": "K N O B S\nX X Y G X\n\nYou have 3 guesses left." + } + }, + { + "reply": "[bobby]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "bobby", + "observation": "B O B B Y\nG G G G G\n\nYou have 2 guesses left." + } + } + ] +} diff --git a/public/traces/wordle/solver-s7.json b/public/traces/wordle/solver-s7.json new file mode 100644 index 0000000..a0bd86a --- /dev/null +++ b/public/traces/wordle/solver-s7.json @@ -0,0 +1,68 @@ +{ + "runId": "solver-s7", + "seed": 7, + "model": "entropy-solver", + "capturedAt": "2026-08-28", + "rewards": { + "solved": 1.0, + "economy": 1.0, + "consistency": 0.6666666666666666 + }, + "metrics": { + "guesses_used": 3.0, + "rejected_replies": 0.0, + "reference_depth": 3.0, + "turns_granted": 6.0, + "inconsistent_guesses": 1.0 + }, + "truncated": false, + "outcome": "solved", + "answer": "clews", + "turns": [ + { + "reply": "[tares]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "tares", + "observation": "T A R E S\nX X X Y G\n\nYou have 5 guesses left." + } + }, + { + "reply": "[wield]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "wield", + "observation": "W I E L D\nY X G Y X\n\nYou have 4 guesses left." + } + }, + { + "reply": "[clews]", + "reasoning": null, + "call": { + "promptTokens": null, + "completionTokens": null, + "reasoningTokens": null, + "durationMs": null, + "finishReason": "generated" + }, + "info": { + "guess": "clews", + "observation": "C L E W S\nG G G G G\n\nYou have 3 guesses left." + } + } + ] +} diff --git a/scripts/_lib.mjs b/scripts/_lib.mjs index eec39d0..bb3a75b 100644 --- a/scripts/_lib.mjs +++ b/scripts/_lib.mjs @@ -294,7 +294,7 @@ function codeSpans(src) { * @param {'{' | '['} open * @returns {string | null} */ -export function literalAfter(src, anchor, open = '{') { +export function literalsAfter(src, anchor, open = '{') { const flags = anchor.flags.includes('g') ? anchor.flags : `${anchor.flags}g`; const re = new RegExp(anchor.source, flags); const spans = codeSpans(src); @@ -302,18 +302,29 @@ export function literalAfter(src, anchor, open = '{') { const hit = spans.find((s) => offset >= s.start && offset < s.end); return hit ? hit.code : true; }; + const found = []; let match; while ((match = re.exec(src)) !== null) { - // The anchor may legally match inside a comment or a string — a docblock + // The anchor may legally match inside a comment or a string \u2014 a docblock // that quotes `const meta = {`. Only a match in real code counts. if (!isCode(match.index)) continue; const from = src.indexOf(open, match.index + Math.max(match[0].length - 1, 0)); if (from === -1) continue; const between = src.slice(match.index + match[0].length, from); if (/[;=}]/.test(between)) continue; - return src.slice(from, scanBalanced(src, from)); + const end = scanBalanced(src, from); + found.push({ text: src.slice(from, end), start: from, end }); + // Skip past this literal so a nested anchor of the same name is not + // reported as a second, overlapping hit. + re.lastIndex = end; } - return null; + return found; +} + +/** First literal following `anchor`, or null. See {@link literalsAfter}. */ +export function literalAfter(src, anchor, open = '{') { + const [first] = literalsAfter(src, anchor, open); + return first ? first.text : null; } /** Marker key on the stand-in an unresolvable identifier evaluates to. */ @@ -501,7 +512,10 @@ const VERTICAL_ANCHORS = [ const asVertical = (id, v) => ({ id, title: v?.title ?? v?.label ?? v?.name ?? null, - description: v?.description ?? v?.blurb ?? v?.tagline ?? v?.summary ?? null, + // `anxiety` is this repo's field for the one line a vertical page leads + // with, and it is the only prose in a VerticalEntry short enough to be a + // meta description. + description: v?.description ?? v?.blurb ?? v?.tagline ?? v?.summary ?? v?.anxiety ?? null, }); /** @@ -559,8 +573,14 @@ export function loadManifest() { } /** @type {Map}>} */ const byDemo = new Map(); - const container = - raw && typeof raw === 'object' && !Array.isArray(raw) && raw.demos && typeof raw.demos === 'object' ? raw.demos : raw; + // Unwrap the two envelopes the shell's own reader accepts. `{runs: [...]}` + // has to be unwrapped BEFORE the record branch below, or "runs" is read as a + // demo slug and every real demo reports zero runs. + let container = raw; + if (raw && typeof raw === 'object' && !Array.isArray(raw)) { + if (raw.demos && typeof raw.demos === 'object') container = raw.demos; + else if (Array.isArray(raw.runs)) container = raw.runs; + } if (Array.isArray(container)) { // A flat array of runs, each carrying its own `demo`/`slug`. @@ -706,12 +726,32 @@ export function discoverRoutePatterns() { return [...patterns]; } -const DEMO_PARAMS = /^(slug|demo|demoSlug)$/; -const VERTICAL_PARAMS = /^(vertical|verticalId|sector|category)$/; +/** + * Which set fills a pattern's one parameter. + * + * Decided by the STATIC segment in front of it, not by the parameter's name. + * This router calls both of them `:slug` — `demos/:slug` and `verticals/:slug` + * — so a name-based reader confidently prerenders `/verticals/wordle`, twelve + * 404s, and no vertical pages at all. + */ +function fillFor(pattern) { + const head = pattern.replace(/^\/+/, '').split('/')[0]?.toLowerCase() ?? ''; + if (/^(demos?|d)$/.test(head)) return 'demo'; + if (/^(verticals?|v|sectors?)$/.test(head)) return 'vertical'; + const param = pattern.match(/:([A-Za-z0-9_]+)/)?.[1] ?? ''; + if (/^(vertical|verticalId|sector|category)$/.test(param)) return 'vertical'; + if (/^(slug|demo|demoSlug|id)$/.test(param)) return 'demo'; + return null; +} /** * Expands the router's patterns against the real data into concrete paths. * + * Child paths are taken as written. This router nests everything exactly one + * level under `/`, so a child's `path` is already the full path; a second level + * of nesting would need the parent prefix joined on, and this would silently + * emit the wrong routes rather than fail. If you nest deeper, fix this. + * * @param {{metas: Map, verticals: {id: string}[]}} data * @returns {{routes: {path: string, kind: string, slug?: string, id?: string}[], errors: string[]}} */ @@ -749,18 +789,18 @@ export function expandRoutes(data) { errors.push(`route pattern "${pattern}" has more than one parameter; prerender cannot expand it.`); continue; } - const param = params[0]; - const fill = (value, record) => add(pattern.replace(/:[A-Za-z0-9_]+\??/, value), record); - if (VERTICAL_PARAMS.test(param)) { - if (!verticalIds.length) errors.push(`route "${pattern}" needs verticals, but none were readable from src/content/verticals.ts.`); - for (const id of verticalIds) fill(id, { kind: 'vertical', id }); - } else if (DEMO_PARAMS.test(param) || param === 'id') { + const fill = fillFor(pattern); + const substitute = (value) => pattern.replace(/:[A-Za-z0-9_]+\??/, value); + if (fill === 'vertical') { + if (!verticalIds.length) errors.push(`route "${pattern}" needs verticals, but none were readable from ${rel(VERTICALS_FILE)}.`); + for (const id of verticalIds) add(substitute(id), { kind: 'vertical', id }); + } else if (fill === 'demo') { if (!demoSlugList.length) errors.push(`route "${pattern}" needs demos, but no demo meta was readable under src/demos/.`); - for (const slug of demoSlugList) fill(slug, { kind: 'demo', slug }); + for (const slug of demoSlugList) add(substitute(slug), { kind: 'demo', slug }); } else { errors.push( - `route pattern "${pattern}" uses parameter ":${param}", which prerender cannot fill. ` + - 'Name it :slug (a demo) or :vertical (a vertical), or teach scripts/_lib.mjs about it.', + `route pattern "${pattern}" has a parameter prerender cannot fill. Put it under /demos/ or ` + + '/verticals/, or teach fillFor() in scripts/_lib.mjs about it.', ); } } diff --git a/scripts/check-demos.mjs b/scripts/check-demos.mjs new file mode 100644 index 0000000..b0ff167 --- /dev/null +++ b/scripts/check-demos.mjs @@ -0,0 +1,597 @@ +#!/usr/bin/env node +/** + * The demo contract validator. + * + * `src/lib/demo-kit/types.ts` says what a demo is. TypeScript enforces the + * shapes; this enforces everything a type cannot: that the directory name + * matches the slug, that a `spec` demo is a real published specification rather + * than a coming-soon card, that a `live` demo's traces exist on disk, that the + * shared shell has no idea any particular demo exists, and that the reward has + * something pulling against its objective. + * + * Thirteen numbered rules, each reported with the file to open. Run it with + * `pnpm check`. + * + * Most rules are checked by READING the TypeScript, not by running it — the + * demos import React, `?raw` Python and the Vite `@/` alias, none of which + * survive a bare `node` import. Rule 13 is the exception: it tries the adapter + * for real under tsx first, and says so in the output when it had to settle for + * reading the source. + */ + +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { spawnSync } from 'node:child_process'; + +import { + DEMOS_DIR, + Report, + abs, + allDemoDirs, + demoFiles, + demoSlugs, + die, + exists, + findInDemo, + isUnresolved, + loadManifest, + loadMeta, + parseStringUnion, + read, + rel, + traceFile, + walk, +} from './_lib.mjs'; + +const report = new Report('check-demos'); + +/* ------------------------------------------------ the contract, read once */ + +const TYPES_FILE = abs('src', 'lib', 'demo-kit', 'types.ts'); +if (!exists(TYPES_FILE)) die(`${rel(TYPES_FILE)} is missing. It is the contract; there is nothing to check against.`); +const typesSrc = read(TYPES_FILE); + +// Parsed out of types.ts rather than restated here. A second copy of this list +// would drift, and it would drift silently in the direction of passing. +const VERTICALS = parseStringUnion(typesSrc, 'Vertical'); +const STATUSES = parseStringUnion(typesSrc, 'DemoStatus'); +const ROLES = ['objective', 'counterweight', 'gate']; +if (!VERTICALS) die(`could not parse the \`Vertical\` union out of ${rel(TYPES_FILE)}.`); +if (!STATUSES) die(`could not parse the \`DemoStatus\` union out of ${rel(TYPES_FILE)}.`); + +/** + * Every icon name lucide-react actually exports, from its own type + * declarations. "Plausible PascalCase" would accept `Grid3X3`, which compiles + * to `undefined` and renders as a hole in the header. + */ +const LUCIDE_ICONS = (() => { + const dts = abs('node_modules', 'lucide-react', 'dist', 'lucide-react.d.ts'); + if (!exists(dts)) return null; + const names = new Set(); + for (const m of read(dts).matchAll(/declare const ([A-Za-z][A-Za-z0-9_]*)\s*:/g)) names.add(m[1]); + return names.size > 100 ? names : null; +})(); + +const nonEmptyString = (v) => typeof v === 'string' && v.trim().length > 0; + +/** field -> validator returning `true` or the reason it is wrong. */ +const META_FIELDS = { + slug: (v) => (/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(String(v)) ? true : 'must be a non-empty kebab-case string'), + title: (v) => (nonEmptyString(v) ? true : 'must be a non-empty string'), + tagline: (v) => (nonEmptyString(v) ? true : 'must be a non-empty string'), + vertical: (v) => (VERTICALS.includes(v) ? true : `must be one of the Vertical union: ${VERTICALS.join(', ')}`), + status: (v) => (STATUSES.includes(v) ? true : `must be one of the DemoStatus union: ${STATUSES.join(', ')}`), + order: (v) => (typeof v === 'number' && Number.isFinite(v) ? true : 'must be a finite number'), + icon: (v) => { + if (!nonEmptyString(v)) return 'must be a non-empty string'; + if (!/^[A-Z][A-Za-z0-9]*$/.test(v)) return `"${v}" is not PascalCase, so it is not a lucide export name`; + if (LUCIDE_ICONS && !LUCIDE_ICONS.has(v)) return `lucide-react does not export "${v}" (check the exact casing at lucide.dev)`; + return true; + }, + persona: (v) => (nonEmptyString(v) ? true : 'must be a non-empty string'), + rewardLine: (v) => (nonEmptyString(v) ? true : 'must be a non-empty string'), + ogImage: (v) => (nonEmptyString(v) && v.startsWith('/') ? true : "must be a site-absolute path, e.g. '/og/wordle.png'"), +}; + +/* --------------------------------------------------------- literal anchors */ + +const COMPONENT_ANCHORS = [/\bcomponents\s*:\s*/]; +const ANATOMY_ANCHORS = [/(?:export\s+)?const\s+anatomy\s*(?::\s*[^=]+)?=\s*/, /\banatomy\s*:\s*/]; +const PROVENANCE_ANCHORS = [ + /(?:export\s+)?const\s+provenance\s*(?::\s*[^=]+)?=\s*/, + /\bprovenance\s*:\s*/, +]; + +/* ------------------------------------------------------------- the checks */ + +const slugs = demoSlugs(); +if (slugs.length === 0) die(`no demos found under ${rel(DEMOS_DIR)}. A repo with no demos has nothing to validate.`); + +const manifest = loadManifest(); + +for (const slug of slugs) { + const dir = path.join(DEMOS_DIR, slug); + const found = loadMeta(slug); + + if ('error' in found) { + report.fail(found.file, 'rule 2 (DemoMeta present)', found.error); + continue; + } + const { meta, file: metaFile } = found; + + /* -- 1. directory name is the slug ------------------------------------- */ + report.check( + meta.slug === slug, + metaFile, + 'rule 1 (slug is the directory name)', + `meta.slug is ${JSON.stringify(meta.slug)} but the directory is ${JSON.stringify(slug)}. ` + + 'The registry, the route and the OG card are all keyed on this; rename one of them.', + ); + + /* -- 2. every required field, correctly typed ---------------------------- */ + for (const [field, validate] of Object.entries(META_FIELDS)) { + if (!(field in meta)) { + report.fail(metaFile, 'rule 2 (DemoMeta complete)', `\`${field}\` is missing.`); + continue; + } + const verdict = validate(meta[field]); + report.check(verdict === true, metaFile, 'rule 2 (DemoMeta complete)', `\`${field}\`: ${verdict}`); + } + if (!LUCIDE_ICONS) { + report.staticOnly(`${slug}: lucide-react is not installed, so \`icon\` was only checked for PascalCase shape.`); + } + + /* -- 4. the social card exists ------------------------------------------ */ + if (nonEmptyString(meta.ogImage)) { + const card = abs('public', String(meta.ogImage).replace(/^\/+/, '')); + report.check( + exists(card), + metaFile, + 'rule 4 (ogImage exists)', + `meta.ogImage points at ${meta.ogImage}, which is ${rel(card)} on disk, and that file does not exist. ` + + 'Generate it with `node scripts/og.mjs` (x86 only — it needs Playwright chromium).', + ); + } + + /* -- 5. demo.tsx default-exports via defineDemo ------------------------- */ + const demoFile = path.join(dir, 'demo.tsx'); + if (!exists(demoFile)) { + report.fail(rel(demoFile), 'rule 5 (defineDemo default export)', 'demo.tsx does not exist.'); + } else { + const src = read(demoFile); + report.check( + /export\s+default\s+defineDemo\s*(?:<[^>]*>)?\s*\(/.test(src), + rel(demoFile), + 'rule 5 (defineDemo default export)', + 'no `export default defineDemo(...)` found. The registry loads demos through defineDemo; ' + + 'a bare object default export skips whatever the kit validates.', + ); + report.check( + /\bdefineDemo\b[\s\S]*?from\s*'@\/lib\/demo-kit'/.test(src) || /from\s*'@\/lib\/demo-kit'[\s\S]*?\bdefineDemo\b/.test(src), + rel(demoFile), + 'rule 5 (defineDemo default export)', + "defineDemo must be imported from '@/lib/demo-kit'.", + ); + } + + /* -- 6/10. the specification and the reward ----------------------------- */ + const anatomy = findInDemo(slug, ANATOMY_ANCHORS, '{', 'anatomy', { lenient: true }); + const provenance = findInDemo(slug, PROVENANCE_ANCHORS, '{', 'provenance', { lenient: true }); + const components = findInDemo(slug, COMPONENT_ANCHORS, '[', 'reward.components', { lenient: true }); + + const specRule = 'rule 6 (complete specification)'; + if (anatomy.error) { + report.fail(anatomy.file ?? rel(dir), specRule, anatomy.error); + } else { + for (const field of ['task', 'actions', 'grader', 'score']) { + report.check( + nonEmptyString(anatomy.value?.[field]), + anatomy.file, + specRule, + `anatomy.${field} is empty. A ${meta.status} demo publishes the four boxes in full; ` + + 'an empty one reads as a placeholder and undoes the page it sits on.', + ); + } + } + + if (provenance.error) { + report.fail(provenance.file ?? rel(dir), specRule, provenance.error); + } else { + report.check( + nonEmptyString(provenance.value?.command), + provenance.file, + specRule, + 'provenance.command is empty. The eval command is the reader\'s way to disprove the page; ' + + 'it is the one field that must never be aspirational.', + ); + } + + const weightRule = 'rule 10 (reward weights)'; + if (components.error) { + report.fail(components.file ?? rel(dir), weightRule, components.error); + } else if (!Array.isArray(components.value) || components.value.length === 0) { + report.fail(components.file, weightRule, 'reward.components is not a non-empty array.'); + } else { + const list = components.value; + let sum = 0; + let readable = true; + list.forEach((component, i) => { + const where = `reward.components[${i}]${component?.key ? ` (${component.key})` : ''}`; + for (const field of ['key', 'label', 'description']) { + report.check( + nonEmptyString(component?.[field]), + components.file, + weightRule, + `${where}: \`${field}\` is empty.`, + ); + } + if (typeof component?.weight !== 'number' || !Number.isFinite(component.weight)) { + readable = false; + report.fail( + components.file, + weightRule, + `${where}: \`weight\` is ${isUnresolved(component?.weight) ? 'an imported constant' : JSON.stringify(component?.weight)}, ` + + 'not a literal number. Weights are read off the page beside the code, so they are written out here.', + ); + } else { + sum += component.weight; + } + report.check( + ROLES.includes(component?.role), + components.file, + weightRule, + `${where}: \`role\` must be one of ${ROLES.join(', ')}.`, + ); + }); + + if (readable) { + report.check( + Math.abs(sum - 1) <= 1e-9, + components.file, + weightRule, + `the weights sum to ${sum} and must sum to 1.0 (within 1e-9). ` + + 'A reward whose weights do not sum to one is not the reward the page shows you tuning.', + ); + } + report.check( + list.some((c) => c?.role === 'counterweight'), + components.file, + 'rule 10 (counterweight required)', + 'no component has role "counterweight". An objective with nothing pulling against it is a metric to ' + + 'game, and the demo exists to show the opposite. If the second term is one every good policy also ' + + 'scores 1.0 on, it is a gate — and the demo still needs a real counterweight.', + ); + const counterweight = list.find((c) => c?.role === 'counterweight'); + if (counterweight) { + report.check( + nonEmptyString(counterweight.description), + components.file, + specRule, + `the counterweight (${counterweight.key}) has no description. It is the half of the reward a reader ` + + 'does not expect, so it is the half that must be spelled out.', + ); + } + } + + /* -- 7/11/12. recorded runs --------------------------------------------- */ + if (meta.status === 'live') { + const runRule = 'rule 7 (live demo has runs)'; + if (manifest.error) { + report.fail(rel(abs('public', 'traces', 'manifest.json')), runRule, manifest.error); + } else { + const entry = manifest.byDemo.get(slug); + const runs = entry?.runs ?? []; + if (runs.length === 0) { + report.fail( + rel(abs('public', 'traces', 'manifest.json')), + runRule, + `demo "${slug}" is status "live" but has no runs in the manifest. ` + + 'A live demo asserts recorded evidence; without a run there is nothing to replay.', + ); + } else { + const episodes = []; + for (const run of runs) { + const label = `run ${run?.id ?? '(no id)'}`; + if (!nonEmptyString(run?.path)) { + report.fail(rel(abs('public', 'traces', 'manifest.json')), runRule, `${label}: \`path\` is empty.`); + continue; + } + const file = traceFile(run.path); + if (!report.check(exists(file), rel(file), runRule, `${label} points at ${run.path}, which does not exist on disk.`)) { + continue; + } + try { + const raw = JSON.parse(read(file)); + for (const ep of Array.isArray(raw) ? raw : Array.isArray(raw?.episodes) ? raw.episodes : [raw]) { + episodes.push({ run, file, episode: ep }); + } + } catch (error) { + report.fail(rel(file), runRule, `not valid JSON: ${error.message}`); + } + + /* -- 12. an intervened run must say what was done to it ---------- */ + if (run?.kind === 'intervened') { + report.check( + nonEmptyString(run.intervention), + rel(abs('public', 'traces', 'manifest.json')), + 'rule 12 (intervened runs declare the intervention)', + `${label} has kind "intervened" but no \`intervention\`. Without it a prompt change reads as a ` + + 'training result by omission, which is the single most misleading thing this page could do.', + ); + } + } + + /* -- 11. a failure, or a solve rate the page renders -------------- */ + const failedRun = episodes.find((e) => e.episode?.outcome === 'failed'); + if (!failedRun && episodes.length) { + const rateSources = [entry?.extras ?? {}, ...runs]; + const rate = rateSources + .map((s) => s?.solveRate ?? s?.solve_rate) + .find((v) => typeof v === 'number' && Number.isFinite(v)); + const rendered = walk(abs('src'), (f) => /\.tsx?$/.test(f)).some((f) => /\bsolve[_R]?ate|solveRate|solve_rate/.test(read(f))); + report.check( + typeof rate === 'number' && rendered, + rel(abs('public', 'traces', 'manifest.json')), + 'rule 11 (a clean sweep is reported, not hidden)', + `demo "${slug}" ships no run with outcome "failed". That is allowed only if the page states the ` + + 'solve rate: add a numeric `solveRate` beside the runs in the manifest AND render it. ' + + `Right now solveRate is ${typeof rate === 'number' ? rate : 'absent'} and a renderer for it was ` + + `${rendered ? 'found' : 'not found'} under src/. Hunting for a losing seed to make the demo look ` + + 'honest is the failure mode this rule exists to block.', + ); + } + } + } + } + + /* -- 13. every step announces itself ------------------------------------ */ + checkAnnounce(slug, report, manifest); +} + +/* ---------------------------------------- 8. the shell knows no demo names */ + +const SHELL_DIR = abs('src', 'components', 'demo'); +for (const file of walk(SHELL_DIR, (f) => /\.(tsx?|css)$/.test(f))) { + const lines = read(file).split('\n'); + for (const slug of allDemoDirs()) { + if (slug.startsWith('_')) continue; + // Word-boundary on both sides so a slug like `wordle` is not found inside + // an unrelated identifier, and so `wordle-five` matches as one token. + const needle = new RegExp(`(? needle.test(line)); + if (hit !== -1) { + report.fail( + `${rel(file)}:${hit + 1}`, + 'rule 8 (the shell is generic)', + `mentions the demo slug "${slug}". Nothing under src/components/demo/ may know which demo it is ` + + 'rendering. If the shell needs to special-case something, that is a missing slot on DemoModule or a ' + + 'missing capability in demo-kit — fix it there and every future demo gets it too.', + ); + } else { + report.passed += 1; + } + } +} + +/* --------------------------------- 9. demos reach only for the kit's barrel */ + +const FORBIDDEN_IMPORTS = [ + { + test: (spec) => /^@\/lib\/demo-kit\/.+/.test(spec), + why: "imports a file INSIDE demo-kit. Import '@/lib/demo-kit' — the barrel is the contract, and anything " + + 'you can only get by reaching past it is either private or belongs in the barrel.', + }, + { + test: (spec) => /^@\/components\/demo(\/|$)/.test(spec), + why: 'imports the shared shell. The shell renders demos; demos never render the shell. This is the ' + + 'dependency that, once it exists, makes the shell impossible to change without opening every demo.', + }, +]; + +for (const slug of allDemoDirs()) { + for (const file of demoFiles(slug)) { + for (const spec of importSpecifiers(read(file))) { + const normalised = normaliseSpecifier(spec, file); + for (const rule of FORBIDDEN_IMPORTS) { + if (rule.test(normalised)) { + report.fail( + rel(file), + 'rule 9 (demos import only the demo-kit barrel)', + `\`${spec}\`${normalised === spec ? '' : ` (resolves to ${normalised})`} ${rule.why}`, + ); + } else { + report.passed += 1; + } + } + } + } +} + +report.finish(); + +/* ------------------------------------------------------------- helpers */ + +/** Every module specifier a file imports, static, dynamic or side-effect. */ +function importSpecifiers(src) { + const out = new Set(); + for (const m of src.matchAll(/\bfrom\s*['"]([^'"]+)['"]/g)) out.add(m[1]); + for (const m of src.matchAll(/\bimport\s*\(\s*['"]([^'"]+)['"]\s*\)/g)) out.add(m[1]); + for (const m of src.matchAll(/^\s*import\s+['"]([^'"]+)['"]/gm)) out.add(m[1]); + return [...out]; +} + +/** + * Rewrites a relative specifier into its `@/` form. + * + * Without this, `../../components/demo/Player` walks straight past a lint that + * only looks at the alias — and it is the form an editor's auto-import + * produces, so it is the form the violation actually arrives in. + */ +function normaliseSpecifier(spec, fromFile) { + if (!spec.startsWith('.')) return spec; + const resolved = path.resolve(path.dirname(fromFile), spec); + const inSrc = path.relative(abs('src'), resolved); + if (inSrc.startsWith('..')) return spec; + return `@/${inSrc.split(path.sep).join('/')}`.replace(/\.(tsx?|jsx?)$/, ''); +} + +/** + * Rule 13: every `DemoStep` an adapter produces sets a non-empty `announce`. + * + * Reduced motion clamps the tile flip to nothing, so for a screen-reader user + * the announcement IS the result. A missing one is not a degraded experience, + * it is a blank page that claims to be showing you something. + * + * Tried dynamically first, under tsx, against the demo's own recorded traces — + * a static check cannot see through a helper that builds the step object. When + * the module will not load in bare Node (a `?raw` import, a Vite-only alias, + * JSX pulling in something browser-shaped) it falls back to reading the source + * and SAYS SO in the report rather than quietly downgrading. + */ +function checkAnnounce(slug, report, manifest) { + const rule = 'rule 13 (every step announces itself)'; + const dir = path.join(DEMOS_DIR, slug); + const candidates = ['adapt.ts', 'adapter.ts', 'adapters.ts', 'adapt.tsx', 'demo.tsx'] + .map((n) => path.join(dir, n)) + .filter((p) => exists(p) && /\badapt\b\s*[:=(]|function\s+adapt\b/.test(read(p))); + + if (candidates.length === 0) { + report.fail( + rel(dir), + rule, + 'no `adapt` implementation found. DemoModule.adapt is required; it is what turns a recorded episode ' + + 'into the steps the player renders.', + ); + return; + } + + const traces = (manifest.byDemo.get(slug)?.runs ?? []) + .map((run) => (run?.path ? traceFile(run.path) : null)) + .filter((p) => p && exists(p)); + + if (traces.length > 0) { + for (const module of candidates) { + const result = probeAdapter(module, traces); + if (result.loaded) { + if (result.ok) { + report.passed += 1; + } else { + for (const problem of result.problems) report.fail(rel(module), rule, problem); + } + return; + } + } + } + + // Static fallback. It cannot prove `announce` is set on every step, only that + // the adapter sets it at all and never sets it to nothing. + const module = candidates[0]; + const src = read(module); + const assigns = /\bannounce\s*:/.test(src); + report.check( + assigns, + rel(module), + rule, + 'the adapter never assigns `announce`. Every DemoStep must carry one — with motion reduced, it is the ' + + 'only thing that reports the result.', + ); + if (assigns) { + const empty = /\bannounce\s*:\s*(?:''|""|``|null|undefined)\s*[,}\n]/.test(src); + report.check(!empty, rel(module), rule, 'the adapter assigns an empty `announce` somewhere. An empty announcement is a missing one.'); + } + report.staticOnly( + `${slug}: rule 13 was checked by reading ${rel(module)}, not by running the adapter` + + `${traces.length === 0 ? ' (no recorded traces to feed it)' : ' (the module would not load under tsx in bare Node)'}.`, + ); +} + +/** Runs one adapter over real episodes under tsx. Never throws. */ +function probeAdapter(modulePath, traces) { + const tsx = abs('node_modules', '.bin', 'tsx'); + if (!exists(tsx)) return { loaded: false, reason: 'tsx is not installed' }; + + const probe = path.join(fs.mkdtempSync(path.join(os.tmpdir(), 'pig-adapt-')), 'probe.mts'); + fs.writeFileSync(probe, adapterProbeSource(), 'utf8'); + try { + const run = spawnSync(process.execPath, [tsx, probe, modulePath, ...traces], { + cwd: abs('.'), + encoding: 'utf8', + timeout: 30_000, + // tsx resolves the `@/` alias from a tsconfig, and the probe lives in + // a temp dir where it would never find ours. + env: { ...process.env, TSX_TSCONFIG_PATH: abs('tsconfig.json') }, + }); + if (run.status !== 0 || !run.stdout) return { loaded: false, reason: (run.stderr || '').trim().split('\n').slice(-1)[0] }; + const line = run.stdout.trim().split('\n').pop(); + const parsed = JSON.parse(line); + return parsed.loaded === false ? { loaded: false, reason: parsed.reason } : { loaded: true, ...parsed }; + } catch { + return { loaded: false, reason: 'the probe did not produce parseable output' }; + } finally { + fs.rmSync(path.dirname(probe), { recursive: true, force: true }); + } +} + +/** + * The probe, written to a temp file and executed by tsx. + * + * It lives here as a string rather than as a checked-in `.mts` because it is an + * implementation detail of this script, and a stray TypeScript file in + * `scripts/` would get swept into the typecheck it is deliberately outside of. + */ +function adapterProbeSource() { + return ` +import { readFileSync } from 'node:fs'; + +const [modulePath, ...traces] = process.argv.slice(2); +const say = (o: unknown) => console.log(JSON.stringify(o)); + +let mod: any; +try { + mod = await import(modulePath); +} catch (error: any) { + say({ loaded: false, reason: String(error?.message ?? error).split('\\n')[0] }); + process.exit(0); +} + +const adapt = + typeof mod.adapt === 'function' ? mod.adapt : + typeof mod.default?.adapt === 'function' ? mod.default.adapt : + typeof mod.default === 'function' ? mod.default : null; + +if (!adapt) { + say({ loaded: false, reason: 'no callable \\\`adapt\\\` export' }); + process.exit(0); +} + +const problems: string[] = []; +let steps = 0; + +for (const trace of traces) { + const raw = JSON.parse(readFileSync(trace, 'utf8')); + const episodes = Array.isArray(raw) ? raw : Array.isArray(raw?.episodes) ? raw.episodes : [raw]; + for (const episode of episodes) { + let produced: any; + try { + produced = adapt(episode); + } catch (error: any) { + problems.push(trace + ': adapt() threw ' + String(error?.message ?? error).split('\\n')[0]); + continue; + } + if (!Array.isArray(produced)) { + problems.push(trace + ': adapt() returned ' + typeof produced + ', not an array of DemoStep'); + continue; + } + produced.forEach((step: any, i: number) => { + steps += 1; + if (typeof step?.announce !== 'string' || step.announce.trim() === '') { + problems.push(trace + ': step ' + i + ' has an empty \\\`announce\\\`'); + } + }); + } +} + +say({ loaded: true, ok: problems.length === 0, problems: problems.slice(0, 10), steps }); +`; +} diff --git a/scripts/check-licenses.mjs b/scripts/check-licenses.mjs new file mode 100644 index 0000000..09cbc42 --- /dev/null +++ b/scripts/check-licenses.mjs @@ -0,0 +1,181 @@ +#!/usr/bin/env node +/** + * The attribution checker. + * + * This repo is public and Apache-2.0, and almost none of what makes the demo + * work is ours: the primitives are shadcn/ui over Radix, the answer list is + * Wordnik's filtered through SCOWL, the typeface is Manrope under the SIL Open + * Font License, the icons are lucide, the charts are recharts. Every one of + * those licences is permissive, and every one of them requires attribution. + * + * So `NOTICE` is not paperwork, it is a build artifact with a test. This script + * is that test: it fails if NOTICE is missing any source we actually ship, and + * it fails if a source is named without its licence beside it — because + * "uses Manrope" without "OFL-1.1" is not attribution, it is a mention. + * + * The OFL additionally requires that the licence text travel WITH the font, so + * `public/fonts/OFL.txt` is checked for separately. Shipping the .woff2 out of + * node_modules and leaving the licence behind is the single easiest way to + * violate the one licence on this list that has teeth. + */ + +import fs from 'node:fs'; + +import { Report, abs, exists, read, rel } from './_lib.mjs'; + +const report = new Report('check-licenses'); + +const NOTICE = abs('NOTICE'); +const OFL = abs('public', 'fonts', 'OFL.txt'); + +/** + * Every third-party source this site ships, and the licence each one must be + * named with. `used` is a cheap sanity check in the other direction: an entry + * whose artifact has left the repo is stale attribution, and stale attribution + * quietly becomes wrong attribution. + */ +const SOURCES = [ + { + label: 'shadcn/ui', + match: /shadcn/i, + licence: /\bMIT\b/, + why: 'the UI primitives in src/components/ui are shadcn/ui components, hand-copied into this repo.', + used: () => exists(abs('components.json')), + }, + { + label: 'Radix UI', + match: /radix/i, + licence: /\bMIT\b/, + why: 'every primitive with behaviour — dialog, tabs, tooltip, slider — is Radix underneath.', + used: () => exists(abs('node_modules', '@radix-ui')), + }, + { + label: 'the Wordnik word list', + match: /wordnik/i, + licence: /\bMIT\b/, + why: 'the guess list is derived from Wordnik.', + used: () => hasWordFile(/wordnik/i), + }, + { + label: 'SCOWL', + match: /\bSCOWL\b/i, + licence: /permissive|attribution|BSD|Kevin\s+Atkinson/i, + why: 'the answer list is filtered through SCOWL.', + used: () => hasWordFile(/scowl/i), + }, + { + label: 'Manrope', + match: /manrope/i, + // The OFL is version-specific and the version matters: OFL-1.1 is the one + // Manrope ships under, and it is the one whose terms are quoted in OFL.txt. + licence: /OFL[-\s]?1\.1|SIL\s+Open\s+Font\s+License/i, + why: 'Manrope is the typeface, embedded as a variable woff2.', + used: () => exists(abs('node_modules', '@fontsource-variable', 'manrope')), + }, + { + label: 'lucide', + match: /lucide/i, + licence: /\bISC\b/, + why: 'every icon on the site is a lucide icon.', + used: () => exists(abs('node_modules', 'lucide-react')), + }, + { + label: 'recharts', + match: /recharts/i, + licence: /\bMIT\b/, + why: 'the reward and metric charts are recharts.', + used: () => exists(abs('node_modules', 'recharts')), + }, + { + label: "PIG's own token layer", + match: /\bPIG\b|Prime\s+Intellect\s+Growth/i, + licence: /Apache[-\s]?2\.0/i, + why: 'the colour and motion tokens in src/index.css come from PIG and ship under Apache-2.0.', + used: () => exists(abs('src', 'index.css')), + }, +]; + +/** True when the words directory still carries a file from this source. */ +function hasWordFile(pattern) { + const dir = abs('envs', 'wordle_five', 'words'); + return exists(dir) && fs.readdirSync(dir).some((name) => pattern.test(name)); +} + +if (!exists(NOTICE)) { + report.fail( + 'NOTICE', + 'NOTICE exists', + 'there is no NOTICE file at the repo root. Every source below is shipped by this site and each of ' + + `their licences requires attribution: ${SOURCES.map((s) => s.label).join(', ')}.`, + ); + report.finish(); +} + +const notice = read(NOTICE); +const lines = notice.split('\n'); + +for (const source of SOURCES) { + // Every line that names the source, not just the first. "PIG-Demo" in the + // copyright header matches the PIG entry three dozen lines before its actual + // attribution block, and a first-match-wins reader fails on a NOTICE that is + // completely correct. + const hits = []; + lines.forEach((line, i) => { + if (source.match.test(line)) hits.push(i); + }); + + if (hits.length === 0) { + report.fail( + 'NOTICE', + 'attribution complete', + `${source.label} is not named. It is shipped by this site — ${source.why} — and its licence requires ` + + 'attribution. Add it with its licence identifier.', + ); + continue; + } + report.passed += 1; + + // The licence has to sit with the name, not merely somewhere in the file: + // a NOTICE that says "MIT" once at the top and lists nine projects under it + // is attributing all nine to whichever licence happens to be first. + const near = (i) => lines.slice(Math.max(0, i - 2), i + 6).join('\n'); + report.check( + hits.some((i) => source.licence.test(near(i))), + `NOTICE:${hits[0] + 1}`, + 'attribution names the licence', + `${source.label} is named on line(s) ${hits.map((i) => i + 1).join(', ')} but no matching licence appears ` + + `beside any of them (looking for ${source.licence}). A name without a licence is a mention, not an ` + + 'attribution.', + ); + + if (!source.used()) { + report.warn( + `NOTICE names ${source.label}, but nothing in the repo appears to use it any more. ` + + 'Stale attribution is how a NOTICE stops being trustworthy.', + ); + } +} + +/* -------------------------------------------------------- the OFL's own rule */ + +report.check( + exists(OFL), + rel(OFL), + 'OFL text ships with the font', + 'Manrope is under SIL OFL-1.1, which requires the licence text to travel with the font files. ' + + 'Copy node_modules/@fontsource-variable/manrope/LICENSE to public/fonts/OFL.txt. ' + + 'Shipping the woff2 and leaving the licence behind is the one violation on this list with teeth.', +); + +if (exists(OFL)) { + const text = read(OFL); + report.check( + /SIL OPEN FONT LICENSE/i.test(text) && /Version 1\.1/i.test(text), + rel(OFL), + 'OFL text ships with the font', + 'the file exists but does not look like the SIL Open Font License 1.1. It must be the licence text ' + + 'itself, not a pointer to it.', + ); +} + +report.finish(); diff --git a/scripts/check-receipts.mjs b/scripts/check-receipts.mjs new file mode 100644 index 0000000..ac63eec --- /dev/null +++ b/scripts/check-receipts.mjs @@ -0,0 +1,196 @@ +#!/usr/bin/env node +/** + * The receipt checker. + * + * Every demo quotes the code that computes its reward, verbatim, next to the + * number that code produced. That quote is the page's whole claim to being + * evidence rather than marketing, so the thing it quotes has to exist and the + * region it quotes has to resolve. A receipt panel that silently renders empty, + * or renders the wrong forty lines because someone inserted a function above + * the marker, is worse than no receipt at all. + * + * ── MARKER SYNTAX ────────────────────────────────────────────────────────── + * + * `RewardSpec.source` is `{ path, code, marker? }`. Without a `marker` the + * whole file is the receipt. With one, the receipt is the region delimited by + * a matched pair of comment lines in the source file itself: + * + * # region: pig-demo/ + * ... the quoted lines ... + * # endregion: pig-demo/ + * + * Rules, all enforced here: + * + * · The marker lines are EXCLUSIVE — neither appears in the quoted region. + * · EXACTLY ONE pair per file per marker. Zero is a broken receipt; two or + * more is ambiguous, and an ambiguous receipt silently quotes whichever + * region the reader's implementation happened to find first. Both are + * fatal. + * · `region` must come before `endregion`, and the region must be non-empty + * once the marker lines are removed. + * · The comment token is `#` (Python) or `//` (TypeScript), then optional + * whitespace. Everything else on the line is ignored, so + * `# region: pig-demo/consistency (see PR #41)` is legal. + * · The `pig-demo/` prefix is required. It is what makes these greppable and + * stops an editor's own `#region` folding markers from being read as + * receipts. + * + * Marker names are `[a-z0-9][a-z0-9-]*`: they end up in a grep, a CI message + * and a code comment, and mixed case in all three is a bug factory. + */ + +import path from 'node:path'; + +import { + Report, + abs, + allDemoDirs, + demoFiles, + evalLiteral, + exists, + isUnresolved, + literalsAfter, + read, + rel, +} from './_lib.mjs'; + +const report = new Report('check-receipts'); + +const RULE_PATH = 'reward.source.path'; +const RULE_MARKER = 'reward.source.marker'; + +const MARKER_NAME = /^[a-z0-9][a-z0-9-]*$/; +const markerLine = (kind, marker) => + new RegExp(`^[ \\t]*(?:#|//)[ \\t]*${kind}:[ \\t]*pig-demo/${marker.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}(?![\\w-])`, 'm'); + +/** Every `source: { ... }` literal in a demo, wherever the demo chose to put it. */ +function receipts(slug) { + const out = []; + for (const file of demoFiles(slug)) { + const src = read(file); + for (const literal of literalsAfter(src, /\bsource\s*:\s*/, '{')) { + // `code` is legitimately an identifier here — the Python arrives through + // a Vite `?raw` import, which cannot resolve in plain Node — so the + // literal is read leniently and only `path` and `marker` are trusted. + const evaluated = evalLiteral(literal.text, 'RewardSpec.source', { lenient: true }); + if (!evaluated.ok) { + out.push({ file, error: evaluated.error }); + continue; + } + const value = evaluated.value; + if (!value || typeof value !== 'object' || !('path' in value)) continue; // some other `source:` key + const line = src.slice(0, literal.start).split('\n').length; + out.push({ file, line, value }); + } + } + return out; +} + +const slugs = allDemoDirs().filter((s) => !s.startsWith('_')); +if (slugs.length === 0) { + console.log('check-receipts: no demos to check.'); + process.exit(0); +} + +for (const slug of slugs) { + const found = receipts(slug); + + if (found.length === 0) { + report.fail( + rel(path.join(abs('src', 'demos'), slug)), + RULE_PATH, + 'no `reward.source` found. Every demo quotes the code that computes its reward; without a source ' + + 'the receipt panel has nothing to show and the page is asserting its numbers rather than proving them.', + ); + continue; + } + + for (const receipt of found) { + const where = `${rel(receipt.file)}${receipt.line ? `:${receipt.line}` : ''}`; + if (receipt.error) { + report.fail(where, RULE_PATH, receipt.error); + continue; + } + const { path: sourcePath, marker } = receipt.value; + + if (typeof sourcePath !== 'string' || sourcePath.trim() === '') { + report.fail( + where, + RULE_PATH, + `\`path\` is ${isUnresolved(sourcePath) ? 'an imported constant' : JSON.stringify(sourcePath)}. ` + + 'It must be a literal repo-relative path, e.g. "envs/wordle_five/wordle_five/rubric.py", ' + + 'so a reader can open the same file this page quotes.', + ); + continue; + } + + // Repo-relative by contract. A leading slash would mean the site root, + // which is not where the Python lives, and an absolute disk path would + // only work on the machine that wrote it. + const clean = sourcePath.replace(/^\.\//, ''); + if (path.isAbsolute(clean)) { + report.fail(where, RULE_PATH, `\`path\` is absolute ("${sourcePath}"). Use a repo-relative path.`); + continue; + } + const onDisk = abs(clean); + if (!report.check(exists(onDisk), where, RULE_PATH, `\`path\` is "${sourcePath}", and ${rel(onDisk)} does not exist.`)) { + continue; + } + + if (marker === undefined || marker === null) continue; // whole file is the receipt + + if (typeof marker !== 'string' || !MARKER_NAME.test(marker)) { + report.fail( + where, + RULE_MARKER, + `\`marker\` is ${JSON.stringify(marker)}. Markers are lower-kebab-case (${MARKER_NAME}) because they ` + + 'appear in a grep, a CI message and a code comment, and mixed case in all three is a bug factory.', + ); + continue; + } + + const body = read(onDisk); + const lines = body.split('\n'); + const opens = []; + const closes = []; + lines.forEach((line, i) => { + if (markerLine('region', marker).test(line)) opens.push(i); + if (markerLine('endregion', marker).test(line)) closes.push(i); + }); + + const syntax = `Expected exactly one "# region: pig-demo/${marker}" and one "# endregion: pig-demo/${marker}".`; + if (opens.length !== 1 || closes.length !== 1) { + report.fail( + rel(onDisk), + RULE_MARKER, + `marker "${marker}" (referenced from ${where}) resolves to ${opens.length} region marker(s) and ` + + `${closes.length} endregion marker(s). ${syntax} ` + + (opens.length > 1 || closes.length > 1 + ? 'More than one pair is ambiguous: the receipt would quote whichever region the reader found first.' + : 'A missing marker renders the receipt panel empty, which reads as the code not existing.'), + ); + continue; + } + + const [open] = opens; + const [close] = closes; + if (close <= open) { + report.fail( + rel(onDisk), + RULE_MARKER, + `marker "${marker}": the endregion is on line ${close + 1}, at or before the region on line ${open + 1}.`, + ); + continue; + } + const region = lines.slice(open + 1, close); + report.check( + region.some((line) => line.trim() !== ''), + rel(onDisk), + RULE_MARKER, + `marker "${marker}" delimits an empty region (lines ${open + 2}-${close}). The markers are exclusive, ` + + 'so a pair on adjacent lines quotes nothing.', + ); + } +} + +report.finish(); diff --git a/src/components/demo/BlindCompare.tsx b/src/components/demo/BlindCompare.tsx new file mode 100644 index 0000000..21f5b13 --- /dev/null +++ b/src/components/demo/BlindCompare.tsx @@ -0,0 +1,192 @@ +import { useState } from 'react'; +import type { ComponentType } from 'react'; +import { Eye, Trophy } from 'lucide-react'; +import type { DemoStep } from '@/lib/demo-kit/types'; +import { cn } from '@/lib/utils'; +import { formatOrDash } from './format'; + +export interface BlindCompareRun { + runId: string; + /** The identity, revealed only after the visitor commits. */ + label: string; + model: string; + /** Required by the contract on an `intervened` run; shown at reveal. */ + intervention?: string; + steps: DemoStep[]; + total: number | null; +} + +export type BlindVote = 'A' | 'B' | 'tie'; + +export interface BlindCompareProps { + /** Both runs must be the same seed or the comparison is meaningless. */ + seed: number; + a: BlindCompareRun; + b: BlindCompareRun; + Surface: ComponentType<{ state: T; compact?: boolean }>; + question?: string; + onVote?: (vote: BlindVote) => void; + className?: string; +} + +/** + * Two runs on the same seed, unlabelled, until you commit. + * + * The point is not the vote. The point is that the visitor forms an opinion + * from the behaviour BEFORE they learn which one had the better prompt, the + * bigger model or the training run — because once they know, they cannot + * unknow it, and every "obviously the trained one looks better" is worthless + * after the fact. + * + * There is a "reveal without voting" escape on purpose: a visitor who does not + * want to play should not be held hostage by a modal-shaped page. + */ +export function BlindCompare({ + seed, + a, + b, + Surface, + question = 'Which agent would you rather have running this?', + onVote, + className, +}: BlindCompareProps) { + const [vote, setVote] = useState(null); + const [revealed, setRevealed] = useState(false); + + const commit = (choice: BlindVote) => { + setVote(choice); + setRevealed(true); + onVote?.(choice); + }; + + const winner: 'A' | 'B' | 'tie' = + a.total === null || b.total === null + ? 'tie' + : a.total > b.total + ? 'A' + : b.total > a.total + ? 'B' + : 'tie'; + + const sides: { id: 'A' | 'B'; run: BlindCompareRun }[] = [ + { id: 'A', run: a }, + { id: 'B', run: b }, + ]; + + return ( +
+
+

{question}

+

Same puzzle, same seed ({seed}).

+
+ +
+ {sides.map(({ id, run }) => { + const last = run.steps[run.steps.length - 1]; + const picked = vote === id; + return ( +
+
+

Agent {id}

+ {run.steps.length} steps +
+ +
+ {last ? ( + + ) : ( +

This run recorded no steps.

+ )} +
+ + {!revealed ? ( + + ) : ( +
+
+
Identity
+
{run.label}
+
+
+
Model
+
{run.model}
+
+
+
Intervention
+
+ {run.intervention ?? ( + none — plain rollout + )} +
+
+
+
Reward
+
+ {revealed && winner === id ? ( +
+
+
+ )} +
+ ); + })} +
+ + {!revealed ? ( +
+ + +
+ ) : ( +

+ {vote === null ? ( + <>Revealed without a vote. + ) : vote === winner ? ( + <> + You picked the higher-scoring run.{' '} + + ) : ( + <> + + You picked the lower-scoring run. + {' '} + + )} + The environment scored these two with the same grader, on the same seed. The difference + between them is stated above — and if it is a prompt change rather than a training run, + it says so, because a prompt change presented as a training result is the oldest trick + in this business. +

+ )} +
+ ); +} diff --git a/src/components/demo/CodeReceipt.tsx b/src/components/demo/CodeReceipt.tsx new file mode 100644 index 0000000..d53865d --- /dev/null +++ b/src/components/demo/CodeReceipt.tsx @@ -0,0 +1,188 @@ +import { useMemo, useState } from 'react'; +import { ExternalLink } from 'lucide-react'; +import { cn } from '@/lib/utils'; + +export interface CodeReceiptProps { + /** The file's text, imported with `?raw` so it cannot drift from the source. */ + code: string; + /** Repo-relative path, shown as the receipt's header. */ + path: string; + /** + * A literal string that appears in the source and marks the interesting part. + * See `resolveMarkedRange` for the two conventions it supports. + */ + marker?: string; + /** Link to the whole file — GitHub, usually. */ + href?: string; + /** Lines of context kept around the marked range while collapsed. */ + context?: number; + className?: string; +} + +export interface MarkedRange { + /** 0-based, inclusive. */ + start: number; + /** 0-based, inclusive. */ + end: number; +} + +/** + * Two marker conventions, because both exist in real repos: + * + * TWICE — the marker brackets a region (`# --8<-- reward` … `# --8<--`). + * The marked range is the lines BETWEEN them; the fences themselves + * are not interesting code. + * + * ONCE — the marker sits on a definition line (`def compute_reward`). The + * marked range is that line plus its indented body, which is what + * you actually meant. Blank lines inside the body are kept; trailing + * blank lines are not, or the highlight runs on past the function. + * + * A marker that matches nothing returns null and the whole file renders. That + * is the right failure: a stale marker must not hide the source. + */ +export function resolveMarkedRange(lines: string[], marker?: string): MarkedRange | null { + if (!marker) return null; + const hits: number[] = []; + lines.forEach((line, index) => { + if (line.includes(marker)) hits.push(index); + }); + + const first = hits[0]; + if (first === undefined) return null; + + if (hits.length >= 2) { + const last = hits[hits.length - 1] as number; + return last - first > 1 ? { start: first + 1, end: last - 1 } : { start: first, end: last }; + } + + const anchor = lines[first] ?? ''; + const indent = anchor.length - anchor.trimStart().length; + let end = first; + for (let i = first + 1; i < lines.length; i += 1) { + const line = lines[i] ?? ''; + if (line.trim() === '') continue; + const lineIndent = line.length - line.trimStart().length; + if (lineIndent <= indent) break; + end = i; + } + return { start: first, end }; +} + +/** + * Source, with the part that matters marked. + * + * Syntax highlighting is deliberately absent: it costs a highlighter in the + * bundle and buys nothing an exec can use, while the marked range — the thing + * that says "this, right here, is the grader" — costs nothing and is the whole + * reason the panel exists. + * + * The block scrolls horizontally inside itself. Long Python lines must never + * make the PAGE scroll sideways; on a phone that turns every vertical swipe + * into a fight. + */ +export function CodeReceipt({ + code, + path, + marker, + href, + context = 3, + className, +}: CodeReceiptProps) { + const lines = useMemo(() => code.replace(/\n$/, '').split('\n'), [code]); + const range = useMemo(() => resolveMarkedRange(lines, marker), [lines, marker]); + const [expanded, setExpanded] = useState(range === null); + + const shown = useMemo(() => { + if (range === null || expanded) return { from: 0, to: lines.length - 1 }; + return { + from: Math.max(range.start - context, 0), + to: Math.min(range.end + context, lines.length - 1), + }; + }, [range, expanded, context, lines.length]); + + const visible = lines.slice(shown.from, shown.to + 1); + const gutterWidth = `${String(lines.length).length + 1}ch`; + + return ( +
+
+

+ {path} +

+ {range && !expanded ? ( + + ) : null} + {range && expanded ? ( + + ) : null} + {href ? ( + + Read the whole file + + ) : null} +
+ +
+
+          
+            {shown.from > 0 ? (
+              
+            ) : null}
+            {visible.map((line, offset) => {
+              const number = shown.from + offset;
+              const marked = range !== null && number >= range.start && number <= range.end;
+              return (
+                
+                  {/* Sticky so the line numbers survive a horizontal scroll —
+                      without it they slide out of view exactly when a long
+                      line makes you want them. */}
+                  
+                  {line === '' ? ' ' : line}
+                
+              );
+            })}
+            {shown.to < lines.length - 1 ? (
+              
+            ) : null}
+          
+        
+
+
+ ); +} diff --git a/src/components/demo/DemoCard.tsx b/src/components/demo/DemoCard.tsx new file mode 100644 index 0000000..3d0f3db --- /dev/null +++ b/src/components/demo/DemoCard.tsx @@ -0,0 +1,100 @@ +import type { ComponentType } from 'react'; +import { Link } from 'react-router-dom'; +import { ArrowRight } from 'lucide-react'; +import type { DemoMeta } from '@/lib/demo-kit/types'; +import { cn } from '@/lib/utils'; +import { DemoIcon } from './icons'; + +export interface DemoCardProps { + meta: DemoMeta; + /** Defaults to the canonical demo route. */ + href?: string; + /** + * The demo's OWN board, drawn compact, as the thumbnail. A screenshot would + * go stale the first time the board changes and nobody would notice; this + * cannot, because it is the same component the demo page renders. + */ + Surface?: ComponentType<{ state: T; compact?: boolean }>; + /** A representative state for the thumbnail — usually a solved board. */ + thumbnailState?: T; + className?: string; +} + +export function DemoCard({ + meta, + href, + Surface, + thumbnailState, + className, +}: DemoCardProps) { + const to = href ?? `/demo/${meta.slug}`; + const isSpec = meta.status === 'spec'; + const showSurface = Surface !== undefined && thumbnailState !== undefined; + + return ( +
+
+ + + +
+

+ {/* Stretched link: the whole card is the hit target, but there is + still exactly ONE link in the accessibility tree for it. */} + + {meta.title} + +

+

{meta.tagline}

+
+ {isSpec ? ( + + Spec + + ) : null} +
+ + {showSurface ? ( +
+ {/* Decorative here: the title and tagline already name the demo, and + a screen reader has no use for a board with no run behind it. */} + +
+ ) : null} + +
+
+
For
+
{meta.persona}
+
+
+
Vertical
+
{meta.vertical.replace(/-/g, ' ')}
+
+
+ +

+ Reward: + {meta.rewardLine} +

+ +

+ {isSpec ? 'Read the specification' : 'Open the demo'} +

+
+ ); +} diff --git a/src/components/demo/LimitsCallout.tsx b/src/components/demo/LimitsCallout.tsx new file mode 100644 index 0000000..fd471c6 --- /dev/null +++ b/src/components/demo/LimitsCallout.tsx @@ -0,0 +1,66 @@ +import { ArrowRight, ShieldQuestion } from 'lucide-react'; +import type { Limit } from '@/lib/demo-kit/types'; +import { cn } from '@/lib/utils'; + +export interface LimitsCalloutProps { + limits: Limit[]; + /** + * Turns a demo slug into something linkable. The shell has no registry + * dependency of its own, so the page that knows the routes supplies this. + */ + resolveDemo?: (slug: string) => { title: string; href: string } | undefined; + title?: string; + className?: string; +} + +function defaultResolve(slug: string) { + return { title: slug, href: `/demo/${slug}` }; +} + +/** + * What this demo does not teach. + * + * Styled as confidence, not apology. A vendor who tells you the limits of their + * own demo before you find them is a vendor you believe about everything else + * on the page — and every limit here names the demo that closes it, so the list + * reads as a roadmap rather than a disclaimer. + */ +export function LimitsCallout({ + limits, + resolveDemo = defaultResolve, + title = 'What this demo does not teach', + className, +}: LimitsCalloutProps) { + if (limits.length === 0) return null; + return ( +
+
+
+
    + {limits.map((limit) => { + const target = limit.answeredBy ? resolveDemo(limit.answeredBy) : undefined; + return ( +
  • +

    {limit.text}

    + {target ? ( + + Answered by {target.title} + + ) : ( +

    + No demo answers this one yet. It is a real gap, not a rhetorical one. +

    + )} +
  • + ); + })} +
+
+ ); +} diff --git a/src/components/demo/MetricChart.tsx b/src/components/demo/MetricChart.tsx new file mode 100644 index 0000000..21ff6fe --- /dev/null +++ b/src/components/demo/MetricChart.tsx @@ -0,0 +1,116 @@ +import { + CartesianGrid, + Line, + LineChart, + ReferenceLine, + ResponsiveContainer, + Tooltip, + XAxis, + YAxis, +} from 'recharts'; +import type { TooltipProps } from 'recharts'; +import { formatNumber } from './format'; + +export interface MetricPoint { + /** Whatever the x axis is counting: checkpoint, step, arm name. */ + x: string | number; + y: number; +} + +export interface MetricChartProps { + data: MetricPoint[]; + /** The rule the headline number is being compared against. */ + baseline?: { value: number; label: string }; + height: number; + /** Off under `prefers-reduced-motion`; recharts animates on mount by default. */ + animate: boolean; + digits?: number; +} + +/** + * The chart, in its own module so `React.lazy` can hold recharts out of the + * entry chunk. Nothing else in the shell may import this file directly — an + * ordinary import here defeats the whole arrangement and the entry chunk grows + * by ~100 kB without anyone noticing. + * + * Every colour is read from the token layer at paint time rather than passed in + * as a literal, so the chart follows the theme toggle without a re-render. + */ +export default function MetricChart({ + data, + baseline, + height, + animate, + digits = 3, +}: MetricChartProps) { + return ( +
+ + + + + formatNumber(value, digits > 2 ? 2 : digits)} + /> + {baseline ? ( + + ) : null} + ) => ( + + )} + /> + + + +
+ ); +} + +function ChartTooltip({ + active, + payload, + label, + digits, +}: TooltipProps & { digits: number }) { + if (!active || !payload || payload.length === 0) return null; + const value = payload[0]?.value; + return ( +
+

{String(label)}

+

+ {typeof value === 'number' ? formatNumber(value, digits) : '—'} +

+
+ ); +} diff --git a/src/components/demo/MetricMover.tsx b/src/components/demo/MetricMover.tsx new file mode 100644 index 0000000..74a8b99 --- /dev/null +++ b/src/components/demo/MetricMover.tsx @@ -0,0 +1,126 @@ +import { Suspense, lazy, useMemo } from 'react'; +import { cn } from '@/lib/utils'; +import { formatNumber, formatSigned, usePrefersReducedMotion } from './format'; +import type { MetricPoint } from './MetricChart'; +import { EditedChip } from './StatStrip'; + +export type { MetricPoint } from './MetricChart'; + +// Lazy, and deliberately not a static import: recharts is ~100 kB gzipped and +// exactly one surface on the site uses it. `vite.config.ts` also names it as +// its own manual chunk, so this stays out of the entry bundle in both dev and +// production builds. +const MetricChart = lazy(() => import('./MetricChart')); + +const CHART_HEIGHT = 168; + +export interface MetricMoverProps { + /** The one number a reader would repeat in a meeting. */ + label: string; + value: number; + digits?: number; + unit?: string; + /** The dashed rule: what the number was before, or what counts as par. */ + baseline?: { value: number; label: string }; + series?: MetricPoint[]; + /** One sentence on what moved it. Not a caveat — the mechanism. */ + caption?: string; + edited?: boolean; + className?: string; +} + +/** + * The headline number, with the line that shows it moving. + * + * The reference rule is not decoration: a number on its own is a claim, and a + * number against the line it used to sit on is evidence. If a demo has no + * baseline to draw, it should not be using this component. + */ +export function MetricMover({ + label, + value, + digits = 3, + unit, + baseline, + series, + caption, + edited = false, + className, +}: MetricMoverProps) { + const reducedMotion = usePrefersReducedMotion(); + const delta = baseline ? value - baseline.value : null; + const points = useMemo(() => series ?? [], [series]); + + return ( +
+
+

{label}

+ {edited ? : null} +
+

+ + {formatNumber(value, digits)} + + {unit ? {unit} : null} + {delta !== null ? ( + 0 ? 'text-positive' : delta < 0 ? 'text-danger' : 'text-muted', + )} + > + {formatSigned(delta, digits)} vs {baseline?.label} + + ) : null} +

+ + {points.length > 1 ? ( +
+ + {/* The chart is a picture of the table; the table is the accessible + version of the picture. Both are the same numbers. */} +
+ + Show these points as a table + + + + {points.map((point) => ( + + + + + ))} + +
+ {String(point.x)} + {formatNumber(point.y, digits)}
+
+
+ ) : null} + + {caption ? ( +

{caption}

+ ) : null} +
+ ); +} diff --git a/src/components/demo/ProvenanceCard.tsx b/src/components/demo/ProvenanceCard.tsx new file mode 100644 index 0000000..8ee9f04 --- /dev/null +++ b/src/components/demo/ProvenanceCard.tsx @@ -0,0 +1,128 @@ +import { useState } from 'react'; +import { Check, Copy, ExternalLink } from 'lucide-react'; +import type { Provenance, RunRef } from '@/lib/demo-kit/types'; +import { cn } from '@/lib/utils'; +import { formatDate } from './format'; + +export interface ProvenanceCardProps { + provenance: Provenance; + /** The run currently on screen, if the page is showing one. */ + run?: RunRef; + className?: string; +} + +type CopyState = 'idle' | 'copied' | 'failed'; + +/** + * Where these numbers came from, and the command that reproduces them. + * + * The command is the load-bearing element. Everything else on this page is us + * telling you what happened; this is the line you paste into your own terminal + * to find out for yourself, which is why it is verbatim and copy-pasteable + * rather than prettified into something that would not actually run. + */ +export function ProvenanceCard({ provenance, run, className }: ProvenanceCardProps) { + const [copyState, setCopyState] = useState('idle'); + + const copy = async () => { + try { + // `navigator.clipboard` is undefined on a non-secure origin, which is + // exactly what a colleague testing over a LAN IP will hit. Fail visibly. + await navigator.clipboard.writeText(provenance.command); + setCopyState('copied'); + window.setTimeout(() => setCopyState('idle'), 2000); + } catch { + setCopyState('failed'); + } + }; + + const rows: { label: string; value: string; mono?: boolean }[] = [ + { label: 'Environment package', value: provenance.envPackage, mono: true }, + { label: 'Taskset', value: provenance.tasksetId, mono: true }, + { label: 'verifiers version', value: provenance.verifiersVersion, mono: true }, + ]; + if (run) { + rows.push( + { label: 'Model', value: run.model, mono: true }, + { label: 'Captured', value: formatDate(run.capturedAt) }, + { label: 'Seed', value: String(run.seed), mono: true }, + ); + if (run.intervention) rows.push({ label: 'Intervention', value: run.intervention }); + } + + return ( +
+
+

Provenance

+
+ +
+ {rows.map((row) => ( +
+
{row.label}
+
+ {row.value} +
+
+ ))} +
+ +
+
+

+ Run it yourself +

+ +
+
+          {provenance.command}
+        
+

+ {copyState === 'copied' + ? 'Command copied to your clipboard.' + : copyState === 'failed' + ? 'Your browser blocked clipboard access — select the command above and copy it manually.' + : ''} +

+
+ + {provenance.credits.length > 0 ? ( +
+

+ Built on +

+ +
+ ) : null} +
+ ); +} diff --git a/src/components/demo/RewardEditor.tsx b/src/components/demo/RewardEditor.tsx new file mode 100644 index 0000000..e7bf9ae --- /dev/null +++ b/src/components/demo/RewardEditor.tsx @@ -0,0 +1,340 @@ +import { useMemo, useState } from 'react'; +import * as Slider from '@radix-ui/react-slider'; +import { ArrowDown, ArrowUp, Minus, RotateCcw } from 'lucide-react'; +import type { RewardSpec, RewardValues } from '@/lib/demo-kit/types'; +import { cn } from '@/lib/utils'; +import { formatNumber, formatOrDash } from './format'; +import { scoreReward, shippedWeights } from './reward-math'; +import { EditedChip } from './StatStrip'; + +/** One recorded arm — a run, or a group of runs already reduced to one score. */ +export interface RewardArm { + id: string; + label: string; + /** The environment's per-component scores. Re-weighted, never re-run. */ + values: RewardValues; + note?: string; +} + +export interface RewardPreset { + id: string; + label: string; + description?: string; + weights: Record; +} + +export interface RewardEditorProps { + spec: RewardSpec; + arms: RewardArm[]; + /** Two is the right number. More and the visitor reads instead of playing. */ + presets?: RewardPreset[]; + onWeightsChange?: (weights: Record, edited: boolean) => void; + className?: string; +} + +const STEP = 0.05; + +/** + * Presets built from the spec's own labels, for a demo that does not supply + * its own. Both are stated as the choice a buyer would actually argue for in a + * meeting, not as "preset A" and "preset B". + */ +function derivePresets(spec: RewardSpec): RewardPreset[] { + const shipped = shippedWeights(spec); + const counterweights = spec.components.filter((c) => c.role === 'counterweight'); + const objectives = spec.components.filter((c) => c.role === 'objective'); + const presets: RewardPreset[] = [ + { + id: 'shipped', + label: 'What we ship', + description: 'The weights in the environment as committed.', + weights: shipped, + }, + ]; + + const firstObjective = objectives[0]; + if (counterweights.length > 0 && firstObjective) { + const onlyObjective = { ...shipped }; + for (const component of counterweights) onlyObjective[component.key] = 0; + presets.push({ + id: 'objective-only', + label: `${firstObjective.label} at any cost`, + description: `Drops ${counterweights + .map((c) => c.label.toLowerCase()) + .join(' and ')} to zero.`, + weights: onlyObjective, + }); + + const doubled = { ...shipped }; + for (const component of counterweights) doubled[component.key] = component.weight * 2; + presets.push({ + id: 'counterweight-heavy', + label: `Double ${counterweights[0]?.label.toLowerCase() ?? 'the counterweight'}`, + description: 'What a risk-averse buyer would ask for.', + weights: doubled, + }); + } + return presets; +} + +function weightsEqual(a: Record, b: Record): boolean { + const keys = new Set([...Object.keys(a), ...Object.keys(b)]); + for (const key of keys) { + if (Math.abs((a[key] ?? 0) - (b[key] ?? 0)) > 1e-9) return false; + } + return true; +} + +/** + * Change what "good" means and watch the ranking move. + * + * The honesty problem this component has to solve: re-weighting recorded scores + * is NOT training. It shows you the ranking a different reward would have + * produced over these exact attempts; it cannot show you the different attempts + * a model trained on that reward would have made. That distinction is the + * permanent caption at the bottom, and it is not collapsible. + */ +export function RewardEditor({ + spec, + arms, + presets, + onWeightsChange, + className, +}: RewardEditorProps) { + const shipped = useMemo(() => shippedWeights(spec), [spec]); + const [weights, setWeights] = useState>(shipped); + const effectivePresets = useMemo(() => presets ?? derivePresets(spec), [presets, spec]); + + const bounds = useMemo(() => { + const values = spec.components.map((c) => c.weight); + const max = Math.max(2, ...values.map((v) => Math.ceil(Math.abs(v) * 2))); + const min = Math.min(0, ...values.map((v) => Math.floor(v))); + return { min, max }; + }, [spec]); + + const edited = !weightsEqual(weights, shipped); + + const apply = (next: Record) => { + setWeights(next); + onWeightsChange?.(next, !weightsEqual(next, shipped)); + }; + + const ranked = useMemo(() => { + const shippedTotals = new Map( + arms.map((arm) => [arm.id, scoreReward(spec, arm.values).total]), + ); + const rows = arms.map((arm) => ({ + arm, + total: scoreReward(spec, arm.values, weights).total, + shippedTotal: shippedTotals.get(arm.id) ?? null, + })); + // Nulls sort last: an unscored arm is not a zero-scoring arm. + const byTotal = (a: { total: number | null }, b: { total: number | null }) => { + if (a.total === null && b.total === null) return 0; + if (a.total === null) return 1; + if (b.total === null) return -1; + return b.total - a.total; + }; + const shippedOrder = [...rows] + .sort((a, b) => byTotal({ total: a.shippedTotal }, { total: b.shippedTotal })) + .map((row) => row.arm.id); + return [...rows].sort(byTotal).map((row, index) => ({ + ...row, + rank: index + 1, + shippedRank: shippedOrder.indexOf(row.arm.id) + 1, + })); + }, [arms, spec, weights]); + + const span = useMemo(() => { + const totals = ranked.map((row) => row.total).filter((t): t is number => t !== null); + if (totals.length === 0) return { lo: 0, hi: 1 }; + const lo = Math.min(0, ...totals); + const hi = Math.max(...totals); + return { lo, hi: hi === lo ? lo + 1 : hi }; + }, [ranked]); + + const leader = ranked[0]; + + return ( +
+
+
+

Change what good means

+ {edited ? : null} + +
+ +
+ {effectivePresets.map((preset) => { + const active = weightsEqual(weights, preset.weights); + return ( + + ); + })} +
+ +
+ {spec.components.map((component) => { + const value = weights[component.key] ?? component.weight; + const changed = Math.abs(value - component.weight) > 1e-9; + return ( +
+
+ {/* A
+

{component.description}

+ + apply({ ...weights, [component.key]: next[0] ?? component.weight }) + } + > + + + + {/* 44px of hit area around a 16px dot: the visible thumb is + small enough to read the track under it, and still catches + a thumb on a phone. */} + + +
+ ); + })} +
+
+ +
+

+ Ranking under this reward {edited ? : null} +

+ +
    + {ranked.map((row) => { + const moved = row.rank - row.shippedRank; + const width = + row.total === null + ? 0 + : Math.max(2, ((row.total - span.lo) / (span.hi - span.lo)) * 100); + return ( +
  1. +
    + {row.rank} + + {row.arm.label} + + + + {formatOrDash(row.total)} + +
    +
  2. + ); + })} +
+ + {/* User-initiated, so it is safe to announce here without fighting the + shell's step-change region: the two never fire from one action. */} +

+ {leader + ? `Leading under this reward: ${leader.arm.label}, ${formatOrDash(leader.total)}.` + : 'No arms to rank.'} +

+ +

+ We re-scored the same recorded attempts under your reward. Training on it would change + the behaviour, not just the ranking. +

+
+
+ ); +} + +function RankMove({ moved }: { moved: number }) { + if (moved === 0) { + return ( + + + ); + } + const up = moved < 0; + return ( + + {up ? ( + + ); +} diff --git a/src/components/demo/mock.tsx b/src/components/demo/mock.tsx new file mode 100644 index 0000000..8b4428d --- /dev/null +++ b/src/components/demo/mock.tsx @@ -0,0 +1,384 @@ +/** + * A hand-written demo, used to build and exercise the shell without waiting for + * real fixtures. + * + * It is NOT a fixture and it is not shipped as a demo: nothing in + * `src/demos/` imports it, and the shell mounts it only for the reserved slug + * `__mock` in a dev build. It exists so that every surface in this directory + * has something complete to render — including the awkward cases a real trace + * eventually produces: a step with no reasoning, a null model call, a + * not-scored reward component, and a truncated run. + */ +import type { + DemoEpisode, + DemoModule, + DemoStep, + RewardValues, + RunRef, +} from '@/lib/demo-kit/types'; +import { cn } from '@/lib/utils'; + +export type MarkKind = 'exact' | 'present' | 'absent'; + +export interface MockGuess { + word: string; + marks: MarkKind[]; +} + +export interface MockState { + answer: string; + guesses: MockGuess[]; + solved: boolean; +} + +const ANSWER = 'CRANE'; +const MAX_GUESSES = 6; + +/** Standard Wordle marking, duplicates and all. */ +function mark(guess: string, answer: string): MarkKind[] { + const marks: MarkKind[] = Array.from({ length: guess.length }, () => 'absent'); + const pool = new Map(); + for (let i = 0; i < answer.length; i += 1) { + const letter = answer[i] as string; + if (guess[i] === letter) { + marks[i] = 'exact'; + } else { + pool.set(letter, (pool.get(letter) ?? 0) + 1); + } + } + for (let i = 0; i < guess.length; i += 1) { + if (marks[i] === 'exact') continue; + const letter = guess[i] as string; + const left = pool.get(letter) ?? 0; + if (left > 0) { + marks[i] = 'present'; + pool.set(letter, left - 1); + } + } + return marks; +} + +const TILE: Record = { + exact: 'bg-tile-exact text-tile-exact-fg', + present: 'bg-tile-present text-tile-present-fg', + absent: 'bg-tile-absent text-tile-absent-fg', +}; + +const GLYPH: Record = { exact: '●', present: '◐', absent: '○' }; + +function MockSurface({ state, compact = false }: { state: MockState; compact?: boolean }) { + const rows = Array.from({ length: MAX_GUESSES }, (_, index) => state.guesses[index]); + return ( +
guess.word) + .join(', ')}` + } + > + {rows.map((guess, rowIndex) => ( +
+ {Array.from({ length: 5 }, (_unused, colIndex) => { + const letter = guess?.word[colIndex]; + const kind = guess?.marks[colIndex]; + return ( + + {compact ? (kind ? GLYPH[kind] : '') : (letter ?? '')} + + ); + })} +
+ ))} +
+ ); +} + +const REWARD_SOURCE = `import verifiers as vf + + +def reward_solved(state) -> float: + """1.0 if the answer was found, else 0.0.""" + return 1.0 if state["solved"] else 0.0 + + +# --8<-- efficiency +def reward_efficiency(state) -> float: + """Pays for finding it early, and pays nothing for finding it late. + + This is the counterweight. Without it the highest-scoring policy is one + that burns every guess it is allowed, because the objective alone cannot + tell a lucky third guess from a grudging sixth. + """ + if not state["solved"]: + return 0.0 + used = len(state["guesses"]) + return max(0.0, (MAX_GUESSES - used + 1) / MAX_GUESSES) +# --8<-- + + +def reward_legal(state) -> float: + """Every guess was a real five-letter word in the allowed list.""" + return 1.0 if all(g in ALLOWED for g in state["guesses"]) else 0.0 +`; + +function turnBoard(episode: DemoEpisode, upto: number): MockState { + const guesses: MockGuess[] = []; + for (let i = 0; i <= upto && i < episode.turns.length; i += 1) { + const info = episode.turns[i]?.info; + const word = typeof info?.['guess'] === 'string' ? info['guess'] : null; + if (!word) continue; + guesses.push({ word, marks: mark(word, ANSWER) }); + } + const last = guesses[guesses.length - 1]; + return { answer: ANSWER, guesses, solved: last?.word === ANSWER }; +} + +function adapt(episode: DemoEpisode): DemoStep[] { + return episode.turns.map((turn, index) => { + const state = turnBoard(episode, index); + const last = state.guesses[state.guesses.length - 1]; + const word = last?.word ?? 'no guess'; + const exact = last?.marks.filter((m) => m === 'exact').length ?? 0; + return { + index, + state, + reply: turn.reply, + reasoning: turn.reasoning, + call: turn.call, + announce: state.solved + ? `Guess ${index + 1}: ${word}. Solved.` + : `Guess ${index + 1}: ${word}. ${exact} letters in the right place.`, + caption: word.toLowerCase(), + }; + }); +} + +function verify(episode: DemoEpisode): RewardValues | null { + if (episode.truncated) return null; + const final = turnBoard(episode, episode.turns.length - 1); + const used = final.guesses.length; + return { + solved: final.solved ? 1 : 0, + efficiency: final.solved ? Math.max(0, (MAX_GUESSES - used + 1) / MAX_GUESSES) : 0, + legal: 1, + }; +} + +export const mockDemo: DemoModule = { + meta: { + slug: '__mock', + title: 'Five-letter word game (mock)', + tagline: 'An agent guesses a hidden five-letter word from letter feedback.', + vertical: 'reference', + status: 'live', + order: 0, + icon: 'Grid3x3', + persona: 'Anyone deciding whether to fund an environment', + rewardLine: 'Pays for solving; takes away for stalling.', + ogImage: '/og/mock.png', + }, + narrative: { + thesis: + 'An environment is an eval you can take the gradient of. The task, the legal moves and the grader are all code — so you can change what "good" means and watch the number move.', + anxiety: 'Is this a benchmark I read, or a thing I can actually change?', + beats: [ + { id: 'hero', title: 'The run', claim: 'This is a recorded rollout, not a live request.', surface: 'hero' }, + { id: 'anatomy', title: 'The machine', claim: 'Four boxes: task, legal actions, grader, score.', surface: 'anatomy' }, + { id: 'play', title: 'Watch it think', claim: 'Every move has a reason and a cost, both recorded.', surface: 'split-play' }, + { id: 'reward', title: 'Change what good means', claim: 'Move a weight and the ranking moves with it.', surface: 'reward-editor' }, + { id: 'metric', title: 'The number that moves', claim: 'The score is a measurement, not a claim.', surface: 'metric' }, + { id: 'receipt', title: 'The receipts', claim: 'Every number here has a command that reproduces it.', surface: 'receipt' }, + { id: 'limits', title: 'What this does not teach', claim: 'A word game is not your business process.', surface: 'limits' }, + ], + limits: [ + { + text: 'A five-letter word has one right answer. Most business decisions do not, and a grader that pretends otherwise scores confidence rather than correctness.', + answeredBy: 'claims-triage', + }, + { + text: 'Nothing here has a cost of being wrong. A real reward has to price the mistake, not just count the win.', + }, + ], + }, + reward: { + components: [ + { + key: 'solved', + label: 'Found the word', + description: 'One point if the hidden word was guessed within six tries.', + weight: 1, + role: 'objective', + }, + { + key: 'efficiency', + label: 'Found it early', + description: 'Pays more the fewer guesses it took. Zero if it never got there.', + weight: 0.5, + role: 'counterweight', + }, + { + key: 'legal', + label: 'Played legal words', + description: 'Every guess was a real five-letter word from the allowed list.', + weight: 0.25, + role: 'gate', + }, + ], + metrics: [ + { key: 'guesses_used', label: 'Guesses used', description: 'How many of the six were spent.' }, + { key: 'unique_letters', label: 'Unique letters tried', description: 'Breadth of the search.' }, + ], + source: { path: 'envs/wordle_five/wordle_five/rewards.py', code: REWARD_SOURCE, marker: '--8<-- efficiency' }, + }, + provenance: { + envPackage: 'wordle_five', + tasksetId: 'wordle-five-v0-mock', + verifiersVersion: '0.1.0', + command: 'uv run vf-eval wordle-five -n 8 -m gpt-4.1-mini --seed 7', + credits: [ + { label: 'verifiers', href: 'https://github.com/PrimeIntellect-ai/verifiers' }, + ], + }, + anatomy: { + task: 'Guess a hidden five-letter word in six tries, using the coloured feedback from each guess.', + actions: 'One legal five-letter word per turn, drawn from the allowed list. Nothing else is a move.', + grader: 'Deterministic Python: it marks the guess against the answer, checks legality, and scores the episode. No model judges it.', + score: 'A weighted sum: one point for finding the word, half a point scaled by how early, a quarter as a legality gate.', + }, + adapt, + Surface: MockSurface, + verify, +}; + +function call(prompt: number, completion: number, reasoning: number | null, ms: number) { + return { + promptTokens: prompt, + completionTokens: completion, + reasoningTokens: reasoning, + durationMs: ms, + finishReason: 'stop', + }; +} + +export const mockEpisodes: Record = { + 'mock-base': { + runId: 'mock-base', + seed: 7, + model: 'gpt-4.1-mini', + capturedAt: '2026-08-12', + rewards: { solved: 1, efficiency: 0.5, legal: 1 }, + metrics: { guesses_used: 4, unique_letters: 14 }, + outcome: 'solved', + turns: [ + { + reply: 'SLATE', + reasoning: + 'Opening with a word that covers three of the five most common letters and two common consonants. I want information, not a lucky hit.', + call: call(412, 6, 84, 910), + info: { guess: 'SLATE' }, + }, + { + reply: 'TRACE', + reasoning: null, + call: call(486, 6, null, 640), + info: { guess: 'TRACE' }, + }, + { + reply: 'BRACE', + reasoning: + 'A and E are placed. R is in the word but not where I put it. That leaves the first two positions to resolve.', + call: call(551, 6, 121, 1180), + info: { guess: 'BRACE' }, + }, + { + reply: 'CRANE', + reasoning: 'Only one word fits every constraint now.', + call: call(618, 6, 96, 720), + info: { guess: 'CRANE' }, + }, + ], + }, + 'mock-prompted': { + runId: 'mock-prompted', + seed: 7, + model: 'gpt-4.1-mini', + capturedAt: '2026-08-12', + // Written as the expression the environment computes, not as a decimal + // literal: a hand-rounded 0.8333 would fail our own verifier. + rewards: { solved: 1, efficiency: 5 / 6, legal: 1 }, + metrics: { guesses_used: 2, unique_letters: 9 }, + outcome: 'solved', + turns: [ + { + reply: 'TRACE', + reasoning: 'Told to open with maximum letter coverage and to commit early once a word fits.', + call: call(455, 6, 140, 1020), + info: { guess: 'TRACE' }, + }, + { + reply: 'CRANE', + reasoning: 'Every constraint is satisfied by exactly one candidate.', + call: call(512, 6, 88, 660), + info: { guess: 'CRANE' }, + }, + ], + }, + 'mock-truncated': { + runId: 'mock-truncated', + seed: 11, + model: 'gpt-4.1-mini', + capturedAt: '2026-08-12', + // A run that hit the turn cap: the environment never reached a terminal + // state, so two of the three components were never scored. + rewards: { solved: null, efficiency: null, legal: 1 }, + metrics: { guesses_used: 2, unique_letters: 8 }, + truncated: true, + outcome: 'aborted', + turns: [ + { reply: 'AUDIO', reasoning: null, call: call(410, 6, null, 880), info: { guess: 'AUDIO' } }, + { reply: null, reasoning: null, call: null, info: {} }, + ], + }, +}; + +export const mockRuns: RunRef[] = [ + { + id: 'mock-base', + label: 'Out of the box', + path: '/traces/__mock/base.json', + kind: 'recorded', + model: 'gpt-4.1-mini', + capturedAt: '2026-08-12', + seed: 7, + }, + { + id: 'mock-prompted', + label: 'With a strategy prompt', + path: '/traces/__mock/prompted.json', + kind: 'intervened', + intervention: 'System prompt rewritten', + model: 'gpt-4.1-mini', + capturedAt: '2026-08-12', + seed: 7, + }, + { + id: 'mock-truncated', + label: 'Hit the turn cap', + path: '/traces/__mock/truncated.json', + kind: 'recorded', + model: 'gpt-4.1-mini', + capturedAt: '2026-08-12', + seed: 11, + }, +]; diff --git a/src/components/site/ContrastToggle.tsx b/src/components/site/ContrastToggle.tsx index ebda0f0..88d0936 100644 --- a/src/components/site/ContrastToggle.tsx +++ b/src/components/site/ContrastToggle.tsx @@ -8,9 +8,9 @@ const STORAGE_KEY = 'pig-demo:contrast'; /** * `localStorage` is not always readable. In a cross-origin iframe with third- - * party storage blocked, and in Safari private mode, the getter itself THROWS - * rather than returning null — so every access has to be wrapped, not just - * null-checked. Unreadable storage means "off", never a crash. + * party storage blocked, and in Safari's private mode, the accessor itself + * THROWS rather than returning null — so every access has to be wrapped, not + * just null-checked. Unreadable storage means "off", never a crash. */ function readStored(): boolean { try { @@ -24,51 +24,72 @@ function writeStored(high: boolean): void { try { window.localStorage.setItem(STORAGE_KEY, high ? 'high' : 'normal'); } catch { - /* Preference is session-only here. The toggle still works. */ + /* Preference is session-only in this context. The toggle still works. */ } } +/** + * One value shared by every mounted toggle, not one `useState` each. + * + * The header renders this control twice — once in the desktop bar, once inside + * the mobile sheet — and only one of them is ever visible. With local state the + * hidden one keeps a stale `aria-pressed`, so a visitor who toggles on a phone + * and then rotates into the desktop layout is told the setting is off while the + * page is plainly showing it on. + */ +let high = false; +let initialised = false; +const listeners = new Set<() => void>(); + +function applyToRoot(next: boolean): void { + const root = document.documentElement; + // Removed rather than set to "normal": the CSS keys off + // `:root[data-contrast='high']`, and a leftover attribute makes the DOM lie + // about which palette is actually applied. + if (next) root.setAttribute('data-contrast', 'high'); + else root.removeAttribute('data-contrast'); +} + +function setHigh(next: boolean): void { + high = next; + applyToRoot(next); + writeStored(next); + for (const listener of listeners) listener(); +} + +function subscribe(listener: () => void): () => void { + if (!initialised) { + initialised = true; + high = readStored(); + applyToRoot(high); + } + listeners.add(listener); + return () => { + listeners.delete(listener); + }; +} + export function ContrastToggle({ className }: { className?: string }) { - const [high, setHigh] = React.useState(false); - const [mounted, setMounted] = React.useState(false); - - React.useEffect(() => { - const stored = readStored(); - setHigh(stored); - setMounted(true); - }, []); - - React.useEffect(() => { - if (!mounted) return; - const root = document.documentElement; - // Removing the attribute rather than setting it to "normal": the CSS keys - // off `:root[data-contrast='high']`, and leaving a stale attribute behind - // makes the DOM lie about the palette that is actually applied. - if (high) root.setAttribute('data-contrast', 'high'); - else root.removeAttribute('data-contrast'); - }, [high, mounted]); + // The server snapshot is `false` so a prerendered page never claims a + // preference it cannot know; the real value lands on the first subscribe. + const isHigh = React.useSyncExternalStore( + subscribe, + () => high, + () => false, + ); return ( ); } diff --git a/src/components/site/SiteHeader.tsx b/src/components/site/SiteHeader.tsx index 6645a73..044811e 100644 --- a/src/components/site/SiteHeader.tsx +++ b/src/components/site/SiteHeader.tsx @@ -247,7 +247,14 @@ function TasksetSource({ className }: { className?: string }) {

The actual taskset

-
+      {/*
+        Wrapped, not scrolled. The source is quoted verbatim — reformatting it
+        to fit would make it stop being a quote — and the widest line is a third
+        wider than this column at any font size worth reading. In a panel whose
+        entire job is "this is real code", a visible wrap beats a third of the
+        line hidden behind a scrollbar nobody in a boardroom will drag.
+      */}
+      
         {TASKSET_SOURCE}
       
-
+ // 820px, not more: the panel hangs off the LEFT edge of the nav, which + // starts about 100px in, so anything wider than this pushes past the right + // edge of a 1024px laptop and gives the whole page a horizontal scrollbar. +
+
{CONCEPTS.map((concept) => (
diff --git a/src/content/lineup.ts b/src/content/lineup.ts index 57ac253..a1421a8 100644 --- a/src/content/lineup.ts +++ b/src/content/lineup.ts @@ -6,7 +6,7 @@ * or path moves, it moves in one line instead of in four pages. */ -import { demos } from '@/lib/demo-kit'; +import { demos } from '@/lib/demo-kit/registry'; import type { DemoMeta, Vertical } from '@/lib/demo-kit/types'; import { VERTICALS, verticalByKey } from '@/content/verticals'; import type { VerticalEntry } from '@/content/verticals'; @@ -26,13 +26,15 @@ export const routes = { galleryFiltered: (key: Vertical) => `/gallery?vertical=${key}`, } as const; -/** The public repository. Every claim on the site is meant to end up here. */ -export const REPO_URL = 'https://github.com/karti-ai/PIG-Demo'; +/** + * The repository link belongs to the chrome, which already owns it. Re-exported + * rather than repeated so the marketing pages and the footer cannot end up + * pointing at two different repositories. + */ +export { REPO_URL } from '@/components/site/links'; -/** Registry order is authorial; this is the order every list renders in. */ -export const allDemos: readonly DemoMeta[] = [...demos].sort( - (a, b) => a.order - b.order || a.slug.localeCompare(b.slug), -); +/** Already sorted by `order` then slug by the registry. Never sort it again. */ +export const allDemos: readonly DemoMeta[] = demos; export const liveDemos: readonly DemoMeta[] = allDemos.filter((d) => d.status === 'live'); diff --git a/src/lib/demo-kit/index.ts b/src/lib/demo-kit/index.ts index b296e5d..d204503 100644 --- a/src/lib/demo-kit/index.ts +++ b/src/lib/demo-kit/index.ts @@ -1,24 +1,24 @@ /** - * The demo-kit public barrel. + * The demo-kit barrel. * - * This is the ONLY module a demo under `src/demos/` is allowed to import from - * the shared shell, and it deliberately exposes a small surface: the contract - * types, the two `define*` wrappers, and the two pure helpers a demo's own - * surface might need to render a score honestly. + * Two audiences, and the split between them matters. * - * The player, the registry, the verifier and the reward editor's arithmetic are - * NOT here. They are shell machinery — a demo that reaches for `usePlayer` is a - * demo that has started rendering its own chrome, and the whole point of the - * contract is that the shell owns chrome so every demo gets the same one. The - * shell imports those from their own modules: + * A DEMO under `src/demos/` may import from `@/lib/demo-kit` and from nothing + * else in the shell. What it should actually reach for is the first block + * below: the contract types, the two `define*` wrappers, and the two pure + * helpers it needs to render a score without inventing one. That restriction is + * enforced by `scripts/check-demos.mjs`, not by what this file happens to + * export — the shell's own surfaces import through here too, and splitting them + * into a second barrel would only mean the check script had two paths to allow + * instead of one. * - * import { listDemos, loadDemoModule } from '@/lib/demo-kit/registry'; - * import { usePlayer } from '@/lib/demo-kit/player'; - * import { loadEpisode, listRuns } from '@/lib/demo-kit/episode'; - * import { decompose, reweight } from '@/lib/demo-kit/reward'; - * import { verifyEpisode } from '@/lib/demo-kit/verify'; + * The SHELL may import anything here, or reach into the individual modules + * (`@/lib/demo-kit/player`, `/registry`, `/episode`, `/reward`, `/verify`) when + * it wants one thing without the rest. */ +/* -- The contract. Demos live here. --------------------------------------- */ + export type { DemoEpisode, DemoMeta, @@ -39,5 +39,52 @@ export type { export { defineDemo, defineMeta } from './define'; -/** `null` is "not scored", never 0.0. Demos render absences with these two. */ +/** `null` is "not scored", never 0.0. Every absence goes through these two. */ export { isNotScored, rewardTotal } from './episode'; + +/* -- Shell machinery. ------------------------------------------------------ */ + +export { + demos, + getDemo, + getVertical, + hasDemo, + listDemos, + listVerticals, + loadDemoModule, + VERTICAL_LABELS, + VERTICAL_ORDER, +} from './registry'; +export type { AnyDemoModule, VerticalGroup } from './registry'; + +export { + clearEpisodeCache, + listRuns, + loadEpisode, + loadManifest, + MANIFEST_PATH, + scoredCount, +} from './episode'; +export type { RunManifest } from './episode'; + +export { + FALLBACK_STEP_MS, + isPlaybackSpeed, + PLAYBACK_SPEEDS, + usePlayer, + usePrefersReducedMotion, +} from './player'; +export type { PlaybackSpeed, Player, PlayerOptions } from './player'; + +export { + decompose, + isEdited, + pruneOverrides, + reweight, + WEIGHT_EPSILON, + weightsAreDegenerate, +} from './reward'; +export type { RewardBreakdown, RewardRow, WeightOverrides } from './reward'; + +export { recomputeRewards, verifyEpisode, verifySummary, VERIFY_TOLERANCE } from './verify'; +export type { ComponentComparison, VerifyResult, VerifyStatus } from './verify'; diff --git a/src/lib/demo-kit/registry.ts b/src/lib/demo-kit/registry.ts index 89d49f8..d143578 100644 --- a/src/lib/demo-kit/registry.ts +++ b/src/lib/demo-kit/registry.ts @@ -179,6 +179,14 @@ const demosBySlug: ReadonlyMap = (() => { const orderedDemos: readonly DemoMeta[] = [...demosBySlug.values()].sort(byOrderThenSlug); +/** + * The same list as `listDemos()`, as a frozen constant. + * + * Handy where a module wants the lineup at import time rather than in a render. + * It is the SAME array every caller sees — never sort or splice it in place. + */ +export const demos: readonly DemoMeta[] = orderedDemos; + /** Every demo that survived validation, sorted by `order` then slug. */ export function listDemos(): DemoMeta[] { return [...orderedDemos]; diff --git a/src/lib/demo-kit/verify.ts b/src/lib/demo-kit/verify.ts index 9b911c2..b714c5a 100644 --- a/src/lib/demo-kit/verify.ts +++ b/src/lib/demo-kit/verify.ts @@ -137,6 +137,29 @@ export function verifyEpisode(module: AnyDemoModule, episode: DemoEpisode): Veri return result; } +/** + * Just the recomputed numbers, with none of the comparison. + * + * `verifyEpisode` is the one to reach for — it does the comparing, and the + * comparing is where the honesty rules live. This exists for a surface that + * wants to render the recomputed values itself and only needs the safe call: + * a grader that is missing, throws, or declines returns `null` rather than + * propagating, so no caller can turn a broken verifier into a zero. + */ +export function recomputeRewards( + module: AnyDemoModule, + episode: DemoEpisode, +): RewardValues | null { + if (typeof module.verify !== 'function') return null; + if (episode.truncated === true) return null; + try { + return module.verify(episode); + } catch (error: unknown) { + console.error('[demo-kit] in-browser grader threw', error); + return null; + } +} + /** One line an exec can read, given a result. Keeps the wording in one place. */ export function verifySummary(result: VerifyResult): string { switch (result.status) { diff --git a/src/pages/Gallery.tsx b/src/pages/Gallery.tsx index f2dd229..965de58 100644 --- a/src/pages/Gallery.tsx +++ b/src/pages/Gallery.tsx @@ -3,21 +3,26 @@ import { Link, useSearchParams } from 'react-router-dom'; import { ArrowRight, FileText } from 'lucide-react'; import { iconFor } from '@/content/icons'; -import { allDemos, routes, verticalForDemo, verticalKeysInUse } from '@/content/lineup'; -import { PROPOSAL_NOTICE, verticalByKey } from '@/content/verticals'; +import { + allDemos, + demosForVertical, + lineup, + routes, + verticalForDemo, + verticalKeysInUse, +} from '@/content/lineup'; +import { PROPOSAL_NOTICE } from '@/content/verticals'; +// The registry owns the taxonomy's exec-facing names. The lineup's own titles +// are longer marketing headings ("Customer Support Resolution") and would wrap +// two lines inside a filter chip on a phone, so chips use the registry label. +import { VERTICAL_LABELS } from '@/lib/demo-kit/registry'; import type { Vertical } from '@/lib/demo-kit/types'; import * as s from '@/content/styles'; const ALL = 'all'; -/** - * A label for a `Vertical` key. Everything except `reference` has a vertical - * entry to borrow the title from; `reference` is the hello-world demo and - * belongs to no industry, so it is named for what it is. - */ function verticalLabel(key: Vertical): string { - if (key === 'reference') return 'Reference'; - return verticalByKey(key)?.title ?? key; + return VERTICAL_LABELS[key]; } export default function Gallery() { @@ -44,6 +49,8 @@ export default function Gallery() { const filters: readonly (Vertical | typeof ALL)[] = [ALL, ...verticalKeysInUse]; + const unbuilt = lineup.filter((v) => demosForVertical(v.key).length === 0).length; + return (

Gallery

@@ -114,6 +121,7 @@ export default function Gallery() { // A spec is dimmed but never disabled: it goes to a real // page with a real specification on it, which is the only // thing that makes dimming it honest rather than teasing. + aria-label={`${demo.title} — ${isSpec ? 'read the specification' : 'play it'}`} className={`${s.cardLink} h-full ${isSpec ? 'opacity-70 hover:opacity-100' : ''}`} to={routes.demo(demo.slug)} > @@ -169,8 +177,12 @@ export default function Gallery() { )} -
-

The eleven we have not built

+
+ {/* Counted, not typed. A hard-coded "eleven" on a site about checkable + numbers goes stale the first time a demo ships. */} +

+ The {unbuilt} we have not built +

The lineup is a set of proposals, written to the same four-part shape as the live one. Reading one takes a minute and tells you whether the idea survives contact with your own diff --git a/src/pages/Home.tsx b/src/pages/Home.tsx index 2e25543..a5ad6c4 100644 --- a/src/pages/Home.tsx +++ b/src/pages/Home.tsx @@ -77,6 +77,7 @@ export default function Home() { {helloWorldCitations.map((c) => (

  • - + Live @@ -253,7 +258,11 @@ export default function Home() { const Icon = iconFor(v.icon); return (
  • - +