The 600-second per-model-call ceiling that capped the thinking-on run at n=6 is gone: DEFAULT_TIMEOUT is read=None in all nine venvs. There is NO verifiers 0.4.0. The `version = "0.4.0"` in the vendored tree belongs to a [[tool.uv.dependency-metadata]] block for nemo-gym, and reading it as verifiers' own sends you hunting a release that does not exist and concluding the bump is blocked. The vendored checkout is v0.3.0-59-g4bcb48e5, published to PyPI as 0.3.1.dev59, and the installed wheel diffs clean against the whole vendored v1 tree. Pinned as an ordinary index dependency: a path source is reproducible only on this box and a git URL pins a commit no wheel matches. ⚠️ dev59 changes the default agent runtime from subprocess to `prime`, a REMOTE sandbox. Without an explicit runtime every eval here exits "not authenticated with prime" before a single rollout — having already created its output directory and a zero-line traces.jsonl, so it reads as a failed eval rather than an unauthenticated one. All nine configs now say subprocess. Nothing here needs a sandbox: the reward is a pure function of the trace and the harness is null. Output layout is flat now — <env>--<model>--<harness>--<hex8>/ with configs/, traces.jsonl and a logs/eval.log that 0.3.0 never wrote. `-o` is the output_dir, not the run dir; `--run.dir` pins the leaf. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
33 lines
1.6 KiB
TOML
33 lines
1.6 KiB
TOML
# grand-exchange-live — see environments/grand_exchange_live for what it measures.
|
|
#
|
|
# uv run --project environments/grand_exchange_live eval @ configs/grand_exchange_live.toml \
|
|
# --model <model-id>
|
|
#
|
|
# Eight looks per episode against one basket, so a rollout here costs roughly eight times a
|
|
# one-shot `grand-exchange` rollout in requests and rather more than that in tokens: the
|
|
# whole conversation is replayed to the model on every segment. `num_tasks` is half the
|
|
# one-shot config's for that reason, not because fewer baskets measure better — realised
|
|
# P&L over sixty ticks is the noisiest number in this repository and the ladder is read
|
|
# over five blocks of twenty-four.
|
|
num_tasks = 8
|
|
num_rollouts = 2
|
|
|
|
[env.taskset]
|
|
id = "grand-exchange-live"
|
|
num_items = 5 # items in the basket, one per price tier
|
|
visible = 56 # warmup ticks, read before the first look
|
|
# The graded window is not configurable: 8 looks x 7 ticks + a 4-tick tail = 60, and every
|
|
# constant in the reward was measured against that shape (`measure_ladder.py --sweep-shape`).
|
|
|
|
[env.agent.harness]
|
|
id = "null" # pure-trace reward: a chat loop, no tools, no runtime
|
|
|
|
[env.agent.runtime]
|
|
# ⚠️ REQUIRED since verifiers 0.3.1.dev59, which changed the default agent runtime from
|
|
# `subprocess` to `prime` — a REMOTE sandbox. Without this the run exits
|
|
# `SystemExit: not authenticated with prime` before a single rollout, having already created
|
|
# its output directory and a zero-line traces.jsonl, so it reads as a failed eval rather than
|
|
# an unauthenticated one. Nothing here needs a sandbox: the reward is a pure function of the
|
|
# trace and the harness is `null`.
|
|
type = "subprocess"
|