# grand-exchange-live — see environments/grand_exchange_live for what it measures. # # uv run --project environments/grand_exchange_live eval @ configs/grand_exchange_live.toml \ # --model # # Eight looks per episode against one basket, so a rollout here costs roughly eight times a # one-shot `grand-exchange` rollout in requests and rather more than that in tokens: the # whole conversation is replayed to the model on every segment. `num_tasks` is half the # one-shot config's for that reason, not because fewer baskets measure better — realised # P&L over sixty ticks is the noisiest number in this repository and the ladder is read # over five blocks of twenty-four. num_tasks = 8 num_rollouts = 2 [env.taskset] id = "grand-exchange-live" num_items = 5 # items in the basket, one per price tier visible = 56 # warmup ticks, read before the first look # The graded window is not configurable: 8 looks x 7 ticks + a 4-tick tail = 60, and every # constant in the reward was measured against that shape (`measure_ladder.py --sweep-shape`). [env.agent.harness] id = "null" # pure-trace reward: a chat loop, no tools, no runtime