Files
arena/environments/tera_spatial/measure_hold.py
T
kartiandClaude Opus 5 4147de0282 tera-crow-nav: the first spatial taskset, and the hold sweep that sized its budget
The bridge becomes a runnable taskset. The crux is that these are continuous
control environments and the agent emits text: the model issues a control
decision that is held for N steps, and the hold length is measured rather than
guessed. At one decision every 32 steps the scripted baseline returns -2.807 and
reaches the waypoint 0% of the time, against 5.484 and 100% at every frame.

Open loop — one decision for the whole flight — returns -3.764 against an
inaction floor of -4.161. A bird that sets a course and leaves does no better
than one that does nothing, which is why crow-nav carries the strongest rule-4
claim of the five spatial environments. The budget bites from both ends: 8 turns
reach the goal 44% of the time, 12 reach 88%, 14 reach 100%.

32 episodes against Qwen3.8-27B: mean 0.4603, no malformed replies in 412, every
episode replaying to an identical checksum, and 16 of 32 spending all 16 turns.
The gate fires 43.75% of the time and separates the four scenarios cleanly —
whatever is wrong with the four dead data-environment gates, it is not that
binary gates cannot discriminate.

The oracle comes from the worker's own op. Nothing recomputes in Python a number
that came out of TypeScript, least of all the reward denominator.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-21 17:44:46 -07:00

161 lines
6.4 KiB
Python

"""How long a crow can be flown blind — the measurement the turn budget is set from.
A language model does not emit sixty control vectors a second. It emits a few
dozen replies in a whole episode. So a spatial taskset has to decide how many
simulation steps one reply is held for, and that decision is the environment:
hold too little and the model cannot reach the waypoint inside any budget worth
running; hold too much and the episode is open-loop, which is to say the turn
budget does not bind and rule 4 has nothing to say.
This file measures the frontier rather than guessing at it. It flies Tera's own
scripted baseline — the reward denominator, unmodified — at hold lengths from 1
to the whole episode, and reports for each one the return, the steps, the
decisions and whether the waypoint was reached. Only the refresh rate changes;
the policy, the scenario and the seed do not. The number that matters is where
the return falls off, because that is the boundary between "the model is
flying" and "the model set a course and left".
uv run --no-project python environments/tera_spatial/measure_hold.py
uv run --no-project python environments/tera_spatial/measure_hold.py --env drive-101-v1
Nothing here is a reward and nothing here runs a model. It is the evidence for
two integers in `configs/tera_crow_nav.toml`.
"""
from __future__ import annotations
import argparse
import statistics
from collections.abc import Sequence
from tera_spatial import ENVIRONMENT_IDS, TeraWorker
HOLDS = (1, 2, 3, 5, 8, 12, 16, 24, 32, 48, 64, 100, 150, 300)
SEEDS = (115, 2718, 40_902, 7_331)
BUDGETS = (1, 2, 4, 6, 8, 10, 12, 14, 16, 20, 24, 32)
def sweep(
worker: TeraWorker, env_id: str, holds: Sequence[int], seeds: Sequence[int]
) -> list[dict[str, float]]:
"""One row per hold length, averaged over every public scenario at every seed."""
manifest = next(m for m in worker.manifests() if m["id"] == env_id)
scenarios = [
(split, scenario_id)
for split in ("train", "dev")
for scenario_id in manifest["scenarioIds"][split]
]
rows = []
for hold in holds:
returns, steps, decisions, reached = [], [], [], 0
for split, scenario_id in scenarios:
for seed in seeds:
result = worker.oracle(
env_id, seed, {"split": split, "id": scenario_id}, hold=hold
)
returns.append(result["cumulativeReward"])
steps.append(result["steps"])
decisions.append(result["decisions"])
reached += result["terminalReason"] == "goal"
rows.append(
{
"hold": hold,
"mean_return": statistics.fmean(returns),
"worst_return": min(returns),
"mean_steps": statistics.fmean(steps),
"mean_decisions": statistics.fmean(decisions),
"goal_rate": reached / len(returns),
"episodes": len(returns),
}
)
return rows
def budget_ladder(
worker: TeraWorker,
env_id: str,
turns: Sequence[int],
max_hold: int,
seeds: Sequence[int],
) -> list[dict[str, float]]:
"""The best the baseline can do under a turn budget — the rule-4 ladder.
For each budget the baseline is flown at every legal hold length and the best
result is kept, with the episode cut off at `turns x hold` steps because that
is all the budget buys. A budget that does not change the answer is a budget
that does not bind, and this table is where that shows.
"""
manifest = next(m for m in worker.manifests() if m["id"] == env_id)
scenarios = [
(split, scenario_id)
for split in ("train", "dev")
for scenario_id in manifest["scenarioIds"][split]
]
rows = []
for budget in turns:
best = None
for hold in range(1, max_hold + 1):
returns, reached = [], 0
for split, scenario_id in scenarios:
for seed in seeds:
result = worker.oracle(
env_id,
seed,
{"split": split, "id": scenario_id},
max_steps=budget * hold,
hold=hold,
)
returns.append(result["cumulativeReward"])
reached += result["terminalReason"] == "goal"
mean = statistics.fmean(returns)
if best is None or mean > best["mean_return"]:
best = {
"turns": budget,
"hold": hold,
"mean_return": mean,
"worst_return": min(returns),
"goal_rate": reached / len(returns),
}
rows.append(best)
return rows
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--env", default="crow-nav-v1", choices=ENVIRONMENT_IDS)
parser.add_argument("--seeds", type=int, nargs="+", default=list(SEEDS))
parser.add_argument("--max-hold", type=int, default=12)
args = parser.parse_args()
with TeraWorker() as worker:
rows = sweep(worker, args.env, HOLDS, args.seeds)
floor = worker.oracle(args.env, args.seeds[0], {"split": "train"}, policy="inaction")
ladder = budget_ladder(worker, args.env, BUDGETS, args.max_hold, args.seeds)
print(f"{args.env} — Tera's scripted baseline, flown at one decision every `hold` steps")
print(f"{'hold':>5} {'return':>9} {'worst':>9} {'steps':>7} {'calls':>7} {'goal':>6}")
for row in rows:
print(
f"{row['hold']:>5} {row['mean_return']:>9.3f} {row['worst_return']:>9.3f} "
f"{row['mean_steps']:>7.1f} {row['mean_decisions']:>7.1f} {row['goal_rate']:>6.0%}"
)
print(
f"\ninaction floor (seed {args.seeds[0]}, train): "
f"{floor['cumulativeReward']:.3f} over {floor['steps']} steps "
f"({floor['terminalReason']})"
)
print(f"{rows[0]['episodes']} episodes per row.")
print(f"\nUnder a turn budget, hold capped at {args.max_hold} — best constant hold per budget")
print(f"{'turns':>6} {'hold':>5} {'return':>9} {'worst':>9} {'goal':>6}")
for row in ladder:
print(
f"{row['turns']:>6} {row['hold']:>5} {row['mean_return']:>9.3f} "
f"{row['worst_return']:>9.3f} {row['goal_rate']:>6.0%}"
)
return 0
if __name__ == "__main__":
raise SystemExit(main())