The bridge becomes a runnable taskset. The crux is that these are continuous control environments and the agent emits text: the model issues a control decision that is held for N steps, and the hold length is measured rather than guessed. At one decision every 32 steps the scripted baseline returns -2.807 and reaches the waypoint 0% of the time, against 5.484 and 100% at every frame. Open loop — one decision for the whole flight — returns -3.764 against an inaction floor of -4.161. A bird that sets a course and leaves does no better than one that does nothing, which is why crow-nav carries the strongest rule-4 claim of the five spatial environments. The budget bites from both ends: 8 turns reach the goal 44% of the time, 12 reach 88%, 14 reach 100%. 32 episodes against Qwen3.8-27B: mean 0.4603, no malformed replies in 412, every episode replaying to an identical checksum, and 16 of 32 spending all 16 turns. The gate fires 43.75% of the time and separates the four scenarios cleanly — whatever is wrong with the four dead data-environment gates, it is not that binary gates cannot discriminate. The oracle comes from the worker's own op. Nothing recomputes in Python a number that came out of TypeScript, least of all the reward denominator. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
161 lines
6.4 KiB
Python
161 lines
6.4 KiB
Python
"""How long a crow can be flown blind — the measurement the turn budget is set from.
|
|
|
|
A language model does not emit sixty control vectors a second. It emits a few
|
|
dozen replies in a whole episode. So a spatial taskset has to decide how many
|
|
simulation steps one reply is held for, and that decision is the environment:
|
|
hold too little and the model cannot reach the waypoint inside any budget worth
|
|
running; hold too much and the episode is open-loop, which is to say the turn
|
|
budget does not bind and rule 4 has nothing to say.
|
|
|
|
This file measures the frontier rather than guessing at it. It flies Tera's own
|
|
scripted baseline — the reward denominator, unmodified — at hold lengths from 1
|
|
to the whole episode, and reports for each one the return, the steps, the
|
|
decisions and whether the waypoint was reached. Only the refresh rate changes;
|
|
the policy, the scenario and the seed do not. The number that matters is where
|
|
the return falls off, because that is the boundary between "the model is
|
|
flying" and "the model set a course and left".
|
|
|
|
uv run --no-project python environments/tera_spatial/measure_hold.py
|
|
uv run --no-project python environments/tera_spatial/measure_hold.py --env drive-101-v1
|
|
|
|
Nothing here is a reward and nothing here runs a model. It is the evidence for
|
|
two integers in `configs/tera_crow_nav.toml`.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import statistics
|
|
from collections.abc import Sequence
|
|
|
|
from tera_spatial import ENVIRONMENT_IDS, TeraWorker
|
|
|
|
HOLDS = (1, 2, 3, 5, 8, 12, 16, 24, 32, 48, 64, 100, 150, 300)
|
|
SEEDS = (115, 2718, 40_902, 7_331)
|
|
BUDGETS = (1, 2, 4, 6, 8, 10, 12, 14, 16, 20, 24, 32)
|
|
|
|
|
|
def sweep(
|
|
worker: TeraWorker, env_id: str, holds: Sequence[int], seeds: Sequence[int]
|
|
) -> list[dict[str, float]]:
|
|
"""One row per hold length, averaged over every public scenario at every seed."""
|
|
manifest = next(m for m in worker.manifests() if m["id"] == env_id)
|
|
scenarios = [
|
|
(split, scenario_id)
|
|
for split in ("train", "dev")
|
|
for scenario_id in manifest["scenarioIds"][split]
|
|
]
|
|
rows = []
|
|
for hold in holds:
|
|
returns, steps, decisions, reached = [], [], [], 0
|
|
for split, scenario_id in scenarios:
|
|
for seed in seeds:
|
|
result = worker.oracle(
|
|
env_id, seed, {"split": split, "id": scenario_id}, hold=hold
|
|
)
|
|
returns.append(result["cumulativeReward"])
|
|
steps.append(result["steps"])
|
|
decisions.append(result["decisions"])
|
|
reached += result["terminalReason"] == "goal"
|
|
rows.append(
|
|
{
|
|
"hold": hold,
|
|
"mean_return": statistics.fmean(returns),
|
|
"worst_return": min(returns),
|
|
"mean_steps": statistics.fmean(steps),
|
|
"mean_decisions": statistics.fmean(decisions),
|
|
"goal_rate": reached / len(returns),
|
|
"episodes": len(returns),
|
|
}
|
|
)
|
|
return rows
|
|
|
|
|
|
def budget_ladder(
|
|
worker: TeraWorker,
|
|
env_id: str,
|
|
turns: Sequence[int],
|
|
max_hold: int,
|
|
seeds: Sequence[int],
|
|
) -> list[dict[str, float]]:
|
|
"""The best the baseline can do under a turn budget — the rule-4 ladder.
|
|
|
|
For each budget the baseline is flown at every legal hold length and the best
|
|
result is kept, with the episode cut off at `turns x hold` steps because that
|
|
is all the budget buys. A budget that does not change the answer is a budget
|
|
that does not bind, and this table is where that shows.
|
|
"""
|
|
manifest = next(m for m in worker.manifests() if m["id"] == env_id)
|
|
scenarios = [
|
|
(split, scenario_id)
|
|
for split in ("train", "dev")
|
|
for scenario_id in manifest["scenarioIds"][split]
|
|
]
|
|
rows = []
|
|
for budget in turns:
|
|
best = None
|
|
for hold in range(1, max_hold + 1):
|
|
returns, reached = [], 0
|
|
for split, scenario_id in scenarios:
|
|
for seed in seeds:
|
|
result = worker.oracle(
|
|
env_id,
|
|
seed,
|
|
{"split": split, "id": scenario_id},
|
|
max_steps=budget * hold,
|
|
hold=hold,
|
|
)
|
|
returns.append(result["cumulativeReward"])
|
|
reached += result["terminalReason"] == "goal"
|
|
mean = statistics.fmean(returns)
|
|
if best is None or mean > best["mean_return"]:
|
|
best = {
|
|
"turns": budget,
|
|
"hold": hold,
|
|
"mean_return": mean,
|
|
"worst_return": min(returns),
|
|
"goal_rate": reached / len(returns),
|
|
}
|
|
rows.append(best)
|
|
return rows
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("--env", default="crow-nav-v1", choices=ENVIRONMENT_IDS)
|
|
parser.add_argument("--seeds", type=int, nargs="+", default=list(SEEDS))
|
|
parser.add_argument("--max-hold", type=int, default=12)
|
|
args = parser.parse_args()
|
|
|
|
with TeraWorker() as worker:
|
|
rows = sweep(worker, args.env, HOLDS, args.seeds)
|
|
floor = worker.oracle(args.env, args.seeds[0], {"split": "train"}, policy="inaction")
|
|
ladder = budget_ladder(worker, args.env, BUDGETS, args.max_hold, args.seeds)
|
|
|
|
print(f"{args.env} — Tera's scripted baseline, flown at one decision every `hold` steps")
|
|
print(f"{'hold':>5} {'return':>9} {'worst':>9} {'steps':>7} {'calls':>7} {'goal':>6}")
|
|
for row in rows:
|
|
print(
|
|
f"{row['hold']:>5} {row['mean_return']:>9.3f} {row['worst_return']:>9.3f} "
|
|
f"{row['mean_steps']:>7.1f} {row['mean_decisions']:>7.1f} {row['goal_rate']:>6.0%}"
|
|
)
|
|
print(
|
|
f"\ninaction floor (seed {args.seeds[0]}, train): "
|
|
f"{floor['cumulativeReward']:.3f} over {floor['steps']} steps "
|
|
f"({floor['terminalReason']})"
|
|
)
|
|
print(f"{rows[0]['episodes']} episodes per row.")
|
|
|
|
print(f"\nUnder a turn budget, hold capped at {args.max_hold} — best constant hold per budget")
|
|
print(f"{'turns':>6} {'hold':>5} {'return':>9} {'worst':>9} {'goal':>6}")
|
|
for row in ladder:
|
|
print(
|
|
f"{row['turns']:>6} {row['hold']:>5} {row['mean_return']:>9.3f} "
|
|
f"{row['worst_return']:>9.3f} {row['goal_rate']:>6.0%}"
|
|
)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|