tera-crow-nav: the first spatial taskset, and the hold sweep that sized its budget

The bridge becomes a runnable taskset. The crux is that these are continuous
control environments and the agent emits text: the model issues a control
decision that is held for N steps, and the hold length is measured rather than
guessed. At one decision every 32 steps the scripted baseline returns -2.807 and
reaches the waypoint 0% of the time, against 5.484 and 100% at every frame.

Open loop — one decision for the whole flight — returns -3.764 against an
inaction floor of -4.161. A bird that sets a course and leaves does no better
than one that does nothing, which is why crow-nav carries the strongest rule-4
claim of the five spatial environments. The budget bites from both ends: 8 turns
reach the goal 44% of the time, 12 reach 88%, 14 reach 100%.

32 episodes against Qwen3.8-27B: mean 0.4603, no malformed replies in 412, every
episode replaying to an identical checksum, and 16 of 32 spending all 16 turns.
The gate fires 43.75% of the time and separates the four scenarios cleanly —
whatever is wrong with the four dead data-environment gates, it is not that
binary gates cannot discriminate.

The oracle comes from the worker's own op. Nothing recomputes in Python a number
that came out of TypeScript, least of all the reward denominator.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-08-21 17:44:46 -07:00
co-authored by Claude Opus 5
parent 94ab513eda
commit 4147de0282
10 changed files with 1482 additions and 18 deletions
@@ -0,0 +1,305 @@
"""The taskset's own gate: the grammar, the arithmetic, and one flight flown for real.
`test_replay.py` proves the bridge carries a simulation faithfully. This file
proves the three things built on top of it are right when no model is watching:
1. **The reply grammar.** A reply is the only thing a model controls, so every way
of getting it wrong has to land somewhere sensible and none of them may raise —
a formatting slip that crashes a rollout scores nothing at all rather than
scoring badly.
2. **The reward.** Floor, ceiling and the two products, computed against episodes
this file actually flies rather than against a fixture.
3. **The rendering.** The strongest claim a spatial observation can make is that
it is enough to fly on, so the test flies on it: a policy that reads nothing
but the rendered panel — the same characters the model is shown, parsed back
out of them — reaches the waypoint inside the shipped budget.
uv run --project environments/tera_spatial python -m unittest discover \\
-s environments/tera_spatial/tests -v
"""
from __future__ import annotations
import json
import math
import re
import unittest
from tera_crow_nav.taskset import CrowNavConfig, CrowNavTaskset
from tera_spatial import TeraWorker
from tera_spatial.spatial import (
Command,
Flight,
SpatialData,
TurnView,
parse_command,
score,
)
FIELDS = ("forward", "turn", "pitch", "climb", "glide")
INACTION = dict.fromkeys(FIELDS, 0.0)
MINIMUM_POWER = {"forward": -1.0, "turn": 0.0, "pitch": 0.0, "climb": 0.03, "glide": False}
_DISTANCE = re.compile(r"^\s+([-\d.]+) m away .* bearing ([+-]\d+)", re.MULTILINE)
_HEIGHT = re.compile(r"([+-][\d.]+) m of height to make up")
def read_the_board(board: str) -> tuple[float, float, float]:
"""Distance, bearing in degrees and height to make up — from the panel text only.
Deliberately parsed out of the rendered characters rather than taken from the
observation dict. If this function can fly the crow, the rendering carries
enough; if the rendering ever stops carrying enough, this stops passing.
"""
away, bearing = _DISTANCE.search(board).groups()
(height,) = _HEIGHT.search(board).groups()
return float(away), float(bearing), float(height)
def reader_policy(board: str) -> str:
"""A reply, in the grammar, from a reader who has seen only the panel.
Proportional guidance and nothing clever: point at the waypoint, trim the
height, and shorten the hold as it closes, because a crow at 12 m/s covers
12 m in one held second and the waypoint is 2.5 m wide.
"""
away, bearing, height = read_the_board(board)
hold = 8 if away > 40 else 4 if away > 12 else 2
return json.dumps(
{
"forward": 0.45,
"turn": max(-1.0, min(1.0, math.radians(bearing) / 0.55)),
"pitch": 0.0,
"climb": max(-0.8, min(0.8, height / 10 - 0.2)),
"glide": False,
"hold": hold,
}
)
class GrammarTests(unittest.TestCase):
"""Every reply a model can emit, and where it lands."""
def parse(self, reply: str, previous: Command | None = None) -> Command:
return parse_command(
reply,
fields=FIELDS,
max_hold=12,
default_hold=6,
previous=previous,
inaction=INACTION,
)
def test_a_fenced_block_is_read(self) -> None:
command = self.parse('```json\n{"forward": 0.5, "turn": -1, "glide": true, "hold": 4}\n```')
self.assertIsNone(command.problem)
self.assertEqual(command.hold, 4)
self.assertEqual(command.action["forward"], 0.5)
self.assertEqual(command.action["turn"], -1.0)
self.assertIs(command.action["glide"], True)
def test_the_last_block_wins(self) -> None:
"""A model that thinks out loud in JSON is answering with its last one."""
command = self.parse(
'```json\n{"forward": 0.1}\n```\nno, better:\n```json\n{"forward": 0.9}\n```'
)
self.assertEqual(command.action["forward"], 0.9)
def test_a_bare_object_is_read(self) -> None:
self.assertEqual(self.parse('{"climb": 0.3}').action["climb"], 0.3)
def test_omitted_controls_keep_flying(self) -> None:
previous = Command({"forward": 0.8, "turn": 0.2, "pitch": 0.0, "climb": 0.1, "glide": False}, 6)
command = self.parse('{"turn": -0.5}', previous)
self.assertEqual(command.action["forward"], 0.8)
self.assertEqual(command.action["turn"], -0.5)
def test_an_unreadable_reply_costs_the_default_hold(self) -> None:
"""Silence is the one thing a reply must not be. A turn that costs no
simulation steps is a free turn, and a free turn is a budget that does not
bind — which is house rule 4 undone by a parser."""
previous = Command({**INACTION, "forward": 0.7}, 5)
for reply in ("", "I think we should climb.", "```json\n{not json}\n```", "[1, 2, 3]"):
with self.subTest(reply=reply):
command = self.parse(reply, previous)
self.assertIsNotNone(command.problem)
self.assertEqual(command.hold, 6)
self.assertEqual(command.action, previous.action)
def test_json_that_names_no_control_is_flagged_and_still_flies(self) -> None:
command = self.parse('{"thinking": "left a bit"}')
self.assertEqual(command.problem, "your JSON named none of the controls")
self.assertEqual(command.action, INACTION)
def test_hold_is_clipped_never_raised(self) -> None:
for raw, expected in ((0, 1), (-9, 1), (999, 12), (3.7, 3)):
with self.subTest(hold=raw):
self.assertEqual(self.parse(json.dumps({"forward": 0, "hold": raw})).hold, expected)
def test_non_finite_numbers_are_dropped_not_carried(self) -> None:
"""`json.loads` accepts bare Infinity and NaN and overflows 1e309 to inf.
Either one reaching a reward is a corrupted training signal rather than a
bad answer, and both are reachable from a reply a model can emit."""
previous = Command({**INACTION, "forward": 0.4}, 6)
command = self.parse('{"forward": NaN, "turn": Infinity, "climb": 1e309, "hold": NaN}', previous)
self.assertEqual(command.action["forward"], 0.4)
self.assertEqual(command.action["turn"], 0.0)
self.assertEqual(command.hold, 6)
for value in command.action.values():
if isinstance(value, float):
self.assertTrue(math.isfinite(value))
def test_twenty_thousand_nested_arrays_do_not_crash_the_decoder(self) -> None:
self.assertIsNotNone(self.parse("[" * 20_000).problem)
class FlightTests(unittest.TestCase):
"""One scenario, flown three ways, graded on what the simulator recorded."""
worker: TeraWorker
@classmethod
def setUpClass(cls) -> None:
cls.worker = TeraWorker()
cls.taskset = CrowNavTaskset(CrowNavConfig(id="tera-crow-nav", num_tasks=4))
cls.tasks = list(cls.taskset)
@classmethod
def tearDownClass(cls) -> None:
cls.worker.close()
def fly(self, data: SpatialData, replies) -> tuple[Flight, dict]:
"""Drive a whole episode from Python, exactly as the env does."""
episode = self.worker.open(
data.env_id, data.seed, {"split": data.split, "id": data.scenario_id}
)
flight = Flight(
episode=episode,
max_turns=data.max_turns,
budget=data.max_turns * data.max_hold,
observation=episode.observation,
)
view = TurnView(
observation=flight.observation,
turn=1,
turns_left=data.max_turns,
steps=0,
budget=flight.budget,
max_hold=data.max_hold,
last=None,
)
while not flight.done:
command = parse_command(
replies(self.taskset.board(view)),
fields=FIELDS,
max_hold=data.max_hold,
default_hold=data.default_hold,
previous=flight.last,
inaction=INACTION,
)
flight.fly(command)
view = TurnView(
observation=flight.observation,
turn=flight.turns + 1,
turns_left=data.max_turns - flight.turns,
steps=flight.steps,
budget=flight.budget,
max_hold=data.max_hold,
last=command,
)
envelope = episode.trace()
episode.close()
return flight, envelope
def test_a_reader_of_the_panel_reaches_every_waypoint(self) -> None:
"""The rendering's real claim: it is enough to fly on.
Nothing in `reader_policy` sees the observation dict — it sees the same
characters the model is shown and parses them back out. Four scenarios,
four waypoints, inside the shipped budget.
"""
for task in self.tasks:
with self.subTest(task=task.data.name):
flight, envelope = self.fly(task.data, reader_policy)
self.assertEqual(flight.terminal_reason, "goal")
self.assertLessEqual(flight.turns, task.data.max_turns)
rewards, metrics, info = score(task.data, flight, envelope, self.worker)
self.assertEqual(metrics["gate"], 1.0)
self.assertEqual(metrics["replay_ok"], 1.0)
self.assertGreater(rewards["flight"][0], 0.0)
self.assertEqual(info["terminal_reason"], "goal")
def test_minimum_power_scores_exactly_zero(self) -> None:
"""House rule 3's floor. A crow that will not fly does not fly, and the
reward for not flying is 0.000 and not 0.001 — the clip is against a
negative return, so there is nothing to round."""
data = self.tasks[0].data
flight, envelope = self.fly(data, lambda _board: json.dumps({**MINIMUM_POWER, "hold": 12}))
rewards, metrics, _ = score(data, flight, envelope, self.worker)
self.assertNotEqual(flight.terminal_reason, "goal")
self.assertLess(metrics["ts_return"], 0.0)
self.assertEqual(rewards["flight"][0], 0.0)
self.assertEqual(rewards["economy"][0], 0.0)
def test_a_silent_model_still_spends_its_budget(self) -> None:
"""Sixteen unreadable replies cost sixteen turns and ninety-six steps."""
data = self.tasks[0].data
flight, _ = self.fly(data, lambda _board: "I would rather not.")
self.assertEqual(flight.turns, data.max_turns)
self.assertEqual(flight.malformed, data.max_turns)
self.assertEqual(flight.steps, data.max_turns * data.default_hold)
def test_the_ceiling_is_reachable_inside_the_budget(self) -> None:
"""House rule 3's ceiling, taken from Tera and not from Python.
The scripted controller — the same function the denominator is measured
with — flown at a hold this taskset permits, cut off at the turns it
grants, must still clear the band. If it cannot, the budget is not tight,
it is impossible.
"""
data = self.tasks[0].data
best = self.worker.oracle(
data.env_id,
data.seed,
{"split": data.split, "id": data.scenario_id},
max_steps=data.max_turns * 5,
hold=5,
)
self.assertEqual(best["terminalReason"], "goal")
self.assertGreaterEqual(
best["cumulativeReward"] / (data.band * data.oracle_return), 1.0
)
def test_the_turn_budget_binds(self) -> None:
"""House rule 4, in the shipped engine rather than in a spreadsheet.
The same policy at the same refresh rate, allowed one turn and allowed
sixteen. If the two agree, the budget buys nothing and crow-nav has no
rule-4 claim to make.
"""
data = self.tasks[0].data
request = {"split": data.split, "id": data.scenario_id}
one = self.worker.oracle(data.env_id, data.seed, request, max_steps=12, hold=12)
many = self.worker.oracle(data.env_id, data.seed, request, max_steps=16 * 5, hold=5)
self.assertNotEqual(one["terminalReason"], "goal")
self.assertEqual(many["terminalReason"], "goal")
self.assertGreater(many["cumulativeReward"] - one["cumulativeReward"], 4.0)
def test_a_tampered_trace_fails_the_episode_replay(self) -> None:
"""`replay_ok` is not decoration: it is the bridge's claim, asserted once
per episode. An envelope whose reward was edited and re-sealed with a
checksum Tera itself computed still has to fail."""
data = self.tasks[0].data
flight, envelope = self.fly(data, reader_policy)
forged = json.loads(json.dumps(envelope))
forged["steps"][0]["reward"] += 1.0
forged["cumulativeReward"] += 1.0
forged["checksum"] = self.worker.checksum(
{k: v for k, v in forged.items() if k != "checksum"}
)
_, metrics, _ = score(data, flight, forged, self.worker)
self.assertEqual(metrics["replay_ok"], 0.0)
if __name__ == "__main__":
unittest.main()