The 600-second per-model-call ceiling that capped the thinking-on run at n=6 is gone: DEFAULT_TIMEOUT is read=None in all nine venvs. There is NO verifiers 0.4.0. The `version = "0.4.0"` in the vendored tree belongs to a [[tool.uv.dependency-metadata]] block for nemo-gym, and reading it as verifiers' own sends you hunting a release that does not exist and concluding the bump is blocked. The vendored checkout is v0.3.0-59-g4bcb48e5, published to PyPI as 0.3.1.dev59, and the installed wheel diffs clean against the whole vendored v1 tree. Pinned as an ordinary index dependency: a path source is reproducible only on this box and a git URL pins a commit no wheel matches. ⚠️ dev59 changes the default agent runtime from subprocess to `prime`, a REMOTE sandbox. Without an explicit runtime every eval here exits "not authenticated with prime" before a single rollout — having already created its output directory and a zero-line traces.jsonl, so it reads as a failed eval rather than an unauthenticated one. All nine configs now say subprocess. Nothing here needs a sandbox: the reward is a pure function of the trace and the harness is null. Output layout is flat now — <env>--<model>--<harness>--<hex8>/ with configs/, traces.jsonl and a logs/eval.log that 0.3.0 never wrote. `-o` is the output_dir, not the run dir; `--run.dir` pins the leaf. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
26 lines
1.1 KiB
TOML
26 lines
1.1 KiB
TOML
[project]
|
|
name = "tera-spatial"
|
|
version = "0.1.0"
|
|
description = "tera-spatial — Arena's bridge to Tera's four renderer-independent spatial environments, executed in TypeScript and graded in Python, and the tasksets that sit on it."
|
|
requires-python = ">=3.11"
|
|
dependencies = ["verifiers==0.3.1.dev59"]
|
|
|
|
# One directory, several tasksets: the environments share a vendored simulator and
|
|
# a single worker, so splitting them into four wheels would ship the same 283 KB
|
|
# of TypeScript four times. Discovery reads this key instead of assuming one wheel
|
|
# is one taskset.
|
|
#
|
|
# Only the ones that exist are listed. `tera-office-nav` and `tera-california-flight`
|
|
# follow; `tera-drive-101` ships labelled a bridge-correctness environment or not at
|
|
# all, because its return is monotone in throttle and any positive constant scores
|
|
# ~0.95 of the baseline — a number with no house rule 4 behind it.
|
|
[tool.arena]
|
|
tasksets = ["tera-crow-nav"]
|
|
|
|
[build-system]
|
|
requires = ["hatchling"]
|
|
build-backend = "hatchling.build"
|
|
|
|
[tool.hatch.build.targets.wheel]
|
|
packages = ["tera_spatial", "tera_crow_nav"]
|