Files
compute/src/main.rs
T
Karti Tripathi a4490ec80e
ci / rust (push) Successful in 2m26s
Lumbridge Compute
Governed compute for unified-memory AI hardware — machines where CPU and GPU
share one pool and there is no separate VRAM allocation to bounce off.
Over-commit that pool and the box thrashes and wedges, SSH and ping included,
before the OOM killer gets a turn.

Compute does not run inference. It supervises the servers that do: admission
control against both a declared budget and what the machine actually has free,
a 1 Hz watchdog that stops the newest model before thrash, Scenes activated as
one transactional unit with rollback, process ownership bound to
(boot_id, pid, start_time_ticks, pgid) so a reused PID can never be
group-killed, a protocol-transparent gateway, an MCP server, and a read-only
HTTP API for dashboards.

Registry footprints in this release are measured on a live node rather than
estimated.

One binary, six direct dependencies. Apache-2.0.

Generated by scripts/publish-compute.sh, which refuses to publish a tree it
cannot prove clean.
2026-08-03 22:23:56 -07:00

432 lines
13 KiB
Rust

//! Lumbridge Compute — safe AI workload orchestration for accelerator nodes.
mod agent;
mod config;
mod eval;
mod gateway;
mod governor;
mod http;
mod lifecycle;
mod mcp;
mod mem;
mod proc;
use anyhow::Result;
use clap::{Parser, Subcommand};
use std::path::{Path, PathBuf};
use std::time::Duration;
use config::{find_scene, load_scenes, Registry, Scene};
use proc::State;
#[derive(Parser)]
#[command(
name = "lumbridge-compute",
version,
about = "Lumbridge Compute — safe AI workload orchestration for accelerator nodes."
)]
struct Cli {
/// Root dir containing registry/ and scenes/ (default: $LUMBRIDGE_COMPUTE_ROOT, $KUDA_ROOT, or current dir)
#[arg(long, global = true)]
root: Option<PathBuf>,
#[command(subcommand)]
cmd: Cmd,
}
#[derive(Subcommand)]
enum Cmd {
/// Governor status: memory, budget, running set, headroom
Status,
/// Manage the model registry
Model {
#[command(subcommand)]
cmd: ModelCmd,
},
/// Manage scenes (bundles of models)
Scene {
#[command(subcommand)]
cmd: SceneCmd,
},
/// Run reproducible capability and serving-performance evaluations
Eval {
#[command(subcommand)]
cmd: EvalCmd,
},
/// Run the memory watchdog in the foreground (kills the newest model before OOM-wedge)
Watchdog {
/// Kill the newest model if MemAvailable dips below this many GB
#[arg(long, default_value_t = governor::WATCHDOG_FLOOR_GB)]
floor: f64,
},
/// Run the stable streaming gateway in the foreground
Gateway {
#[arg(long, default_value = "127.0.0.1:8011")]
listen: String,
#[arg(long, default_value = "127.0.0.1:8001")]
upstream: String,
},
/// Serve the Governor as a read-only JSON API over HTTP
///
/// A second transport over the same operations the MCP server exposes, for a web UI. It
/// never mutates: no activation, no stop, no eval run.
Api {
/// Loopback by default. This API has no authentication unless --token is set, and it
/// reveals which models this node runs, so widening the bind is an explicit act.
#[arg(long, default_value = "127.0.0.1:8012")]
listen: String,
/// Browser origin permitted to read this API. Repeatable. Empty means no browser may
/// read it; `*` is deliberately not supported, because any page you visit would then
/// be able to fingerprint this machine.
#[arg(long = "allow-origin")]
allow_origin: Vec<String>,
/// Require `Authorization: Bearer <token>` on every request.
#[arg(long)]
token: Option<String>,
},
/// Serve the Governor to AI agents as MCP tools over stdio
Mcp {
/// Let an agent perform a real Scene transition, not just plan one.
/// Off by default: the agent writes its own tool arguments, so the only
/// meaningful gate on a production switch is one the operator sets here.
#[arg(long)]
allow_activate: bool,
},
/// Run the resident supervisor: resume desired Scene, gateway, and memory floor
Agent {
#[arg(long, default_value = "127.0.0.1:8011")]
listen: String,
#[arg(long, default_value = "127.0.0.1:8001")]
upstream: String,
#[arg(long, default_value_t = governor::WATCHDOG_FLOOR_GB)]
floor: f64,
},
}
#[derive(Subcommand)]
enum ModelCmd {
/// List registered models and their live state
Ls,
}
#[derive(Subcommand)]
enum SceneCmd {
/// List scenes with total footprint
Ls,
/// Show a scene: models, footprints, and the Governor's admission verdict
Show { name: String },
/// Activate a scene: stop what's not in it, admit + start what is
Activate {
name: String,
/// Print the plan without changing anything
#[arg(long)]
dry_run: bool,
},
/// Adopt an already-running exact Scene and bind legacy PIDs to process identities
Adopt { name: String },
/// Resume the persisted desired Scene, falling back to the previous known-good Scene
Resume,
/// Stop all Lumbridge Compute-managed models
Deactivate,
}
#[derive(Subcommand)]
enum EvalCmd {
/// List available evaluation suites
Ls,
/// Run a suite against an OpenAI-compatible endpoint
Run {
suite: String,
#[arg(long, default_value = "http://127.0.0.1:8001/v1")]
base_url: String,
#[arg(long, default_value = "brain")]
model: String,
/// Override the suite's repetitions per case
#[arg(long)]
repeat: Option<u32>,
},
}
fn root_dir(cli: &Cli) -> PathBuf {
cli.root
.clone()
.or_else(|| {
std::env::var("LUMBRIDGE_COMPUTE_ROOT")
.ok()
.map(PathBuf::from)
})
.or_else(|| std::env::var("KUDA_ROOT").ok().map(PathBuf::from))
.unwrap_or_else(|| PathBuf::from("."))
}
fn main() -> Result<()> {
let cli = Cli::parse();
let root = root_dir(&cli);
match &cli.cmd {
Cmd::Status => cmd_status(&root),
Cmd::Model { cmd } => match cmd {
ModelCmd::Ls => cmd_model_ls(&root),
},
Cmd::Scene { cmd } => match cmd {
SceneCmd::Ls => cmd_scene_ls(&root),
SceneCmd::Show { name } => cmd_scene_show(&root, name),
SceneCmd::Activate { name, dry_run } => lifecycle::activate(&root, name, *dry_run),
SceneCmd::Adopt { name } => lifecycle::adopt(&root, name),
SceneCmd::Resume => lifecycle::resume(&root),
SceneCmd::Deactivate => lifecycle::deactivate(&root),
},
Cmd::Eval { cmd } => match cmd {
EvalCmd::Ls => eval::list(&root),
EvalCmd::Run {
suite,
base_url,
model,
repeat,
} => eval::run(&root, suite, base_url, model, *repeat),
},
Cmd::Watchdog { floor } => cmd_watchdog(&root, *floor),
Cmd::Gateway { listen, upstream } => gateway::run(listen, upstream),
Cmd::Api {
listen,
allow_origin,
token,
} => http::run(
listen,
http::Config {
root: root.clone(),
allowed_origins: allow_origin.clone(),
token: token.clone(),
},
),
Cmd::Mcp { allow_activate } => mcp::run(&root, *allow_activate),
Cmd::Agent {
listen,
upstream,
floor,
} => agent::run(&root, listen, upstream, *floor),
}
}
fn cmd_status(root: &Path) -> Result<()> {
let reg = Registry::load(root)?;
let m = mem::read()?;
let committed = governor::committed_gb(&reg);
let running = governor::running_ids(&reg);
let budget = governor::DEFAULT_BUDGET_GB;
// The binding constraint, not the generous one: admission enforces the declared
// budget AND observed memory, so reporting only the declared headroom would promise
// room the next `scene activate` is going to refuse.
let headroom = governor::headroom_gb(committed, budget, Some(m.available_gb)).max(0.0);
let declared_headroom = (budget - committed - governor::SAFETY_MARGIN_GB).max(0.0);
let managed = State::load_checked(root)?;
println!("Lumbridge Compute · governor");
println!(
" memory {:.1} GB total · {:.1} GB available",
m.total_gb, m.available_gb
);
println!(
" budget {:.0} GB (safety margin {:.0} · watchdog floor {:.0})",
budget,
governor::SAFETY_MARGIN_GB,
governor::WATCHDOG_FLOOR_GB
);
println!(
" committed {committed:.1} GB across {} model(s)",
running.len()
);
if headroom < declared_headroom {
println!(
" headroom {headroom:.1} GB admittable (budget allows {declared_headroom:.1}; \
MemAvailable is the tighter limit)"
);
} else {
println!(" headroom {headroom:.1} GB admittable");
}
println!();
println!(
" scene active={} desired={} fallback={}",
managed.active_scene.as_deref().unwrap_or("-"),
managed.desired_scene.as_deref().unwrap_or("-"),
managed.last_known_good_scene.as_deref().unwrap_or("-")
);
if let Some(error) = &managed.last_error {
println!(" last error {error}");
}
if running.is_empty() {
println!(" (no registered models currently serving)");
} else {
println!(" running:");
for id in &running {
let mdl = &reg.models[id];
let tag = if managed
.procs
.get(id)
.is_some_and(|process| process.owned_alive())
{
"compute"
} else {
"ext "
};
println!(
" ● [{tag}] {:<12} {:>5.0} GB :{:<5} {}",
id,
mdl.footprint_gb,
port_str(mdl.serve.port),
mdl.name
);
}
}
Ok(())
}
fn cmd_model_ls(root: &Path) -> Result<()> {
let reg = Registry::load(root)?;
println!(
"{:<12} {:>6} {:<6} {:<6} MODEL",
"ID", "GB", "STATE", "PORT"
);
for (id, m) in &reg.models {
println!(
"{:<12} {:>6.0} {:<6} {:<6} {}",
id,
m.footprint_gb,
if governor::is_running(m) {
"up"
} else {
"down"
},
port_str(m.serve.port),
m.name
);
}
Ok(())
}
fn cmd_scene_ls(root: &Path) -> Result<()> {
let reg = Registry::load(root)?;
let scenes = load_scenes(root)?;
if scenes.is_empty() {
println!("no scenes in {}/scenes", root.display());
return Ok(());
}
println!("{:<12} {:>6} {:<30} MODELS", "SCENE", "GB", "DESCRIPTION");
for s in &scenes {
let total = scene_footprint(s, &reg);
println!(
"{:<12} {:>6.0} {:<30} {}",
s.metadata.name,
total,
truncate(&s.metadata.description, 30),
s.models.join(", ")
);
}
Ok(())
}
fn cmd_scene_show(root: &Path, name: &str) -> Result<()> {
let reg = Registry::load(root)?;
let scene = find_scene(root, name)?;
let budget = scene.budget_gb.unwrap_or(governor::DEFAULT_BUDGET_GB);
println!(
"scene {} (v{})",
scene.metadata.name, scene.metadata.version
);
if !scene.metadata.description.is_empty() {
println!(" {}", scene.metadata.description);
}
println!(" budget {budget:.0} GB\n models:");
let mut total = 0.0;
let mut missing = Vec::new();
for id in &scene.models {
match reg.models.get(id) {
Some(m) => {
total += m.footprint_gb;
let state = if governor::is_running(m) {
"up"
} else {
"down"
};
println!(
" {:<12} {:>5.0} GB [{:<4}] {}",
id, m.footprint_gb, state, m.name
);
}
None => {
missing.push(id.clone());
println!(" {id:<12} ? [MISSING from registry]");
}
}
}
let needed = total + governor::SAFETY_MARGIN_GB;
println!(
"\n total footprint {total:.1} GB (+{:.0} safety = {needed:.1} GB)",
governor::SAFETY_MARGIN_GB
);
let verdict = if !missing.is_empty() {
format!("{} model(s) missing from registry", missing.len())
} else if needed <= budget {
format!("✓ fits — {:.1} GB to spare", budget - needed)
} else {
format!("✗ exceeds budget by {:.1} GB", needed - budget)
};
println!(" admission {verdict}");
Ok(())
}
fn cmd_watchdog(root: &Path, floor: f64) -> Result<()> {
println!(
"Lumbridge Compute watchdog — killing the newest model if MemAvailable < {floor:.1} GB. Ctrl-C to stop."
);
loop {
let m = mem::read()?;
if m.available_gb < floor {
let mut state = State::load_checked(root)?;
match state.newest_alive() {
Some((id, p)) => {
eprintln!(
"watchdog: MemAvailable {:.1} GB < floor {:.1} — killing newest '{id}' (pid {})",
m.available_gb, floor, p.pid
);
proc::stop_owned(&p)?;
state.procs.remove(&id);
state.save(root)?;
}
None => eprintln!(
"watchdog: MemAvailable {:.1} GB < floor {:.1} but no Lumbridge Compute-managed model to kill!",
m.available_gb, floor
),
}
}
std::thread::sleep(Duration::from_secs(1));
}
}
// ---- helpers ---------------------------------------------------------------
fn scene_footprint(s: &Scene, reg: &Registry) -> f64 {
s.models
.iter()
.filter_map(|id| reg.models.get(id))
.map(|m| m.footprint_gb)
.sum()
}
fn port_str(p: Option<u16>) -> String {
p.map(|p| p.to_string()).unwrap_or_else(|| "-".to_string())
}
fn truncate(s: &str, n: usize) -> String {
if s.chars().count() <= n {
s.to_string()
} else {
format!(
"{}",
s.chars().take(n.saturating_sub(1)).collect::<String>()
)
}
}