import assert from 'node:assert/strict'; import test from 'node:test'; import { loadPiggyConfig, loadPiggyTurnLimits } from '../src/config'; const minimum = { DATABASE_URL: 'postgres://pig:pig@localhost:54330/pig', PIGGY_INFERENCE_API_KEY: 'test-key', PIGGY_INTERNAL_TOKEN: 'test-internal-token-for-piggy-000000', }; test('the chat budget is separate from the worker budget, and larger', () => { const config = loadPiggyConfig(minimum); // The worker extracts; the chat has to quote aggregates back. Sharing one // budget meant tuning either one moved both. assert.equal(config.PIGGY_MAX_TOKENS, 1_024); assert.equal(config.PIGGY_CHAT_MAX_TOKENS, 2_048); assert.equal(config.PIGGY_MAX_TURNS, 4); }); test('a turn has a ceiling on both axes, generous against the measured turn', () => { const config = loadPiggyConfig(minimum); // Measured on the live stack against the default model: a one-tool turn is // 2 model calls and 4,922 tokens, a two-tool turn is 3 and 12,265. The // ceilings are roughly three times the busiest of those, which leaves a real // multi-step question room to breathe and still stops a `while (true)` in // seconds rather than in dollars. assert.equal(config.PIGGY_CHAT_MAX_MODEL_CALLS, 8); assert.equal(config.PIGGY_CHAT_MAX_TURN_TOKENS, 40_000); assert.equal(config.PIGGY_CHAT_DAILY_LIMIT_CENTS, 200); // PIGGY_MAX_TURNS is the queue worker's own budget and reaches nothing in the // chat path. Keeping them distinct is the point: raising one used to look // like it raised the other, which is how the chat came to have no ceiling at // all. assert.notEqual(config.PIGGY_MAX_TURNS, config.PIGGY_CHAT_MAX_MODEL_CALLS); }); test('the ceilings can be read without the rest of the environment', () => { // The chat server is handed a socket and a token and builds the rest from // defaults; it must not start demanding a DATABASE_URL it never uses. assert.deepEqual(loadPiggyTurnLimits({}), { maxModelCalls: 8, maxTurnTokens: 40_000, dailyLimitCents: 200, }); assert.deepEqual( loadPiggyTurnLimits({ PIGGY_CHAT_MAX_MODEL_CALLS: '3', PIGGY_CHAT_MAX_TURN_TOKENS: '9000', PIGGY_CHAT_DAILY_LIMIT_CENTS: '0', }), { maxModelCalls: 3, maxTurnTokens: 9_000, dailyLimitCents: 0 }, ); // A ceiling of zero model calls would answer nothing at all, so it is a // configuration error rather than a very strict deployment. assert.throws( () => loadPiggyTurnLimits({ PIGGY_CHAT_MAX_MODEL_CALLS: '0' }), /PIGGY_CHAT_MAX_MODEL_CALLS/, ); assert.throws( () => loadPiggyTurnLimits({ PIGGY_CHAT_MAX_TURN_TOKENS: 'plenty' }), /PIGGY_CHAT_MAX_TURN_TOKENS/, ); }); test('reasoning stays off by default', () => { // Reasoning tokens are billed like any other and nemotron-nano's are // verbose. The knob exists for debugging, not for the default deployment. assert.equal(loadPiggyConfig(minimum).PIGGY_REASONING_EFFORT, 'none'); assert.equal( loadPiggyConfig({ ...minimum, PIGGY_REASONING_EFFORT: 'low' }).PIGGY_REASONING_EFFORT, 'low', ); assert.throws( () => loadPiggyConfig({ ...minimum, PIGGY_REASONING_EFFORT: 'maximum' }), /PIGGY_REASONING_EFFORT/, ); }); test('the default token prices are the published price of the default model', () => { const config = loadPiggyConfig(minimum); // $0.05/$0.20 per million tokens, carried as cents per million so that // tokens x price is already micro-cents. assert.equal(config.PIGGY_PRICE_INPUT_CENTS_PER_MTOK, 5); assert.equal(config.PIGGY_PRICE_OUTPUT_CENTS_PER_MTOK, 20); assert.equal(config.PIGGY_MODEL, 'nvidia/nemotron-3-nano-30b-a3b'); });