diff --git a/src/autonomy/llm.js b/src/autonomy/llm.js index 50af7d8..d532629 100644 --- a/src/autonomy/llm.js +++ b/src/autonomy/llm.js @@ -8,7 +8,19 @@ function extractJson(text) { } } -async function callCoordinator(config, prompt) { +// OpenRouter reserves max_tokens against the key's remaining budget up front, and +// that affordable ceiling shrinks as the balance depletes. An unbounded request is +// refused outright, so "no cap" is not an option on a limited key -- it produces no +// output at all rather than truncated output. This pulls the real ceiling out of the +// refusal so we can retry just under it instead of guessing a fixed number. +function affordableTokens(message) { + const match = /can only afford (\d+)/i.exec(String(message || '')); + if (!match) return null; + const affordable = Number(match[1]); + return Number.isFinite(affordable) && affordable > 256 ? affordable : null; +} + +async function callCoordinator(config, prompt, options = {}) { const apiKey = String(config?.openRouter?.apiKey || '').trim(); if (!apiKey) throw new Error('OpenRouter API key is not configured'); const timeoutMs = Math.max(1000, Number(config?.openRouter?.timeoutMs || process.env.OPEN_ROUTER_TIMEOUT_MS) || 60000); @@ -23,11 +35,12 @@ async function callCoordinator(config, prompt) { body: JSON.stringify({ model: config.openRouter.llmModel, temperature: 0, - // Without this OpenRouter reserves the model's entire output window against - // the key's remaining budget and rejects the call with a 402 before it ever - // runs -- 131k tokens reserved to produce a few hundred. Reasoning models - // also spend completion tokens on reasoning, so leave real headroom. - max_tokens: Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 8000), + // Sized against reality rather than guessed: the largest proposal this has + // ever produced was ~1,258 tokens carrying 12 predictions, the average is + // ~50. Reasoning models spend completion tokens thinking first, so this is + // still several times the worst case actually observed. + max_tokens: options.maxTokens + || Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 6000), response_format: { type: 'json_object' }, messages: [ { role: 'system', content: 'You are a coordinator. Extract only evidence-backed categorical hypotheses. Never output probabilities, expected returns, confidence scores, position sizes, or trade actions.' }, @@ -43,10 +56,28 @@ async function callCoordinator(config, prompt) { } if (!response.ok) { const body = await response.text().catch(() => ''); + // A 402 names the ceiling the key can currently afford. Retry once just under + // it rather than failing the event, but only downwards, so a shrinking budget + // degrades output length instead of stopping the pipeline dead. + const affordable = response.status === 402 ? affordableTokens(body) : null; + if (affordable && !options.retriedForBudget) { + const retryTokens = Math.floor(affordable * 0.9); + console.warn(`[llm] budget only affords ${affordable} tokens, retrying with max_tokens=${retryTokens}`); + return callCoordinator(config, prompt, { maxTokens: retryTokens, retriedForBudget: true }); + } throw new Error(`coordinator request failed with ${response.status}: ${body.slice(0, 300)}`); } + const body = await response.json(); - return extractJson(body?.choices?.[0]?.message?.content); + const choice = body?.choices?.[0]; + + // Truncated json is worse than no json, because a partial object can occasionally + // still parse and quietly lose predictions. Fail loudly on the reason field rather + // than letting extractJson guess at a half-written response. + if (choice?.finish_reason === 'length') { + throw new Error('coordinator response was truncated by max_tokens, raise OPEN_ROUTER_MAX_TOKENS'); + } + return extractJson(choice?.message?.content); } module.exports = { extractJson, callCoordinator };