From e8fb9e3b1c2107abbad42748ca206a7866c0c6d8 Mon Sep 17 00:00:00 2001 From: ImBenji Date: Fri, 4 Sep 2026 18:43:23 +0100 Subject: [PATCH] fix: no token ceiling unless the budget forces one Removing the caps rather than tuning them. Every number I picked was a number I invented, and 6000 was already tight enough to truncate a real replay article, which is the failure that was called out when the cap first went in. The coordinator now sends no max_tokens at all. The 402 handler supplies one only when openrouter says the budget cannot cover an open ended request, so the ceiling exists exactly when it has to and never otherwise. Verified unbounded is accepted against the live key before making this the default. The caps on the signal, augor, consolidation and graph workers are gone too. They were added to work around an empty account, not because any of them ever produced too much, and unlike the coordinator none of them detect truncation, so an invented ceiling there risked silently corrupting company facts. A budget failure in those is at least loud. The 220 token cap in crawlerClassifier is left alone, it predates this and bounds a genuinely tiny classification. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb --- src/autonomy/llm.js | 18 ++++++++++-------- workers/augorWorker.js | 3 --- workers/consolidationWorker.js | 3 --- workers/graphWorker.js | 4 ---- workers/signalWorker.js | 3 --- 5 files changed, 10 insertions(+), 21 deletions(-) diff --git a/src/autonomy/llm.js b/src/autonomy/llm.js index 9a4a583..b3aad3f 100644 --- a/src/autonomy/llm.js +++ b/src/autonomy/llm.js @@ -24,6 +24,9 @@ async function callCoordinator(config, prompt, options = {}) { const apiKey = String(config?.openRouter?.apiKey || '').trim(); if (!apiKey) throw new Error('OpenRouter API key is not configured'); const timeoutMs = Math.max(1000, Number(config?.openRouter?.timeoutMs || process.env.OPEN_ROUTER_TIMEOUT_MS) || 60000); + // only set when a budget retry forced one, or an operator asked for one + const maxTokens = options.maxTokens + || Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || null; const controller = new AbortController(); const timeout = setTimeout(() => controller.abort(), timeoutMs); let response; @@ -35,15 +38,14 @@ async function callCoordinator(config, prompt, options = {}) { body: JSON.stringify({ model: config.openRouter.llmModel, temperature: 0, - // Deliberately far above any output we have seen, because the cap exists to - // satisfy openrouter's affordability check, not to ration tokens: you are - // billed for what is used, not what is reserved. 6000 was tight enough to - // truncate a real replay article. The 402 handler below walks this down - // automatically when the budget cannot cover it, so a high ceiling costs - // nothing while funded and degrades on its own when not. - max_tokens: options.maxTokens - || Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 32000), response_format: { type: 'json_object' }, + // No ceiling by default. Any number we pick is a number we invented, and + // 6000 was already tight enough to truncate a real replay article. A cap + // only exists to satisfy openrouter's affordability reservation, so it is + // supplied by the 402 handler below when the budget genuinely cannot cover + // an open ended request, and never otherwise. Set OPEN_ROUTER_MAX_TOKENS + // if you ever want one imposed deliberately. + ...(maxTokens ? { max_tokens: maxTokens } : {}), messages: [ { role: 'system', content: 'You are a coordinator. Extract only evidence-backed categorical hypotheses. Never output probabilities, expected returns, confidence scores, position sizes, or trade actions.' }, { role: 'user', content: prompt }, diff --git a/workers/augorWorker.js b/workers/augorWorker.js index 3e44754..2eb424b 100644 --- a/workers/augorWorker.js +++ b/workers/augorWorker.js @@ -306,9 +306,6 @@ async function callLlm(llmConfig, prompt) { model: llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0.1, - // Unbounded requests get a 402 for reserving the model's whole output - // window against the remaining key budget, before running anything. - max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 6000), }); const url = new URL("https://openrouter.ai/api/v1/chat/completions"); diff --git a/workers/consolidationWorker.js b/workers/consolidationWorker.js index f17a4a4..559f815 100644 --- a/workers/consolidationWorker.js +++ b/workers/consolidationWorker.js @@ -220,9 +220,6 @@ Rules: model: llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0.1, - // Unbounded requests get a 402 for reserving the model's whole output - // window against the remaining key budget, before running anything. - max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 6000), }); const url = new URL("https://openrouter.ai/api/v1/chat/completions"); diff --git a/workers/graphWorker.js b/workers/graphWorker.js index c173fbf..a892ea6 100644 --- a/workers/graphWorker.js +++ b/workers/graphWorker.js @@ -101,10 +101,6 @@ Reply with just the number of the match, or "none" if none apply. No explanation model: llmConfig.cheapModel || llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0, - // Same reason as the coordinator: an unbounded request gets a 402 for reserving - // the whole output window. This one only ever replies with a number or "none", - // the headroom is for models that think before answering. - max_tokens: Math.max(256, Number(process.env.OPEN_ROUTER_GRAPH_MAX_TOKENS) || 2000), }); if (Date.now() < llmCooldownUntil) return null; diff --git a/workers/signalWorker.js b/workers/signalWorker.js index 1da1cb1..78aaf61 100644 --- a/workers/signalWorker.js +++ b/workers/signalWorker.js @@ -297,9 +297,6 @@ async function callLlm(llmConfig, prompt) { model: llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0.1, - // Unbounded requests get a 402 for reserving the model's whole output - // window against the remaining key budget, before running anything. - max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 4000), }); const url = new URL("https://openrouter.ai/api/v1/chat/completions");