diff --git a/src/autonomy/llm.js b/src/autonomy/llm.js index 9a4a583..b3aad3f 100644 --- a/src/autonomy/llm.js +++ b/src/autonomy/llm.js @@ -24,6 +24,9 @@ async function callCoordinator(config, prompt, options = {}) { const apiKey = String(config?.openRouter?.apiKey || '').trim(); if (!apiKey) throw new Error('OpenRouter API key is not configured'); const timeoutMs = Math.max(1000, Number(config?.openRouter?.timeoutMs || process.env.OPEN_ROUTER_TIMEOUT_MS) || 60000); + // only set when a budget retry forced one, or an operator asked for one + const maxTokens = options.maxTokens + || Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || null; const controller = new AbortController(); const timeout = setTimeout(() => controller.abort(), timeoutMs); let response; @@ -35,15 +38,14 @@ async function callCoordinator(config, prompt, options = {}) { body: JSON.stringify({ model: config.openRouter.llmModel, temperature: 0, - // Deliberately far above any output we have seen, because the cap exists to - // satisfy openrouter's affordability check, not to ration tokens: you are - // billed for what is used, not what is reserved. 6000 was tight enough to - // truncate a real replay article. The 402 handler below walks this down - // automatically when the budget cannot cover it, so a high ceiling costs - // nothing while funded and degrades on its own when not. - max_tokens: options.maxTokens - || Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 32000), response_format: { type: 'json_object' }, + // No ceiling by default. Any number we pick is a number we invented, and + // 6000 was already tight enough to truncate a real replay article. A cap + // only exists to satisfy openrouter's affordability reservation, so it is + // supplied by the 402 handler below when the budget genuinely cannot cover + // an open ended request, and never otherwise. Set OPEN_ROUTER_MAX_TOKENS + // if you ever want one imposed deliberately. + ...(maxTokens ? { max_tokens: maxTokens } : {}), messages: [ { role: 'system', content: 'You are a coordinator. Extract only evidence-backed categorical hypotheses. Never output probabilities, expected returns, confidence scores, position sizes, or trade actions.' }, { role: 'user', content: prompt }, diff --git a/workers/augorWorker.js b/workers/augorWorker.js index 3e44754..2eb424b 100644 --- a/workers/augorWorker.js +++ b/workers/augorWorker.js @@ -306,9 +306,6 @@ async function callLlm(llmConfig, prompt) { model: llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0.1, - // Unbounded requests get a 402 for reserving the model's whole output - // window against the remaining key budget, before running anything. - max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 6000), }); const url = new URL("https://openrouter.ai/api/v1/chat/completions"); diff --git a/workers/consolidationWorker.js b/workers/consolidationWorker.js index f17a4a4..559f815 100644 --- a/workers/consolidationWorker.js +++ b/workers/consolidationWorker.js @@ -220,9 +220,6 @@ Rules: model: llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0.1, - // Unbounded requests get a 402 for reserving the model's whole output - // window against the remaining key budget, before running anything. - max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 6000), }); const url = new URL("https://openrouter.ai/api/v1/chat/completions"); diff --git a/workers/graphWorker.js b/workers/graphWorker.js index c173fbf..a892ea6 100644 --- a/workers/graphWorker.js +++ b/workers/graphWorker.js @@ -101,10 +101,6 @@ Reply with just the number of the match, or "none" if none apply. No explanation model: llmConfig.cheapModel || llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0, - // Same reason as the coordinator: an unbounded request gets a 402 for reserving - // the whole output window. This one only ever replies with a number or "none", - // the headroom is for models that think before answering. - max_tokens: Math.max(256, Number(process.env.OPEN_ROUTER_GRAPH_MAX_TOKENS) || 2000), }); if (Date.now() < llmCooldownUntil) return null; diff --git a/workers/signalWorker.js b/workers/signalWorker.js index 1da1cb1..78aaf61 100644 --- a/workers/signalWorker.js +++ b/workers/signalWorker.js @@ -297,9 +297,6 @@ async function callLlm(llmConfig, prompt) { model: llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0.1, - // Unbounded requests get a 402 for reserving the model's whole output - // window against the remaining key budget, before running anything. - max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 4000), }); const url = new URL("https://openrouter.ai/api/v1/chat/completions");