diff --git a/src/autonomy/llm.js b/src/autonomy/llm.js index d532629..9a4a583 100644 --- a/src/autonomy/llm.js +++ b/src/autonomy/llm.js @@ -35,12 +35,14 @@ async function callCoordinator(config, prompt, options = {}) { body: JSON.stringify({ model: config.openRouter.llmModel, temperature: 0, - // Sized against reality rather than guessed: the largest proposal this has - // ever produced was ~1,258 tokens carrying 12 predictions, the average is - // ~50. Reasoning models spend completion tokens thinking first, so this is - // still several times the worst case actually observed. + // Deliberately far above any output we have seen, because the cap exists to + // satisfy openrouter's affordability check, not to ration tokens: you are + // billed for what is used, not what is reserved. 6000 was tight enough to + // truncate a real replay article. The 402 handler below walks this down + // automatically when the budget cannot cover it, so a high ceiling costs + // nothing while funded and degrades on its own when not. max_tokens: options.maxTokens - || Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 6000), + || Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 32000), response_format: { type: 'json_object' }, messages: [ { role: 'system', content: 'You are a coordinator. Extract only evidence-backed categorical hypotheses. Never output probabilities, expected returns, confidence scores, position sizes, or trade actions.' },