From 4718635c759eebca83355462a23c2999d3892159 Mon Sep 17 00:00:00 2001 From: ImBenji Date: Tue, 1 Sep 2026 21:32:22 +0100 Subject: [PATCH] fix: bound max_tokens on the openrouter calls Neither the coordinator nor the graph resolver set max_tokens, so openrouter reserved the model's entire output window against the key's remaining budget and returned 402 before running anything: 131k tokens reserved to produce a few hundred. It never showed up with the old model because its output window is small enough to fit under the limit. Both are now bounded and overridable by env. The headroom is deliberate, the newer reasoning models spend completion tokens thinking before they answer. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb --- .env.example | 2 +- src/autonomy/llm.js | 5 +++++ workers/graphWorker.js | 4 ++++ 3 files changed, 10 insertions(+), 1 deletion(-) diff --git a/.env.example b/.env.example index b8786d2..cb69a18 100644 --- a/.env.example +++ b/.env.example @@ -12,7 +12,7 @@ ALPHA_VANTAGE_API_KEY= FINNHUB_API_KEY= OPEN_ROUTER_API_KEY= -OPEN_ROUTER_LLM_MODEL=qwen/qwen3-235b-a22b-2507 +OPEN_ROUTER_LLM_MODEL=~deepseek/deepseek-v4-flash-latest OPEN_ROUTER_EMBED_MODEL=qwen/qwen3-embedding-8b # Paper execution is disabled unless AUTONOMY_EXECUTION_MODE=paper. diff --git a/src/autonomy/llm.js b/src/autonomy/llm.js index ebe81b6..50af7d8 100644 --- a/src/autonomy/llm.js +++ b/src/autonomy/llm.js @@ -23,6 +23,11 @@ async function callCoordinator(config, prompt) { body: JSON.stringify({ model: config.openRouter.llmModel, temperature: 0, + // Without this OpenRouter reserves the model's entire output window against + // the key's remaining budget and rejects the call with a 402 before it ever + // runs -- 131k tokens reserved to produce a few hundred. Reasoning models + // also spend completion tokens on reasoning, so leave real headroom. + max_tokens: Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 8000), response_format: { type: 'json_object' }, messages: [ { role: 'system', content: 'You are a coordinator. Extract only evidence-backed categorical hypotheses. Never output probabilities, expected returns, confidence scores, position sizes, or trade actions.' }, diff --git a/workers/graphWorker.js b/workers/graphWorker.js index a892ea6..c173fbf 100644 --- a/workers/graphWorker.js +++ b/workers/graphWorker.js @@ -101,6 +101,10 @@ Reply with just the number of the match, or "none" if none apply. No explanation model: llmConfig.cheapModel || llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0, + // Same reason as the coordinator: an unbounded request gets a 402 for reserving + // the whole output window. This one only ever replies with a number or "none", + // the headroom is for models that think before answering. + max_tokens: Math.max(256, Number(process.env.OPEN_ROUTER_GRAPH_MAX_TOKENS) || 2000), }); if (Date.now() < llmCooldownUntil) return null;