fix: bound max_tokens on the openrouter calls

Neither the coordinator nor the graph resolver set max_tokens, so openrouter
reserved the model's entire output window against the key's remaining budget and
returned 402 before running anything: 131k tokens reserved to produce a few
hundred. It never showed up with the old model because its output window is
small enough to fit under the limit.

Both are now bounded and overridable by env. The headroom is deliberate, the
newer reasoning models spend completion tokens thinking before they answer.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-09-01 21:32:22 +01:00
co-authored by Claude Opus 5
parent 9fc443579f
commit 4718635c75
3 changed files with 10 additions and 1 deletions
+1 -1
View File
@@ -12,7 +12,7 @@ ALPHA_VANTAGE_API_KEY=
FINNHUB_API_KEY= FINNHUB_API_KEY=
OPEN_ROUTER_API_KEY= OPEN_ROUTER_API_KEY=
OPEN_ROUTER_LLM_MODEL=qwen/qwen3-235b-a22b-2507 OPEN_ROUTER_LLM_MODEL=~deepseek/deepseek-v4-flash-latest
OPEN_ROUTER_EMBED_MODEL=qwen/qwen3-embedding-8b OPEN_ROUTER_EMBED_MODEL=qwen/qwen3-embedding-8b
# Paper execution is disabled unless AUTONOMY_EXECUTION_MODE=paper. # Paper execution is disabled unless AUTONOMY_EXECUTION_MODE=paper.
+5
View File
@@ -23,6 +23,11 @@ async function callCoordinator(config, prompt) {
body: JSON.stringify({ body: JSON.stringify({
model: config.openRouter.llmModel, model: config.openRouter.llmModel,
temperature: 0, temperature: 0,
// Without this OpenRouter reserves the model's entire output window against
// the key's remaining budget and rejects the call with a 402 before it ever
// runs -- 131k tokens reserved to produce a few hundred. Reasoning models
// also spend completion tokens on reasoning, so leave real headroom.
max_tokens: Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 8000),
response_format: { type: 'json_object' }, response_format: { type: 'json_object' },
messages: [ messages: [
{ role: 'system', content: 'You are a coordinator. Extract only evidence-backed categorical hypotheses. Never output probabilities, expected returns, confidence scores, position sizes, or trade actions.' }, { role: 'system', content: 'You are a coordinator. Extract only evidence-backed categorical hypotheses. Never output probabilities, expected returns, confidence scores, position sizes, or trade actions.' },
+4
View File
@@ -101,6 +101,10 @@ Reply with just the number of the match, or "none" if none apply. No explanation
model: llmConfig.cheapModel || llmConfig.llmModel || llmConfig.model, model: llmConfig.cheapModel || llmConfig.llmModel || llmConfig.model,
messages: [{ role: "user", content: prompt }], messages: [{ role: "user", content: prompt }],
temperature: 0, temperature: 0,
// Same reason as the coordinator: an unbounded request gets a 402 for reserving
// the whole output window. This one only ever replies with a number or "none",
// the headroom is for models that think before answering.
max_tokens: Math.max(256, Number(process.env.OPEN_ROUTER_GRAPH_MAX_TOKENS) || 2000),
}); });
if (Date.now() < llmCooldownUntil) return null; if (Date.now() < llmCooldownUntil) return null;