From ce1ce4dd5701eaa45d66fd626e4c571dd3aa4486 Mon Sep 17 00:00:00 2001 From: ImBenji Date: Tue, 1 Sep 2026 21:36:13 +0100 Subject: [PATCH] fix: bound max_tokens on the remaining llm callers signal, augor and consolidation had the same unbounded request as the coordinator, so switching to a model with a 131k output window made all three 402 on every call while the coordinator itself was fine. Found them by grepping for the endpoint rather than waiting for each one to surface in the logs. Sized per worker rather than one global number, since these produce more than the coordinator's small json object. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb --- workers/augorWorker.js | 3 +++ workers/consolidationWorker.js | 3 +++ workers/signalWorker.js | 3 +++ 3 files changed, 9 insertions(+) diff --git a/workers/augorWorker.js b/workers/augorWorker.js index 2eb424b..3e44754 100644 --- a/workers/augorWorker.js +++ b/workers/augorWorker.js @@ -306,6 +306,9 @@ async function callLlm(llmConfig, prompt) { model: llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0.1, + // Unbounded requests get a 402 for reserving the model's whole output + // window against the remaining key budget, before running anything. + max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 6000), }); const url = new URL("https://openrouter.ai/api/v1/chat/completions"); diff --git a/workers/consolidationWorker.js b/workers/consolidationWorker.js index 559f815..f17a4a4 100644 --- a/workers/consolidationWorker.js +++ b/workers/consolidationWorker.js @@ -220,6 +220,9 @@ Rules: model: llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0.1, + // Unbounded requests get a 402 for reserving the model's whole output + // window against the remaining key budget, before running anything. + max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 6000), }); const url = new URL("https://openrouter.ai/api/v1/chat/completions"); diff --git a/workers/signalWorker.js b/workers/signalWorker.js index 78aaf61..1da1cb1 100644 --- a/workers/signalWorker.js +++ b/workers/signalWorker.js @@ -297,6 +297,9 @@ async function callLlm(llmConfig, prompt) { model: llmConfig.llmModel || llmConfig.model, messages: [{ role: "user", content: prompt }], temperature: 0.1, + // Unbounded requests get a 402 for reserving the model's whole output + // window against the remaining key budget, before running anything. + max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 4000), }); const url = new URL("https://openrouter.ai/api/v1/chat/completions");