fix: bound max_tokens on the remaining llm callers

signal, augor and consolidation had the same unbounded request as the
coordinator, so switching to a model with a 131k output window made all three
402 on every call while the coordinator itself was fine. Found them by grepping
for the endpoint rather than waiting for each one to surface in the logs.

Sized per worker rather than one global number, since these produce more than
the coordinator's small json object.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-09-01 21:36:13 +01:00
co-authored by Claude Opus 5
parent 4718635c75
commit ce1ce4dd57
3 changed files with 9 additions and 0 deletions
+3
View File
@@ -306,6 +306,9 @@ async function callLlm(llmConfig, prompt) {
model: llmConfig.llmModel || llmConfig.model, model: llmConfig.llmModel || llmConfig.model,
messages: [{ role: "user", content: prompt }], messages: [{ role: "user", content: prompt }],
temperature: 0.1, temperature: 0.1,
// Unbounded requests get a 402 for reserving the model's whole output
// window against the remaining key budget, before running anything.
max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 6000),
}); });
const url = new URL("https://openrouter.ai/api/v1/chat/completions"); const url = new URL("https://openrouter.ai/api/v1/chat/completions");
+3
View File
@@ -220,6 +220,9 @@ Rules:
model: llmConfig.llmModel || llmConfig.model, model: llmConfig.llmModel || llmConfig.model,
messages: [{ role: "user", content: prompt }], messages: [{ role: "user", content: prompt }],
temperature: 0.1, temperature: 0.1,
// Unbounded requests get a 402 for reserving the model's whole output
// window against the remaining key budget, before running anything.
max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 6000),
}); });
const url = new URL("https://openrouter.ai/api/v1/chat/completions"); const url = new URL("https://openrouter.ai/api/v1/chat/completions");
+3
View File
@@ -297,6 +297,9 @@ async function callLlm(llmConfig, prompt) {
model: llmConfig.llmModel || llmConfig.model, model: llmConfig.llmModel || llmConfig.model,
messages: [{ role: "user", content: prompt }], messages: [{ role: "user", content: prompt }],
temperature: 0.1, temperature: 0.1,
// Unbounded requests get a 402 for reserving the model's whole output
// window against the remaining key budget, before running anything.
max_tokens: Math.max(512, Number(process.env.OPEN_ROUTER_MAX_TOKENS) || 4000),
}); });
const url = new URL("https://openrouter.ai/api/v1/chat/completions"); const url = new URL("https://openrouter.ai/api/v1/chat/completions");