fix: adapt to the budget ceiling and never parse truncated json

Two things wrong with the fixed cap I added earlier.

It was guessed rather than measured. The largest proposal this has ever produced
was about 1,258 tokens carrying 12 predictions and the average is around 50, so
8000 was arbitrary, and worse, it was above what the key could afford by the
time it deployed. The ceiling openrouter will accept shrinks as the balance
depletes: 15,666 earlier today, 3,921 an hour later.

So the cap is now adaptive. A 402 names the ceiling, and we retry once just under
it, downwards only. A shrinking budget shortens the allowed answer instead of
stopping the pipeline dead. Worth being clear that removing the cap is not an
option on a limited key, an unbounded request is refused outright and produces
no output at all rather than a truncated one.

And truncation is no longer silent. finish_reason length now throws instead of
handing a half written response to extractJson, which could occasionally parse a
partial object and quietly drop predictions. That was the real risk in capping.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-09-01 21:54:34 +01:00
co-authored by Claude Opus 5
parent ce1ce4dd57
commit 73d09c943f
+38 -7
View File
@@ -8,7 +8,19 @@ function extractJson(text) {
}
}
async function callCoordinator(config, prompt) {
// OpenRouter reserves max_tokens against the key's remaining budget up front, and
// that affordable ceiling shrinks as the balance depletes. An unbounded request is
// refused outright, so "no cap" is not an option on a limited key -- it produces no
// output at all rather than truncated output. This pulls the real ceiling out of the
// refusal so we can retry just under it instead of guessing a fixed number.
function affordableTokens(message) {
const match = /can only afford (\d+)/i.exec(String(message || ''));
if (!match) return null;
const affordable = Number(match[1]);
return Number.isFinite(affordable) && affordable > 256 ? affordable : null;
}
async function callCoordinator(config, prompt, options = {}) {
const apiKey = String(config?.openRouter?.apiKey || '').trim();
if (!apiKey) throw new Error('OpenRouter API key is not configured');
const timeoutMs = Math.max(1000, Number(config?.openRouter?.timeoutMs || process.env.OPEN_ROUTER_TIMEOUT_MS) || 60000);
@@ -23,11 +35,12 @@ async function callCoordinator(config, prompt) {
body: JSON.stringify({
model: config.openRouter.llmModel,
temperature: 0,
// Without this OpenRouter reserves the model's entire output window against
// the key's remaining budget and rejects the call with a 402 before it ever
// runs -- 131k tokens reserved to produce a few hundred. Reasoning models
// also spend completion tokens on reasoning, so leave real headroom.
max_tokens: Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 8000),
// Sized against reality rather than guessed: the largest proposal this has
// ever produced was ~1,258 tokens carrying 12 predictions, the average is
// ~50. Reasoning models spend completion tokens thinking first, so this is
// still several times the worst case actually observed.
max_tokens: options.maxTokens
|| Math.max(512, Number(config?.openRouter?.maxTokens || process.env.OPEN_ROUTER_MAX_TOKENS) || 6000),
response_format: { type: 'json_object' },
messages: [
{ role: 'system', content: 'You are a coordinator. Extract only evidence-backed categorical hypotheses. Never output probabilities, expected returns, confidence scores, position sizes, or trade actions.' },
@@ -43,10 +56,28 @@ async function callCoordinator(config, prompt) {
}
if (!response.ok) {
const body = await response.text().catch(() => '');
// A 402 names the ceiling the key can currently afford. Retry once just under
// it rather than failing the event, but only downwards, so a shrinking budget
// degrades output length instead of stopping the pipeline dead.
const affordable = response.status === 402 ? affordableTokens(body) : null;
if (affordable && !options.retriedForBudget) {
const retryTokens = Math.floor(affordable * 0.9);
console.warn(`[llm] budget only affords ${affordable} tokens, retrying with max_tokens=${retryTokens}`);
return callCoordinator(config, prompt, { maxTokens: retryTokens, retriedForBudget: true });
}
throw new Error(`coordinator request failed with ${response.status}: ${body.slice(0, 300)}`);
}
const body = await response.json();
return extractJson(body?.choices?.[0]?.message?.content);
const choice = body?.choices?.[0];
// Truncated json is worse than no json, because a partial object can occasionally
// still parse and quietly lose predictions. Fail loudly on the reason field rather
// than letting extractJson guess at a half-written response.
if (choice?.finish_reason === 'length') {
throw new Error('coordinator response was truncated by max_tokens, raise OPEN_ROUTER_MAX_TOKENS');
}
return extractJson(choice?.message?.content);
}
module.exports = { extractJson, callCoordinator };