From 9b490c4d39a9d7735bc8b898c6844c62f8bcf093 Mon Sep 17 00:00:00 2001 From: ImBenji Date: Fri, 4 Sep 2026 21:57:17 +0100 Subject: [PATCH] fix: honour OPEN_ROUTER_CHEAP_MODEL, it was set and never read The env var has been set to google/gemma-4-31b-it all along and nothing ever mapped it onto openRouter.cheapModel, so graphWorker's fallback chain silently used the main model instead. Graph entity resolution is the highest volume llm call in the system and its entire job is to reply with the number of a match. Measured on the live key, same prompt: deepseek v4 flash 7564 completion tokens, 7558 of them reasoning $0.0013708 gemma-4-31b-it 14 completion tokens, 0 reasoning $0.0000121 113x. Gemma is actually the more expensive model per token, which is why this was worth measuring rather than reasoning about prices: the cost is not the price of the tokens, it is a reasoning model spending seven thousand tokens thinking about a multiple choice question. This also explains the reasoning tokens dominating the usage dashboard, and why the daily spend roughly doubled today rather than yesterday. Removing the token ceiling let a trivial prompt reason without bound. The ceiling was never the right control for that, the model choice is. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb --- src/config.js | 4 ++++ workers/index.js | 4 ++++ 2 files changed, 8 insertions(+) diff --git a/src/config.js b/src/config.js index 1291858..e9f3c75 100644 --- a/src/config.js +++ b/src/config.js @@ -29,6 +29,10 @@ if (process.env.FINNHUB_API_KEY) config.finnhub.apiKey = process.env.FINNH if (process.env.OPEN_ROUTER_API_KEY) config.openRouter.apiKey = process.env.OPEN_ROUTER_API_KEY; if (process.env.OPEN_ROUTER_LLM_MODEL) config.openRouter.llmModel = process.env.OPEN_ROUTER_LLM_MODEL; if (process.env.OPEN_ROUTER_EMBED_MODEL) config.openRouter.embeddingModel = process.env.OPEN_ROUTER_EMBED_MODEL; +// OPEN_ROUTER_CHEAP_MODEL was already set in the environment and nothing read it, +// so graph entity resolution ran on the reasoning model: 7,558 reasoning tokens to +// answer "reply with just the number", 113x the cost of a model that just answers. +if (process.env.OPEN_ROUTER_CHEAP_MODEL) config.openRouter.cheapModel = process.env.OPEN_ROUTER_CHEAP_MODEL; if (process.env.GDELT_BQ_PROJECT) config.gdelt.bigQueryProject = process.env.GDELT_BQ_PROJECT; if (process.env.GDELT_BQ_KEY_FILE) config.gdelt.bigQueryKeyFile = process.env.GDELT_BQ_KEY_FILE; diff --git a/workers/index.js b/workers/index.js index 769a406..a8804b2 100644 --- a/workers/index.js +++ b/workers/index.js @@ -26,6 +26,10 @@ const openRouter = { ...rawConfig.openRouter }; if (process.env.OPEN_ROUTER_API_KEY) openRouter.apiKey = process.env.OPEN_ROUTER_API_KEY; if (process.env.OPEN_ROUTER_LLM_MODEL) openRouter.llmModel = process.env.OPEN_ROUTER_LLM_MODEL; if (process.env.OPEN_ROUTER_EMBED_MODEL) openRouter.embeddingModel = process.env.OPEN_ROUTER_EMBED_MODEL; +// OPEN_ROUTER_CHEAP_MODEL was already set in the environment and nothing read it, +// so graph entity resolution ran on the reasoning model: 7,558 reasoning tokens to +// answer "reply with just the number", 113x the cost of a model that just answers. +if (process.env.OPEN_ROUTER_CHEAP_MODEL) openRouter.cheapModel = process.env.OPEN_ROUTER_CHEAP_MODEL; const config = { duriin_db: process.env.DURIIN_DB || resolvePath(rawConfig.duriin_db, path.resolve(configDir, "archive.sqlite")),