From 1c75e4171e67433fc0adf4902243fe64a4486267 Mon Sep 17 00:00:00 2001 From: ImBenji Date: Sun, 20 Sep 2026 15:39:51 +0100 Subject: [PATCH] feat: let jev answer the crawler's classification, and keep its confidence The crawler asked gpt-4.1-mini one three-way question per undecided page, and then threw away the confidence it asked for in the same prompt. Jev answers that question as a typed choice for $0.042 per million in and nothing out, with a confidence that falls out of the distribution rather than the model's opinion of itself. It cannot write learnedSignals though -- rule_value is free text -- so the expensive model is still what teaches a new site. Once a site has banked enough rules and jev is sure, we stop paying for it. A dead signal call now falls back to the jev answer instead of losing the page. confidence is stored on crawler_page_classifications, additively, in both dialects. Pattern, rule and signal-model rows keep a null rather than borrowing a number that belonged to a different answer. The openrouter slug is unverified -- jev is beta there and absent from /api/v1/models -- so both the slug and the endpoint are env overridable, and a bad slug degrades to the old path instead of breaking the crawl. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01EZ6bEoR6m6vrksSYRiXwVD --- src/db/index.js | 18 ++++ src/db/postgres.js | 9 ++ src/sources/crawlerClassifier.js | 140 ++++++++++++++++++++++++++++--- 3 files changed, 156 insertions(+), 11 deletions(-) diff --git a/src/db/index.js b/src/db/index.js index a3751e6..9313969 100644 --- a/src/db/index.js +++ b/src/db/index.js @@ -132,6 +132,7 @@ db.exec(` url TEXT PRIMARY KEY, site_name TEXT NOT NULL, classification TEXT NOT NULL, + confidence REAL, pattern TEXT, classified_at TEXT NOT NULL DEFAULT (datetime('now')) ); @@ -173,4 +174,21 @@ db.exec(` ); `); +// CREATE TABLE IF NOT EXISTS does nothing to a database that already has the table, +// so anything added to one above also needs an alter here. sqlite has no +// ADD COLUMN IF NOT EXISTS, hence swallowing the duplicate and shouting about +// everything else. Additive only -- nothing in here may drop or rewrite a column. +for (const statement of [ + 'ALTER TABLE crawler_page_classifications ADD COLUMN confidence REAL', +]) { + try { + db.exec(statement); + } catch (error) { + const message = String(error && error.message || '').toLowerCase(); + if (!message.includes('duplicate column') && !message.includes('already exists')) { + console.error(`[db] migration failed: ${statement}`, error.message, error.stack); + } + } +} + module.exports = db; diff --git a/src/db/postgres.js b/src/db/postgres.js index cf70a69..fb48e49 100644 --- a/src/db/postgres.js +++ b/src/db/postgres.js @@ -182,6 +182,7 @@ exec(` url TEXT PRIMARY KEY, site_name TEXT NOT NULL, classification TEXT NOT NULL, + confidence DOUBLE PRECISION, pattern TEXT, classified_at TIMESTAMPTZ NOT NULL DEFAULT NOW() ) @@ -223,4 +224,12 @@ exec(` ) `); +// same story as the sqlite side: the creates above skip a database that already +// has the table, so additions land here too. postgres does have the IF NOT EXISTS +// form so this one stays boring. Additive only. +exec(` + ALTER TABLE crawler_page_classifications + ADD COLUMN IF NOT EXISTS confidence DOUBLE PRECISION +`); + module.exports = pgDb; \ No newline at end of file diff --git a/src/sources/crawlerClassifier.js b/src/sources/crawlerClassifier.js index 25e15e8..3b4ca60 100644 --- a/src/sources/crawlerClassifier.js +++ b/src/sources/crawlerClassifier.js @@ -1,6 +1,24 @@ const db = require('../db'); const config = require('../config'); +// Jev only answers the one question the crawler actually cares about, and it answers +// it for basically nothing -- $0.042 per million tokens in, output billed at zero. +// It cannot write the learnedSignals though, those are free text, so gpt-4.1-mini is +// still the model that teaches a new site what its own markup looks like. + +// openrouter carries jev on its own endpoint rather than chat/completions, and it +// is still beta there -- the model does not show up in /api/v1/models yet, so if the +// slug moves point CRAWLER_JEV_MODEL at it, or CRAWLER_JEV_URL straight at typesafe. +const JEV_URL = process.env.CRAWLER_JEV_URL || "https://openrouter.ai/api/v1/systemone"; +const JEV_MODEL = process.env.CRAWLER_JEV_MODEL || "jev-latest"; +const SIGNAL_MODEL = process.env.CRAWLER_SIGNAL_MODEL || "openai/gpt-4.1-mini"; + +// below this we dont trust jev to write a cached classification on its own +const JEV_MIN_CONFIDENCE = Number(process.env.CRAWLER_JEV_MIN_CONFIDENCE) || 0.75; + +// how many rules a site needs banked before we stop paying for signal extraction +const SITE_RULE_TARGET = Number(process.env.CRAWLER_SITE_RULE_TARGET) || 12; + const POSITIVE_RULE_TYPES = new Set([ 'meta_og_type', 'meta_has_publish_time', @@ -37,11 +55,12 @@ const selectCachedClassification = db.prepare(` WHERE url = ? `); const upsertCachedClassification = db.prepare(` - INSERT INTO crawler_page_classifications (url, site_name, classification, pattern) - VALUES (?, ?, ?, ?) + INSERT INTO crawler_page_classifications (url, site_name, classification, confidence, pattern) + VALUES (?, ?, ?, ?, ?) ON CONFLICT(url) DO UPDATE SET site_name = excluded.site_name, classification = excluded.classification, + confidence = excluded.confidence, pattern = excluded.pattern, classified_at = datetime('now') `); @@ -81,6 +100,11 @@ const upsertRule = db.prepare(` END, updated_at = datetime('now') `); +const countRulesForSite = db.prepare(` + SELECT COUNT(*) AS n + FROM crawler_site_rules + WHERE site_name = ? +`); function normalizePathSegment(segment) { if (/^\d{4}$/.test(segment)) { @@ -657,6 +681,55 @@ function sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, signal return parts.filter(Boolean).join('\n').slice(0, 4200); } +async function requestJevClassification(sanitizedHtml) { + const response = await fetch(JEV_URL, { + method: "POST", + headers: { + Authorization: `Bearer ${String(config.openRouter.apiKey || "").trim()}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ + model: JEV_MODEL, + state: sanitizedHtml, + questions: { + page_kind: { + type: "choice", + instructions: "Classify this page for a news crawler. The URL, title, meta tags, a sample of links and the first few paragraphs are all in the state.", + criteria: { + article: "A single news story page.", + listing: "Homepage, topic page, category page, archive, feature hub, or any page that is mostly links to other stories.", + other: "Anything else -- about and contact pages, video hubs, tag clouds, login walls, utility pages.", + }, + }, + }, + }), + }); + + if (!response.ok) { + const body = await response.text().catch(() => ""); + const requestError = new Error(`jev classification failed with ${response.status}: ${body.slice(0, 200)}`); + requestError.status = response.status; + throw requestError; + } + + const payload = await response.json(); + const answer = payload && payload.answers && payload.answers.page_kind; + const choice = String((answer && answer.choice) || "").trim().toLowerCase(); + + // the criteria keys are the only three things it can possibly come back with, + // but a beta endpoint changing its answer shape shouldnt silently become "other" + if (choice !== "article" && choice !== "listing" && choice !== "other") { + throw new Error(`jev returned an unusable answer: ${JSON.stringify(answer).slice(0, 200)}`); + } + + // this confidence falls out of the shape of the probability distribution rather + // than the model telling us how sure it feels, which is why gating on it works + // at all. the old prompt asked gpt for a confidence and then never read it. + const confidence = Number(answer.confidence); + + return { classification: choice, confidence: Number.isFinite(confidence) ? confidence : 0 }; +} + async function requestLlmClassification(url, sanitizedHtml, heuristic) { const response = await fetch('https://openrouter.ai/api/v1/chat/completions', { method: 'POST', @@ -665,7 +738,7 @@ async function requestLlmClassification(url, sanitizedHtml, heuristic) { 'Content-Type': 'application/json', }, body: JSON.stringify({ - model: 'openai/gpt-4.1-mini', + model: SIGNAL_MODEL, messages: [ { role: 'system', @@ -757,7 +830,7 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h .find((entry) => patternToRegex(entry.pattern).test(pathname)); if (matchedPattern) { - upsertCachedClassification.run(url, siteName, matchedPattern.classification, matchedPattern.pattern); + upsertCachedClassification.run(url, siteName, matchedPattern.classification, null, matchedPattern.pattern); return { classification: matchedPattern.classification, source: 'pattern', learnedSignals: [], negativeSignals: [] }; } } @@ -765,7 +838,7 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h const ruleSignals = buildRuleSignals(url, meta, html, jsonLdArticle, links, heuristic); const matchedRule = selectRulesForSite.all(siteName, minPatternHits).find((rule) => matchRule(rule, ruleSignals)); if (matchedRule) { - upsertCachedClassification.run(url, siteName, matchedRule.classification, pattern); + upsertCachedClassification.run(url, siteName, matchedRule.classification, null, pattern); return { classification: matchedRule.classification, source: 'rule', learnedSignals: [], negativeSignals: [] }; } @@ -773,13 +846,52 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h return { classification: null, source: 'disabled', learnedSignals: [], negativeSignals: [] }; } - const result = await requestLlmClassification( - url, - sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, ruleSignals), - heuristic, - ); + const sanitized = sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, ruleSignals); - upsertCachedClassification.run(url, siteName, result.classification, pattern); + let jev = null; + try { + jev = await requestJevClassification(sanitized); + } catch (error) { + // not fatal on its own, the expensive path below can still answer + console.error(`[crawler-jev] ${siteName} ${url} failed:`, error.message, error.stack); + } + + const bankedRules = Number(countRulesForSite.get(siteName).n) || 0; + const stillLearning = bankedRules < SITE_RULE_TARGET; + const unsure = !jev || jev.confidence < JEV_MIN_CONFIDENCE; + + // steady state. the site has already taught us enough rules and jev is sure, so + // there is nothing left for the expensive model to add here. + if (!stillLearning && !unsure) { + upsertCachedClassification.run(url, siteName, jev.classification, jev.confidence, pattern); + + if (pattern) { + upsertPattern.run(siteName, pattern, jev.classification); + } + + console.log(`[crawler-jev] ${siteName} ${jev.classification.toUpperCase()} conf=${jev.confidence.toFixed(2)} ${url}`); + return { classification: jev.classification, source: 'jev', learnedSignals: [], negativeSignals: [] }; + } + + let result; + try { + result = await requestLlmClassification(url, sanitized, heuristic); + } catch (error) { + console.error(`[crawler-llm] ${siteName} ${url} failed:`, error.message, error.stack); + + // we already paid for jev, so a dead signal call shouldnt also cost us the + // page. nothing is cached or learned off the back of it though. + if (!jev) { + throw error; + } + + console.warn(`[crawler-llm] falling back to jev for ${url} (conf=${jev.confidence.toFixed(2)})`); + return { classification: jev.classification, source: 'jev-fallback', learnedSignals: [], negativeSignals: [] }; + } + + // null, not jev's number -- the row holds the signal model's classification and + // jev's confidence was in its own answer, which may not even be the same one. + upsertCachedClassification.run(url, siteName, result.classification, null, pattern); if (pattern) { upsertPattern.run(siteName, pattern, result.classification); @@ -789,6 +901,12 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h upsertRule.run(siteName, signal.ruleType, signal.ruleValue, result.classification); } + if (jev && jev.classification !== result.classification) { + // if this stays noisy for one site the rules it is banking are probably junk, + // or the sanitized state is cutting off the part that decides it. + console.warn(`[crawler-jev] disagreed on ${url}: jev=${jev.classification} conf=${jev.confidence.toFixed(2)} llm=${result.classification}`); + } + console.log(`[crawler-llm] ${siteName} ${result.classification.toUpperCase()} ${url}`); return { classification: result.classification,