feat: let jev answer the crawler's classification, and keep its confidence

The crawler asked gpt-4.1-mini one three-way question per undecided page, and
then threw away the confidence it asked for in the same prompt. Jev answers that
question as a typed choice for $0.042 per million in and nothing out, with a
confidence that falls out of the distribution rather than the model's opinion of
itself.

It cannot write learnedSignals though -- rule_value is free text -- so the
expensive model is still what teaches a new site. Once a site has banked enough
rules and jev is sure, we stop paying for it. A dead signal call now falls back
to the jev answer instead of losing the page.

confidence is stored on crawler_page_classifications, additively, in both
dialects. Pattern, rule and signal-model rows keep a null rather than borrowing
a number that belonged to a different answer.

The openrouter slug is unverified -- jev is beta there and absent from
/api/v1/models -- so both the slug and the endpoint are env overridable, and a
bad slug degrades to the old path instead of breaking the crawl.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01EZ6bEoR6m6vrksSYRiXwVD
This commit is contained in:
ImBenji
2026-09-20 15:39:51 +01:00
co-authored by Claude Opus 5
parent 20fb021983
commit 1c75e4171e
3 changed files with 156 additions and 11 deletions
+18
View File
@@ -132,6 +132,7 @@ db.exec(`
url TEXT PRIMARY KEY, url TEXT PRIMARY KEY,
site_name TEXT NOT NULL, site_name TEXT NOT NULL,
classification TEXT NOT NULL, classification TEXT NOT NULL,
confidence REAL,
pattern TEXT, pattern TEXT,
classified_at TEXT NOT NULL DEFAULT (datetime('now')) classified_at TEXT NOT NULL DEFAULT (datetime('now'))
); );
@@ -173,4 +174,21 @@ db.exec(`
); );
`); `);
// CREATE TABLE IF NOT EXISTS does nothing to a database that already has the table,
// so anything added to one above also needs an alter here. sqlite has no
// ADD COLUMN IF NOT EXISTS, hence swallowing the duplicate and shouting about
// everything else. Additive only -- nothing in here may drop or rewrite a column.
for (const statement of [
'ALTER TABLE crawler_page_classifications ADD COLUMN confidence REAL',
]) {
try {
db.exec(statement);
} catch (error) {
const message = String(error && error.message || '').toLowerCase();
if (!message.includes('duplicate column') && !message.includes('already exists')) {
console.error(`[db] migration failed: ${statement}`, error.message, error.stack);
}
}
}
module.exports = db; module.exports = db;
+9
View File
@@ -182,6 +182,7 @@ exec(`
url TEXT PRIMARY KEY, url TEXT PRIMARY KEY,
site_name TEXT NOT NULL, site_name TEXT NOT NULL,
classification TEXT NOT NULL, classification TEXT NOT NULL,
confidence DOUBLE PRECISION,
pattern TEXT, pattern TEXT,
classified_at TIMESTAMPTZ NOT NULL DEFAULT NOW() classified_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
) )
@@ -223,4 +224,12 @@ exec(`
) )
`); `);
// same story as the sqlite side: the creates above skip a database that already
// has the table, so additions land here too. postgres does have the IF NOT EXISTS
// form so this one stays boring. Additive only.
exec(`
ALTER TABLE crawler_page_classifications
ADD COLUMN IF NOT EXISTS confidence DOUBLE PRECISION
`);
module.exports = pgDb; module.exports = pgDb;
+129 -11
View File
@@ -1,6 +1,24 @@
const db = require('../db'); const db = require('../db');
const config = require('../config'); const config = require('../config');
// Jev only answers the one question the crawler actually cares about, and it answers
// it for basically nothing -- $0.042 per million tokens in, output billed at zero.
// It cannot write the learnedSignals though, those are free text, so gpt-4.1-mini is
// still the model that teaches a new site what its own markup looks like.
// openrouter carries jev on its own endpoint rather than chat/completions, and it
// is still beta there -- the model does not show up in /api/v1/models yet, so if the
// slug moves point CRAWLER_JEV_MODEL at it, or CRAWLER_JEV_URL straight at typesafe.
const JEV_URL = process.env.CRAWLER_JEV_URL || "https://openrouter.ai/api/v1/systemone";
const JEV_MODEL = process.env.CRAWLER_JEV_MODEL || "jev-latest";
const SIGNAL_MODEL = process.env.CRAWLER_SIGNAL_MODEL || "openai/gpt-4.1-mini";
// below this we dont trust jev to write a cached classification on its own
const JEV_MIN_CONFIDENCE = Number(process.env.CRAWLER_JEV_MIN_CONFIDENCE) || 0.75;
// how many rules a site needs banked before we stop paying for signal extraction
const SITE_RULE_TARGET = Number(process.env.CRAWLER_SITE_RULE_TARGET) || 12;
const POSITIVE_RULE_TYPES = new Set([ const POSITIVE_RULE_TYPES = new Set([
'meta_og_type', 'meta_og_type',
'meta_has_publish_time', 'meta_has_publish_time',
@@ -37,11 +55,12 @@ const selectCachedClassification = db.prepare(`
WHERE url = ? WHERE url = ?
`); `);
const upsertCachedClassification = db.prepare(` const upsertCachedClassification = db.prepare(`
INSERT INTO crawler_page_classifications (url, site_name, classification, pattern) INSERT INTO crawler_page_classifications (url, site_name, classification, confidence, pattern)
VALUES (?, ?, ?, ?) VALUES (?, ?, ?, ?, ?)
ON CONFLICT(url) DO UPDATE SET ON CONFLICT(url) DO UPDATE SET
site_name = excluded.site_name, site_name = excluded.site_name,
classification = excluded.classification, classification = excluded.classification,
confidence = excluded.confidence,
pattern = excluded.pattern, pattern = excluded.pattern,
classified_at = datetime('now') classified_at = datetime('now')
`); `);
@@ -81,6 +100,11 @@ const upsertRule = db.prepare(`
END, END,
updated_at = datetime('now') updated_at = datetime('now')
`); `);
const countRulesForSite = db.prepare(`
SELECT COUNT(*) AS n
FROM crawler_site_rules
WHERE site_name = ?
`);
function normalizePathSegment(segment) { function normalizePathSegment(segment) {
if (/^\d{4}$/.test(segment)) { if (/^\d{4}$/.test(segment)) {
@@ -657,6 +681,55 @@ function sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, signal
return parts.filter(Boolean).join('\n').slice(0, 4200); return parts.filter(Boolean).join('\n').slice(0, 4200);
} }
async function requestJevClassification(sanitizedHtml) {
const response = await fetch(JEV_URL, {
method: "POST",
headers: {
Authorization: `Bearer ${String(config.openRouter.apiKey || "").trim()}`,
"Content-Type": "application/json",
},
body: JSON.stringify({
model: JEV_MODEL,
state: sanitizedHtml,
questions: {
page_kind: {
type: "choice",
instructions: "Classify this page for a news crawler. The URL, title, meta tags, a sample of links and the first few paragraphs are all in the state.",
criteria: {
article: "A single news story page.",
listing: "Homepage, topic page, category page, archive, feature hub, or any page that is mostly links to other stories.",
other: "Anything else -- about and contact pages, video hubs, tag clouds, login walls, utility pages.",
},
},
},
}),
});
if (!response.ok) {
const body = await response.text().catch(() => "");
const requestError = new Error(`jev classification failed with ${response.status}: ${body.slice(0, 200)}`);
requestError.status = response.status;
throw requestError;
}
const payload = await response.json();
const answer = payload && payload.answers && payload.answers.page_kind;
const choice = String((answer && answer.choice) || "").trim().toLowerCase();
// the criteria keys are the only three things it can possibly come back with,
// but a beta endpoint changing its answer shape shouldnt silently become "other"
if (choice !== "article" && choice !== "listing" && choice !== "other") {
throw new Error(`jev returned an unusable answer: ${JSON.stringify(answer).slice(0, 200)}`);
}
// this confidence falls out of the shape of the probability distribution rather
// than the model telling us how sure it feels, which is why gating on it works
// at all. the old prompt asked gpt for a confidence and then never read it.
const confidence = Number(answer.confidence);
return { classification: choice, confidence: Number.isFinite(confidence) ? confidence : 0 };
}
async function requestLlmClassification(url, sanitizedHtml, heuristic) { async function requestLlmClassification(url, sanitizedHtml, heuristic) {
const response = await fetch('https://openrouter.ai/api/v1/chat/completions', { const response = await fetch('https://openrouter.ai/api/v1/chat/completions', {
method: 'POST', method: 'POST',
@@ -665,7 +738,7 @@ async function requestLlmClassification(url, sanitizedHtml, heuristic) {
'Content-Type': 'application/json', 'Content-Type': 'application/json',
}, },
body: JSON.stringify({ body: JSON.stringify({
model: 'openai/gpt-4.1-mini', model: SIGNAL_MODEL,
messages: [ messages: [
{ {
role: 'system', role: 'system',
@@ -757,7 +830,7 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
.find((entry) => patternToRegex(entry.pattern).test(pathname)); .find((entry) => patternToRegex(entry.pattern).test(pathname));
if (matchedPattern) { if (matchedPattern) {
upsertCachedClassification.run(url, siteName, matchedPattern.classification, matchedPattern.pattern); upsertCachedClassification.run(url, siteName, matchedPattern.classification, null, matchedPattern.pattern);
return { classification: matchedPattern.classification, source: 'pattern', learnedSignals: [], negativeSignals: [] }; return { classification: matchedPattern.classification, source: 'pattern', learnedSignals: [], negativeSignals: [] };
} }
} }
@@ -765,7 +838,7 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
const ruleSignals = buildRuleSignals(url, meta, html, jsonLdArticle, links, heuristic); const ruleSignals = buildRuleSignals(url, meta, html, jsonLdArticle, links, heuristic);
const matchedRule = selectRulesForSite.all(siteName, minPatternHits).find((rule) => matchRule(rule, ruleSignals)); const matchedRule = selectRulesForSite.all(siteName, minPatternHits).find((rule) => matchRule(rule, ruleSignals));
if (matchedRule) { if (matchedRule) {
upsertCachedClassification.run(url, siteName, matchedRule.classification, pattern); upsertCachedClassification.run(url, siteName, matchedRule.classification, null, pattern);
return { classification: matchedRule.classification, source: 'rule', learnedSignals: [], negativeSignals: [] }; return { classification: matchedRule.classification, source: 'rule', learnedSignals: [], negativeSignals: [] };
} }
@@ -773,13 +846,52 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
return { classification: null, source: 'disabled', learnedSignals: [], negativeSignals: [] }; return { classification: null, source: 'disabled', learnedSignals: [], negativeSignals: [] };
} }
const result = await requestLlmClassification( const sanitized = sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, ruleSignals);
url,
sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, ruleSignals),
heuristic,
);
upsertCachedClassification.run(url, siteName, result.classification, pattern); let jev = null;
try {
jev = await requestJevClassification(sanitized);
} catch (error) {
// not fatal on its own, the expensive path below can still answer
console.error(`[crawler-jev] ${siteName} ${url} failed:`, error.message, error.stack);
}
const bankedRules = Number(countRulesForSite.get(siteName).n) || 0;
const stillLearning = bankedRules < SITE_RULE_TARGET;
const unsure = !jev || jev.confidence < JEV_MIN_CONFIDENCE;
// steady state. the site has already taught us enough rules and jev is sure, so
// there is nothing left for the expensive model to add here.
if (!stillLearning && !unsure) {
upsertCachedClassification.run(url, siteName, jev.classification, jev.confidence, pattern);
if (pattern) {
upsertPattern.run(siteName, pattern, jev.classification);
}
console.log(`[crawler-jev] ${siteName} ${jev.classification.toUpperCase()} conf=${jev.confidence.toFixed(2)} ${url}`);
return { classification: jev.classification, source: 'jev', learnedSignals: [], negativeSignals: [] };
}
let result;
try {
result = await requestLlmClassification(url, sanitized, heuristic);
} catch (error) {
console.error(`[crawler-llm] ${siteName} ${url} failed:`, error.message, error.stack);
// we already paid for jev, so a dead signal call shouldnt also cost us the
// page. nothing is cached or learned off the back of it though.
if (!jev) {
throw error;
}
console.warn(`[crawler-llm] falling back to jev for ${url} (conf=${jev.confidence.toFixed(2)})`);
return { classification: jev.classification, source: 'jev-fallback', learnedSignals: [], negativeSignals: [] };
}
// null, not jev's number -- the row holds the signal model's classification and
// jev's confidence was in its own answer, which may not even be the same one.
upsertCachedClassification.run(url, siteName, result.classification, null, pattern);
if (pattern) { if (pattern) {
upsertPattern.run(siteName, pattern, result.classification); upsertPattern.run(siteName, pattern, result.classification);
@@ -789,6 +901,12 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
upsertRule.run(siteName, signal.ruleType, signal.ruleValue, result.classification); upsertRule.run(siteName, signal.ruleType, signal.ruleValue, result.classification);
} }
if (jev && jev.classification !== result.classification) {
// if this stays noisy for one site the rules it is banking are probably junk,
// or the sanitized state is cutting off the part that decides it.
console.warn(`[crawler-jev] disagreed on ${url}: jev=${jev.classification} conf=${jev.confidence.toFixed(2)} llm=${result.classification}`);
}
console.log(`[crawler-llm] ${siteName} ${result.classification.toUpperCase()} ${url}`); console.log(`[crawler-llm] ${siteName} ${result.classification.toUpperCase()} ${url}`);
return { return {
classification: result.classification, classification: result.classification,