feat: let jev answer the crawler's classification, and keep its confidence
The crawler asked gpt-4.1-mini one three-way question per undecided page, and then threw away the confidence it asked for in the same prompt. Jev answers that question as a typed choice for $0.042 per million in and nothing out, with a confidence that falls out of the distribution rather than the model's opinion of itself. It cannot write learnedSignals though -- rule_value is free text -- so the expensive model is still what teaches a new site. Once a site has banked enough rules and jev is sure, we stop paying for it. A dead signal call now falls back to the jev answer instead of losing the page. confidence is stored on crawler_page_classifications, additively, in both dialects. Pattern, rule and signal-model rows keep a null rather than borrowing a number that belonged to a different answer. The openrouter slug is unverified -- jev is beta there and absent from /api/v1/models -- so both the slug and the endpoint are env overridable, and a bad slug degrades to the old path instead of breaking the crawl. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EZ6bEoR6m6vrksSYRiXwVD
This commit is contained in:
@@ -132,6 +132,7 @@ db.exec(`
|
||||
url TEXT PRIMARY KEY,
|
||||
site_name TEXT NOT NULL,
|
||||
classification TEXT NOT NULL,
|
||||
confidence REAL,
|
||||
pattern TEXT,
|
||||
classified_at TEXT NOT NULL DEFAULT (datetime('now'))
|
||||
);
|
||||
@@ -173,4 +174,21 @@ db.exec(`
|
||||
);
|
||||
`);
|
||||
|
||||
// CREATE TABLE IF NOT EXISTS does nothing to a database that already has the table,
|
||||
// so anything added to one above also needs an alter here. sqlite has no
|
||||
// ADD COLUMN IF NOT EXISTS, hence swallowing the duplicate and shouting about
|
||||
// everything else. Additive only -- nothing in here may drop or rewrite a column.
|
||||
for (const statement of [
|
||||
'ALTER TABLE crawler_page_classifications ADD COLUMN confidence REAL',
|
||||
]) {
|
||||
try {
|
||||
db.exec(statement);
|
||||
} catch (error) {
|
||||
const message = String(error && error.message || '').toLowerCase();
|
||||
if (!message.includes('duplicate column') && !message.includes('already exists')) {
|
||||
console.error(`[db] migration failed: ${statement}`, error.message, error.stack);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = db;
|
||||
|
||||
@@ -182,6 +182,7 @@ exec(`
|
||||
url TEXT PRIMARY KEY,
|
||||
site_name TEXT NOT NULL,
|
||||
classification TEXT NOT NULL,
|
||||
confidence DOUBLE PRECISION,
|
||||
pattern TEXT,
|
||||
classified_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
)
|
||||
@@ -223,4 +224,12 @@ exec(`
|
||||
)
|
||||
`);
|
||||
|
||||
// same story as the sqlite side: the creates above skip a database that already
|
||||
// has the table, so additions land here too. postgres does have the IF NOT EXISTS
|
||||
// form so this one stays boring. Additive only.
|
||||
exec(`
|
||||
ALTER TABLE crawler_page_classifications
|
||||
ADD COLUMN IF NOT EXISTS confidence DOUBLE PRECISION
|
||||
`);
|
||||
|
||||
module.exports = pgDb;
|
||||
@@ -1,6 +1,24 @@
|
||||
const db = require('../db');
|
||||
const config = require('../config');
|
||||
|
||||
// Jev only answers the one question the crawler actually cares about, and it answers
|
||||
// it for basically nothing -- $0.042 per million tokens in, output billed at zero.
|
||||
// It cannot write the learnedSignals though, those are free text, so gpt-4.1-mini is
|
||||
// still the model that teaches a new site what its own markup looks like.
|
||||
|
||||
// openrouter carries jev on its own endpoint rather than chat/completions, and it
|
||||
// is still beta there -- the model does not show up in /api/v1/models yet, so if the
|
||||
// slug moves point CRAWLER_JEV_MODEL at it, or CRAWLER_JEV_URL straight at typesafe.
|
||||
const JEV_URL = process.env.CRAWLER_JEV_URL || "https://openrouter.ai/api/v1/systemone";
|
||||
const JEV_MODEL = process.env.CRAWLER_JEV_MODEL || "jev-latest";
|
||||
const SIGNAL_MODEL = process.env.CRAWLER_SIGNAL_MODEL || "openai/gpt-4.1-mini";
|
||||
|
||||
// below this we dont trust jev to write a cached classification on its own
|
||||
const JEV_MIN_CONFIDENCE = Number(process.env.CRAWLER_JEV_MIN_CONFIDENCE) || 0.75;
|
||||
|
||||
// how many rules a site needs banked before we stop paying for signal extraction
|
||||
const SITE_RULE_TARGET = Number(process.env.CRAWLER_SITE_RULE_TARGET) || 12;
|
||||
|
||||
const POSITIVE_RULE_TYPES = new Set([
|
||||
'meta_og_type',
|
||||
'meta_has_publish_time',
|
||||
@@ -37,11 +55,12 @@ const selectCachedClassification = db.prepare(`
|
||||
WHERE url = ?
|
||||
`);
|
||||
const upsertCachedClassification = db.prepare(`
|
||||
INSERT INTO crawler_page_classifications (url, site_name, classification, pattern)
|
||||
VALUES (?, ?, ?, ?)
|
||||
INSERT INTO crawler_page_classifications (url, site_name, classification, confidence, pattern)
|
||||
VALUES (?, ?, ?, ?, ?)
|
||||
ON CONFLICT(url) DO UPDATE SET
|
||||
site_name = excluded.site_name,
|
||||
classification = excluded.classification,
|
||||
confidence = excluded.confidence,
|
||||
pattern = excluded.pattern,
|
||||
classified_at = datetime('now')
|
||||
`);
|
||||
@@ -81,6 +100,11 @@ const upsertRule = db.prepare(`
|
||||
END,
|
||||
updated_at = datetime('now')
|
||||
`);
|
||||
const countRulesForSite = db.prepare(`
|
||||
SELECT COUNT(*) AS n
|
||||
FROM crawler_site_rules
|
||||
WHERE site_name = ?
|
||||
`);
|
||||
|
||||
function normalizePathSegment(segment) {
|
||||
if (/^\d{4}$/.test(segment)) {
|
||||
@@ -657,6 +681,55 @@ function sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, signal
|
||||
return parts.filter(Boolean).join('\n').slice(0, 4200);
|
||||
}
|
||||
|
||||
async function requestJevClassification(sanitizedHtml) {
|
||||
const response = await fetch(JEV_URL, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
Authorization: `Bearer ${String(config.openRouter.apiKey || "").trim()}`,
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: JEV_MODEL,
|
||||
state: sanitizedHtml,
|
||||
questions: {
|
||||
page_kind: {
|
||||
type: "choice",
|
||||
instructions: "Classify this page for a news crawler. The URL, title, meta tags, a sample of links and the first few paragraphs are all in the state.",
|
||||
criteria: {
|
||||
article: "A single news story page.",
|
||||
listing: "Homepage, topic page, category page, archive, feature hub, or any page that is mostly links to other stories.",
|
||||
other: "Anything else -- about and contact pages, video hubs, tag clouds, login walls, utility pages.",
|
||||
},
|
||||
},
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const body = await response.text().catch(() => "");
|
||||
const requestError = new Error(`jev classification failed with ${response.status}: ${body.slice(0, 200)}`);
|
||||
requestError.status = response.status;
|
||||
throw requestError;
|
||||
}
|
||||
|
||||
const payload = await response.json();
|
||||
const answer = payload && payload.answers && payload.answers.page_kind;
|
||||
const choice = String((answer && answer.choice) || "").trim().toLowerCase();
|
||||
|
||||
// the criteria keys are the only three things it can possibly come back with,
|
||||
// but a beta endpoint changing its answer shape shouldnt silently become "other"
|
||||
if (choice !== "article" && choice !== "listing" && choice !== "other") {
|
||||
throw new Error(`jev returned an unusable answer: ${JSON.stringify(answer).slice(0, 200)}`);
|
||||
}
|
||||
|
||||
// this confidence falls out of the shape of the probability distribution rather
|
||||
// than the model telling us how sure it feels, which is why gating on it works
|
||||
// at all. the old prompt asked gpt for a confidence and then never read it.
|
||||
const confidence = Number(answer.confidence);
|
||||
|
||||
return { classification: choice, confidence: Number.isFinite(confidence) ? confidence : 0 };
|
||||
}
|
||||
|
||||
async function requestLlmClassification(url, sanitizedHtml, heuristic) {
|
||||
const response = await fetch('https://openrouter.ai/api/v1/chat/completions', {
|
||||
method: 'POST',
|
||||
@@ -665,7 +738,7 @@ async function requestLlmClassification(url, sanitizedHtml, heuristic) {
|
||||
'Content-Type': 'application/json',
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: 'openai/gpt-4.1-mini',
|
||||
model: SIGNAL_MODEL,
|
||||
messages: [
|
||||
{
|
||||
role: 'system',
|
||||
@@ -757,7 +830,7 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
|
||||
.find((entry) => patternToRegex(entry.pattern).test(pathname));
|
||||
|
||||
if (matchedPattern) {
|
||||
upsertCachedClassification.run(url, siteName, matchedPattern.classification, matchedPattern.pattern);
|
||||
upsertCachedClassification.run(url, siteName, matchedPattern.classification, null, matchedPattern.pattern);
|
||||
return { classification: matchedPattern.classification, source: 'pattern', learnedSignals: [], negativeSignals: [] };
|
||||
}
|
||||
}
|
||||
@@ -765,7 +838,7 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
|
||||
const ruleSignals = buildRuleSignals(url, meta, html, jsonLdArticle, links, heuristic);
|
||||
const matchedRule = selectRulesForSite.all(siteName, minPatternHits).find((rule) => matchRule(rule, ruleSignals));
|
||||
if (matchedRule) {
|
||||
upsertCachedClassification.run(url, siteName, matchedRule.classification, pattern);
|
||||
upsertCachedClassification.run(url, siteName, matchedRule.classification, null, pattern);
|
||||
return { classification: matchedRule.classification, source: 'rule', learnedSignals: [], negativeSignals: [] };
|
||||
}
|
||||
|
||||
@@ -773,13 +846,52 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
|
||||
return { classification: null, source: 'disabled', learnedSignals: [], negativeSignals: [] };
|
||||
}
|
||||
|
||||
const result = await requestLlmClassification(
|
||||
url,
|
||||
sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, ruleSignals),
|
||||
heuristic,
|
||||
);
|
||||
const sanitized = sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, ruleSignals);
|
||||
|
||||
upsertCachedClassification.run(url, siteName, result.classification, pattern);
|
||||
let jev = null;
|
||||
try {
|
||||
jev = await requestJevClassification(sanitized);
|
||||
} catch (error) {
|
||||
// not fatal on its own, the expensive path below can still answer
|
||||
console.error(`[crawler-jev] ${siteName} ${url} failed:`, error.message, error.stack);
|
||||
}
|
||||
|
||||
const bankedRules = Number(countRulesForSite.get(siteName).n) || 0;
|
||||
const stillLearning = bankedRules < SITE_RULE_TARGET;
|
||||
const unsure = !jev || jev.confidence < JEV_MIN_CONFIDENCE;
|
||||
|
||||
// steady state. the site has already taught us enough rules and jev is sure, so
|
||||
// there is nothing left for the expensive model to add here.
|
||||
if (!stillLearning && !unsure) {
|
||||
upsertCachedClassification.run(url, siteName, jev.classification, jev.confidence, pattern);
|
||||
|
||||
if (pattern) {
|
||||
upsertPattern.run(siteName, pattern, jev.classification);
|
||||
}
|
||||
|
||||
console.log(`[crawler-jev] ${siteName} ${jev.classification.toUpperCase()} conf=${jev.confidence.toFixed(2)} ${url}`);
|
||||
return { classification: jev.classification, source: 'jev', learnedSignals: [], negativeSignals: [] };
|
||||
}
|
||||
|
||||
let result;
|
||||
try {
|
||||
result = await requestLlmClassification(url, sanitized, heuristic);
|
||||
} catch (error) {
|
||||
console.error(`[crawler-llm] ${siteName} ${url} failed:`, error.message, error.stack);
|
||||
|
||||
// we already paid for jev, so a dead signal call shouldnt also cost us the
|
||||
// page. nothing is cached or learned off the back of it though.
|
||||
if (!jev) {
|
||||
throw error;
|
||||
}
|
||||
|
||||
console.warn(`[crawler-llm] falling back to jev for ${url} (conf=${jev.confidence.toFixed(2)})`);
|
||||
return { classification: jev.classification, source: 'jev-fallback', learnedSignals: [], negativeSignals: [] };
|
||||
}
|
||||
|
||||
// null, not jev's number -- the row holds the signal model's classification and
|
||||
// jev's confidence was in its own answer, which may not even be the same one.
|
||||
upsertCachedClassification.run(url, siteName, result.classification, null, pattern);
|
||||
|
||||
if (pattern) {
|
||||
upsertPattern.run(siteName, pattern, result.classification);
|
||||
@@ -789,6 +901,12 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
|
||||
upsertRule.run(siteName, signal.ruleType, signal.ruleValue, result.classification);
|
||||
}
|
||||
|
||||
if (jev && jev.classification !== result.classification) {
|
||||
// if this stays noisy for one site the rules it is banking are probably junk,
|
||||
// or the sanitized state is cutting off the part that decides it.
|
||||
console.warn(`[crawler-jev] disagreed on ${url}: jev=${jev.classification} conf=${jev.confidence.toFixed(2)} llm=${result.classification}`);
|
||||
}
|
||||
|
||||
console.log(`[crawler-llm] ${siteName} ${result.classification.toUpperCase()} ${url}`);
|
||||
return {
|
||||
classification: result.classification,
|
||||
|
||||
Reference in New Issue
Block a user