|
|
@@ -1,6 +1,24 @@
|
|
|
|
const db = require('../db');
|
|
|
|
const db = require('../db');
|
|
|
|
const config = require('../config');
|
|
|
|
const config = require('../config');
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
// Jev only answers the one question the crawler actually cares about, and it answers
|
|
|
|
|
|
|
|
// it for basically nothing -- $0.042 per million tokens in, output billed at zero.
|
|
|
|
|
|
|
|
// It cannot write the learnedSignals though, those are free text, so gpt-4.1-mini is
|
|
|
|
|
|
|
|
// still the model that teaches a new site what its own markup looks like.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
// openrouter carries jev on its own endpoint rather than chat/completions, and it
|
|
|
|
|
|
|
|
// is still beta there -- the model does not show up in /api/v1/models yet, so if the
|
|
|
|
|
|
|
|
// slug moves point CRAWLER_JEV_MODEL at it, or CRAWLER_JEV_URL straight at typesafe.
|
|
|
|
|
|
|
|
const JEV_URL = process.env.CRAWLER_JEV_URL || "https://openrouter.ai/api/v1/systemone";
|
|
|
|
|
|
|
|
const JEV_MODEL = process.env.CRAWLER_JEV_MODEL || "jev-latest";
|
|
|
|
|
|
|
|
const SIGNAL_MODEL = process.env.CRAWLER_SIGNAL_MODEL || "openai/gpt-4.1-mini";
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
// below this we dont trust jev to write a cached classification on its own
|
|
|
|
|
|
|
|
const JEV_MIN_CONFIDENCE = Number(process.env.CRAWLER_JEV_MIN_CONFIDENCE) || 0.75;
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
// how many rules a site needs banked before we stop paying for signal extraction
|
|
|
|
|
|
|
|
const SITE_RULE_TARGET = Number(process.env.CRAWLER_SITE_RULE_TARGET) || 12;
|
|
|
|
|
|
|
|
|
|
|
|
const POSITIVE_RULE_TYPES = new Set([
|
|
|
|
const POSITIVE_RULE_TYPES = new Set([
|
|
|
|
'meta_og_type',
|
|
|
|
'meta_og_type',
|
|
|
|
'meta_has_publish_time',
|
|
|
|
'meta_has_publish_time',
|
|
|
@@ -37,11 +55,12 @@ const selectCachedClassification = db.prepare(`
|
|
|
|
WHERE url = ?
|
|
|
|
WHERE url = ?
|
|
|
|
`);
|
|
|
|
`);
|
|
|
|
const upsertCachedClassification = db.prepare(`
|
|
|
|
const upsertCachedClassification = db.prepare(`
|
|
|
|
INSERT INTO crawler_page_classifications (url, site_name, classification, pattern)
|
|
|
|
INSERT INTO crawler_page_classifications (url, site_name, classification, confidence, pattern)
|
|
|
|
VALUES (?, ?, ?, ?)
|
|
|
|
VALUES (?, ?, ?, ?, ?)
|
|
|
|
ON CONFLICT(url) DO UPDATE SET
|
|
|
|
ON CONFLICT(url) DO UPDATE SET
|
|
|
|
site_name = excluded.site_name,
|
|
|
|
site_name = excluded.site_name,
|
|
|
|
classification = excluded.classification,
|
|
|
|
classification = excluded.classification,
|
|
|
|
|
|
|
|
confidence = excluded.confidence,
|
|
|
|
pattern = excluded.pattern,
|
|
|
|
pattern = excluded.pattern,
|
|
|
|
classified_at = datetime('now')
|
|
|
|
classified_at = datetime('now')
|
|
|
|
`);
|
|
|
|
`);
|
|
|
@@ -81,6 +100,11 @@ const upsertRule = db.prepare(`
|
|
|
|
END,
|
|
|
|
END,
|
|
|
|
updated_at = datetime('now')
|
|
|
|
updated_at = datetime('now')
|
|
|
|
`);
|
|
|
|
`);
|
|
|
|
|
|
|
|
const countRulesForSite = db.prepare(`
|
|
|
|
|
|
|
|
SELECT COUNT(*) AS n
|
|
|
|
|
|
|
|
FROM crawler_site_rules
|
|
|
|
|
|
|
|
WHERE site_name = ?
|
|
|
|
|
|
|
|
`);
|
|
|
|
|
|
|
|
|
|
|
|
function normalizePathSegment(segment) {
|
|
|
|
function normalizePathSegment(segment) {
|
|
|
|
if (/^\d{4}$/.test(segment)) {
|
|
|
|
if (/^\d{4}$/.test(segment)) {
|
|
|
@@ -657,6 +681,55 @@ function sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, signal
|
|
|
|
return parts.filter(Boolean).join('\n').slice(0, 4200);
|
|
|
|
return parts.filter(Boolean).join('\n').slice(0, 4200);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
async function requestJevClassification(sanitizedHtml) {
|
|
|
|
|
|
|
|
const response = await fetch(JEV_URL, {
|
|
|
|
|
|
|
|
method: "POST",
|
|
|
|
|
|
|
|
headers: {
|
|
|
|
|
|
|
|
Authorization: `Bearer ${String(config.openRouter.apiKey || "").trim()}`,
|
|
|
|
|
|
|
|
"Content-Type": "application/json",
|
|
|
|
|
|
|
|
},
|
|
|
|
|
|
|
|
body: JSON.stringify({
|
|
|
|
|
|
|
|
model: JEV_MODEL,
|
|
|
|
|
|
|
|
state: sanitizedHtml,
|
|
|
|
|
|
|
|
questions: {
|
|
|
|
|
|
|
|
page_kind: {
|
|
|
|
|
|
|
|
type: "choice",
|
|
|
|
|
|
|
|
instructions: "Classify this page for a news crawler. The URL, title, meta tags, a sample of links and the first few paragraphs are all in the state.",
|
|
|
|
|
|
|
|
criteria: {
|
|
|
|
|
|
|
|
article: "A single news story page.",
|
|
|
|
|
|
|
|
listing: "Homepage, topic page, category page, archive, feature hub, or any page that is mostly links to other stories.",
|
|
|
|
|
|
|
|
other: "Anything else -- about and contact pages, video hubs, tag clouds, login walls, utility pages.",
|
|
|
|
|
|
|
|
},
|
|
|
|
|
|
|
|
},
|
|
|
|
|
|
|
|
},
|
|
|
|
|
|
|
|
}),
|
|
|
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if (!response.ok) {
|
|
|
|
|
|
|
|
const body = await response.text().catch(() => "");
|
|
|
|
|
|
|
|
const requestError = new Error(`jev classification failed with ${response.status}: ${body.slice(0, 200)}`);
|
|
|
|
|
|
|
|
requestError.status = response.status;
|
|
|
|
|
|
|
|
throw requestError;
|
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
const payload = await response.json();
|
|
|
|
|
|
|
|
const answer = payload && payload.answers && payload.answers.page_kind;
|
|
|
|
|
|
|
|
const choice = String((answer && answer.choice) || "").trim().toLowerCase();
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
// the criteria keys are the only three things it can possibly come back with,
|
|
|
|
|
|
|
|
// but a beta endpoint changing its answer shape shouldnt silently become "other"
|
|
|
|
|
|
|
|
if (choice !== "article" && choice !== "listing" && choice !== "other") {
|
|
|
|
|
|
|
|
throw new Error(`jev returned an unusable answer: ${JSON.stringify(answer).slice(0, 200)}`);
|
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
// this confidence falls out of the shape of the probability distribution rather
|
|
|
|
|
|
|
|
// than the model telling us how sure it feels, which is why gating on it works
|
|
|
|
|
|
|
|
// at all. the old prompt asked gpt for a confidence and then never read it.
|
|
|
|
|
|
|
|
const confidence = Number(answer.confidence);
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
return { classification: choice, confidence: Number.isFinite(confidence) ? confidence : 0 };
|
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
async function requestLlmClassification(url, sanitizedHtml, heuristic) {
|
|
|
|
async function requestLlmClassification(url, sanitizedHtml, heuristic) {
|
|
|
|
const response = await fetch('https://openrouter.ai/api/v1/chat/completions', {
|
|
|
|
const response = await fetch('https://openrouter.ai/api/v1/chat/completions', {
|
|
|
|
method: 'POST',
|
|
|
|
method: 'POST',
|
|
|
@@ -665,7 +738,7 @@ async function requestLlmClassification(url, sanitizedHtml, heuristic) {
|
|
|
|
'Content-Type': 'application/json',
|
|
|
|
'Content-Type': 'application/json',
|
|
|
|
},
|
|
|
|
},
|
|
|
|
body: JSON.stringify({
|
|
|
|
body: JSON.stringify({
|
|
|
|
model: 'openai/gpt-4.1-mini',
|
|
|
|
model: SIGNAL_MODEL,
|
|
|
|
messages: [
|
|
|
|
messages: [
|
|
|
|
{
|
|
|
|
{
|
|
|
|
role: 'system',
|
|
|
|
role: 'system',
|
|
|
@@ -757,7 +830,7 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
|
|
|
|
.find((entry) => patternToRegex(entry.pattern).test(pathname));
|
|
|
|
.find((entry) => patternToRegex(entry.pattern).test(pathname));
|
|
|
|
|
|
|
|
|
|
|
|
if (matchedPattern) {
|
|
|
|
if (matchedPattern) {
|
|
|
|
upsertCachedClassification.run(url, siteName, matchedPattern.classification, matchedPattern.pattern);
|
|
|
|
upsertCachedClassification.run(url, siteName, matchedPattern.classification, null, matchedPattern.pattern);
|
|
|
|
return { classification: matchedPattern.classification, source: 'pattern', learnedSignals: [], negativeSignals: [] };
|
|
|
|
return { classification: matchedPattern.classification, source: 'pattern', learnedSignals: [], negativeSignals: [] };
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
@@ -765,7 +838,7 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
|
|
|
|
const ruleSignals = buildRuleSignals(url, meta, html, jsonLdArticle, links, heuristic);
|
|
|
|
const ruleSignals = buildRuleSignals(url, meta, html, jsonLdArticle, links, heuristic);
|
|
|
|
const matchedRule = selectRulesForSite.all(siteName, minPatternHits).find((rule) => matchRule(rule, ruleSignals));
|
|
|
|
const matchedRule = selectRulesForSite.all(siteName, minPatternHits).find((rule) => matchRule(rule, ruleSignals));
|
|
|
|
if (matchedRule) {
|
|
|
|
if (matchedRule) {
|
|
|
|
upsertCachedClassification.run(url, siteName, matchedRule.classification, pattern);
|
|
|
|
upsertCachedClassification.run(url, siteName, matchedRule.classification, null, pattern);
|
|
|
|
return { classification: matchedRule.classification, source: 'rule', learnedSignals: [], negativeSignals: [] };
|
|
|
|
return { classification: matchedRule.classification, source: 'rule', learnedSignals: [], negativeSignals: [] };
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
@@ -773,13 +846,52 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
|
|
|
|
return { classification: null, source: 'disabled', learnedSignals: [], negativeSignals: [] };
|
|
|
|
return { classification: null, source: 'disabled', learnedSignals: [], negativeSignals: [] };
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
const result = await requestLlmClassification(
|
|
|
|
const sanitized = sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, ruleSignals);
|
|
|
|
url,
|
|
|
|
|
|
|
|
sanitizeForLlm(url, html, meta, jsonLdArticle, links, heuristic, ruleSignals),
|
|
|
|
|
|
|
|
heuristic,
|
|
|
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
upsertCachedClassification.run(url, siteName, result.classification, pattern);
|
|
|
|
let jev = null;
|
|
|
|
|
|
|
|
try {
|
|
|
|
|
|
|
|
jev = await requestJevClassification(sanitized);
|
|
|
|
|
|
|
|
} catch (error) {
|
|
|
|
|
|
|
|
// not fatal on its own, the expensive path below can still answer
|
|
|
|
|
|
|
|
console.error(`[crawler-jev] ${siteName} ${url} failed:`, error.message, error.stack);
|
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
const bankedRules = Number(countRulesForSite.get(siteName).n) || 0;
|
|
|
|
|
|
|
|
const stillLearning = bankedRules < SITE_RULE_TARGET;
|
|
|
|
|
|
|
|
const unsure = !jev || jev.confidence < JEV_MIN_CONFIDENCE;
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
// steady state. the site has already taught us enough rules and jev is sure, so
|
|
|
|
|
|
|
|
// there is nothing left for the expensive model to add here.
|
|
|
|
|
|
|
|
if (!stillLearning && !unsure) {
|
|
|
|
|
|
|
|
upsertCachedClassification.run(url, siteName, jev.classification, jev.confidence, pattern);
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if (pattern) {
|
|
|
|
|
|
|
|
upsertPattern.run(siteName, pattern, jev.classification);
|
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
console.log(`[crawler-jev] ${siteName} ${jev.classification.toUpperCase()} conf=${jev.confidence.toFixed(2)} ${url}`);
|
|
|
|
|
|
|
|
return { classification: jev.classification, source: 'jev', learnedSignals: [], negativeSignals: [] };
|
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
let result;
|
|
|
|
|
|
|
|
try {
|
|
|
|
|
|
|
|
result = await requestLlmClassification(url, sanitized, heuristic);
|
|
|
|
|
|
|
|
} catch (error) {
|
|
|
|
|
|
|
|
console.error(`[crawler-llm] ${siteName} ${url} failed:`, error.message, error.stack);
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
// we already paid for jev, so a dead signal call shouldnt also cost us the
|
|
|
|
|
|
|
|
// page. nothing is cached or learned off the back of it though.
|
|
|
|
|
|
|
|
if (!jev) {
|
|
|
|
|
|
|
|
throw error;
|
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
console.warn(`[crawler-llm] falling back to jev for ${url} (conf=${jev.confidence.toFixed(2)})`);
|
|
|
|
|
|
|
|
return { classification: jev.classification, source: 'jev-fallback', learnedSignals: [], negativeSignals: [] };
|
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
// null, not jev's number -- the row holds the signal model's classification and
|
|
|
|
|
|
|
|
// jev's confidence was in its own answer, which may not even be the same one.
|
|
|
|
|
|
|
|
upsertCachedClassification.run(url, siteName, result.classification, null, pattern);
|
|
|
|
|
|
|
|
|
|
|
|
if (pattern) {
|
|
|
|
if (pattern) {
|
|
|
|
upsertPattern.run(siteName, pattern, result.classification);
|
|
|
|
upsertPattern.run(siteName, pattern, result.classification);
|
|
|
@@ -789,6 +901,12 @@ async function classifyPageWithLlm({ siteName, url, html, meta, jsonLdArticle, h
|
|
|
|
upsertRule.run(siteName, signal.ruleType, signal.ruleValue, result.classification);
|
|
|
|
upsertRule.run(siteName, signal.ruleType, signal.ruleValue, result.classification);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if (jev && jev.classification !== result.classification) {
|
|
|
|
|
|
|
|
// if this stays noisy for one site the rules it is banking are probably junk,
|
|
|
|
|
|
|
|
// or the sanitized state is cutting off the part that decides it.
|
|
|
|
|
|
|
|
console.warn(`[crawler-jev] disagreed on ${url}: jev=${jev.classification} conf=${jev.confidence.toFixed(2)} llm=${result.classification}`);
|
|
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
console.log(`[crawler-llm] ${siteName} ${result.classification.toUpperCase()} ${url}`);
|
|
|
|
console.log(`[crawler-llm] ${siteName} ${result.classification.toUpperCase()} ${url}`);
|
|
|
|
return {
|
|
|
|
return {
|
|
|
|
classification: result.classification,
|
|
|
|
classification: result.classification,
|
|
|
|