add Google News integration and enhance crawler capabilities

This commit is contained in:
ImBenji
2026-04-18 06:35:12 +01:00
parent 1a8504389a
commit c3f9e59c5e
16 changed files with 3020 additions and 904 deletions
+147 -32
View File
@@ -1,6 +1,12 @@
const config = require('../config');
const { fetchWithPolicy } = require('../http');
const { createBrowserSession, shouldUseBrowser } = require('./browserCrawler');
const { classifyPageWithLlm } = require('./crawlerClassifier');
const { getRssSources } = require('./sourceCatalog');
function sleep(ms) {
return new Promise((resolve) => setTimeout(resolve, ms));
}
const TRACKING_PARAM_PATTERNS = [
/^utm_/i,
@@ -22,7 +28,7 @@ const ARTICLE_DATE_PATH = /\/\d{4}\/\d{2}\/\d{2}(?:\/|$)|\/\d{4}\/\d{2}(?:\/|$)/
const ARTICLE_PATH_HINT = /(\/article\/|\/articles\/|\/news\/|\/story\/|\/stories\/)/i;
const ARTICLE_PATH_STRONG_HINT = /\/\d{4}\/\d{2}\/\d{2}\//;
const LISTING_ARTICLE_FALSE_POSITIVE_PATH = /(\/category\/|\/tag\/|\/latest(?:\/|$)|\/topics?(?:\/|$)|\/sections?(?:\/|$))/i;
const BLOCKED_PATH_HINT = /(\/search(?:\/|$)|\/login(?:\/|$)|\/account(?:\/|$)|\/video(?:\/|$)|\/videos(?:\/|$)|\/podcast(?:\/|$)|\/podcasts(?:\/|$)|\/live(?:\/|$))/i;
const BLOCKED_PATH_HINT = /(\/search(?:\/|$)|\/login(?:\/|$)|\/account(?:\/|$)|\/video(?:\/|$)|\/videos(?:\/|$)|\/podcast(?:\/|$)|\/podcasts(?:\/|$)|\/live(?:\/|$)|\/subscribe(?:\/|$)|\/subscription(?:\/|$)|\/newsletters?(?:\/|$)|\/privacy(?:\/|$)|\/terms(?:\/|$)|\/about(?:\/|$)|\/contact(?:\/|$))/i;
const EXPLORATION_PATH_HINT = /(\/page\/\d+(?:\/|$)|[?&]page=\d+|\/archive(?:s)?(?:\/|$)|\/latest(?:\/|$)|\/news(?:\/|$)|\/world(?:\/|$)|\/business(?:\/|$)|\/politics(?:\/|$)|\/technology(?:\/|$)|\/tech(?:\/|$)|\/markets(?:\/|$)|\/economy(?:\/|$)|\/topic(?:s)?(?:\/|$)|\/section(?:s)?(?:\/|$)|\/category(?:ies)?(?:\/|$)|\/tag(?:s)?(?:\/|$))/i;
function decodeHtmlEntities(value) {
@@ -149,9 +155,14 @@ function extractTimeDatetime(html) {
return match ? decodeHtmlEntities(match[2]).trim() : null;
}
function extractParagraphTextLength(html) {
function extractParagraphStats(html) {
const paragraphs = html.match(/<p\b[^>]*>[\s\S]*?<\/p>/gi) || [];
return paragraphs.slice(0, 10).reduce((total, paragraph) => total + normalizeText(paragraph).length, 0);
const normalizedParagraphs = paragraphs.slice(0, 12).map((paragraph) => normalizeText(paragraph)).filter(Boolean);
return {
textLength: normalizedParagraphs.reduce((total, paragraph) => total + paragraph.length, 0),
substantialCount: normalizedParagraphs.filter((paragraph) => paragraph.length >= 80).length,
};
}
function extractJsonLdBlocks(html) {
@@ -266,31 +277,43 @@ function scorePage(pageUrl, meta, html, jsonLdArticle, links) {
const hasArticlePathHint = ARTICLE_PATH_HINT.test(pageUrl);
const hasStrongArticlePath = ARTICLE_PATH_STRONG_HINT.test(pathname);
const hasListingFalsePositivePath = LISTING_ARTICLE_FALSE_POSITIVE_PATH.test(pathname);
const paragraphTextLength = extractParagraphTextLength(html);
const { textLength: paragraphTextLength, substantialCount: substantialParagraphCount } = extractParagraphStats(html);
const headlineLinks = links.filter(({ text }) => text.length >= 25 && text.length <= 180).length;
const h1 = extractH1(html);
const titleTag = extractTitleTag(html);
const ogTitle = normalizeText(meta.get('og:title') || '');
const twitterTitle = normalizeText(meta.get('twitter:title') || '');
const titleSignalsLower = [h1, titleTag, ogTitle, twitterTitle].filter(Boolean).map((value) => value.toLowerCase());
const looksLikeSectionTitle = titleSignalsLower.some((value) => /^(category:|section:)/.test(value)
|| /(market[s]?|business|technology|tech|science|health|gaming|culture|news|world|economy)/.test(value) && value.length <= 80);
const looksLikeCommercialPage = titleSignalsLower.some((value) => /(subscribe|subscription|sign in|log in|newsletter|advertis)/.test(value));
const jsonLdHeadline = normalizeText(jsonLdArticle && jsonLdArticle.headline);
const jsonLdMatchesPage = jsonLdHeadline
&& ((h1 && (h1.includes(jsonLdHeadline) || jsonLdHeadline.includes(h1)))
|| (ogTitle && (ogTitle.includes(jsonLdHeadline) || jsonLdHeadline.includes(ogTitle))));
const hasJsonLdArticle = Boolean(jsonLdArticle && jsonLdMatchesPage);
const titleSignals = [h1, ogTitle, twitterTitle, titleTag].filter(Boolean);
const matchingTitleSignals = titleSignals.filter((value) => jsonLdHeadline && (value.includes(jsonLdHeadline) || jsonLdHeadline.includes(value)));
const hasJsonLdArticle = Boolean(jsonLdArticle && matchingTitleSignals.length > 0);
const hasPublishTime = Boolean(meta.get('article:published_time') || meta.get('og:article:published_time') || extractTimeDatetime(html));
const hasOgArticle = String(meta.get('og:type') || '').toLowerCase() === 'article';
const hasArticleTag = /<article\b/i.test(html);
const hasLongBody = paragraphTextLength >= 600 && substantialParagraphCount >= 3;
const hasArticleStructure = Boolean(h1) && (hasArticleTag || hasLongBody);
const hasMetadataArticleSignal = hasOgArticle || hasPublishTime || hasJsonLdArticle;
const hasUrlArticleSignal = hasArticleDatePath || hasStrongArticlePath || hasArticlePathHint;
const hasPrimaryArticleSignal = hasMetadataArticleSignal || hasUrlArticleSignal;
const hasSecondaryArticleSignal = hasArticlePathHint || hasArticleStructure;
if (hasJsonLdArticle) {
articleScore += 4;
articleScore += 3;
}
if (hasOgArticle && !hasListingFalsePositivePath) {
articleScore += 1;
articleScore += 2;
}
if (hasPublishTime && !hasListingFalsePositivePath) {
articleScore += 1;
articleScore += 2;
}
if (/<article\b/i.test(html)) {
if (hasArticleTag) {
articleScore += 1;
}
@@ -298,7 +321,7 @@ function scorePage(pageUrl, meta, html, jsonLdArticle, links) {
articleScore += 2;
}
if (h1 && paragraphTextLength >= 500) {
if (hasLongBody) {
articleScore += 2;
}
@@ -318,19 +341,41 @@ function scorePage(pageUrl, meta, html, jsonLdArticle, links) {
listingScore += 3;
}
if (articleScore > 0) {
listingScore -= 1;
if (looksLikeSectionTitle) {
listingScore += 3;
}
const hasArticleSignalsBeyondJsonLd = hasOgArticle || hasPublishTime || hasStrongArticlePath || hasArticlePathHint || paragraphTextLength >= 500;
const looksLikeListingPage = headlineLinks >= 15;
const isArticleCandidate = !looksLikeListingPage
&& articleScore >= 5
&& articleScore > listingScore
&& hasArticleSignalsBeyondJsonLd
&& (!jsonLdArticle || hasJsonLdArticle || hasStrongArticlePath || hasArticlePathHint || paragraphTextLength >= 500);
if (looksLikeCommercialPage) {
listingScore += 4;
}
return { articleScore, listingScore, isArticleCandidate };
if ((pathname === '/' || LISTING_PATH_HINT.test(pathname)) && headlineLinks >= 8 && links.length >= 20) {
listingScore += 4;
}
if (headlineLinks >= 15 && !hasPrimaryArticleSignal && !hasLongBody) {
listingScore += 3;
}
if (!hasPrimaryArticleSignal && headlineLinks >= 12 && links.length >= 25) {
listingScore += 2;
}
const shouldAskLlm = !looksLikeCommercialPage && (
articleScore >= 2
|| hasPrimaryArticleSignal
|| hasSecondaryArticleSignal
|| (headlineLinks <= 10 && links.length <= 40)
);
return {
articleScore,
listingScore,
shouldAskLlm,
headlineLinks,
paragraphCount: substantialParagraphCount,
paragraphTextLength,
};
}
function shouldQueueLink(url) {
@@ -458,16 +503,23 @@ function normalizeSite(site) {
const seeds = unique((site.seeds || [])
.map((seed) => canonicalizeUrl(seed, seed, allowedHosts))
.filter(Boolean));
const renderMode = String(site.renderMode || 'http').trim().toLowerCase() === 'browser' ? 'browser' : 'http';
const maxPages = normalizeLimit(site.maxPages, 15, 1, 500);
return {
name: String(site.name || '').trim(),
label: String(site.label || '').trim(),
allowedHosts,
seeds,
renderMode: String(site.renderMode || 'http').trim().toLowerCase() === 'browser' ? 'browser' : 'http',
maxPages: normalizeLimit(site.maxPages, 15, 1, 500),
renderMode,
maxPages,
maxDepth: normalizeLimit(site.maxDepth, 1, 0, 5),
pageConcurrency: normalizeLimit(site.pageConcurrency, String(site.renderMode || 'http').trim().toLowerCase() === 'browser' ? 3 : 4, 1, 12),
pageConcurrency: normalizeLimit(site.pageConcurrency, renderMode === 'browser' ? 2 : 4, 1, renderMode === 'browser' ? 4 : 12),
maxQueuedPages: normalizeLimit(site.maxQueuedPages, Math.min(maxPages * 3, 1000), maxPages, 2000),
memorySoftLimitMb: normalizeLimit(site.memorySoftLimitMb, 800, 128, 8192),
memoryHardLimitMb: normalizeLimit(site.memoryHardLimitMb, 1400, 256, 16384),
memoryThrottleDelayMs: normalizeLimit(site.memoryThrottleDelayMs, 1500, 100, 10000),
llmPatternMinHits: normalizeLimit(site.llmPatternMinHits, 3, 1, 20),
requestTimeout: Math.max(1000, Math.min(Number(site.requestTimeout) || 15000, 30000)),
};
}
@@ -483,7 +535,7 @@ function getConfiguredCrawlerSites() {
const explicitLabels = new Set(explicitSites.map((site) => site.label).filter(Boolean));
const derivedSites = [];
for (const feed of config.rssFeeds || []) {
for (const feed of getRssSources()) {
const label = String(feed.label || '').trim();
if (!label || disabledLabels.has(label) || explicitLabels.has(label)) {
continue;
@@ -491,7 +543,7 @@ function getConfiguredCrawlerSites() {
let hostname = '';
try {
hostname = new URL(feed.url).hostname;
hostname = new URL(feed.feedUrl).hostname;
} catch {
continue;
}
@@ -501,11 +553,16 @@ function getConfiguredCrawlerSites() {
label,
name: override.name || `crawler_${slugifyLabel(label)}`,
allowedHosts: override.allowedHosts || buildAllowedHosts(hostname),
seeds: override.seeds || buildDefaultSeeds(feed.url),
seeds: override.seeds || buildDefaultSeeds(feed.feedUrl),
renderMode: override.renderMode || defaults.renderMode,
maxPages: override.maxPages || defaults.maxPages,
maxDepth: override.maxDepth || defaults.maxDepth,
pageConcurrency: override.pageConcurrency,
maxQueuedPages: override.maxQueuedPages,
memorySoftLimitMb: override.memorySoftLimitMb || defaults.memorySoftLimitMb,
memoryHardLimitMb: override.memoryHardLimitMb || defaults.memoryHardLimitMb,
memoryThrottleDelayMs: override.memoryThrottleDelayMs || defaults.memoryThrottleDelayMs,
llmPatternMinHits: override.llmPatternMinHits || defaults.llmPatternMinHits,
requestTimeout: override.requestTimeout || defaults.requestTimeout,
});
@@ -555,6 +612,28 @@ async function crawlSite(site) {
const discoveredArticleUrls = new Set();
const articles = [];
function getResidentSetMb() {
return Math.round(process.memoryUsage().rss / (1024 * 1024));
}
async function throttleForMemory() {
let residentSetMb = getResidentSetMb();
while (residentSetMb >= normalizedSite.memorySoftLimitMb) {
if (residentSetMb >= normalizedSite.memoryHardLimitMb) {
console.error(`Crawler memory hard limit reached for ${normalizedSite.name}: ${residentSetMb}MB >= ${normalizedSite.memoryHardLimitMb}MB`);
queue.length = 0;
return false;
}
console.warn(`Crawler memory throttle for ${normalizedSite.name}: ${residentSetMb}MB >= ${normalizedSite.memorySoftLimitMb}MB`);
await sleep(normalizedSite.memoryThrottleDelayMs);
residentSetMb = getResidentSetMb();
}
return true;
}
async function processPage(current) {
let html;
try {
@@ -575,7 +654,35 @@ async function crawlSite(site) {
? canonicalizeUrl(canonicalHref, current.url, normalizedSite.allowedHosts) || current.url
: current.url;
const links = extractLinks(html, canonicalUrl, normalizedSite.allowedHosts);
const { listingScore, isArticleCandidate } = scorePage(canonicalUrl, meta, html, jsonLdArticle, links);
const heuristic = scorePage(canonicalUrl, meta, html, jsonLdArticle, links);
let isArticleCandidate = false;
let effectiveListingScore = heuristic.listingScore;
if (heuristic.shouldAskLlm) {
try {
const llmDecision = await classifyPageWithLlm({
siteName: normalizedSite.name,
url: canonicalUrl,
html,
meta,
jsonLdArticle,
heuristic,
links,
minPatternHits: normalizedSite.llmPatternMinHits,
});
if (llmDecision.classification === 'article') {
isArticleCandidate = true;
effectiveListingScore = Math.max(0, heuristic.listingScore - 2);
} else if (llmDecision.classification === 'listing') {
effectiveListingScore = Math.max(heuristic.listingScore, 3);
} else if (llmDecision.classification === 'other') {
effectiveListingScore = Math.max(0, heuristic.listingScore - 1);
}
} catch (error) {
console.error(`Crawler LLM classification failed for ${normalizedSite.name}: ${canonicalUrl}`, error);
}
}
if (isArticleCandidate && !discoveredArticleUrls.has(canonicalUrl)) {
const title = normalizeText(selectTitle(meta, jsonLdArticle, html));
@@ -587,16 +694,20 @@ async function crawlSite(site) {
url: canonicalUrl,
source: normalizedSite.name,
pubDate: selectPubDate(meta, jsonLdArticle, html),
isIndexPage: !isArticleCandidate,
isIndexPage: false,
});
}
}
if (current.depth >= normalizedSite.maxDepth || !shouldContinueExploring(current, listingScore, links)) {
if (current.depth >= normalizedSite.maxDepth || !shouldContinueExploring(current, effectiveListingScore, links)) {
return;
}
for (const link of links) {
if (queuedUrls.size >= normalizedSite.maxQueuedPages) {
break;
}
if (!shouldQueueLink(link.url) || visitedUrls.has(link.url) || queuedUrls.has(link.url)) {
continue;
}
@@ -608,6 +719,10 @@ async function crawlSite(site) {
try {
while (queue.length && visitedUrls.size < normalizedSite.maxPages) {
if (!await throttleForMemory()) {
break;
}
const batch = [];
while (queue.length && batch.length < normalizedSite.pageConcurrency && visitedUrls.size + batch.length < normalizedSite.maxPages) {