add Google News integration and enhance crawler capabilities
This commit is contained in:
+147
-32
@@ -1,6 +1,12 @@
|
||||
const config = require('../config');
|
||||
const { fetchWithPolicy } = require('../http');
|
||||
const { createBrowserSession, shouldUseBrowser } = require('./browserCrawler');
|
||||
const { classifyPageWithLlm } = require('./crawlerClassifier');
|
||||
const { getRssSources } = require('./sourceCatalog');
|
||||
|
||||
function sleep(ms) {
|
||||
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
}
|
||||
|
||||
const TRACKING_PARAM_PATTERNS = [
|
||||
/^utm_/i,
|
||||
@@ -22,7 +28,7 @@ const ARTICLE_DATE_PATH = /\/\d{4}\/\d{2}\/\d{2}(?:\/|$)|\/\d{4}\/\d{2}(?:\/|$)/
|
||||
const ARTICLE_PATH_HINT = /(\/article\/|\/articles\/|\/news\/|\/story\/|\/stories\/)/i;
|
||||
const ARTICLE_PATH_STRONG_HINT = /\/\d{4}\/\d{2}\/\d{2}\//;
|
||||
const LISTING_ARTICLE_FALSE_POSITIVE_PATH = /(\/category\/|\/tag\/|\/latest(?:\/|$)|\/topics?(?:\/|$)|\/sections?(?:\/|$))/i;
|
||||
const BLOCKED_PATH_HINT = /(\/search(?:\/|$)|\/login(?:\/|$)|\/account(?:\/|$)|\/video(?:\/|$)|\/videos(?:\/|$)|\/podcast(?:\/|$)|\/podcasts(?:\/|$)|\/live(?:\/|$))/i;
|
||||
const BLOCKED_PATH_HINT = /(\/search(?:\/|$)|\/login(?:\/|$)|\/account(?:\/|$)|\/video(?:\/|$)|\/videos(?:\/|$)|\/podcast(?:\/|$)|\/podcasts(?:\/|$)|\/live(?:\/|$)|\/subscribe(?:\/|$)|\/subscription(?:\/|$)|\/newsletters?(?:\/|$)|\/privacy(?:\/|$)|\/terms(?:\/|$)|\/about(?:\/|$)|\/contact(?:\/|$))/i;
|
||||
const EXPLORATION_PATH_HINT = /(\/page\/\d+(?:\/|$)|[?&]page=\d+|\/archive(?:s)?(?:\/|$)|\/latest(?:\/|$)|\/news(?:\/|$)|\/world(?:\/|$)|\/business(?:\/|$)|\/politics(?:\/|$)|\/technology(?:\/|$)|\/tech(?:\/|$)|\/markets(?:\/|$)|\/economy(?:\/|$)|\/topic(?:s)?(?:\/|$)|\/section(?:s)?(?:\/|$)|\/category(?:ies)?(?:\/|$)|\/tag(?:s)?(?:\/|$))/i;
|
||||
|
||||
function decodeHtmlEntities(value) {
|
||||
@@ -149,9 +155,14 @@ function extractTimeDatetime(html) {
|
||||
return match ? decodeHtmlEntities(match[2]).trim() : null;
|
||||
}
|
||||
|
||||
function extractParagraphTextLength(html) {
|
||||
function extractParagraphStats(html) {
|
||||
const paragraphs = html.match(/<p\b[^>]*>[\s\S]*?<\/p>/gi) || [];
|
||||
return paragraphs.slice(0, 10).reduce((total, paragraph) => total + normalizeText(paragraph).length, 0);
|
||||
const normalizedParagraphs = paragraphs.slice(0, 12).map((paragraph) => normalizeText(paragraph)).filter(Boolean);
|
||||
|
||||
return {
|
||||
textLength: normalizedParagraphs.reduce((total, paragraph) => total + paragraph.length, 0),
|
||||
substantialCount: normalizedParagraphs.filter((paragraph) => paragraph.length >= 80).length,
|
||||
};
|
||||
}
|
||||
|
||||
function extractJsonLdBlocks(html) {
|
||||
@@ -266,31 +277,43 @@ function scorePage(pageUrl, meta, html, jsonLdArticle, links) {
|
||||
const hasArticlePathHint = ARTICLE_PATH_HINT.test(pageUrl);
|
||||
const hasStrongArticlePath = ARTICLE_PATH_STRONG_HINT.test(pathname);
|
||||
const hasListingFalsePositivePath = LISTING_ARTICLE_FALSE_POSITIVE_PATH.test(pathname);
|
||||
const paragraphTextLength = extractParagraphTextLength(html);
|
||||
const { textLength: paragraphTextLength, substantialCount: substantialParagraphCount } = extractParagraphStats(html);
|
||||
const headlineLinks = links.filter(({ text }) => text.length >= 25 && text.length <= 180).length;
|
||||
const h1 = extractH1(html);
|
||||
const titleTag = extractTitleTag(html);
|
||||
const ogTitle = normalizeText(meta.get('og:title') || '');
|
||||
const twitterTitle = normalizeText(meta.get('twitter:title') || '');
|
||||
const titleSignalsLower = [h1, titleTag, ogTitle, twitterTitle].filter(Boolean).map((value) => value.toLowerCase());
|
||||
const looksLikeSectionTitle = titleSignalsLower.some((value) => /^(category:|section:)/.test(value)
|
||||
|| /(market[s]?|business|technology|tech|science|health|gaming|culture|news|world|economy)/.test(value) && value.length <= 80);
|
||||
const looksLikeCommercialPage = titleSignalsLower.some((value) => /(subscribe|subscription|sign in|log in|newsletter|advertis)/.test(value));
|
||||
const jsonLdHeadline = normalizeText(jsonLdArticle && jsonLdArticle.headline);
|
||||
const jsonLdMatchesPage = jsonLdHeadline
|
||||
&& ((h1 && (h1.includes(jsonLdHeadline) || jsonLdHeadline.includes(h1)))
|
||||
|| (ogTitle && (ogTitle.includes(jsonLdHeadline) || jsonLdHeadline.includes(ogTitle))));
|
||||
const hasJsonLdArticle = Boolean(jsonLdArticle && jsonLdMatchesPage);
|
||||
const titleSignals = [h1, ogTitle, twitterTitle, titleTag].filter(Boolean);
|
||||
const matchingTitleSignals = titleSignals.filter((value) => jsonLdHeadline && (value.includes(jsonLdHeadline) || jsonLdHeadline.includes(value)));
|
||||
const hasJsonLdArticle = Boolean(jsonLdArticle && matchingTitleSignals.length > 0);
|
||||
const hasPublishTime = Boolean(meta.get('article:published_time') || meta.get('og:article:published_time') || extractTimeDatetime(html));
|
||||
const hasOgArticle = String(meta.get('og:type') || '').toLowerCase() === 'article';
|
||||
const hasArticleTag = /<article\b/i.test(html);
|
||||
const hasLongBody = paragraphTextLength >= 600 && substantialParagraphCount >= 3;
|
||||
const hasArticleStructure = Boolean(h1) && (hasArticleTag || hasLongBody);
|
||||
const hasMetadataArticleSignal = hasOgArticle || hasPublishTime || hasJsonLdArticle;
|
||||
const hasUrlArticleSignal = hasArticleDatePath || hasStrongArticlePath || hasArticlePathHint;
|
||||
const hasPrimaryArticleSignal = hasMetadataArticleSignal || hasUrlArticleSignal;
|
||||
const hasSecondaryArticleSignal = hasArticlePathHint || hasArticleStructure;
|
||||
|
||||
if (hasJsonLdArticle) {
|
||||
articleScore += 4;
|
||||
articleScore += 3;
|
||||
}
|
||||
|
||||
if (hasOgArticle && !hasListingFalsePositivePath) {
|
||||
articleScore += 1;
|
||||
articleScore += 2;
|
||||
}
|
||||
|
||||
if (hasPublishTime && !hasListingFalsePositivePath) {
|
||||
articleScore += 1;
|
||||
articleScore += 2;
|
||||
}
|
||||
|
||||
if (/<article\b/i.test(html)) {
|
||||
if (hasArticleTag) {
|
||||
articleScore += 1;
|
||||
}
|
||||
|
||||
@@ -298,7 +321,7 @@ function scorePage(pageUrl, meta, html, jsonLdArticle, links) {
|
||||
articleScore += 2;
|
||||
}
|
||||
|
||||
if (h1 && paragraphTextLength >= 500) {
|
||||
if (hasLongBody) {
|
||||
articleScore += 2;
|
||||
}
|
||||
|
||||
@@ -318,19 +341,41 @@ function scorePage(pageUrl, meta, html, jsonLdArticle, links) {
|
||||
listingScore += 3;
|
||||
}
|
||||
|
||||
if (articleScore > 0) {
|
||||
listingScore -= 1;
|
||||
if (looksLikeSectionTitle) {
|
||||
listingScore += 3;
|
||||
}
|
||||
|
||||
const hasArticleSignalsBeyondJsonLd = hasOgArticle || hasPublishTime || hasStrongArticlePath || hasArticlePathHint || paragraphTextLength >= 500;
|
||||
const looksLikeListingPage = headlineLinks >= 15;
|
||||
const isArticleCandidate = !looksLikeListingPage
|
||||
&& articleScore >= 5
|
||||
&& articleScore > listingScore
|
||||
&& hasArticleSignalsBeyondJsonLd
|
||||
&& (!jsonLdArticle || hasJsonLdArticle || hasStrongArticlePath || hasArticlePathHint || paragraphTextLength >= 500);
|
||||
if (looksLikeCommercialPage) {
|
||||
listingScore += 4;
|
||||
}
|
||||
|
||||
return { articleScore, listingScore, isArticleCandidate };
|
||||
if ((pathname === '/' || LISTING_PATH_HINT.test(pathname)) && headlineLinks >= 8 && links.length >= 20) {
|
||||
listingScore += 4;
|
||||
}
|
||||
|
||||
if (headlineLinks >= 15 && !hasPrimaryArticleSignal && !hasLongBody) {
|
||||
listingScore += 3;
|
||||
}
|
||||
|
||||
if (!hasPrimaryArticleSignal && headlineLinks >= 12 && links.length >= 25) {
|
||||
listingScore += 2;
|
||||
}
|
||||
|
||||
const shouldAskLlm = !looksLikeCommercialPage && (
|
||||
articleScore >= 2
|
||||
|| hasPrimaryArticleSignal
|
||||
|| hasSecondaryArticleSignal
|
||||
|| (headlineLinks <= 10 && links.length <= 40)
|
||||
);
|
||||
|
||||
return {
|
||||
articleScore,
|
||||
listingScore,
|
||||
shouldAskLlm,
|
||||
headlineLinks,
|
||||
paragraphCount: substantialParagraphCount,
|
||||
paragraphTextLength,
|
||||
};
|
||||
}
|
||||
|
||||
function shouldQueueLink(url) {
|
||||
@@ -458,16 +503,23 @@ function normalizeSite(site) {
|
||||
const seeds = unique((site.seeds || [])
|
||||
.map((seed) => canonicalizeUrl(seed, seed, allowedHosts))
|
||||
.filter(Boolean));
|
||||
const renderMode = String(site.renderMode || 'http').trim().toLowerCase() === 'browser' ? 'browser' : 'http';
|
||||
const maxPages = normalizeLimit(site.maxPages, 15, 1, 500);
|
||||
|
||||
return {
|
||||
name: String(site.name || '').trim(),
|
||||
label: String(site.label || '').trim(),
|
||||
allowedHosts,
|
||||
seeds,
|
||||
renderMode: String(site.renderMode || 'http').trim().toLowerCase() === 'browser' ? 'browser' : 'http',
|
||||
maxPages: normalizeLimit(site.maxPages, 15, 1, 500),
|
||||
renderMode,
|
||||
maxPages,
|
||||
maxDepth: normalizeLimit(site.maxDepth, 1, 0, 5),
|
||||
pageConcurrency: normalizeLimit(site.pageConcurrency, String(site.renderMode || 'http').trim().toLowerCase() === 'browser' ? 3 : 4, 1, 12),
|
||||
pageConcurrency: normalizeLimit(site.pageConcurrency, renderMode === 'browser' ? 2 : 4, 1, renderMode === 'browser' ? 4 : 12),
|
||||
maxQueuedPages: normalizeLimit(site.maxQueuedPages, Math.min(maxPages * 3, 1000), maxPages, 2000),
|
||||
memorySoftLimitMb: normalizeLimit(site.memorySoftLimitMb, 800, 128, 8192),
|
||||
memoryHardLimitMb: normalizeLimit(site.memoryHardLimitMb, 1400, 256, 16384),
|
||||
memoryThrottleDelayMs: normalizeLimit(site.memoryThrottleDelayMs, 1500, 100, 10000),
|
||||
llmPatternMinHits: normalizeLimit(site.llmPatternMinHits, 3, 1, 20),
|
||||
requestTimeout: Math.max(1000, Math.min(Number(site.requestTimeout) || 15000, 30000)),
|
||||
};
|
||||
}
|
||||
@@ -483,7 +535,7 @@ function getConfiguredCrawlerSites() {
|
||||
const explicitLabels = new Set(explicitSites.map((site) => site.label).filter(Boolean));
|
||||
const derivedSites = [];
|
||||
|
||||
for (const feed of config.rssFeeds || []) {
|
||||
for (const feed of getRssSources()) {
|
||||
const label = String(feed.label || '').trim();
|
||||
if (!label || disabledLabels.has(label) || explicitLabels.has(label)) {
|
||||
continue;
|
||||
@@ -491,7 +543,7 @@ function getConfiguredCrawlerSites() {
|
||||
|
||||
let hostname = '';
|
||||
try {
|
||||
hostname = new URL(feed.url).hostname;
|
||||
hostname = new URL(feed.feedUrl).hostname;
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
@@ -501,11 +553,16 @@ function getConfiguredCrawlerSites() {
|
||||
label,
|
||||
name: override.name || `crawler_${slugifyLabel(label)}`,
|
||||
allowedHosts: override.allowedHosts || buildAllowedHosts(hostname),
|
||||
seeds: override.seeds || buildDefaultSeeds(feed.url),
|
||||
seeds: override.seeds || buildDefaultSeeds(feed.feedUrl),
|
||||
renderMode: override.renderMode || defaults.renderMode,
|
||||
maxPages: override.maxPages || defaults.maxPages,
|
||||
maxDepth: override.maxDepth || defaults.maxDepth,
|
||||
pageConcurrency: override.pageConcurrency,
|
||||
maxQueuedPages: override.maxQueuedPages,
|
||||
memorySoftLimitMb: override.memorySoftLimitMb || defaults.memorySoftLimitMb,
|
||||
memoryHardLimitMb: override.memoryHardLimitMb || defaults.memoryHardLimitMb,
|
||||
memoryThrottleDelayMs: override.memoryThrottleDelayMs || defaults.memoryThrottleDelayMs,
|
||||
llmPatternMinHits: override.llmPatternMinHits || defaults.llmPatternMinHits,
|
||||
requestTimeout: override.requestTimeout || defaults.requestTimeout,
|
||||
});
|
||||
|
||||
@@ -555,6 +612,28 @@ async function crawlSite(site) {
|
||||
const discoveredArticleUrls = new Set();
|
||||
const articles = [];
|
||||
|
||||
function getResidentSetMb() {
|
||||
return Math.round(process.memoryUsage().rss / (1024 * 1024));
|
||||
}
|
||||
|
||||
async function throttleForMemory() {
|
||||
let residentSetMb = getResidentSetMb();
|
||||
|
||||
while (residentSetMb >= normalizedSite.memorySoftLimitMb) {
|
||||
if (residentSetMb >= normalizedSite.memoryHardLimitMb) {
|
||||
console.error(`Crawler memory hard limit reached for ${normalizedSite.name}: ${residentSetMb}MB >= ${normalizedSite.memoryHardLimitMb}MB`);
|
||||
queue.length = 0;
|
||||
return false;
|
||||
}
|
||||
|
||||
console.warn(`Crawler memory throttle for ${normalizedSite.name}: ${residentSetMb}MB >= ${normalizedSite.memorySoftLimitMb}MB`);
|
||||
await sleep(normalizedSite.memoryThrottleDelayMs);
|
||||
residentSetMb = getResidentSetMb();
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
async function processPage(current) {
|
||||
let html;
|
||||
try {
|
||||
@@ -575,7 +654,35 @@ async function crawlSite(site) {
|
||||
? canonicalizeUrl(canonicalHref, current.url, normalizedSite.allowedHosts) || current.url
|
||||
: current.url;
|
||||
const links = extractLinks(html, canonicalUrl, normalizedSite.allowedHosts);
|
||||
const { listingScore, isArticleCandidate } = scorePage(canonicalUrl, meta, html, jsonLdArticle, links);
|
||||
const heuristic = scorePage(canonicalUrl, meta, html, jsonLdArticle, links);
|
||||
let isArticleCandidate = false;
|
||||
let effectiveListingScore = heuristic.listingScore;
|
||||
|
||||
if (heuristic.shouldAskLlm) {
|
||||
try {
|
||||
const llmDecision = await classifyPageWithLlm({
|
||||
siteName: normalizedSite.name,
|
||||
url: canonicalUrl,
|
||||
html,
|
||||
meta,
|
||||
jsonLdArticle,
|
||||
heuristic,
|
||||
links,
|
||||
minPatternHits: normalizedSite.llmPatternMinHits,
|
||||
});
|
||||
|
||||
if (llmDecision.classification === 'article') {
|
||||
isArticleCandidate = true;
|
||||
effectiveListingScore = Math.max(0, heuristic.listingScore - 2);
|
||||
} else if (llmDecision.classification === 'listing') {
|
||||
effectiveListingScore = Math.max(heuristic.listingScore, 3);
|
||||
} else if (llmDecision.classification === 'other') {
|
||||
effectiveListingScore = Math.max(0, heuristic.listingScore - 1);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error(`Crawler LLM classification failed for ${normalizedSite.name}: ${canonicalUrl}`, error);
|
||||
}
|
||||
}
|
||||
|
||||
if (isArticleCandidate && !discoveredArticleUrls.has(canonicalUrl)) {
|
||||
const title = normalizeText(selectTitle(meta, jsonLdArticle, html));
|
||||
@@ -587,16 +694,20 @@ async function crawlSite(site) {
|
||||
url: canonicalUrl,
|
||||
source: normalizedSite.name,
|
||||
pubDate: selectPubDate(meta, jsonLdArticle, html),
|
||||
isIndexPage: !isArticleCandidate,
|
||||
isIndexPage: false,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (current.depth >= normalizedSite.maxDepth || !shouldContinueExploring(current, listingScore, links)) {
|
||||
if (current.depth >= normalizedSite.maxDepth || !shouldContinueExploring(current, effectiveListingScore, links)) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (const link of links) {
|
||||
if (queuedUrls.size >= normalizedSite.maxQueuedPages) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (!shouldQueueLink(link.url) || visitedUrls.has(link.url) || queuedUrls.has(link.url)) {
|
||||
continue;
|
||||
}
|
||||
@@ -608,6 +719,10 @@ async function crawlSite(site) {
|
||||
|
||||
try {
|
||||
while (queue.length && visitedUrls.size < normalizedSite.maxPages) {
|
||||
if (!await throttleForMemory()) {
|
||||
break;
|
||||
}
|
||||
|
||||
const batch = [];
|
||||
|
||||
while (queue.length && batch.length < normalizedSite.pageConcurrency && visitedUrls.size + batch.length < normalizedSite.maxPages) {
|
||||
|
||||
Reference in New Issue
Block a user