add Google News integration and enhance crawler capabilities

This commit is contained in:
ImBenji
2026-04-18 06:35:12 +01:00
parent 1a8504389a
commit c3f9e59c5e
16 changed files with 3020 additions and 904 deletions
+5 -44
View File
@@ -1,8 +1,9 @@
const { extract } = require('@extractus/article-extractor');
const { extractFromHtml } = require('@extractus/article-extractor');
const sharp = require('sharp');
const db = require('./db');
const { generateAndStoreEmbedding } = require('./embeddings');
const { fetchWithPolicy } = require('./http');
const { getSharedBrowserSession } = require('./sources/browserCrawler');
const updateArticleAssets = db.prepare(`
UPDATE articles
@@ -40,32 +41,7 @@ const selectArticlesMissingContent = db.prepare(`
LIMIT ?
`);
const blockedContentDomains = [
'axios.com',
'bizjournals.com',
'fastcompany.com',
'gurufocus.com',
'investing.com',
'rbc.ru',
'stocktitan.net',
];
const loggedBlockedDomains = new Set();
const articleFetchHeaders = {
'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36',
Accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8',
'Accept-Language': 'en-US,en;q=0.9',
'Cache-Control': 'no-cache',
Pragma: 'no-cache',
'Upgrade-Insecure-Requests': '1',
'sec-ch-ua': '"Google Chrome";v="135", "Chromium";v="135", "Not.A/Brand";v="24"',
'sec-ch-ua-mobile': '?0',
'sec-ch-ua-platform': '"macOS"',
'Sec-Fetch-Dest': 'document',
'Sec-Fetch-Mode': 'navigate',
'Sec-Fetch-Site': 'none',
'Sec-Fetch-User': '?1',
};
let contentBackfillRunning = false;
function getHostname(url) {
@@ -76,10 +52,6 @@ function getHostname(url) {
}
}
function isBlockedContentUrl(url) {
const hostname = getHostname(url);
return blockedContentDomains.some((domain) => hostname === domain || hostname.endsWith(`.${domain}`));
}
function getErrorStatus(error) {
if (error && Number.isInteger(error.status)) {
@@ -147,20 +119,9 @@ async function fetchCompressedImage(url) {
async function fetchAndStoreContent(id, url) {
try {
if (isBlockedContentUrl(url)) {
const hostname = getHostname(url);
if (hostname && !loggedBlockedDomains.has(hostname)) {
loggedBlockedDomains.add(hostname);
console.warn(`content extraction skipped for blocked domain ${hostname}`);
}
markArticleStatus(markContentSkipped, id, `blocked domain: ${hostname || 'unknown'}`);
return;
}
const article = await extract(url, {}, {
headers: articleFetchHeaders,
signal: AbortSignal.timeout(20000),
});
const browserSession = await getSharedBrowserSession({ requestTimeout: 20000, maxConcurrentPages: 2 });
const html = await browserSession.fetchRenderedHtml(url, { timeout: 20000 });
const article = await extractFromHtml(html, url);
if (!article) {
markArticleStatus(markContentSkipped, id, 'extractor returned no article');
return;