add Google News integration and enhance crawler capabilities
This commit is contained in:
+5
-44
@@ -1,8 +1,9 @@
|
||||
const { extract } = require('@extractus/article-extractor');
|
||||
const { extractFromHtml } = require('@extractus/article-extractor');
|
||||
const sharp = require('sharp');
|
||||
const db = require('./db');
|
||||
const { generateAndStoreEmbedding } = require('./embeddings');
|
||||
const { fetchWithPolicy } = require('./http');
|
||||
const { getSharedBrowserSession } = require('./sources/browserCrawler');
|
||||
|
||||
const updateArticleAssets = db.prepare(`
|
||||
UPDATE articles
|
||||
@@ -40,32 +41,7 @@ const selectArticlesMissingContent = db.prepare(`
|
||||
LIMIT ?
|
||||
`);
|
||||
|
||||
const blockedContentDomains = [
|
||||
'axios.com',
|
||||
'bizjournals.com',
|
||||
'fastcompany.com',
|
||||
'gurufocus.com',
|
||||
'investing.com',
|
||||
'rbc.ru',
|
||||
'stocktitan.net',
|
||||
];
|
||||
const loggedBlockedDomains = new Set();
|
||||
const articleFetchHeaders = {
|
||||
'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36',
|
||||
Accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8',
|
||||
'Accept-Language': 'en-US,en;q=0.9',
|
||||
'Cache-Control': 'no-cache',
|
||||
Pragma: 'no-cache',
|
||||
'Upgrade-Insecure-Requests': '1',
|
||||
'sec-ch-ua': '"Google Chrome";v="135", "Chromium";v="135", "Not.A/Brand";v="24"',
|
||||
'sec-ch-ua-mobile': '?0',
|
||||
'sec-ch-ua-platform': '"macOS"',
|
||||
'Sec-Fetch-Dest': 'document',
|
||||
'Sec-Fetch-Mode': 'navigate',
|
||||
'Sec-Fetch-Site': 'none',
|
||||
'Sec-Fetch-User': '?1',
|
||||
};
|
||||
|
||||
let contentBackfillRunning = false;
|
||||
|
||||
function getHostname(url) {
|
||||
@@ -76,10 +52,6 @@ function getHostname(url) {
|
||||
}
|
||||
}
|
||||
|
||||
function isBlockedContentUrl(url) {
|
||||
const hostname = getHostname(url);
|
||||
return blockedContentDomains.some((domain) => hostname === domain || hostname.endsWith(`.${domain}`));
|
||||
}
|
||||
|
||||
function getErrorStatus(error) {
|
||||
if (error && Number.isInteger(error.status)) {
|
||||
@@ -147,20 +119,9 @@ async function fetchCompressedImage(url) {
|
||||
|
||||
async function fetchAndStoreContent(id, url) {
|
||||
try {
|
||||
if (isBlockedContentUrl(url)) {
|
||||
const hostname = getHostname(url);
|
||||
if (hostname && !loggedBlockedDomains.has(hostname)) {
|
||||
loggedBlockedDomains.add(hostname);
|
||||
console.warn(`content extraction skipped for blocked domain ${hostname}`);
|
||||
}
|
||||
markArticleStatus(markContentSkipped, id, `blocked domain: ${hostname || 'unknown'}`);
|
||||
return;
|
||||
}
|
||||
|
||||
const article = await extract(url, {}, {
|
||||
headers: articleFetchHeaders,
|
||||
signal: AbortSignal.timeout(20000),
|
||||
});
|
||||
const browserSession = await getSharedBrowserSession({ requestTimeout: 20000, maxConcurrentPages: 2 });
|
||||
const html = await browserSession.fetchRenderedHtml(url, { timeout: 20000 });
|
||||
const article = await extractFromHtml(html, url);
|
||||
if (!article) {
|
||||
markArticleStatus(markContentSkipped, id, 'extractor returned no article');
|
||||
return;
|
||||
|
||||
Reference in New Issue
Block a user