initialize project structure with core modules and configuration

This commit is contained in:
ImBenji
2026-04-16 22:47:34 +01:00
parent 82d856a21d
commit 7724fafbdc
20 changed files with 3644 additions and 0 deletions
+203
View File
@@ -0,0 +1,203 @@
const { extract } = require('@extractus/article-extractor');
const sharp = require('sharp');
const db = require('./db');
const { generateAndStoreEmbedding } = require('./embeddings');
const { fetchWithPolicy } = require('./http');
const updateArticleAssets = db.prepare(`
UPDATE articles
SET content = ?, image = ?, content_status = 'ready', content_error = NULL, content_attempted_at = ?
WHERE id = ?
`);
const markContentSkipped = db.prepare(`
UPDATE articles
SET content_status = 'skipped', content_error = ?, content_attempted_at = ?
WHERE id = ?
`);
const markContentFailed = db.prepare(`
UPDATE articles
SET content_status = 'failed', content_error = ?, content_attempted_at = ?
WHERE id = ?
`);
const markContentPending = db.prepare(`
UPDATE articles
SET content_status = NULL, content_error = NULL, content_attempted_at = ?
WHERE id = ?
`);
const selectArticlesMissingContent = db.prepare(`
SELECT id, url
FROM articles
WHERE (content IS NULL OR TRIM(content) = '')
AND (content_status IS NULL OR content_status = 'pending')
ORDER BY ingested_at DESC, id DESC
LIMIT ?
`);
const blockedContentDomains = [
'axios.com',
'bizjournals.com',
'fastcompany.com',
'gurufocus.com',
'investing.com',
'rbc.ru',
'stocktitan.net',
];
const loggedBlockedDomains = new Set();
const articleFetchHeaders = {
Accept: 'text/html,application/xhtml+xml',
'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36',
};
let contentBackfillRunning = false;
function getHostname(url) {
try {
return new URL(url).hostname.toLowerCase();
} catch {
return '';
}
}
function isBlockedContentUrl(url) {
const hostname = getHostname(url);
return blockedContentDomains.some((domain) => hostname === domain || hostname.endsWith(`.${domain}`));
}
function getErrorStatus(error) {
if (error && Number.isInteger(error.status)) {
return error.status;
}
const match = String(error && error.message || '').match(/\b(401|403|404|408|429|5\d\d)\b/);
return match ? Number(match[1]) : null;
}
function getErrorMessage(error, fallback) {
const message = String(error && error.message || fallback || '').trim();
return message ? message.slice(0, 500) : null;
}
function markArticleStatus(statement, id, message) {
statement.run(message, new Date().toISOString(), id);
}
async function fetchCompressedImage(url) {
const response = await fetchWithPolicy(url, {
retries: 1,
headers: {
Accept: 'image/*',
},
});
if (!response.ok) {
const error = new Error(`image request failed with ${response.status}`);
error.status = response.status;
throw error;
}
const contentType = String(response.headers.get('content-type') || '').toLowerCase();
if (!contentType.startsWith('image/')) {
throw new Error(`image request returned ${contentType || 'unknown content-type'}`);
}
const input = Buffer.from(await response.arrayBuffer());
if (input.length === 0) {
throw new Error('image request returned an empty body');
}
const output = await sharp(input)
.rotate()
.resize({ width: 320, height: 320, fit: 'inside', withoutEnlargement: true })
.webp({ quality: 25 })
.toBuffer();
return output.toString('base64');
}
async function fetchAndStoreContent(id, url) {
try {
if (isBlockedContentUrl(url)) {
const hostname = getHostname(url);
if (hostname && !loggedBlockedDomains.has(hostname)) {
loggedBlockedDomains.add(hostname);
console.warn(`content extraction skipped for blocked domain ${hostname}`);
}
markArticleStatus(markContentSkipped, id, `blocked domain: ${hostname || 'unknown'}`);
return;
}
const article = await extract(url, {}, {
headers: articleFetchHeaders,
signal: AbortSignal.timeout(20000),
});
if (!article) {
markArticleStatus(markContentSkipped, id, 'extractor returned no article');
return;
}
const content = typeof article.content === 'string'
? article.content.replace(/<[^>]+>/g, ' ').replace(/\s+/g, ' ').trim() || null
: null;
let image = null;
if (article.image) {
try {
image = await fetchCompressedImage(article.image);
} catch (error) {
const status = getErrorStatus(error);
if (status === 401 || status === 403 || status === 404 || status === 429) {
console.warn(`image fetch skipped for ${url}: upstream returned ${status}`);
} else {
console.error(`image fetch failed for ${url}:`, error);
}
}
}
if (!content && !image) {
markArticleStatus(markContentSkipped, id, 'article had no extractable content or image');
return;
}
updateArticleAssets.run(content, image, new Date().toISOString(), id);
await generateAndStoreEmbedding(id);
} catch (error) {
const status = getErrorStatus(error);
if (status === 401 || status === 403 || status === 404) {
console.warn(`content fetch skipped for ${url}: upstream returned ${status}`);
markArticleStatus(markContentSkipped, id, `upstream returned ${status}`);
return;
}
if (status === 408 || status === 429 || (status && status >= 500)) {
console.warn(`content fetch deferred for ${url}: upstream returned ${status}`);
markArticleStatus(markContentPending, id, null);
return;
}
markArticleStatus(markContentFailed, id, getErrorMessage(error, 'content fetch failed'));
console.error(`content fetch failed for ${url}:`, error);
}
}
async function backfillMissingContent(limit = 10) {
if (contentBackfillRunning) {
return;
}
contentBackfillRunning = true;
try {
const rows = selectArticlesMissingContent.all(limit);
for (const row of rows) {
await fetchAndStoreContent(row.id, row.url);
}
} finally {
contentBackfillRunning = false;
}
}
module.exports = {
fetchAndStoreContent,
backfillMissingContent,
};