diff --git a/src/content.js b/src/content.js index 4ff14b2..188eba0 100644 --- a/src/content.js +++ b/src/content.js @@ -68,8 +68,11 @@ const selectPartitionedArticlesMissingContent = db.prepare(` SELECT id, url, title, description, source, pub_date_effective, ROW_NUMBER() OVER (PARTITION BY source ORDER BY pub_date_effective DESC, id DESC) AS rn FROM articles - WHERE (content IS NULL OR TRIM(content) = '') - AND (content_status IS NULL OR content_status = 'pending') + -- content_status is the authority on whether a row has been fetched, and it + -- agrees with the content column on all 2.2M rows. Testing TRIM(content) here + -- as well meant reading a 4GB blob column just to find out which rows to skip, + -- and it stopped the partial index below being usable at all. + WHERE (content_status IS NULL OR content_status = 'pending') AND (content_retry_after IS NULL OR content_retry_after <= datetime('now')) AND (id % ?) = ? ) @@ -409,8 +412,7 @@ async function runBackfillWorker({ workerIndex, workerCount, perSource, batchSiz function hasPendingContent() { return Boolean(db.prepare(` SELECT 1 FROM articles - WHERE (content IS NULL OR TRIM(content) = '') - AND (content_status IS NULL OR content_status = 'pending') + WHERE (content_status IS NULL OR content_status = 'pending') AND (content_retry_after IS NULL OR content_retry_after <= datetime('now')) LIMIT 1 `).get()); diff --git a/src/db/index.js b/src/db/index.js index 9ea1cb2..a3751e6 100644 --- a/src/db/index.js +++ b/src/db/index.js @@ -54,8 +54,26 @@ db.exec(` AND content != '' AND is_index_page = 0 AND has_embedding = 1; + + -- The content backfill picker partitions by source and orders by pub date, and + -- without this it built two temp b-trees over every unfetched row: 194 seconds + -- per call on a 2.2M row archive, synchronously, which froze the whole process. + -- Column order matches PARTITION BY source ORDER BY pub_date_effective DESC, id DESC + -- so the window function can just walk it. + CREATE INDEX IF NOT EXISTS idx_articles_pending_fetch + ON articles(source, pub_date_effective DESC, id DESC) + WHERE content_status IS NULL OR content_status = 'pending'; `); +// Without stats the planner ignores the partial index above and falls back to the +// content_status index plus a temp b-tree, which is roughly 70% slower. optimize +// only re-analyses what has actually drifted, so this is cheap after the first run. +try { + db.exec('PRAGMA optimize;'); +} catch (error) { + console.error('[db] PRAGMA optimize failed, query plans may be stale:', error.message); +} + db.exec(` CREATE TABLE IF NOT EXISTS article_embedding_store ( article_id INTEGER NOT NULL,