The crawler asked gpt-4.1-mini one three-way question per undecided page, and then threw away the confidence it asked for in the same prompt. Jev answers that question as a typed choice for $0.042 per million in and nothing out, with a confidence that falls out of the distribution rather than the model's opinion of itself. It cannot write learnedSignals though -- rule_value is free text -- so the expensive model is still what teaches a new site. Once a site has banked enough rules and jev is sure, we stop paying for it. A dead signal call now falls back to the jev answer instead of losing the page. confidence is stored on crawler_page_classifications, additively, in both dialects. Pattern, rule and signal-model rows keep a null rather than borrowing a number that belonged to a different answer. The openrouter slug is unverified -- jev is beta there and absent from /api/v1/models -- so both the slug and the endpoint are env overridable, and a bad slug degrades to the old path instead of breaking the crawl. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EZ6bEoR6m6vrksSYRiXwVD
235 lines
6.4 KiB
JavaScript
235 lines
6.4 KiB
JavaScript
// postgres adapter that mimics the better-sqlite3 synchronous interface.
|
|
// uses deasync to block the event loop until queries resolve so all
|
|
// existing consumers (which call .get()/.all()/.run() synchronously) keep working.
|
|
|
|
const { Pool } = require('pg');
|
|
const deasync = require('deasync');
|
|
const config = require('../config');
|
|
|
|
const pool = new Pool(config.database.postgres);
|
|
|
|
function querySync(sql, params) {
|
|
let done = false;
|
|
let result, err;
|
|
|
|
pool.query(sql, params).then(r => { result = r; done = true; }).catch(e => { err = e; done = true; });
|
|
deasync.loopWhile(() => !done);
|
|
|
|
if (err) throw err;
|
|
return result;
|
|
}
|
|
|
|
// translate ? placeholders to $1, $2, ... for postgres
|
|
function toPositional(sql) {
|
|
let i = 0;
|
|
return sql.replace(/\?/g, () => `$${++i}`);
|
|
}
|
|
|
|
|
|
// sqlite-vec uses: WHERE embedding MATCH ? AND k = ?
|
|
// pgvector uses: ORDER BY embedding <-> $1::vector LIMIT $2
|
|
const NEAREST_NEIGHBORS_RE = /SELECT\s+article_id\s*,\s*distance\s+FROM\s+article_embeddings\s+WHERE\s+embedding\s+MATCH\s+\?\s+AND\s+k\s*=\s*\?\s+ORDER\s+BY\s+distance/is;
|
|
|
|
function rewriteNearestNeighbors(sql) {
|
|
if (!NEAREST_NEIGHBORS_RE.test(sql)) return null;
|
|
return `
|
|
SELECT article_id, (embedding <-> $1::vector) AS distance
|
|
FROM article_embeddings
|
|
ORDER BY embedding <-> $1::vector
|
|
LIMIT $2
|
|
`;
|
|
}
|
|
|
|
function Statement(sql) {
|
|
const pgSql = rewriteNearestNeighbors(sql) || toPositional(sql);
|
|
|
|
return {
|
|
get(...params) {
|
|
const r = querySync(pgSql, params.flat());
|
|
return r.rows[0] ?? undefined;
|
|
},
|
|
|
|
all(...params) {
|
|
const r = querySync(pgSql, params.flat());
|
|
return r.rows;
|
|
},
|
|
|
|
run(...params) {
|
|
const r = querySync(pgSql, params.flat());
|
|
return { changes: r.rowCount, lastInsertRowid: null };
|
|
},
|
|
};
|
|
}
|
|
|
|
function exec(sql) {
|
|
// split on ; and run each statement individually (DDL blocks)
|
|
const stmts = sql.split(';').map(s => s.trim()).filter(Boolean);
|
|
for (const stmt of stmts) {
|
|
querySync(stmt, []);
|
|
}
|
|
}
|
|
|
|
function transaction(fn) {
|
|
return (...args) => {
|
|
querySync('BEGIN', []);
|
|
try {
|
|
const result = fn(...args);
|
|
querySync('COMMIT', []);
|
|
return result;
|
|
} catch (e) {
|
|
querySync('ROLLBACK', []);
|
|
throw e;
|
|
}
|
|
};
|
|
}
|
|
|
|
function prepare(sql) {
|
|
return Statement(sql);
|
|
}
|
|
|
|
const pgDb = { prepare, exec, transaction, pool };
|
|
|
|
// create the schema in postgres, mirroring src/db.js
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS events (
|
|
id SERIAL PRIMARY KEY,
|
|
title TEXT NOT NULL,
|
|
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
|
)
|
|
`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS articles (
|
|
id SERIAL PRIMARY KEY,
|
|
title TEXT NOT NULL,
|
|
description TEXT,
|
|
content TEXT,
|
|
image TEXT,
|
|
content_status TEXT,
|
|
content_error TEXT,
|
|
content_attempted_at TEXT,
|
|
content_attempt_count INTEGER NOT NULL DEFAULT 0,
|
|
content_retry_after TEXT,
|
|
is_index_page INTEGER NOT NULL DEFAULT 0,
|
|
has_embedding INTEGER NOT NULL DEFAULT 0,
|
|
url TEXT NOT NULL UNIQUE,
|
|
normalized_title TEXT NOT NULL,
|
|
source TEXT NOT NULL,
|
|
pub_date TEXT,
|
|
pub_date_effective TEXT,
|
|
language TEXT,
|
|
event_id INTEGER REFERENCES events(id),
|
|
ingested_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
|
)
|
|
`);
|
|
|
|
exec(`CREATE INDEX IF NOT EXISTS idx_articles_source ON articles(source)`);
|
|
exec(`CREATE INDEX IF NOT EXISTS idx_articles_pub_date ON articles(pub_date)`);
|
|
exec(`CREATE INDEX IF NOT EXISTS idx_articles_ingested_at ON articles(ingested_at)`);
|
|
exec(`CREATE INDEX IF NOT EXISTS idx_articles_normalized_title ON articles(normalized_title)`);
|
|
exec(`CREATE INDEX IF NOT EXISTS idx_articles_event_id ON articles(event_id)`);
|
|
exec(`CREATE INDEX IF NOT EXISTS idx_articles_has_embedding ON articles(has_embedding)`);
|
|
exec(`CREATE INDEX IF NOT EXISTS idx_articles_pub_date_effective ON articles(pub_date_effective DESC)`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS article_embedding_store (
|
|
article_id INTEGER NOT NULL,
|
|
model TEXT NOT NULL,
|
|
embedding BYTEA NOT NULL,
|
|
embedded_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
|
PRIMARY KEY (article_id, model)
|
|
)
|
|
`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS article_embedding_meta (
|
|
article_id INTEGER PRIMARY KEY,
|
|
model TEXT NOT NULL,
|
|
embedded_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
|
)
|
|
`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS article_embeddings (
|
|
article_id INTEGER PRIMARY KEY,
|
|
embedding vector(8192)
|
|
)
|
|
`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS query_embeddings (
|
|
query TEXT NOT NULL,
|
|
model TEXT NOT NULL,
|
|
embedding BYTEA NOT NULL,
|
|
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
|
PRIMARY KEY (query, model)
|
|
)
|
|
`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS gdelt_backfill_windows (
|
|
source_id TEXT NOT NULL,
|
|
window_start TEXT NOT NULL,
|
|
window_end TEXT NOT NULL,
|
|
completed_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
|
PRIMARY KEY (source_id, window_start, window_end)
|
|
)
|
|
`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS crawler_page_classifications (
|
|
url TEXT PRIMARY KEY,
|
|
site_name TEXT NOT NULL,
|
|
classification TEXT NOT NULL,
|
|
confidence DOUBLE PRECISION,
|
|
pattern TEXT,
|
|
classified_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
|
)
|
|
`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS crawler_url_patterns (
|
|
site_name TEXT NOT NULL,
|
|
pattern TEXT NOT NULL,
|
|
classification TEXT NOT NULL,
|
|
hit_count INTEGER NOT NULL DEFAULT 1,
|
|
updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
|
PRIMARY KEY (site_name, pattern)
|
|
)
|
|
`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS crawler_site_rules (
|
|
site_name TEXT NOT NULL,
|
|
rule_type TEXT NOT NULL,
|
|
rule_value TEXT NOT NULL,
|
|
classification TEXT NOT NULL,
|
|
hit_count INTEGER NOT NULL DEFAULT 1,
|
|
updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
|
PRIMARY KEY (site_name, rule_type, rule_value)
|
|
)
|
|
`);
|
|
|
|
exec(`
|
|
CREATE TABLE IF NOT EXISTS domain_fetch_policy (
|
|
domain TEXT PRIMARY KEY,
|
|
policy TEXT NOT NULL DEFAULT 'auto',
|
|
consecutive_plain_failures INTEGER NOT NULL DEFAULT 0,
|
|
consecutive_browser_failures INTEGER NOT NULL DEFAULT 0,
|
|
plain_success_count INTEGER NOT NULL DEFAULT 0,
|
|
browser_success_count INTEGER NOT NULL DEFAULT 0,
|
|
expires_at TEXT,
|
|
updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
|
)
|
|
`);
|
|
|
|
// same story as the sqlite side: the creates above skip a database that already
|
|
// has the table, so additions land here too. postgres does have the IF NOT EXISTS
|
|
// form so this one stays boring. Additive only.
|
|
exec(`
|
|
ALTER TABLE crawler_page_classifications
|
|
ADD COLUMN IF NOT EXISTS confidence DOUBLE PRECISION
|
|
`);
|
|
|
|
module.exports = pgDb; |