fix: restart the stalled autonomy pipeline and make calibration honest

Archive ingestion had been dead since 2026-08-02 because nothing in the
compose stack actually ran it. Everything downstream starved from there.

- add ingest + enrichment services. server.js only starts the scheduler when
  DURIIN_RUN_SCHEDULER is not "false", and workers/index.js was not running at
  all, so articles never got event_id/content/has_embedding and the coordinator
  had nothing to lease.
- pass an explicit origin from coordinatorWorker. it was never passed, so
  acceptProposal defaulted to 'live' and 464 historical backfill predictions
  were recorded as live. that also meant verifyEvidence got a null cutoff and
  skipped its date check entirely.
- coarsen cohortKey to event families + horizon buckets. 201 free text event
  types produced 221 cohorts averaging 2.76 samples, so the n>=30 gate could
  never be reached and everything abstained for the wrong reason.
- gate on cohort diversity, not just sample count. one ticker was roughly half
  of all resolved outcomes, so a pure count gate was measuring one company.
  unknown diversity abstains rather than passing.
- resolve the admin archive db explicitly and probe it. it relied on a
  Dockerfile symlink, and without it better-sqlite3 quietly creates an empty
  file and serves a phantom archive.
- clamp implausible future publication dates at ingest.
- pin the db backend to sqlite by default. compose hardcoded postgres "true",
  which would have overridden the operator's own .env on the next redeploy and
  pointed everything at a stale snapshot.

scripts/repair-autonomy-labels.js relabels the affected rows. it is dry run by
default and has not been applied.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-08-29 21:43:24 +01:00
co-authored by Claude Opus 5
parent d778a02bfb
commit 6f1d1eee2d
19 changed files with 1497 additions and 100 deletions
+92 -1
View File
@@ -14,10 +14,77 @@ function quantile(values, q) {
return sorted[lower] + (sorted[upper] - sorted[lower]) * (position - lower);
}
// The coordinator emits event_type as free text, so production ended up with 200+
// distinct values across ~600 predictions. Keying calibration on the raw string
// gave cohorts of ~2.7 samples each, which can never clear any honest sample gate.
// These families are a closed set: order matters, first match wins, and anything
// we don't recognise lands in `other` rather than inventing its own cohort.
const EVENT_FAMILIES = [
['analyst_action', /\b(analysts?|upgrades?|downgrades?|price[_ ]?targets?|ratings?|initiations?|coverage|overweight|underweight|outperform)\b/],
['guidance', /\b(guidance|outlooks?|forecasts?|pre[_ ]?announce\w*|warns?|warning|raises?[_ ]guid\w*|cuts?[_ ]guid\w*|projections?)\b/],
['earnings', /\b(earnings?|results?|quarterly|eps|revenues?|margins?|beat|miss(ed|es)?|q[1-4]|fy\d{2,4}|financials?)\b/],
['m_and_a', /\b(m&a|merger|mergers|acquisitions?|acquires?|acquired|takeovers?|buyouts?|divestitures?|divests?|spin[_ ]?offs?|stake[_ ]sales?|tender[_ ]offers?)\b/],
['legal', /\b(lawsuits?|litigations?|courts?|patents?|settlements?|verdicts?|injunctions?|class[_ ]actions?|subpoenas?|infringements?|appeals?)\b/],
['regulatory', /\b(regulat\w*|antitrust|probes?|investigations?|sanctions?|export[_ ]controls?|tariffs?|bans?|banned|approvals?|approved|licens\w*|compliance|fda|ftc|doj|sec[_ ]filing|policy)\b/],
['leadership', /\b(ceo|cfo|coo|cto|chairman|executives?|resign\w*|appoint\w*|steps?[_ ]down|boards?|successions?|layoffs?|restructur\w*|hiring|departures?)\b/],
['supply_chain', /\b(supply|suppliers?|shortages?|capacity|production|fabs?|foundry|inventor\w+|logistics?|shipments?|recalls?|manufactur\w*|yields?|backlog)\b/],
['contract', /\b(contracts?|orders?|partnerships?|partners?|agreements?|collaborations?|deals?|customers?|wins?|awards?)\b/],
['product', /\b(products?|launch\w*|unveil\w*|releases?|announcements?|chips?|models?|features?|roadmaps?|platforms?)\b/],
['capital', /\b(buybacks?|repurchases?|dividends?|offerings?|debt|capital[_ ]raise|stock[_ ]splits?|ipos?|financing|bonds?)\b/],
['security_incident', /\b(hacks?|hacked|breach\w*|cyber\w*|ransomware|outages?|downtime|vulnerabilit\w+|exploits?)\b/],
['macro', /\b(macro\w*|fed|federal[_ ]reserve|interest[_ ]rates?|inflation|gdp|econom\w+|recession|currenc\w+|geopolit\w+|war|elections?|demand)\b/],
];
const EVENT_FAMILY_NAMES = EVENT_FAMILIES.map(([name]) => name).concat('other');
// snake_case, camelCase, "Supply Constraint" and "supply-constraint" all have to
// collapse onto the same token stream before we try to match anything.
function normalizeEventType(raw) {
if (raw === null || raw === undefined) return 'other';
const text = String(raw)
.replace(/([a-z0-9])([A-Z])/g, '$1 $2')
.toLowerCase()
.replace(/[^a-z0-9&]+/g, ' ')
.trim();
if (!text) return 'other';
for (const [family, pattern] of EVENT_FAMILIES) {
if (pattern.test(text)) return family;
}
return 'other';
}
// ALLOWED_HORIZONS is 1/5/10/20/30/60/90 in the coordinator. Seven horizons times
// two directions was another multiplier on the cohort explosion, and a 10 day and
// a 20 day call on the same event are not really different populations.
const HORIZON_BUCKETS = ['short', 'medium', 'long'];
function horizonBucket(horizonDays) {
const days = Number(horizonDays);
if (!Number.isFinite(days) || days <= 0) return 'unknown';
if (days <= 5) return 'short';
if (days <= 20) return 'medium';
return 'long';
}
const COHORT_KEY_VERSION = 'v2';
function cohortKey({ direction, eventType, horizonDays, sector = 'unknown' }) {
return [COHORT_KEY_VERSION, sector, normalizeEventType(eventType), horizonBucket(horizonDays), direction].join('|');
}
// Snapshots written before the taxonomy change still carry the raw key, this keeps
// them readable/joinable without a migration.
function legacyCohortKey({ direction, eventType, horizonDays, sector = 'unknown' }) {
return [sector, eventType || 'unknown', horizonDays, direction].join('|');
}
function instrumentOf(row) {
const symbol = row.instrument ?? row.symbol ?? null;
if (symbol === null || symbol === undefined) return null;
const trimmed = String(symbol).trim().toUpperCase();
return trimmed || null;
}
function calibrateOutcomes(rows, parent = null) {
const clean = rows.filter((row) => Number.isFinite(Number(row.excess_return)));
const wins = clean.filter((row) => Number(row.direction_correct) === 1).length;
@@ -26,6 +93,17 @@ function calibrateOutcomes(rows, parent = null) {
const priorStrength = parent ? Math.max(2, Math.min(20, parent.effectiveSampleSize / 10)) : 2;
const probability = (wins + priorProbability * priorStrength) / (total + priorStrength);
const returns = clean.map((row) => Number(row.excess_return));
// Concentration matters as much as raw n here. A cohort of 300 outcomes that is
// 95% one ticker is one bet repeated, not 300 independant observations.
const counts = new Map();
for (const row of clean) {
const symbol = instrumentOf(row);
if (!symbol) continue;
counts.set(symbol, (counts.get(symbol) || 0) + 1);
}
const topCount = counts.size ? Math.max(...counts.values()) : 0;
return {
sampleSize: total,
effectiveSampleSize: total + priorStrength,
@@ -33,7 +111,20 @@ function calibrateOutcomes(rows, parent = null) {
expectedExcessReturn: returns.length ? returns.reduce((sum, value) => sum + value, 0) / returns.length : null,
lowerReturn: quantile(returns, 0.1),
upperReturn: quantile(returns, 0.9),
distinctInstruments: counts.size,
topInstrumentShare: total ? topCount / total : null,
};
}
module.exports = { betaMean, cohortKey, calibrateOutcomes, quantile };
module.exports = {
betaMean,
cohortKey,
legacyCohortKey,
calibrateOutcomes,
quantile,
normalizeEventType,
horizonBucket,
EVENT_FAMILY_NAMES,
HORIZON_BUCKETS,
COHORT_KEY_VERSION,
};
+11 -7
View File
@@ -36,18 +36,22 @@ function normalizeProposal(raw, { informationCutoff, model = 'unknown', promptVe
function verifyEvidence(archiveDb, articleIds, informationCutoff = null) {
const placeholders = articleIds.map(() => '?').join(',');
// A replay must only see material which existed at its information cutoff.
// Live proposals retain the simpler existence check.
// No proposal, whatever lane produced it, may cite material which did not yet
// exist at its own information cutoff. This used to be a replay-only rule and
// that was a lookahead hole for every other origin.
const cutoffClause = informationCutoff ? ' AND datetime(COALESCE(pub_date_effective, pub_date, ingested_at)) <= datetime(?)' : '';
let rows;
try {
rows = archiveDb.prepare(`SELECT id FROM articles WHERE id IN (${placeholders})${cutoffClause}`)
.all(...articleIds, ...(informationCutoff ? [informationCutoff] : []));
} catch (error) {
// Minimal/test archives may not retain publication metadata. A production
// replay archive is required to have it, so this fallback is only for the
// existing live evidence contract.
if (informationCutoff) throw error;
// Minimal/test archives may not retain publication metadata at all, in which
// case the cutoff clause cannot even be prepared. We degrade to a plain
// existence check rather than blocking the pipeline, but the degredation is
// never silent - a production archive missing these columns is a real bug.
console.warn('[coordinator] evidence cutoff check unavailable, falling back to existence only.',
`cutoff=${informationCutoff} articles=${JSON.stringify(articleIds)} reason=${error && error.message}`);
if (error && error.stack) console.warn(error.stack);
rows = archiveDb.prepare(`SELECT id FROM articles WHERE id IN (${placeholders})`).all(...articleIds);
}
const found = new Set(rows.map((row) => row.id));
@@ -57,7 +61,7 @@ function verifyEvidence(archiveDb, articleIds, informationCutoff = null) {
function acceptProposal(intelligenceDb, archiveDb, raw, metadata = {}) {
const proposal = normalizeProposal(raw, metadata);
for (const prediction of proposal.predictions) {
if (!verifyEvidence(archiveDb, prediction.evidenceArticleIds, metadata.origin === 'replay' ? proposal.informationCutoff : null)) {
if (!verifyEvidence(archiveDb, prediction.evidenceArticleIds, proposal.informationCutoff)) {
throw new Error(`proposal references missing evidence for ${prediction.instrument}`);
}
const instrument = intelligenceDb.prepare(
+57 -7
View File
@@ -1,15 +1,65 @@
function decide({ direction = 'positive', probability, expectedExcessReturn, lowerReturn, upperReturn, sampleSize }, rules = {}) {
const minSampleSize = Number(rules.minSampleSize ?? 30);
const minProbability = Number(rules.minProbability ?? 0.58);
const minExpectedReturn = Number(rules.minExpectedReturn ?? 0.005);
const maxDownside = Number(rules.maxDownside ?? -0.08);
// Thresholds live here so the worker, the replay evaluator and the tests all
// argue from the same numbers instead of sprinkling magic 30s around.
const DEFAULT_POLICY_RULES = {
minSampleSize: 30,
// A cohort has to be built from more than a handful of tickers. In production
// one name (NVDA) accounted for roughly half of every resolved outcome, so a
// pure sample-size gate was measuring one company, not an edge.
minDistinctInstruments: 5,
maxInstrumentConcentration: 0.5,
minProbability: 0.58,
minExpectedReturn: 0.005,
maxDownside: -0.08,
};
function decide({
direction = 'positive',
probability,
expectedExcessReturn,
lowerReturn,
upperReturn,
sampleSize,
distinctInstruments,
topInstrumentShare,
}, rules = {}) {
const minSampleSize = Number(rules.minSampleSize ?? DEFAULT_POLICY_RULES.minSampleSize);
const minDistinctInstruments = Number(rules.minDistinctInstruments ?? DEFAULT_POLICY_RULES.minDistinctInstruments);
const maxInstrumentConcentration = Number(rules.maxInstrumentConcentration ?? DEFAULT_POLICY_RULES.maxInstrumentConcentration);
const minProbability = Number(rules.minProbability ?? DEFAULT_POLICY_RULES.minProbability);
const minExpectedReturn = Number(rules.minExpectedReturn ?? DEFAULT_POLICY_RULES.minExpectedReturn);
const maxDownside = Number(rules.maxDownside ?? DEFAULT_POLICY_RULES.maxDownside);
if (![probability, expectedExcessReturn].every(Number.isFinite)) {
return { action: 'ABSTAIN', rationale: 'calibration unavailable' };
}
if (sampleSize < minSampleSize) {
if (!Number.isFinite(Number(sampleSize)) || Number(sampleSize) < minSampleSize) {
return { action: 'ABSTAIN', rationale: `insufficient calibration sample (${sampleSize}/${minSampleSize})` };
}
// Snapshots written before diversification was tracked come back with the count
// missing. Unknown diversity is not the same as adequate diversity, abstain.
const instruments = distinctInstruments === null || distinctInstruments === undefined ? NaN : Number(distinctInstruments);
if (!Number.isFinite(instruments)) {
return { action: 'ABSTAIN', rationale: 'cohort instrument diversity unknown' };
}
if (instruments < minDistinctInstruments) {
return { action: 'ABSTAIN', rationale: `insufficient cohort diversity (${instruments}/${minDistinctInstruments} instruments)` };
}
// Same rule as the count above: a missing share is unknown, not safe. Number(null)
// is 0, which would sail straight through the cap, so check for absence first.
const concentration = topInstrumentShare === null || topInstrumentShare === undefined
? NaN
: Number(topInstrumentShare);
if (!Number.isFinite(concentration)) {
return { action: 'ABSTAIN', rationale: 'cohort instrument concentration unknown' };
}
if (concentration > maxInstrumentConcentration) {
return {
action: 'ABSTAIN',
rationale: `cohort dominated by a single instrument (${(concentration * 100).toFixed(0)}% > ${(maxInstrumentConcentration * 100).toFixed(0)}%)`,
};
}
const signedExpectedReturn = direction === 'negative' ? -expectedExcessReturn : expectedExcessReturn;
const signedLowerReturn = direction === 'negative'
? (Number.isFinite(upperReturn) ? -upperReturn : null)
@@ -23,4 +73,4 @@ function decide({ direction = 'positive', probability, expectedExcessReturn, low
return { action: 'HOLD', rationale: 'calibrated edge does not clear policy thresholds' };
}
module.exports = { decide };
module.exports = { decide, DEFAULT_POLICY_RULES };
+18 -1
View File
@@ -1,5 +1,12 @@
const AUTONOMY_SCHEMA_VERSION = 2;
// sqlite and postgres word this differently, and we re-run every ALTER on each
// boot, so a re-add is the expected case rather than a failure.
function isDuplicateColumn(error) {
const message = String(error && error.message || '').toLowerCase();
return message.includes('duplicate column') || message.includes('already exists');
}
function initAutonomySchema(db) {
if (db.dialect === 'postgres') return;
db.exec(`
@@ -225,8 +232,18 @@ function initAutonomySchema(db) {
'ALTER TABLE autonomy_replay_runs ADD COLUMN cursor_effective_at TEXT',
"ALTER TABLE autonomy_calibration_snapshots ADD COLUMN source TEXT NOT NULL DEFAULT 'live'",
'ALTER TABLE autonomy_calibration_snapshots ADD COLUMN replay_run_id INTEGER',
'ALTER TABLE autonomy_calibration_snapshots ADD COLUMN distinct_instruments INTEGER',
'ALTER TABLE autonomy_calibration_snapshots ADD COLUMN top_instrument_share REAL',
]) {
try { db.exec(statement); } catch (_) {}
try {
db.exec(statement);
} catch (error) {
// Re-running these is normal, the column is already there. Anything else
// means a migration genuinely failed and we want to hear about it.
if (!isDuplicateColumn(error)) {
console.error(`[autonomy-schema] migration failed: ${statement}`, error.message, error.stack);
}
}
}
}
+7 -1
View File
@@ -1,6 +1,7 @@
const db = require('./db');
const { normalizeTitle } = require('./dedup');
const { markSourceRun } = require('./state');
const { guardEffectivePubDate } = require('./pubDateGuard');
const sourcesById = Object.fromEntries(
require('../sources.json').map((s) => [s.id, s])
@@ -88,6 +89,11 @@ function ingestArticle(article) {
const ingestedAt = new Date().toISOString();
const language = (sourcesById[source] && sourcesById[source].language) || null;
// pub_date keeps whatever the source claimed (it is still useful for
// debugging a broken feed), but the effective date — the one the coordinator
// turns into an information cutoff — refuses anything from the future.
const effectivePubDate = guardEffectivePubDate(pubDate, ingestedAt, { source, url });
try {
const result = insertArticle.run(
title,
@@ -98,7 +104,7 @@ function ingestArticle(article) {
source,
pubDate,
ingestedAt,
pubDate || ingestedAt,
effectivePubDate,
language
);
+61
View File
@@ -0,0 +1,61 @@
// Guard against publication dates that sit in the future.
//
// pub_date_effective is what the autonomy coordinator uses to derive a
// prediction's information_cutoff (max pub_date_effective across an event's
// articles), so a single bogus feed date drags the cutoff forward and quietly
// breaks evidence-cutoff enforcement and outcome scoring. Production currently
// has exactly one such row, but one is enough to poison an event.
//
// Tolerance: 48 hours. It has to swallow the legitimate cases —
// * date only strings ("2026-08-29") are stored as midnight UTC, and a
// publisher in UTC+14 can legitimately stamp tomorrow's date,
// * feeds that emit local time without an offset, worst case ~14h ahead,
// * modest clock skew on the publisher's box.
// 48h covers all of that with room to spare while still catching anything
// genuinely wrong — the offending production row is about four months out.
const DEFAULT_TOLERANCE_MS = 48 * 60 * 60 * 1000;
function toleranceMs() {
const hours = Number(process.env.INGEST_FUTURE_PUB_DATE_HOURS);
if (Number.isFinite(hours) && hours > 0) return hours * 60 * 60 * 1000;
return DEFAULT_TOLERANCE_MS;
}
// Returns { ok, value, skewMs, toleranceMs }. `value` is null when the date is
// implausible so the caller can fall back to ingestion time. The article itself
// is never dropped for this — a bad date is not a bad article.
function checkPubDate(value, now = Date.now(), tolerance = toleranceMs()) {
if (!value) return { ok: true, value: null, skewMs: 0, toleranceMs: tolerance };
const parsed = new Date(value).getTime();
if (Number.isNaN(parsed)) return { ok: true, value: null, skewMs: 0, toleranceMs: tolerance };
const skewMs = parsed - now;
if (skewMs > tolerance) {
return { ok: false, value: null, skewMs, toleranceMs: tolerance };
}
return { ok: true, value, skewMs, toleranceMs: tolerance };
}
// Same check, but it also does the shouting. Keeps ingest.js readable and makes
// sure every clamp lands in the logs with the source and the offending value.
function guardEffectivePubDate(pubDate, fallback, context = {}) {
const verdict = checkPubDate(pubDate);
if (verdict.ok) return pubDate || fallback;
const days = (verdict.skewMs / 86400000).toFixed(1);
console.warn(
`[ingest] refusing future pub date from "${context.source || 'unknown source'}": ${pubDate} is ${days} days ahead ` +
`(tolerance ${Math.round(verdict.toleranceMs / 3600000)}h) — pub_date_effective falls back to ${fallback}. url=${context.url || 'n/a'}`
);
return fallback;
}
module.exports = {
checkPubDate,
guardEffectivePubDate,
DEFAULT_TOLERANCE_MS,
};
+243 -28
View File
@@ -9,11 +9,74 @@ const { openRuntimeDb, isPostgresEnabled } = require('../db/runtime');
const pg = require('../db/pgAsync');
let idb = null;
let adb = null;
let statsSummaryCache = null;
let statsDetailCache = null;
const configDir = path.resolve(__dirname, '..', '..');
// The archive is resolved exactly like the workers do it (workers/index.js:31):
// DURIIN_DB wins, then config, and only then the repo relative default. The old
// code here went straight to config.database.path — a repo relative
// "./archive.sqlite" — which inside the container only ever pointed at the real
// data because of a build time symlink, and which quietly opens a brand new
// empty database when that symlink is not there.
function resolveArchivePath() {
const raw = process.env.DURIIN_DB
|| config.duriin_db
|| (config.database && config.database.path)
|| './archive.sqlite';
return path.isAbsolute(raw) ? raw : path.resolve(configDir, raw);
}
function resolveIntelligencePath() {
return process.env.INTELLIGENCE_DB
|| (config.intelligence_db
? (path.isAbsolute(config.intelligence_db) ? config.intelligence_db : path.resolve(configDir, config.intelligence_db))
: path.resolve(configDir, 'intelligence.sqlite'));
}
// Opens the archive and *proves* it is the archive before handing it back. Any
// failure throws with the resolved target in the message — serving the wrong
// database silently is far worse than an error on the sql console.
function getArchiveDb() {
if (adb) return adb;
const target = isPostgresEnabled() ? 'postgres schema "archive"' : resolveArchivePath();
try {
if (isPostgresEnabled()) {
const handle = openRuntimeDb(resolveArchivePath(), { schema: 'archive' });
const probe = handle.prepare("SELECT to_regclass('archive.articles') AS relation").get();
if (!probe || !probe.relation) throw new Error('the archive schema has no articles table');
adb = handle;
return adb;
}
const filePath = resolveArchivePath();
if (!fs.existsSync(filePath)) throw new Error('no such file');
const handle = new Database(filePath, { fileMustExist: true });
try {
const probe = handle.prepare("SELECT name FROM sqlite_master WHERE type='table' AND name='articles'").get();
if (!probe) throw new Error('this file has no articles table, so it is not the archive');
} catch (probeError) {
handle.close();
throw probeError;
}
adb = handle;
return adb;
} catch (error) {
// never cached — if the volume shows up later the next request recovers
console.error(`[admin] archive database unavailable (${target}):`, error);
throw new Error(`archive database unavailable (${target}): ${error.message}`);
}
}
function calculateArchiveStats() {
const databasePath = path.resolve(__dirname, '..', '..', config.database.path || './archive.sqlite');
const databasePath = resolveArchivePath();
const workerPath = path.resolve(__dirname, '..', 'adminStatsWorker.js');
return new Promise((resolve, reject) => {
const worker = new Worker(workerPath, { workerData: { databasePath } });
@@ -42,18 +105,134 @@ function calculateArchiveStats() {
function getIntelligenceDb() {
if (idb) return idb;
const configDir = path.resolve(__dirname, '..', '..');
const rawPath = process.env.INTELLIGENCE_DB
|| (config.intelligence_db
? (path.isAbsolute(config.intelligence_db) ? config.intelligence_db : path.resolve(configDir, config.intelligence_db))
: path.resolve(configDir, 'intelligence.sqlite'));
const rawPath = resolveIntelligencePath();
if (!isPostgresEnabled() && !fs.existsSync(rawPath)) return null;
if (!isPostgresEnabled() && !fs.existsSync(rawPath)) {
console.error(`[admin] intelligence database unavailable: no such file (${rawPath})`);
return null;
}
idb = isPostgresEnabled() ? openRuntimeDb(rawPath, { schema: 'intelligence' }) : new Database(rawPath);
return idb;
}
// Prediction origins are not interchangeable. 'live' is genuine real time work,
// 'historical' is coordinator backfill over the archive and 'replay' is
// walk-forward replay. Averaging them into a single accuracy number reads like
// live edge when it is nothing of the sort, so the overview reports them side by
// side and lets the page say "nothing live yet" out loud.
const OUTCOME_ORIGINS = ['live', 'historical', 'replay'];
const OUTCOMES_BY_ORIGIN_SQL = `
SELECT p.origin AS origin,
COUNT(*) AS total,
SUM(o.direction_correct) AS correct,
AVG(o.excess_return) AS average_excess_return
FROM autonomy_outcomes o
JOIN autonomy_predictions p ON p.id = o.prediction_id
GROUP BY p.origin
`;
const PREDICTIONS_BY_ORIGIN_SQL = `
SELECT origin, status, COUNT(*) AS count
FROM autonomy_predictions
GROUP BY origin, status
`;
function summarizeOutcomeOrigins(rows) {
const buckets = new Map();
for (const name of OUTCOME_ORIGINS) {
buckets.set(name, { origin: name, total: 0, correct: 0, average_excess_return: null });
}
for (const row of rows || []) {
const origin = String(row.origin || 'unknown').toLowerCase();
if (!buckets.has(origin)) buckets.set(origin, { origin, total: 0, correct: 0, average_excess_return: null });
const bucket = buckets.get(origin);
bucket.total = Number(row.total || 0);
bucket.correct = Number(row.correct || 0);
bucket.average_excess_return = row.average_excess_return == null ? null : Number(row.average_excess_return);
}
const byOrigin = [...buckets.values()];
const live = byOrigin.find((bucket) => bucket.origin === 'live');
return { byOrigin, live };
}
// Diversification gate, mirrored from the policy layer. A cohort only earns the
// right to authorise a trade when it is big enough, spread over enough tickers
// and not dominated by a single one. Missing diversity data counts as a fail —
// the policy treats unknown as disqualifying and the admin view has to agree,
// otherwise the screen says "qualified" while the trader abstains.
const CALIBRATION_MIN_SAMPLES = 30;
const CALIBRATION_MIN_INSTRUMENTS = 5;
const CALIBRATION_MAX_CONCENTRATION = 0.5;
const CALIBRATION_BASE_COLUMNS = [
'cohort_key', 'sample_size', 'effective_sample_size', 'directional_probability',
'expected_excess_return', 'lower_return', 'upper_return', 'created_at',
];
// added by the diversification work; pre-existing rows/deployments may not have
// them yet so they are selected only when they really exist
const CALIBRATION_OPTIONAL_COLUMNS = ['distinct_instruments', 'top_instrument_share', 'source'];
function calibrationSnapshotSql(available) {
const columns = CALIBRATION_BASE_COLUMNS.slice();
for (const name of CALIBRATION_OPTIONAL_COLUMNS) {
columns.push(available.has(name) ? name : `NULL AS ${name}`);
}
return `SELECT ${columns.join(', ')} FROM autonomy_calibration_snapshots ORDER BY id DESC LIMIT 8`;
}
function gate(value, ok, threshold) {
return { value: value == null ? null : Number(value), threshold, ok, known: value != null };
}
function decorateCalibration(rows) {
return (rows || []).map((row) => {
const samples = row.sample_size == null ? null : Number(row.sample_size);
const instruments = row.distinct_instruments == null ? null : Number(row.distinct_instruments);
const share = row.top_instrument_share == null ? null : Number(row.top_instrument_share);
const checks = {
sample_size: gate(samples, samples != null && samples >= CALIBRATION_MIN_SAMPLES, CALIBRATION_MIN_SAMPLES),
distinct_instruments: gate(instruments, instruments != null && instruments >= CALIBRATION_MIN_INSTRUMENTS, CALIBRATION_MIN_INSTRUMENTS),
top_instrument_share: gate(share, share != null && share <= CALIBRATION_MAX_CONCENTRATION, CALIBRATION_MAX_CONCENTRATION),
};
const reasons = [];
if (!checks.sample_size.ok) {
reasons.push(samples == null ? 'sample size unknown' : `only ${samples} samples, needs ${CALIBRATION_MIN_SAMPLES}`);
}
if (!checks.distinct_instruments.ok) {
reasons.push(instruments == null ? 'instrument spread unknown' : `only ${instruments} distinct ticker${instruments === 1 ? '' : 's'}, needs ${CALIBRATION_MIN_INSTRUMENTS}`);
}
if (!checks.top_instrument_share.ok) {
reasons.push(share == null ? 'concentration unknown' : `${Math.round(share * 100)}% sits in one ticker, cap is ${Math.round(CALIBRATION_MAX_CONCENTRATION * 100)}%`);
}
// cohort keys are versioned now. legacy rows use the old key shape and will
// never match a current lookup, so they must not read as live calibration.
const legacy = !String(row.cohort_key || '').startsWith('v2|');
return {
...row,
source: row.source || null,
legacy_cohort_key: legacy,
qualification: { qualified: reasons.length === 0, checks, reasons },
};
});
}
function normalizeOriginCounts(rows) {
return (rows || []).map((row) => ({
origin: String(row.origin || 'unknown').toLowerCase(),
status: row.status,
count: Number(row.count || 0),
}));
}
const adminUser = (config.admin && config.admin.username) || 'admin';
const adminPass = (config.admin && config.admin.password) || 'changeme';
@@ -156,12 +335,18 @@ async function adminRoutes(fastify) {
if (isPostgresEnabled()) {
const hasSchema = await pg.get('intelligence', "SELECT 1 FROM information_schema.tables WHERE table_schema = $1 AND table_name = $2", ['intelligence', 'autonomy_jobs']);
if (!hasSchema) return { enabled: false, reason: 'autonomy schema is not initialized' };
const [jobs, predictionCounts, decisionCounts, proposalCounts, outcomeSummary, instruments, latestRows, latestOrders, account, calibration, replay] = await Promise.all([
const calibrationColumns = new Set((await pg.all('intelligence',
'SELECT column_name FROM information_schema.columns WHERE table_schema = $1 AND table_name = $2',
['intelligence', 'autonomy_calibration_snapshots'])).map((row) => row.column_name));
const [jobs, predictionCounts, predictionOriginRows, decisionCounts, proposalCounts, outcomeRows, instruments, latestRows, latestOrders, account, calibration, replay] = await Promise.all([
pg.all('intelligence', 'SELECT lane, status, COUNT(*) AS count FROM autonomy_jobs GROUP BY lane, status ORDER BY lane, status'),
pg.all('intelligence', 'SELECT status, COUNT(*) AS count FROM autonomy_predictions GROUP BY status'),
pg.all('intelligence', PREDICTIONS_BY_ORIGIN_SQL),
pg.all('intelligence', 'SELECT action, COUNT(*) AS count FROM autonomy_decisions GROUP BY action'),
pg.all('intelligence', 'SELECT status, COUNT(*) AS count FROM autonomy_proposals GROUP BY status'),
pg.get('intelligence', "SELECT COUNT(*) AS total, SUM(direction_correct) AS correct, AVG(excess_return) AS average_excess_return FROM autonomy_outcomes o JOIN autonomy_predictions p ON p.id = o.prediction_id WHERE p.origin = 'live'"),
pg.all('intelligence', OUTCOMES_BY_ORIGIN_SQL),
pg.get('intelligence', 'SELECT COUNT(*) AS count FROM autonomy_instruments WHERE active=1 AND tradable=1'),
pg.all('intelligence', `
SELECT p.id, p.instrument, p.direction, p.event_type, p.causal_channel,
@@ -185,7 +370,7 @@ async function adminRoutes(fastify) {
ORDER BY oi.id DESC LIMIT 12
`),
pg.get('intelligence', 'SELECT broker, equity, cash, buying_power, captured_at FROM autonomy_account_snapshots ORDER BY id DESC LIMIT 1'),
pg.all('intelligence', 'SELECT cohort_key, sample_size, effective_sample_size, directional_probability, expected_excess_return, lower_return, upper_return, created_at FROM autonomy_calibration_snapshots ORDER BY id DESC LIMIT 8'),
pg.all('intelligence', calibrationSnapshotSql(calibrationColumns)),
pg.get('intelligence', `
SELECT r.id, r.status, r.watermark_at, r.cursor_article_id, r.cursor_effective_at,
r.processed_articles, r.updated_at,
@@ -205,13 +390,18 @@ async function adminRoutes(fastify) {
const { evidence_article_ids: ignored, ...safeRow } = row;
return { ...safeRow, evidence_count: evidenceCount };
});
const origins = summarizeOutcomeOrigins(outcomeRows);
return {
enabled: true,
mode: process.env.AUTONOMY_EXECUTION_MODE || 'shadow',
broker: { name: 'Alpaca Paper', configured: Boolean(process.env.ALPACA_PAPER_KEY_ID && process.env.ALPACA_PAPER_SECRET_KEY) },
jobs, predictionCounts, decisionCounts, proposalCounts, outcomes: outcomeSummary,
jobs, predictionCounts, decisionCounts, proposalCounts,
predictionsByOrigin: normalizeOriginCounts(predictionOriginRows),
outcomes: origins.live,
outcomesByOrigin: origins.byOrigin,
hasLiveOutcomes: origins.live.total > 0,
allowlistedInstruments: instruments.count, latestPredictions, latestOrders,
account: account || null, calibration, replay: replay || null,
account: account || null, calibration: decorateCalibration(calibration), replay: replay || null,
generatedAt: new Date().toISOString(),
};
}
@@ -235,13 +425,8 @@ async function adminRoutes(fastify) {
const proposalCounts = intelligenceDb.prepare(`
SELECT status, COUNT(*) AS count FROM autonomy_proposals GROUP BY status
`).all();
const outcomeSummary = intelligenceDb.prepare(`
SELECT COUNT(*) AS total, SUM(direction_correct) AS correct,
AVG(excess_return) AS average_excess_return
FROM autonomy_outcomes o
JOIN autonomy_predictions p ON p.id = o.prediction_id
WHERE p.origin = 'live'
`).get();
const predictionOriginRows = intelligenceDb.prepare(PREDICTIONS_BY_ORIGIN_SQL).all();
const origins = summarizeOutcomeOrigins(intelligenceDb.prepare(OUTCOMES_BY_ORIGIN_SQL).all());
const instruments = intelligenceDb.prepare(`
SELECT COUNT(*) AS count FROM autonomy_instruments WHERE active=1 AND tradable=1
`).get();
@@ -275,11 +460,12 @@ async function adminRoutes(fastify) {
SELECT broker, equity, cash, buying_power, captured_at
FROM autonomy_account_snapshots ORDER BY id DESC LIMIT 1
`).get() || null;
const calibration = intelligenceDb.prepare(`
SELECT cohort_key, sample_size, effective_sample_size, directional_probability,
expected_excess_return, lower_return, upper_return, created_at
FROM autonomy_calibration_snapshots ORDER BY id DESC LIMIT 8
`).all();
const calibrationColumns = new Set(
intelligenceDb.prepare('PRAGMA table_info(autonomy_calibration_snapshots)').all().map((row) => row.name)
);
const calibration = decorateCalibration(
intelligenceDb.prepare(calibrationSnapshotSql(calibrationColumns)).all()
);
const replay = intelligenceDb.prepare(`
SELECT r.id, r.status, r.watermark_at, r.cursor_article_id, r.cursor_effective_at,
r.processed_articles, r.updated_at,
@@ -302,9 +488,12 @@ async function adminRoutes(fastify) {
},
jobs,
predictionCounts,
predictionsByOrigin: normalizeOriginCounts(predictionOriginRows),
decisionCounts,
proposalCounts,
outcomes: outcomeSummary,
outcomes: origins.live,
outcomesByOrigin: origins.byOrigin,
hasLiveOutcomes: origins.live.total > 0,
allowlistedInstruments: instruments.count,
latestPredictions,
latestOrders,
@@ -918,8 +1107,29 @@ async function adminRoutes(fastify) {
const { sql, database } = request.body || {};
if (!sql || !sql.trim()) { reply.code(400); return { error: 'no sql provided' }; }
const target = database === 'intelligence' ? getIntelligenceDb() : db;
if (!target) { reply.code(400); return { error: 'database not available' }; }
// empty/omitted means archive, the historic default. anything else has to be
// spelled correctly — a typo used to silently run against the archive.
const requested = String(database || 'archive').trim().toLowerCase() || 'archive';
if (requested !== 'archive' && requested !== 'intelligence') {
reply.code(400);
return { error: `unknown database "${requested}" — expected "archive" or "intelligence"` };
}
let target = null;
try {
target = requested === 'intelligence' ? getIntelligenceDb() : getArchiveDb();
} catch (error) {
console.error(`[admin] sql console cannot reach the ${requested} database:`, error);
reply.code(503);
return { error: error.message };
}
if (!target) {
const where = isPostgresEnabled() ? `postgres schema "${requested}"` : resolveIntelligencePath();
console.error(`[admin] sql console cannot reach the ${requested} database (${where})`);
reply.code(503);
return { error: `${requested} database unavailable (${where})` };
}
// split on semicolons, drop empty statements
const statements = sql.split(';').map(s => s.trim()).filter(s => s.length > 0);
@@ -930,13 +1140,18 @@ async function adminRoutes(fastify) {
for (const s of statements) {
try {
const stmt = target.prepare(s);
if (stmt.reader) {
// the postgres adapters dont expose better-sqlite3's `reader` flag, so
// without this fallback every SELECT went down the run() path and came
// back as a change count with no rows at all
const reads = typeof stmt.reader === 'boolean' ? stmt.reader : /^\s*(SELECT|WITH|PRAGMA|EXPLAIN|SHOW)\b/i.test(s);
if (reads) {
results.push({ sql: s, rows: stmt.all() });
} else {
const info = stmt.run();
results.push({ sql: s, changes: info.changes, lastInsertRowid: info.lastInsertRowid });
}
} catch (err) {
console.error(`[admin] sql console statement failed on ${requested}:`, s, err);
results.push({ sql: s, error: err.message });
}
}