fix: restart the stalled autonomy pipeline and make calibration honest

Archive ingestion had been dead since 2026-08-02 because nothing in the
compose stack actually ran it. Everything downstream starved from there.

- add ingest + enrichment services. server.js only starts the scheduler when
  DURIIN_RUN_SCHEDULER is not "false", and workers/index.js was not running at
  all, so articles never got event_id/content/has_embedding and the coordinator
  had nothing to lease.
- pass an explicit origin from coordinatorWorker. it was never passed, so
  acceptProposal defaulted to 'live' and 464 historical backfill predictions
  were recorded as live. that also meant verifyEvidence got a null cutoff and
  skipped its date check entirely.
- coarsen cohortKey to event families + horizon buckets. 201 free text event
  types produced 221 cohorts averaging 2.76 samples, so the n>=30 gate could
  never be reached and everything abstained for the wrong reason.
- gate on cohort diversity, not just sample count. one ticker was roughly half
  of all resolved outcomes, so a pure count gate was measuring one company.
  unknown diversity abstains rather than passing.
- resolve the admin archive db explicitly and probe it. it relied on a
  Dockerfile symlink, and without it better-sqlite3 quietly creates an empty
  file and serves a phantom archive.
- clamp implausible future publication dates at ingest.
- pin the db backend to sqlite by default. compose hardcoded postgres "true",
  which would have overridden the operator's own .env on the next redeploy and
  pointed everything at a stale snapshot.

scripts/repair-autonomy-labels.js relabels the affected rows. it is dry run by
default and has not been applied.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-08-29 21:43:24 +01:00
co-authored by Claude Opus 5
parent d778a02bfb
commit 6f1d1eee2d
19 changed files with 1497 additions and 100 deletions
+11 -7
View File
@@ -36,18 +36,22 @@ function normalizeProposal(raw, { informationCutoff, model = 'unknown', promptVe
function verifyEvidence(archiveDb, articleIds, informationCutoff = null) {
const placeholders = articleIds.map(() => '?').join(',');
// A replay must only see material which existed at its information cutoff.
// Live proposals retain the simpler existence check.
// No proposal, whatever lane produced it, may cite material which did not yet
// exist at its own information cutoff. This used to be a replay-only rule and
// that was a lookahead hole for every other origin.
const cutoffClause = informationCutoff ? ' AND datetime(COALESCE(pub_date_effective, pub_date, ingested_at)) <= datetime(?)' : '';
let rows;
try {
rows = archiveDb.prepare(`SELECT id FROM articles WHERE id IN (${placeholders})${cutoffClause}`)
.all(...articleIds, ...(informationCutoff ? [informationCutoff] : []));
} catch (error) {
// Minimal/test archives may not retain publication metadata. A production
// replay archive is required to have it, so this fallback is only for the
// existing live evidence contract.
if (informationCutoff) throw error;
// Minimal/test archives may not retain publication metadata at all, in which
// case the cutoff clause cannot even be prepared. We degrade to a plain
// existence check rather than blocking the pipeline, but the degredation is
// never silent - a production archive missing these columns is a real bug.
console.warn('[coordinator] evidence cutoff check unavailable, falling back to existence only.',
`cutoff=${informationCutoff} articles=${JSON.stringify(articleIds)} reason=${error && error.message}`);
if (error && error.stack) console.warn(error.stack);
rows = archiveDb.prepare(`SELECT id FROM articles WHERE id IN (${placeholders})`).all(...articleIds);
}
const found = new Set(rows.map((row) => row.id));
@@ -57,7 +61,7 @@ function verifyEvidence(archiveDb, articleIds, informationCutoff = null) {
function acceptProposal(intelligenceDb, archiveDb, raw, metadata = {}) {
const proposal = normalizeProposal(raw, metadata);
for (const prediction of proposal.predictions) {
if (!verifyEvidence(archiveDb, prediction.evidenceArticleIds, metadata.origin === 'replay' ? proposal.informationCutoff : null)) {
if (!verifyEvidence(archiveDb, prediction.evidenceArticleIds, proposal.informationCutoff)) {
throw new Error(`proposal references missing evidence for ${prediction.instrument}`);
}
const instrument = intelligenceDb.prepare(