fix: restart the stalled autonomy pipeline and make calibration honest

Archive ingestion had been dead since 2026-08-02 because nothing in the
compose stack actually ran it. Everything downstream starved from there.

- add ingest + enrichment services. server.js only starts the scheduler when
  DURIIN_RUN_SCHEDULER is not "false", and workers/index.js was not running at
  all, so articles never got event_id/content/has_embedding and the coordinator
  had nothing to lease.
- pass an explicit origin from coordinatorWorker. it was never passed, so
  acceptProposal defaulted to 'live' and 464 historical backfill predictions
  were recorded as live. that also meant verifyEvidence got a null cutoff and
  skipped its date check entirely.
- coarsen cohortKey to event families + horizon buckets. 201 free text event
  types produced 221 cohorts averaging 2.76 samples, so the n>=30 gate could
  never be reached and everything abstained for the wrong reason.
- gate on cohort diversity, not just sample count. one ticker was roughly half
  of all resolved outcomes, so a pure count gate was measuring one company.
  unknown diversity abstains rather than passing.
- resolve the admin archive db explicitly and probe it. it relied on a
  Dockerfile symlink, and without it better-sqlite3 quietly creates an empty
  file and serves a phantom archive.
- clamp implausible future publication dates at ingest.
- pin the db backend to sqlite by default. compose hardcoded postgres "true",
  which would have overridden the operator's own .env on the next redeploy and
  pointed everything at a stale snapshot.

scripts/repair-autonomy-labels.js relabels the affected rows. it is dry run by
default and has not been applied.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-08-29 21:43:24 +01:00
co-authored by Claude Opus 5
parent d778a02bfb
commit 6f1d1eee2d
19 changed files with 1497 additions and 100 deletions
+1
View File
@@ -53,6 +53,7 @@
.decision-chip { justify-self: end; padding: 5px 8px; color: var(--warning); background: rgba(243,201,105,.06); border: 1px solid rgba(243,201,105,.16); border-radius: 3px; font-family: var(--mono); font-size: 9px; font-weight: 750; }
.decision-chip.buy { color: var(--positive); border-color: rgba(142,230,168,.18); background: rgba(142,230,168,.06); }
.decision-chip.sell { color: var(--negative); border-color: rgba(255,141,125,.18); background: rgba(255,141,125,.06); }
.decision-chip.qualifies { color: var(--positive); border-color: rgba(142,230,168,.18); background: rgba(142,230,168,.06); }
.performance-panel { padding-bottom: 18px; }
.accuracy-orbit { --accuracy: 0deg; width: 166px; height: 166px; margin: 28px auto 24px; padding: 1px; display: grid; place-items: center; border-radius: 50%; background: conic-gradient(var(--accent) var(--accuracy), #252b24 0); }
+82 -2
View File
@@ -35,11 +35,80 @@
</tr>`).join("");
}
const ORIGIN_LABELS = {
live: "Live",
historical: "Historical backfill",
replay: "Walk-forward replay",
};
// live, historical and replay are deliberately never blended. only the live
// row is an edge claim, the other two are how the model was taught.
function renderOriginSplit(byOrigin, livePredictions) {
const host = byId("origin-split");
if (!host) return;
const rows = (byOrigin || []).filter(row => row.origin !== "live");
if (!rows.length) {
host.innerHTML = "";
return;
}
host.innerHTML = rows.map(row => {
const total = Number(row.total || 0);
const label = ORIGIN_LABELS[row.origin] || row.origin;
const accuracy = total ? formatPercent(Number(row.correct || 0) / total) : "—";
return `<div><span>${escapeHtml(label)}</span><strong>${formatNumber(total)} measured · ${accuracy}</strong></div>`;
}).join("");
byId("origin-note").textContent = livePredictions
? "Historical backfill and walk-forward replay are listed separately. Neither counts toward live edge."
: "No live predictions exist yet, so the numbers above are training and replay only — not evidence of live edge.";
}
function cohortCell(check, format) {
if (!check || !check.known) return '<td class="mono muted">unknown</td>';
return `<td class="mono ${check.ok ? "positive" : "negative"}">${format(check.value)} / ${format(check.threshold)}</td>`;
}
function renderCohorts(rows) {
const host = byId("cohort-list");
if (!host) return;
if (!rows?.length) {
host.innerHTML = '<tr><td colspan="6" class="empty-state">No calibration snapshots yet.</td></tr>';
return;
}
host.innerHTML = rows.map(row => {
const checks = row.qualification?.checks || {};
const qualified = Boolean(row.qualification?.qualified);
const reasons = row.qualification?.reasons || [];
const status = qualified
? '<span class="decision-chip qualifies">QUALIFIES</span>'
: `<span class="decision-chip abstain">ABSTAIN</span><div class="hypothesis-channel">${escapeHtml(reasons.join(" · ") || "does not qualify")}</div>`;
const key = row.legacy_cohort_key
? `${escapeHtml(row.cohort_key || "—")}<div class="hypothesis-channel">legacy key, not comparable to current cohorts</div>`
: escapeHtml(row.cohort_key || "—");
return `<tr>
<td class="mono">${key}</td>
<td class="muted">${escapeHtml(row.source || "unknown")}</td>
${cohortCell(checks.sample_size, value => formatNumber(value))}
${cohortCell(checks.distinct_instruments, value => formatNumber(value))}
${cohortCell(checks.top_instrument_share, value => formatPercent(value, 0))}
<td>${status}</td>
</tr>`;
}).join("");
}
function render(data) {
if (!data.enabled) throw new Error(data.reason || "Autonomy is unavailable");
const mode = String(data.mode || "shadow").toUpperCase();
const open = count(data.predictionCounts, "status", "open");
const resolved = count(data.predictionCounts, "status", "resolved");
const byOrigin = data.outcomesByOrigin || [];
const livePredictions = (data.predictionsByOrigin || []).filter(row => row.origin === "live")
.reduce((total, row) => total + Number(row.count || 0), 0);
const outcomes = Number(data.outcomes?.total || 0);
const correct = Number(data.outcomes?.correct || 0);
const accuracy = outcomes ? correct / outcomes : null;
@@ -61,7 +130,9 @@
byId("metric-open").textContent = formatNumber(open);
byId("metric-resolved").textContent = `${formatNumber(resolved)} resolved`;
byId("metric-accuracy").textContent = formatPercent(accuracy);
byId("metric-sample").textContent = outcomes ? `${formatNumber(outcomes)} measured outcomes` : "Waiting for outcomes";
byId("metric-sample").textContent = outcomes
? `${formatNumber(outcomes)} measured live outcomes`
: (livePredictions ? "Live predictions have not matured yet" : "No live predictions yet");
byId("metric-alpha").textContent = formatPercent(data.outcomes?.average_excess_return, 2);
byId("metric-universe").textContent = formatNumber(data.allowlistedInstruments, true);
@@ -77,7 +148,16 @@
byId("perf-resolved").textContent = formatNumber(outcomes);
byId("perf-correct").textContent = formatNumber(correct);
byId("perf-cohorts").textContent = formatNumber(data.calibration?.length || 0);
if (outcomes) byId("performance-note").textContent = `Measured on ${outcomes} matured predictions. Results remain descriptive until the sample is large enough for stable calibration.`;
if (outcomes) {
byId("performance-note").textContent = `Measured on ${outcomes} matured live predictions. Results remain descriptive until the sample is large enough for stable calibration.`;
} else if (livePredictions) {
byId("performance-note").textContent = `No live prediction has matured yet — ${formatNumber(livePredictions)} are still open. Nothing here is a live track record.`;
} else {
byId("performance-note").textContent = "No live predictions yet. Everything measured so far is historical backfill or replay, which is training, not a live track record.";
}
renderOriginSplit(byOrigin, livePredictions);
renderCohorts(data.calibration);
const replay = data.replay;
byId("replay-status").textContent = replay ? replay.status : "Not started";
+14 -2
View File
@@ -8,7 +8,7 @@
<link rel="stylesheet" href="/admin/assets/css/base.css?v=20260804-3">
<link rel="stylesheet" href="/admin/assets/css/layout.css?v=20260804-3">
<link rel="stylesheet" href="/admin/assets/css/components.css?v=20260804-3">
<link rel="stylesheet" href="/admin/assets/css/autonomy.css?v=20260804-3">
<link rel="stylesheet" href="/admin/assets/css/autonomy.css?v=20260829-1">
</head>
<body class="page-autonomy">
@@ -98,6 +98,8 @@
<div><span>Calibration cohorts</span><strong id="perf-cohorts">0</strong></div>
</div>
<p class="performance-note" id="performance-note">Duriin will only claim an edge after predictions mature and are measured out of sample.</p>
<div class="performance-facts" id="origin-split"></div>
<p class="performance-note" id="origin-note">Historical backfill and walk-forward replay are shown separately. Neither is evidence of live edge.</p>
</aside>
</div>
@@ -138,10 +140,20 @@
<div><span>Watermark</span><strong id="replay-watermark">—</strong></div>
</div>
</section>
<section class="panel" aria-label="Calibration cohorts">
<div class="section-head"><div><span class="section-index">07</span><h3>Calibration cohorts</h3></div><span class="mode-pill">Needs 30 samples · 5 tickers · max 50% in one</span></div>
<div class="table-wrap">
<table>
<thead><tr><th>Cohort</th><th>Source</th><th>Samples</th><th>Distinct tickers</th><th>Top ticker share</th><th>Status</th></tr></thead>
<tbody id="cohort-list"><tr><td colspan="6" class="empty-state">No calibration snapshots yet.</td></tr></tbody>
</table>
</div>
</section>
</main>
<div id="toast"><span class="toast-dot"></span><span id="toast-msg"></span></div>
<script src="/admin/assets/js/app.js?v=20260804-3"></script>
<script src="/admin/assets/js/autonomy.js?v=20260804-3"></script>
<script src="/admin/assets/js/autonomy.js?v=20260829-1"></script>
</body>
</html>