fix: restart the stalled autonomy pipeline and make calibration honest
Archive ingestion had been dead since 2026-08-02 because nothing in the compose stack actually ran it. Everything downstream starved from there. - add ingest + enrichment services. server.js only starts the scheduler when DURIIN_RUN_SCHEDULER is not "false", and workers/index.js was not running at all, so articles never got event_id/content/has_embedding and the coordinator had nothing to lease. - pass an explicit origin from coordinatorWorker. it was never passed, so acceptProposal defaulted to 'live' and 464 historical backfill predictions were recorded as live. that also meant verifyEvidence got a null cutoff and skipped its date check entirely. - coarsen cohortKey to event families + horizon buckets. 201 free text event types produced 221 cohorts averaging 2.76 samples, so the n>=30 gate could never be reached and everything abstained for the wrong reason. - gate on cohort diversity, not just sample count. one ticker was roughly half of all resolved outcomes, so a pure count gate was measuring one company. unknown diversity abstains rather than passing. - resolve the admin archive db explicitly and probe it. it relied on a Dockerfile symlink, and without it better-sqlite3 quietly creates an empty file and serves a phantom archive. - clamp implausible future publication dates at ingest. - pin the db backend to sqlite by default. compose hardcoded postgres "true", which would have overridden the operator's own .env on the next redeploy and pointed everything at a stale snapshot. scripts/repair-autonomy-labels.js relabels the affected rows. it is dry run by default and has not been applied. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
+12
-6
@@ -12,7 +12,7 @@ const { calculateOutcome } = require('../src/autonomy/outcomes');
|
||||
const { createOrderIntent } = require('../src/autonomy/orderIntents');
|
||||
const { enqueueCoordinatorEvent, reconcileArchiveBatch, reconcileLiveBatch } = require('../workers/autonomyWorker');
|
||||
const { scheduleNext } = require('../workers/replayWorker');
|
||||
const { refreshHistoricalCalibration, createDecisions } = require('../workers/calibrationWorker');
|
||||
const { refreshHistoricalCalibration, createDecisions, ensureCalibrationColumns } = require('../workers/calibrationWorker');
|
||||
|
||||
test('autonomy schema and leased jobs are restart-safe', () => {
|
||||
const db = new Database(':memory:');
|
||||
@@ -117,8 +117,11 @@ test('calibration and policy abstain on insufficient evidence', () => {
|
||||
assert.equal(calibration.sampleSize, 2);
|
||||
const result = decide({ ...calibration }, { minSampleSize: 30 });
|
||||
assert.equal(result.action, 'ABSTAIN');
|
||||
assert.equal(cohortKey({ sector: 'tech', eventType: 'earnings', horizonDays: 10, direction: 'positive' }), 'tech|earnings|10|positive');
|
||||
assert.equal(decide({ direction: 'negative', probability: 0.8, expectedExcessReturn: -0.02, lowerReturn: -0.04, sampleSize: 40 }).action, 'SELL');
|
||||
assert.equal(cohortKey({ sector: 'tech', eventType: 'earnings', horizonDays: 10, direction: 'positive' }), 'v2|tech|earnings|medium|positive');
|
||||
assert.equal(decide({
|
||||
direction: 'negative', probability: 0.8, expectedExcessReturn: -0.02, lowerReturn: -0.04,
|
||||
sampleSize: 40, distinctInstruments: 9, topInstrumentShare: 0.25,
|
||||
}).action, 'SELL');
|
||||
});
|
||||
|
||||
test('historical replay outcomes create replay calibration snapshots', () => {
|
||||
@@ -136,8 +139,9 @@ test('historical replay outcomes create replay calibration snapshots', () => {
|
||||
VALUES (?, 0.04, 1)
|
||||
`).run(prediction.lastInsertRowid);
|
||||
|
||||
assert.equal(refreshHistoricalCalibration(db, 'test-cal'), 1);
|
||||
const snapshot = db.prepare("SELECT source, replay_run_id, sample_size, directional_probability FROM autonomy_calibration_snapshots").get();
|
||||
// one per-run replay snapshot plus the pooled historical/replay snapshot
|
||||
assert.equal(refreshHistoricalCalibration(db, 'test-cal'), 2);
|
||||
const snapshot = db.prepare("SELECT source, replay_run_id, sample_size, directional_probability FROM autonomy_calibration_snapshots WHERE source='replay'").get();
|
||||
assert.equal(snapshot.source, 'replay');
|
||||
assert.equal(snapshot.replay_run_id, 7);
|
||||
assert.equal(snapshot.sample_size, 1);
|
||||
@@ -154,12 +158,14 @@ test('live decisions map calibration snapshot fields into policy inputs', () =>
|
||||
learning_eligible, strategy_version, origin, status)
|
||||
VALUES (?, 'NVDA', 'positive', 'earnings', 10, datetime('now'), '[1]', 1, 'test', 'live', 'open')
|
||||
`).run(proposal.lastInsertRowid);
|
||||
ensureCalibrationColumns(db);
|
||||
db.prepare(`
|
||||
INSERT INTO autonomy_calibration_snapshots
|
||||
(cohort_key, sample_size, effective_sample_size, directional_probability, expected_excess_return,
|
||||
lower_return, upper_return, parent_cohort_key, version, source)
|
||||
VALUES ('unknown|earnings|10|positive', 40, 42, 0.7, 0.02, -0.01, 0.06, NULL, 'test-cal', 'replay')
|
||||
VALUES ('v2|unknown|earnings|medium|positive', 40, 42, 0.7, 0.02, -0.01, 0.06, NULL, 'test-cal', 'historical')
|
||||
`).run();
|
||||
db.prepare("UPDATE autonomy_calibration_snapshots SET distinct_instruments = 11, top_instrument_share = 0.2").run();
|
||||
|
||||
assert.equal(createDecisions(db), 1);
|
||||
const decision = db.prepare('SELECT * FROM autonomy_decisions WHERE prediction_id=?').get(prediction.lastInsertRowid);
|
||||
|
||||
Reference in New Issue
Block a user