fix: restart the stalled autonomy pipeline and make calibration honest
Archive ingestion had been dead since 2026-08-02 because nothing in the compose stack actually ran it. Everything downstream starved from there. - add ingest + enrichment services. server.js only starts the scheduler when DURIIN_RUN_SCHEDULER is not "false", and workers/index.js was not running at all, so articles never got event_id/content/has_embedding and the coordinator had nothing to lease. - pass an explicit origin from coordinatorWorker. it was never passed, so acceptProposal defaulted to 'live' and 464 historical backfill predictions were recorded as live. that also meant verifyEvidence got a null cutoff and skipped its date check entirely. - coarsen cohortKey to event families + horizon buckets. 201 free text event types produced 221 cohorts averaging 2.76 samples, so the n>=30 gate could never be reached and everything abstained for the wrong reason. - gate on cohort diversity, not just sample count. one ticker was roughly half of all resolved outcomes, so a pure count gate was measuring one company. unknown diversity abstains rather than passing. - resolve the admin archive db explicitly and probe it. it relied on a Dockerfile symlink, and without it better-sqlite3 quietly creates an empty file and serves a phantom archive. - clamp implausible future publication dates at ingest. - pin the db backend to sqlite by default. compose hardcoded postgres "true", which would have overridden the operator's own .env on the next redeploy and pointed everything at a stale snapshot. scripts/repair-autonomy-labels.js relabels the affected rows. it is dry run by default and has not been applied. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
@@ -0,0 +1,112 @@
|
||||
const test = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const Database = require('better-sqlite3');
|
||||
|
||||
const { initAutonomySchema } = require('../src/autonomy/schema');
|
||||
const { cohortKey } = require('../src/autonomy/calibration');
|
||||
const {
|
||||
refreshCalibration,
|
||||
refreshHistoricalCalibration,
|
||||
createDecisions,
|
||||
calibrationHealth,
|
||||
} = require('../workers/calibrationWorker');
|
||||
|
||||
function seedDb() {
|
||||
const db = new Database(':memory:');
|
||||
initAutonomySchema(db);
|
||||
db.prepare("INSERT INTO autonomy_proposals(payload, information_cutoff, status) VALUES ('{}', '2026-01-01T00:00:00Z', 'accepted')").run();
|
||||
return db;
|
||||
}
|
||||
|
||||
function addPrediction(db, { instrument, direction = 'positive', eventType = 'earnings_beat', horizonDays = 10,
|
||||
origin = 'live', status = 'resolved', learningEligible = 0, replayRunId = null, excessReturn = null, correct = null }) {
|
||||
const prediction = db.prepare(`
|
||||
INSERT INTO autonomy_predictions
|
||||
(proposal_id, instrument, direction, event_type, horizon_days, information_cutoff, evidence_article_ids,
|
||||
learning_eligible, strategy_version, origin, replay_run_id, status)
|
||||
VALUES (1, ?, ?, ?, ?, '2026-01-01T00:00:00Z', '[1]', ?, 'test', ?, ?, ?)
|
||||
`).run(instrument, direction, eventType, horizonDays, learningEligible, origin, replayRunId, status);
|
||||
if (excessReturn !== null) {
|
||||
db.prepare('INSERT INTO autonomy_outcomes(prediction_id, excess_return, direction_correct) VALUES (?, ?, ?)')
|
||||
.run(prediction.lastInsertRowid, excessReturn, correct);
|
||||
}
|
||||
return prediction.lastInsertRowid;
|
||||
}
|
||||
|
||||
test('live calibration no longer starves on the never-set learning_eligible flag', () => {
|
||||
const db = seedDb();
|
||||
addPrediction(db, { instrument: 'NVDA', excessReturn: 0.03, correct: 1 });
|
||||
addPrediction(db, { instrument: 'AMD', excessReturn: -0.01, correct: 0 });
|
||||
|
||||
assert.equal(refreshCalibration(db, 'live-cal'), 1);
|
||||
const snapshot = db.prepare("SELECT * FROM autonomy_calibration_snapshots WHERE source='live'").get();
|
||||
assert.equal(snapshot.sample_size, 2);
|
||||
assert.equal(snapshot.distinct_instruments, 2);
|
||||
|
||||
// the old behaviour is still reachable on purpose, for once the flag is populated
|
||||
assert.equal(refreshCalibration(db, 'strict-cal', { requireLearningEligible: true }), 0);
|
||||
|
||||
// counters report snapshots written, so a steady state poll is genuinely quiet
|
||||
assert.equal(refreshCalibration(db, 'live-cal'), 0);
|
||||
assert.equal(db.prepare("SELECT COUNT(*) c FROM autonomy_calibration_snapshots WHERE source='live'").get().c, 1);
|
||||
});
|
||||
|
||||
test('historical calibration pools origin historical and replay together', () => {
|
||||
const db = seedDb();
|
||||
addPrediction(db, { instrument: 'NVDA', origin: 'historical', excessReturn: 0.02, correct: 1 });
|
||||
addPrediction(db, { instrument: 'AMD', origin: 'historical', excessReturn: 0.01, correct: 1 });
|
||||
addPrediction(db, { instrument: 'INTC', origin: 'replay', replayRunId: 3, excessReturn: -0.02, correct: 0 });
|
||||
|
||||
refreshHistoricalCalibration(db, 'hist-cal');
|
||||
const pooled = db.prepare("SELECT * FROM autonomy_calibration_snapshots WHERE source='historical'").get();
|
||||
assert.equal(pooled.sample_size, 3, 'the historical lane must not drop the relabelled rows');
|
||||
assert.equal(pooled.distinct_instruments, 3);
|
||||
const perRun = db.prepare("SELECT * FROM autonomy_calibration_snapshots WHERE source='replay'").get();
|
||||
assert.equal(perRun.replay_run_id, 3);
|
||||
assert.equal(perRun.sample_size, 1);
|
||||
});
|
||||
|
||||
test('decisions are only written for open live predictions', () => {
|
||||
const db = seedDb();
|
||||
const open = addPrediction(db, { instrument: 'NVDA', status: 'open' });
|
||||
addPrediction(db, { instrument: 'AMD', status: 'resolved', excessReturn: 0.01, correct: 1 });
|
||||
addPrediction(db, { instrument: 'INTC', status: 'open', origin: 'historical' });
|
||||
addPrediction(db, { instrument: 'MU', status: 'open', origin: 'replay', replayRunId: 3 });
|
||||
|
||||
assert.equal(createDecisions(db), 1);
|
||||
const rows = db.prepare('SELECT prediction_id, action FROM autonomy_decisions').all();
|
||||
assert.equal(rows.length, 1);
|
||||
assert.equal(rows[0].prediction_id, open);
|
||||
assert.equal(rows[0].action, 'ABSTAIN');
|
||||
// second pass must not duplicate
|
||||
assert.equal(createDecisions(db), 0);
|
||||
});
|
||||
|
||||
test('a big single ticker historical cohort still cannot authorise a live buy', () => {
|
||||
const db = seedDb();
|
||||
for (let index = 0; index < 60; index++) {
|
||||
addPrediction(db, { instrument: 'NVDA', origin: 'historical', excessReturn: 0.04, correct: 1 });
|
||||
}
|
||||
refreshHistoricalCalibration(db, 'hist-cal');
|
||||
const prediction = addPrediction(db, { instrument: 'NVDA', status: 'open' });
|
||||
|
||||
assert.equal(createDecisions(db), 1);
|
||||
const decision = db.prepare('SELECT * FROM autonomy_decisions WHERE prediction_id=?').get(prediction);
|
||||
assert.equal(decision.action, 'ABSTAIN');
|
||||
assert.match(decision.rationale, /diversity|dominated/);
|
||||
assert.match(decision.rationale, new RegExp(cohortKey({ direction: 'positive', eventType: 'earnings_beat', horizonDays: 10 }).replace(/\|/g, '\\|')));
|
||||
});
|
||||
|
||||
test('calibration health reports the stall instead of staying silent', () => {
|
||||
const db = seedDb();
|
||||
addPrediction(db, { instrument: 'NVDA', origin: 'historical', excessReturn: 0.02, correct: 1 });
|
||||
addPrediction(db, { instrument: 'AMD', status: 'open' });
|
||||
refreshHistoricalCalibration(db, 'hist-cal');
|
||||
|
||||
const health = calibrationHealth(db);
|
||||
assert.equal(health.liveOpen, 1);
|
||||
assert.equal(health.offlineResolved, 1);
|
||||
assert.equal(health.learningEligible, 0);
|
||||
assert.ok(health.cohorts >= 1);
|
||||
assert.equal(health.qualifyingCohorts, 0);
|
||||
});
|
||||
Reference in New Issue
Block a user