Offline evidence can no longer authorise anything. createDecisions used to prefer a live snapshot and fall back to the pooled historical one, so once the live lane woke up a live prediction could have drawn a BUY off backfill data. Backfill and replay are fine evidence that the pipeline works, they are not a live track record. No live snapshot now means ABSTAIN. The abstain says whether offline evidence existed for that cohort, so "we have 60 offline samples but no live ones" stays distinguishable from "we know nothing about this cohort". Also split the health counter. It counted qualifying cohorts across every source, which overstated how close we are to being able to trade now that only live cohorts can authorise. It reports qualifying_live_cohorts and qualifying_offline_cohorts separately, and applies the concentration cap it was previously ignoring. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
127 lines
6.1 KiB
JavaScript
127 lines
6.1 KiB
JavaScript
const test = require('node:test');
|
|
const assert = require('node:assert/strict');
|
|
const Database = require('better-sqlite3');
|
|
|
|
const { initAutonomySchema } = require('../src/autonomy/schema');
|
|
const { cohortKey } = require('../src/autonomy/calibration');
|
|
const {
|
|
refreshCalibration,
|
|
refreshHistoricalCalibration,
|
|
createDecisions,
|
|
calibrationHealth,
|
|
} = require('../workers/calibrationWorker');
|
|
|
|
function seedDb() {
|
|
const db = new Database(':memory:');
|
|
initAutonomySchema(db);
|
|
db.prepare("INSERT INTO autonomy_proposals(payload, information_cutoff, status) VALUES ('{}', '2026-01-01T00:00:00Z', 'accepted')").run();
|
|
return db;
|
|
}
|
|
|
|
function addPrediction(db, { instrument, direction = 'positive', eventType = 'earnings_beat', horizonDays = 10,
|
|
origin = 'live', status = 'resolved', learningEligible = 0, replayRunId = null, excessReturn = null, correct = null }) {
|
|
const prediction = db.prepare(`
|
|
INSERT INTO autonomy_predictions
|
|
(proposal_id, instrument, direction, event_type, horizon_days, information_cutoff, evidence_article_ids,
|
|
learning_eligible, strategy_version, origin, replay_run_id, status)
|
|
VALUES (1, ?, ?, ?, ?, '2026-01-01T00:00:00Z', '[1]', ?, 'test', ?, ?, ?)
|
|
`).run(instrument, direction, eventType, horizonDays, learningEligible, origin, replayRunId, status);
|
|
if (excessReturn !== null) {
|
|
db.prepare('INSERT INTO autonomy_outcomes(prediction_id, excess_return, direction_correct) VALUES (?, ?, ?)')
|
|
.run(prediction.lastInsertRowid, excessReturn, correct);
|
|
}
|
|
return prediction.lastInsertRowid;
|
|
}
|
|
|
|
test('live calibration no longer starves on the never-set learning_eligible flag', () => {
|
|
const db = seedDb();
|
|
addPrediction(db, { instrument: 'NVDA', excessReturn: 0.03, correct: 1 });
|
|
addPrediction(db, { instrument: 'AMD', excessReturn: -0.01, correct: 0 });
|
|
|
|
assert.equal(refreshCalibration(db, 'live-cal'), 1);
|
|
const snapshot = db.prepare("SELECT * FROM autonomy_calibration_snapshots WHERE source='live'").get();
|
|
assert.equal(snapshot.sample_size, 2);
|
|
assert.equal(snapshot.distinct_instruments, 2);
|
|
|
|
// the old behaviour is still reachable on purpose, for once the flag is populated
|
|
assert.equal(refreshCalibration(db, 'strict-cal', { requireLearningEligible: true }), 0);
|
|
|
|
// counters report snapshots written, so a steady state poll is genuinely quiet
|
|
assert.equal(refreshCalibration(db, 'live-cal'), 0);
|
|
assert.equal(db.prepare("SELECT COUNT(*) c FROM autonomy_calibration_snapshots WHERE source='live'").get().c, 1);
|
|
});
|
|
|
|
test('historical calibration pools origin historical and replay together', () => {
|
|
const db = seedDb();
|
|
addPrediction(db, { instrument: 'NVDA', origin: 'historical', excessReturn: 0.02, correct: 1 });
|
|
addPrediction(db, { instrument: 'AMD', origin: 'historical', excessReturn: 0.01, correct: 1 });
|
|
addPrediction(db, { instrument: 'INTC', origin: 'replay', replayRunId: 3, excessReturn: -0.02, correct: 0 });
|
|
|
|
refreshHistoricalCalibration(db, 'hist-cal');
|
|
const pooled = db.prepare("SELECT * FROM autonomy_calibration_snapshots WHERE source='historical'").get();
|
|
assert.equal(pooled.sample_size, 3, 'the historical lane must not drop the relabelled rows');
|
|
assert.equal(pooled.distinct_instruments, 3);
|
|
const perRun = db.prepare("SELECT * FROM autonomy_calibration_snapshots WHERE source='replay'").get();
|
|
assert.equal(perRun.replay_run_id, 3);
|
|
assert.equal(perRun.sample_size, 1);
|
|
});
|
|
|
|
test('decisions are only written for open live predictions', () => {
|
|
const db = seedDb();
|
|
const open = addPrediction(db, { instrument: 'NVDA', status: 'open' });
|
|
addPrediction(db, { instrument: 'AMD', status: 'resolved', excessReturn: 0.01, correct: 1 });
|
|
addPrediction(db, { instrument: 'INTC', status: 'open', origin: 'historical' });
|
|
addPrediction(db, { instrument: 'MU', status: 'open', origin: 'replay', replayRunId: 3 });
|
|
|
|
assert.equal(createDecisions(db), 1);
|
|
const rows = db.prepare('SELECT prediction_id, action FROM autonomy_decisions').all();
|
|
assert.equal(rows.length, 1);
|
|
assert.equal(rows[0].prediction_id, open);
|
|
assert.equal(rows[0].action, 'ABSTAIN');
|
|
// second pass must not duplicate
|
|
assert.equal(createDecisions(db), 0);
|
|
});
|
|
|
|
test('a big single ticker historical cohort still cannot authorise a live buy', () => {
|
|
const db = seedDb();
|
|
for (let index = 0; index < 60; index++) {
|
|
addPrediction(db, { instrument: 'NVDA', origin: 'historical', excessReturn: 0.04, correct: 1 });
|
|
}
|
|
refreshHistoricalCalibration(db, 'hist-cal');
|
|
const prediction = addPrediction(db, { instrument: 'NVDA', status: 'open' });
|
|
|
|
assert.equal(createDecisions(db), 1);
|
|
const decision = db.prepare('SELECT * FROM autonomy_decisions WHERE prediction_id=?').get(prediction);
|
|
assert.equal(decision.action, 'ABSTAIN');
|
|
// offline evidence never authorises a live order, however much of it there is,
|
|
// and the abstain has to say the offline data existed so it isnt mistaken for
|
|
// "we know nothing about this cohort"
|
|
assert.match(decision.rationale, /no live calibration/);
|
|
assert.match(decision.rationale, /offline_only source=historical n=60/);
|
|
assert.match(decision.rationale, new RegExp(cohortKey({ direction: 'positive', eventType: 'earnings_beat', horizonDays: 10 }).replace(/\|/g, '\\|')));
|
|
});
|
|
|
|
test('a cohort with no evidence at all is distinguishable from an offline only one', () => {
|
|
const db = seedDb();
|
|
const prediction = addPrediction(db, { instrument: 'NVDA', status: 'open' });
|
|
|
|
assert.equal(createDecisions(db), 1);
|
|
const decision = db.prepare('SELECT * FROM autonomy_decisions WHERE prediction_id=?').get(prediction);
|
|
assert.equal(decision.action, 'ABSTAIN');
|
|
assert.match(decision.rationale, /no evidence/);
|
|
});
|
|
|
|
test('calibration health reports the stall instead of staying silent', () => {
|
|
const db = seedDb();
|
|
addPrediction(db, { instrument: 'NVDA', origin: 'historical', excessReturn: 0.02, correct: 1 });
|
|
addPrediction(db, { instrument: 'AMD', status: 'open' });
|
|
refreshHistoricalCalibration(db, 'hist-cal');
|
|
|
|
const health = calibrationHealth(db);
|
|
assert.equal(health.liveOpen, 1);
|
|
assert.equal(health.offlineResolved, 1);
|
|
assert.equal(health.learningEligible, 0);
|
|
assert.ok(health.cohorts >= 1);
|
|
assert.equal(health.qualifyingCohorts, 0);
|
|
});
|