From 8f4d3b4ce93cd41c2a15d1f6f1a6c6cbee05b53f Mon Sep 17 00:00:00 2001 From: ImBenji Date: Sat, 29 Aug 2026 22:22:36 +0100 Subject: [PATCH] fix: require live calibration to authorise a live order Offline evidence can no longer authorise anything. createDecisions used to prefer a live snapshot and fall back to the pooled historical one, so once the live lane woke up a live prediction could have drawn a BUY off backfill data. Backfill and replay are fine evidence that the pipeline works, they are not a live track record. No live snapshot now means ABSTAIN. The abstain says whether offline evidence existed for that cohort, so "we have 60 offline samples but no live ones" stays distinguishable from "we know nothing about this cohort". Also split the health counter. It counted qualifying cohorts across every source, which overstated how close we are to being able to trade now that only live cohorts can authorise. It reports qualifying_live_cohorts and qualifying_offline_cohorts separately, and applies the concentration cap it was previously ignoring. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb --- test/autonomy.test.js | 2 +- test/calibrationLanes.test.js | 16 +++++++++- workers/calibrationWorker.js | 56 +++++++++++++++++++++++++---------- 3 files changed, 57 insertions(+), 17 deletions(-) diff --git a/test/autonomy.test.js b/test/autonomy.test.js index f2de22a..de2cc8a 100644 --- a/test/autonomy.test.js +++ b/test/autonomy.test.js @@ -163,7 +163,7 @@ test('live decisions map calibration snapshot fields into policy inputs', () => INSERT INTO autonomy_calibration_snapshots (cohort_key, sample_size, effective_sample_size, directional_probability, expected_excess_return, lower_return, upper_return, parent_cohort_key, version, source) - VALUES ('v2|unknown|earnings|medium|positive', 40, 42, 0.7, 0.02, -0.01, 0.06, NULL, 'test-cal', 'historical') + VALUES ('v2|unknown|earnings|medium|positive', 40, 42, 0.7, 0.02, -0.01, 0.06, NULL, 'test-cal', 'live') `).run(); db.prepare("UPDATE autonomy_calibration_snapshots SET distinct_instruments = 11, top_instrument_share = 0.2").run(); diff --git a/test/calibrationLanes.test.js b/test/calibrationLanes.test.js index 7fcc32c..a6b57ee 100644 --- a/test/calibrationLanes.test.js +++ b/test/calibrationLanes.test.js @@ -93,10 +93,24 @@ test('a big single ticker historical cohort still cannot authorise a live buy', assert.equal(createDecisions(db), 1); const decision = db.prepare('SELECT * FROM autonomy_decisions WHERE prediction_id=?').get(prediction); assert.equal(decision.action, 'ABSTAIN'); - assert.match(decision.rationale, /diversity|dominated/); + // offline evidence never authorises a live order, however much of it there is, + // and the abstain has to say the offline data existed so it isnt mistaken for + // "we know nothing about this cohort" + assert.match(decision.rationale, /no live calibration/); + assert.match(decision.rationale, /offline_only source=historical n=60/); assert.match(decision.rationale, new RegExp(cohortKey({ direction: 'positive', eventType: 'earnings_beat', horizonDays: 10 }).replace(/\|/g, '\\|'))); }); +test('a cohort with no evidence at all is distinguishable from an offline only one', () => { + const db = seedDb(); + const prediction = addPrediction(db, { instrument: 'NVDA', status: 'open' }); + + assert.equal(createDecisions(db), 1); + const decision = db.prepare('SELECT * FROM autonomy_decisions WHERE prediction_id=?').get(prediction); + assert.equal(decision.action, 'ABSTAIN'); + assert.match(decision.rationale, /no evidence/); +}); + test('calibration health reports the stall instead of staying silent', () => { const db = seedDb(); addPrediction(db, { instrument: 'NVDA', origin: 'historical', excessReturn: 0.02, correct: 1 }); diff --git a/workers/calibrationWorker.js b/workers/calibrationWorker.js index dfce363..0627f66 100644 --- a/workers/calibrationWorker.js +++ b/workers/calibrationWorker.js @@ -123,8 +123,8 @@ function refreshHistoricalCalibration(db, version = `replay-cal-${Date.now()}`) }); } - // Pooled historical view across both offline origins. This is the snapshot the - // live lane falls back on before it has any live evidence of its own. + // Pooled historical view across both offline origins. Reporting and research + // only, decisions never read it: live orders require live calibration. written += refreshCalibration(db, version, { origins: HISTORICAL_ORIGINS, source: 'historical', @@ -145,12 +145,22 @@ function createDecisions(db, strategyVersion = 'autonomy-1', rules = {}) { WHERE d.prediction_id IS NULL AND p.status = 'open' AND p.origin = 'live' `).all(); - // Prefer calibration built from live outcomes, fall back to the pooled historical - // snapshot, and always record which one we used in the rationale. + // Only calibration built from live outcomes may authorise a live order. Backfill + // and replay are legitimate evidence that the pipeline works, but they are not a + // live track record, and an order placed off them would be exactly the confusion + // this whole thing exists to avoid. No live snapshot means abstain, full stop. const latest = db.prepare(` SELECT * FROM autonomy_calibration_snapshots - WHERE cohort_key = ? - ORDER BY (source = 'live') DESC, created_at DESC, id DESC LIMIT 1 + WHERE cohort_key = ? AND source = 'live' + ORDER BY created_at DESC, id DESC LIMIT 1 + `); + + // Looked up purely so an abstain can say whether offline evidence exists for the + // cohort. It never feeds decide(). + const offline = db.prepare(` + SELECT source, sample_size FROM autonomy_calibration_snapshots + WHERE cohort_key = ? AND source != 'live' + ORDER BY sample_size DESC, created_at DESC, id DESC LIMIT 1 `); const insert = db.prepare(` INSERT INTO autonomy_decisions @@ -162,12 +172,18 @@ function createDecisions(db, strategyVersion = 'autonomy-1', rules = {}) { for (const prediction of predictions) { const key = cohortKey({ direction: prediction.direction, eventType: prediction.event_type, horizonDays: prediction.horizon_days }); const calibration = latest.get(key); - const decision = calibration - ? decide(snapshotToDecisionInput(calibration, prediction.direction), rules) - : { action: 'ABSTAIN', rationale: 'calibration unavailable' }; - const rationale = calibration - ? `${decision.rationale} [cohort=${key} source=${calibration.source} n=${calibration.sample_size}]` - : `${decision.rationale} [cohort=${key}]`; + let decision; + let rationale; + if (calibration) { + decision = decide(snapshotToDecisionInput(calibration, prediction.direction), rules); + rationale = `${decision.rationale} [cohort=${key} source=live n=${calibration.sample_size}]`; + } else { + const fallback = offline.get(key); + decision = { action: 'ABSTAIN', rationale: 'no live calibration for this cohort' }; + rationale = fallback + ? `${decision.rationale} [cohort=${key} offline_only source=${fallback.source} n=${fallback.sample_size}]` + : `${decision.rationale} [cohort=${key} no evidence]`; + } insert.run(prediction.id, decision.action, calibration?.directional_probability || null, calibration?.expected_excess_return || null, rationale, strategyVersion); created++; @@ -229,12 +245,20 @@ function calibrationHealth(db, rules = {}) { SUM(CASE WHEN learning_eligible = 1 THEN 1 ELSE 0 END) AS learning_eligible FROM autonomy_predictions `).get() || {}; + // Only live snapshots can authorise anything, so counting offline cohorts as + // "qualifying" would overstate how close we are to being able to trade. They + // are still worth reporting, just in their own bucket. + const maxConcentration = Number(rules.maxInstrumentConcentration ?? DEFAULT_POLICY_RULES.maxInstrumentConcentration); + const gate = `sample_size >= ? AND COALESCE(distinct_instruments, 0) >= ? + AND COALESCE(top_instrument_share, 1) <= ?`; const cohorts = db.prepare(` SELECT COUNT(*) AS total, - SUM(CASE WHEN sample_size >= ? AND COALESCE(distinct_instruments, 0) >= ? THEN 1 ELSE 0 END) AS qualifying + SUM(CASE WHEN source = 'live' AND ${gate} THEN 1 ELSE 0 END) AS qualifying, + SUM(CASE WHEN source != 'live' AND ${gate} THEN 1 ELSE 0 END) AS offline_qualifying FROM autonomy_calibration_snapshots - `).get(minSampleSize, minDistinctInstruments) || {}; + `).get(minSampleSize, minDistinctInstruments, maxConcentration, + minSampleSize, minDistinctInstruments, maxConcentration) || {}; return { liveOpen: Number(predictions.live_open || 0), liveResolved: Number(predictions.live_resolved || 0), @@ -242,6 +266,7 @@ function calibrationHealth(db, rules = {}) { learningEligible: Number(predictions.learning_eligible || 0), cohorts: Number(cohorts.total || 0), qualifyingCohorts: Number(cohorts.qualifying || 0), + offlineQualifyingCohorts: Number(cohorts.offline_qualifying || 0), }; } catch (error) { console.error('[calibration] health probe failed:', error.message, error.stack); @@ -252,7 +277,8 @@ function calibrationHealth(db, rules = {}) { function formatHealth(health) { if (!health) return 'health=unavailable'; return `live_open=${health.liveOpen} live_resolved=${health.liveResolved} offline_resolved=${health.offlineResolved}` - + ` learning_eligible=${health.learningEligible} cohorts=${health.cohorts} qualifying_cohorts=${health.qualifyingCohorts}`; + + ` learning_eligible=${health.learningEligible} cohorts=${health.cohorts}` + + ` qualifying_live_cohorts=${health.qualifyingCohorts} qualifying_offline_cohorts=${health.offlineQualifyingCohorts}`; } async function runCalibrationWorker({