#!/usr/bin/env node /* * Scoreboard and paired comparison for replay runs. * * Two jobs: * 1. score one run against constant null models, so "50% accuracy" has to * answer the question "compared to what". A model that always says * negative scores whatever share of the sample actually went down, and if * the system cannot beat that it has no directional skill at all. * 2. compare two runs over the SAME articles. Replay walks the archive in * cursor order, so different runs are the only way to hold article * vintage fixed. Comparing two calendar periods of one run compares two * market regimes, not two prompts. * * Read only. Opens intelligence read only and writes nothing. * * PRE REGISTERED TESTS (declared here so the buckets cannot be tuned later): * T1 is run accuracy above the BEST constant baseline on the same sample? * one sided binomial z. the best constant is used as the bar because it * is the hardest of the two, which is conservative for us. * T2 is the direction signed excess return above the BEST constant on the * same bars? paired two sided t. testing it against zero was the first * version and it flattered us: a unit short in everything also earns a * positive number on this sample, so zero is not the bar. * T3 DISCRIMINATION. is P(up | it said positive) above P(up | it said * negative)? two proportion z. this is the only one of the four that a * change of prior cannot fake: it asks whether the choice of direction * carries information, separately from how often it picks each one. * comparing a direction group's accuracy to "always that direction" on * the same rows is an identity and tests nothing, which is what the * first version of this file printed. * T4 paired: on articles both runs answered, is the candidate's per article * accuracy above the baseline's? two sided paired t on the differences. * Everything under "descriptive" is NOT a test. Slice p values carry a * bonferroni factor and are there to generate hypotheses, not confirm them. * * node scripts/score-replay-runs.js * node scripts/score-replay-runs.js --run 1 * node scripts/score-replay-runs.js --baseline 1 --candidate 2 * node scripts/score-replay-runs.js --baseline 1 --candidate 2 --since 2026-02-01 */ const Database = require("better-sqlite3"); const INTELLIGENCE = process.env.INTELLIGENCE_DB || "/data/intelligence.sqlite"; function args() { const out = {}; const argv = process.argv.slice(2); for (let i = 0; i < argv.length; i += 1) { if (!argv[i].startsWith("--")) continue; const key = argv[i].slice(2); const next = argv[i + 1]; out[key] = (next && !next.startsWith("--")) ? (i += 1, next) : true; } return out; } // Abramowitz and Stegun 7.1.26. The last analysis used a logistic shortcut and // it returned p > 1 for negative z, which is nonsense that survived because // nobody looks at a p value and asks whether it is even in range. function erf(x) { const sign = x < 0 ? -1 : 1; const z = Math.abs(x); const t = 1 / (1 + 0.3275911 * z); const y = 1 - ((((1.061405429 * t - 1.453152027) * t + 1.421413741) * t - 0.284496736) * t + 0.254829592) * t * Math.exp(-z * z); return sign * y; } function normalCdf(z) { return 0.5 * (1 + erf(z / Math.SQRT2)); } function twoSided(z) { return 2 * (1 - normalCdf(Math.abs(z))); } function oneSidedUpper(z) { return 1 - normalCdf(z); } function pct(x, digits = 2) { return Number.isFinite(x) ? `${(x * 100).toFixed(digits)}%` : "n/a"; } // accuracy of `hits` out of `n` against a fixed reference rate function binomialZ(hits, n, p0) { if (!n || p0 <= 0 || p0 >= 1) return { z: NaN, p: NaN }; const phat = hits / n; const z = (phat - p0) / Math.sqrt(p0 * (1 - p0) / n); return { z, p: oneSidedUpper(z) }; } function tStat(values) { const n = values.length; if (n < 2) return { n, mean: NaN, z: NaN, p: NaN }; const mean = values.reduce((a, b) => a + b, 0) / n; const variance = values.reduce((a, b) => a + (b - mean) ** 2, 0) / (n - 1); const se = Math.sqrt(variance / n); const z = se > 0 ? mean / se : 0; return { n, mean, se, z, p: twoSided(z) }; } // does the choice of direction carry information at all. invariant to how // often it picks each side, unlike raw accuracy. function discrimination(rows) { const pos = rows.filter((r) => r.direction === "positive"); const neg = rows.filter((r) => r.direction === "negative"); const upPos = pos.filter((r) => r.excess_return > 0).length; const upNeg = neg.filter((r) => r.excess_return > 0).length; const p1 = pos.length ? upPos / pos.length : NaN; const p2 = neg.length ? upNeg / neg.length : NaN; const pooled = (upPos + upNeg) / (pos.length + neg.length); const se = Math.sqrt(pooled * (1 - pooled) * (1 / pos.length + 1 / neg.length)); const z = se > 0 ? (p1 - p2) / se : 0; return { pUpGivenPositive: p1, pUpGivenNegative: p2, nPositive: pos.length, nNegative: neg.length, spread: p1 - p2, z, p: twoSided(z), positiveShare: pos.length / rows.length }; } function signed(row) { // what a unit position in the predicted direction actually earned return row.direction === "negative" ? -row.excess_return : row.excess_return; } function articleOf(row) { try { const parsed = JSON.parse(row.evidence_article_ids || "[]"); return Array.isArray(parsed) && parsed.length ? String(parsed[0]) : null; } catch (error) { console.error(`[score] unparseable evidence on prediction ${row.id}:`, error.message); return null; } } function load(db, { runId, since, until, createdSince }) { const where = ["p.origin = 'replay'", "o.prediction_id IS NOT NULL"]; const params = []; if (runId) { where.push("p.replay_run_id = ?"); params.push(runId); } if (since) { where.push("date(p.information_cutoff) >= date(?)"); params.push(since); } if (until) { where.push("date(p.information_cutoff) <= date(?)"); params.push(until); } // the train/test split is on when the prediction was MADE, because that is // what fixes which prompt produced it. cutoff dates only correlate with it. if (createdSince) { where.push("p.created_at >= ?"); params.push(createdSince); } return db.prepare(` SELECT p.id, p.instrument, p.direction, p.event_type, p.horizon_days, p.information_cutoff, p.evidence_article_ids, p.replay_run_id, p.strategy_version, pr.prompt_version, pr.coordinator_model, o.direction_correct, o.excess_return FROM autonomy_predictions p JOIN autonomy_proposals pr ON pr.id = p.proposal_id JOIN autonomy_outcomes o ON o.prediction_id = p.id WHERE ${where.join(" AND ")} ORDER BY p.information_cutoff ASC, p.id ASC `).all(...params); } function baselines(rows) { const n = rows.length; const up = rows.filter((r) => r.excess_return > 0).length; return { n, alwaysPositive: up / n, alwaysNegative: (n - up) / n, // mean of a unit long in every name, which is what always_positive earns alwaysPositiveExcess: rows.reduce((a, r) => a + r.excess_return, 0) / n, }; } function scoreRun(rows, label) { const n = rows.length; if (!n) { console.log(`\n${label}: no scored predictions\n`); return null; } const hits = rows.filter((r) => r.direction_correct).length; const acc = hits / n; const base = baselines(rows); const best = Math.max(base.alwaysPositive, base.alwaysNegative); const bestName = base.alwaysNegative >= base.alwaysPositive ? "always_negative" : "always_positive"; const t1 = binomialZ(hits, n, best); // pair against the constant on the identical bars. where the system already // agrees with the constant the difference is zero and contributes nothing, // which is exactly right. const constantSign = bestName === "always_negative" ? -1 : 1; const t2 = tStat(rows.map((r) => signed(r) - constantSign * r.excess_return)); const rawSigned = tStat(rows.map(signed)); const constantExcess = rows.reduce((a, r) => a + constantSign * r.excess_return, 0) / n; console.log(`\n=== ${label} ===`); const models = [...new Set(rows.map((r) => r.coordinator_model))]; const prompts = [...new Set(rows.map((r) => r.prompt_version))]; console.log(` span ${rows[0].information_cutoff.slice(0, 10)} .. ${rows[n - 1].information_cutoff.slice(0, 10)}`); console.log(` models ${models.join(", ")}`); console.log(` prompts ${prompts.join(", ")}`); console.log(` scored ${n} predictions over ${new Set(rows.map(articleOf)).size} articles`); console.log(""); console.log(` system accuracy ${pct(acc)} (${hits}/${n})`); console.log(` always_negative ${pct(base.alwaysNegative)} <- share of bars that underperformed SPY`); console.log(` always_positive ${pct(base.alwaysPositive)}`); console.log(` coin flip 50.00%`); console.log(""); console.log(` T1 vs ${bestName}: z=${t1.z.toFixed(3)} p=${t1.p.toFixed(4)} (one sided, does the system beat the bar)`); if (t1.z < 0) console.log(` the system is BELOW the constant. p=${twoSided(t1.z).toFixed(4)} two sided on being different from it.`); console.log(` edge over the bar ${((acc - best) * 100).toFixed(2)} points`); console.log(` T2 signed excess vs ${bestName}: ${pct(t2.mean, 3)} per prediction,` + ` t=${t2.z.toFixed(3)} p=${t2.p.toFixed(4)}`); console.log(` system ${pct(rawSigned.mean, 3)} ${bestName} ${pct(constantExcess, 3)}` + ` always_positive ${pct(base.alwaysPositiveExcess, 3)}`); const t3 = discrimination(rows); console.log(""); console.log(` T3 discrimination: P(up | said positive) ${pct(t3.pUpGivenPositive)} (n=${t3.nPositive})`); console.log(` P(up | said negative) ${pct(t3.pUpGivenNegative)} (n=${t3.nNegative})`); console.log(` spread ${(t3.spread * 100).toFixed(2)} points, z=${t3.z.toFixed(3)} p=${t3.p.toFixed(4)}`); console.log(` it says positive on ${pct(t3.positiveShare)} of calls while ${pct(base.alwaysPositive)}` + ` of the bars went up, so the prior is off by ${((t3.positiveShare - base.alwaysPositive) * 100).toFixed(1)} points`); return { n, acc, hits, best, bestName, t1, t2, t3, base }; } function slice(rows, key, label, minimum = 40) { const groups = new Map(); for (const row of rows) { const k = String(row[key]); if (!groups.has(k)) groups.set(k, []); groups.get(k).push(row); } const kept = [...groups.entries()].filter(([, v]) => v.length >= minimum); if (!kept.length) return; const factor = kept.length; console.log(`\n descriptive by ${label} (n>=${minimum}, bonferroni x${factor}, NOT a test)`); const scored = kept.map(([k, v]) => { const hits = v.filter((r) => r.direction_correct).length; const base = baselines(v); const bar = Math.max(base.alwaysPositive, base.alwaysNegative); const { z } = binomialZ(hits, v.length, bar); return { k, n: v.length, acc: hits / v.length, bar, z, p: Math.min(1, twoSided(z) * factor), excess: tStat(v.map(signed)).mean }; }).sort((a, b) => b.acc - a.acc); for (const s of scored) { // a small p here can mean significantly WORSE than the constant, which read // like good news the first time this printed. say which side it fell on. const side = s.acc >= s.bar ? "above" : "below"; const flag = s.p < 0.05 ? ` * ${side} bar` : ""; console.log(` ${s.k.padEnd(24)} n=${String(s.n).padEnd(5)} acc=${pct(s.acc).padEnd(8)}` + ` bar=${pct(s.bar).padEnd(8)} signed_excess=${pct(s.excess, 3).padEnd(9)} p_adj=${s.p.toFixed(3)}${flag}`); } } // per article accuracy, so an article that produced 11 predictions does not // count eleven times against one that produced a single call function byArticle(rows) { const map = new Map(); for (const row of rows) { const id = articleOf(row); if (!id) continue; if (!map.has(id)) map.set(id, []); map.get(id).push(row); } const out = new Map(); for (const [id, list] of map) { out.set(id, { accuracy: list.filter((r) => r.direction_correct).length / list.length, signed: list.reduce((a, r) => a + signed(r), 0) / list.length, count: list.length, }); } return out; } function paired(baseRows, candRows) { const a = byArticle(baseRows); const b = byArticle(candRows); const shared = [...a.keys()].filter((id) => b.has(id)); console.log(`\n=== T4 paired comparison ===`); console.log(` baseline articles ${a.size}, candidate articles ${b.size}, shared ${shared.length}`); if (shared.length < 30) { console.log(" not enough shared articles to say anything. run the candidate over the baseline's articles first."); return; } const accDiff = shared.map((id) => b.get(id).accuracy - a.get(id).accuracy); const excessDiff = shared.map((id) => b.get(id).signed - a.get(id).signed); const baseAcc = shared.reduce((s, id) => s + a.get(id).accuracy, 0) / shared.length; const candAcc = shared.reduce((s, id) => s + b.get(id).accuracy, 0) / shared.length; const tAcc = tStat(accDiff); const tExc = tStat(excessDiff); const better = shared.filter((id) => b.get(id).accuracy > a.get(id).accuracy).length; const worse = shared.filter((id) => b.get(id).accuracy < a.get(id).accuracy).length; console.log(` baseline per article accuracy ${pct(baseAcc)}`); console.log(` candidate per article accuracy ${pct(candAcc)}`); console.log(` articles improved ${better}, degraded ${worse}, unchanged ${shared.length - better - worse}`); console.log(` T4 accuracy delta ${pct(tAcc.mean)} t=${tAcc.z.toFixed(3)} p=${tAcc.p.toFixed(4)}`); console.log(` signed excess delta ${pct(tExc.mean, 3)} t=${tExc.z.toFixed(3)} p=${tExc.p.toFixed(4)}`); console.log(tAcc.p < 0.05 ? (tAcc.mean > 0 ? " VERDICT: the candidate is better on the same articles." : " VERDICT: the candidate is WORSE on the same articles.") : " VERDICT: no detectable difference on the same articles."); } function main() { const opts = args(); const db = new Database(INTELLIGENCE, { readonly: true }); db.pragma("busy_timeout = 20000"); const runs = db.prepare("SELECT * FROM autonomy_replay_runs ORDER BY id").all(); console.log("replay runs on record:"); for (const run of runs) { console.log(` #${run.id} ${run.status.padEnd(9)} watermark=${String(run.watermark_at).slice(0, 10)}` + ` ${run.strategy_version}/${run.prompt_version} ${run.coordinator_model}` + ` processed=${run.processed_articles}`); } if (opts.baseline && opts.candidate) { const window = { since: opts.since, until: opts.until, createdSince: opts["created-since"] }; const baseRows = load(db, { runId: Number(opts.baseline), ...window }); const candRows = load(db, { runId: Number(opts.candidate), ...window }); scoreRun(baseRows, `run ${opts.baseline} (baseline)`); scoreRun(candRows, `run ${opts.candidate} (candidate)`); paired(baseRows, candRows); db.close(); return; } const runId = opts.run && opts.run !== true ? Number(opts.run) : null; const rows = load(db, { runId, since: opts.since, until: opts.until, createdSince: opts["created-since"] }); const summary = scoreRun(rows, runId ? `run ${runId}` : "all replay runs"); if (summary) { slice(rows, "direction", "direction"); slice(rows, "horizon_days", "horizon"); slice(rows, "event_type", "event family"); slice(rows, "instrument", "instrument"); console.log(""); } db.close(); } main();