feat: let the generator read its own results, and measure it honestly

Nothing in the pipeline has ever fed an outcome back to the thing that makes
predictions. Calibration reads autonomy_outcomes, but calibration only gates
whether to act on a prediction, never what the prediction is. So the only thing
that has ever changed this system's output is a human editing the prompt.

score-replay-runs.js asks "compared to what". The answer is not flattering:
over 2,114 scored replay predictions the system is right 50.05% of the time
while answering "negative" to every one of the same bars scores 54.45%. It is
4.4 points below a constant, z=-4.06. The whole deficit is the prior. It says
positive on 65% of calls when 45.5% of bars beat SPY, a 19 point skew. Its
discrimination, P(up|positive) minus P(up|negative), is +3.2 points with
p=0.15, so the direction it picks is weakly informative and completely buried
by how often it defaults to positive.

The first version of that script compared each direction group's accuracy to
"always that direction" on the same rows, which is an identity and tests
nothing. T3 replaces it with the two proportion test that actually asks whether
the choice of direction carries information.

build-feedback-brief.js turns a run's scored outcomes into a memo the next run
reads before predicting. Generated from the data, not written by hand, or it is
just me editing the prompt again with extra steps.

Replay can now be pinned to an explicit article set, which is what makes two
runs comparable at all. Comparing two calendar windows of one run compares two
market regimes: the epochs in run 1 line up exactly with article vintage, E0 is
late 2024 and E2 is 2026, so nothing could be attributed. A new run also
inherits its parent's watermark instead of recomputing it from today, which
silently guaranteed a different archive slice every time.

prompt_version never moved across four material prompt changes, so every
proposal on record claims to come from the first prompt. coordinator-2 and
replay-coordinator-2.

docs/replay-run-2-preregistration.md fixes the bar before the run exists,
including which result counts as learning and which is only calibration.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-09-08 00:56:43 +01:00
co-authored by Claude Opus 5
parent 028c07688d
commit ec29e64e96
8 changed files with 881 additions and 12 deletions
+163
View File
@@ -0,0 +1,163 @@
#!/usr/bin/env node
/*
* Turn a replay run's own scored outcomes into a memo the next run is told
* before it predicts anything.
*
* This is the piece that was missing. The generator has never once seen its own
* results: nothing in coordinatorWorker, replayWorker, llm.js or graphContext
* reads autonomy_outcomes. Calibration reads them, but calibration only decides
* whether to ACT on a prediction, it never changes what gets predicted. So the
* only thing that has ever altered this system's output is a human editing the
* prompt. A memo generated from the data is not a human editing the prompt.
*
* Everything here is computed from a TRAIN slice bounded by --until. Nothing
* from the evaluation window may appear in the text or the comparison is just
* fitting to the answer sheet.
*
* node scripts/build-feedback-brief.js --run 1 --until "2026-09-04 19:30"
*/
const Database = require("better-sqlite3");
const INTELLIGENCE = process.env.INTELLIGENCE_DB || "/data/intelligence.sqlite";
function pct(x, digits = 1) { return `${(x * 100).toFixed(digits)}%`; }
function erf(x) {
const sign = x < 0 ? -1 : 1;
const z = Math.abs(x);
const t = 1 / (1 + 0.3275911 * z);
const y = 1 - ((((1.061405429 * t - 1.453152027) * t + 1.421413741) * t - 0.284496736) * t + 0.254829592) * t * Math.exp(-z * z);
return sign * y;
}
function twoSided(z) { return 2 * (1 - 0.5 * (1 + erf(Math.abs(z) / Math.SQRT2))); }
function loadTrain(db, { runId, createdBefore }) {
return db.prepare(`
SELECT p.direction, p.event_type, p.horizon_days, p.instrument,
o.direction_correct, o.excess_return
FROM autonomy_predictions p
JOIN autonomy_outcomes o ON o.prediction_id = p.id
WHERE p.origin = 'replay' AND p.replay_run_id = ? AND p.created_at < ?
`).all(runId, createdBefore);
}
// families and horizons that sit far enough below the constant to be worth
// naming. n floor keeps a handful of unlucky calls out of the memo.
function weakSlices(rows, key, { minimum = 40, factor }) {
const groups = new Map();
for (const row of rows) {
const k = String(row[key]);
if (!groups.has(k)) groups.set(k, []);
groups.get(k).push(row);
}
const kept = [...groups.entries()].filter(([, v]) => v.length >= minimum);
const adjust = factor || kept.length || 1;
return kept.map(([k, v]) => {
const hits = v.filter((r) => r.direction_correct).length;
const acc = hits / v.length;
const down = v.filter((r) => r.excess_return <= 0).length / v.length;
const bar = Math.max(down, 1 - down);
const se = Math.sqrt(bar * (1 - bar) / v.length);
const z = se > 0 ? (acc - bar) / se : 0;
return { key: k, n: v.length, acc, bar, p: Math.min(1, twoSided(z) * adjust) };
}).sort((a, b) => a.acc - b.acc);
}
function buildFeedbackBrief(db, { runId, createdBefore }) {
const rows = loadTrain(db, { runId, createdBefore });
if (rows.length < 200) {
throw new Error(`only ${rows.length} scored training predictions for run ${runId}, refusing to write a brief off that`);
}
const n = rows.length;
const positives = rows.filter((r) => r.direction === "positive");
const positiveShare = positives.length / n;
const actuallyUp = rows.filter((r) => r.excess_return > 0).length / n;
const acc = rows.filter((r) => r.direction_correct).length / n;
const alwaysNegative = 1 - actuallyUp;
const upGivenPositive = positives.filter((r) => r.excess_return > 0).length / (positives.length || 1);
const negatives = rows.filter((r) => r.direction === "negative");
const upGivenNegative = negatives.filter((r) => r.excess_return > 0).length / (negatives.length || 1);
const families = weakSlices(rows, "event_type", { minimum: 40 });
const horizons = weakSlices(rows, "horizon_days", { minimum: 40 });
const weakFamilies = families.filter((f) => f.acc < f.bar - 0.08).slice(0, 5);
const strongFamilies = [...families].reverse().filter((f) => f.acc > f.bar + 0.05).slice(0, 4);
const weakHorizons = horizons.filter((h) => h.acc < h.bar - 0.08).slice(0, 3);
const lines = [];
lines.push(`CALIBRATION FEEDBACK. The following comes from ${n} of your own earlier predictions on this`);
lines.push(`archive, every one of them scored on realised excess return against SPY. It describes how you`);
lines.push(`have actually performed, not how you think you perform. Use it.`);
lines.push("");
lines.push(`1. Your directional prior is wrong. You said "positive" on ${pct(positiveShare)} of those predictions.`);
lines.push(` Only ${pct(actuallyUp)} of the same bars actually beat SPY. Most individual names underperform a`);
lines.push(` cap weighted index over any horizon, so "positive" is the minority answer, not the default.`);
lines.push(` You were right ${pct(acc)} of the time. Answering "negative" to every single one of those bars`);
lines.push(` would have scored ${pct(alwaysNegative)}. You are currently below a constant.`);
lines.push("");
lines.push(`2. Beating SPY is the bar, not the company doing well. Good news that the index already had`);
lines.push(` priced, or that lifts the whole sector, is not a positive excess return. Ask whether this name`);
lines.push(` outperforms the market, never whether the story is upbeat.`);
lines.push("");
// a spread under two points is not worth telling it to keep, it would just be
// flattering noise back at itself
if (upGivenPositive - upGivenNegative >= 0.02) {
lines.push(`3. Your instinct on WHICH way is weakly right: bars you called positive beat SPY ${pct(upGivenPositive)}`);
lines.push(` of the time versus ${pct(upGivenNegative)} for the ones you called negative. That separation is small`);
lines.push(` enough that it could still be luck, and it is buried by how often you default to positive.`);
} else {
lines.push(`3. Your choice of direction carries no information yet: bars you called positive beat SPY`);
lines.push(` ${pct(upGivenPositive)} of the time versus ${pct(upGivenNegative)} for the ones you called negative. Only predict when`);
lines.push(` the evidence gives you a genuine mechanism, and return nothing otherwise.`);
}
lines.push("");
if (weakFamilies.length) {
lines.push(`4. Event families you read worst, accuracy against the constant on the same bars:`);
for (const f of weakFamilies) {
lines.push(` ${f.key}: you ${pct(f.acc)}, constant ${pct(f.bar)}, n=${f.n}`);
}
lines.push(` On these, prefer returning an empty predictions array over a weak call.`);
lines.push("");
}
if (strongFamilies.length) {
lines.push(`5. Event families you read best, where a confident call is warranted:`);
for (const f of strongFamilies) {
lines.push(` ${f.key}: you ${pct(f.acc)}, constant ${pct(f.bar)}, n=${f.n}`);
}
lines.push("");
}
if (weakHorizons.length) {
lines.push(`6. Horizons that went worst for you: ${weakHorizons.map((h) => `${h.key}d (${pct(h.acc)}, n=${h.n})`).join(", ")}.`);
lines.push(` The longer the horizon the more of the move is market and sector rather than the event.`);
lines.push("");
}
lines.push(`None of this tells you what to answer for the evidence below. It tells you which of your habits`);
lines.push(`have already cost you. An empty predictions array is always available and costs nothing.`);
return { text: lines.join("\n"), stats: { n, positiveShare, actuallyUp, acc, alwaysNegative,
upGivenPositive, upGivenNegative, weakFamilies, strongFamilies, weakHorizons } };
}
function main() {
const argv = process.argv.slice(2);
const opts = {};
for (let i = 0; i < argv.length; i += 1) {
if (!argv[i].startsWith("--")) continue;
const key = argv[i].slice(2);
const next = argv[i + 1];
opts[key] = (next && !next.startsWith("--")) ? (i += 1, next) : true;
}
const db = new Database(INTELLIGENCE, { readonly: true });
db.pragma("busy_timeout = 20000");
const { text, stats } = buildFeedbackBrief(db, {
runId: Number(opts.run || 1),
createdBefore: String(opts.until || "2026-09-04 19:30"),
});
console.log(text);
console.log(`\n--- derived from ${stats.n} scored training predictions ---`);
db.close();
}
if (require.main === module) main();
module.exports = { buildFeedbackBrief };