Files
Duriin-API/scripts/build-feedback-brief.js
ImBenjiandClaude Opus 5 b246bd9d4b fix: a recovered replay job can no longer hijack the active run
leaseNextJob hands back any pending replay_article job, it has no idea about
runs, and the worker was attributing whatever came back to whichever run was
active. One recovered dead letter from run 1 would have been stamped with run
2's id, given run 2's feedback brief, and dragged run 2's cursor to wherever
that old article sits in the archive. A pinned run would then decide its set was
finished after a couple of articles. There are 139 dead letters and they are
built to recover, so this was not hypothetical.

The idempotency key already says which run enqueued the job. Ask it.

Also: refuse to inherit the parent's model label when starting a run. Inheriting
is exactly how run 1 came to be labelled qwen for predictions deepseek made.

The split moves to the replay container's actual restart time rather than the
commit timestamp five minutes later. Verified the running container really does
have the instrument rules, the de-anchoring and the enum before trusting it as
the boundary. It makes no difference to the partition, there are no replay
predictions at all between 15:57 and midnight that day, but the boundary should
be the thing that actually changed the prompt.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
2026-09-08 01:01:36 +01:00

171 lines
8.5 KiB
JavaScript

#!/usr/bin/env node
/*
* Turn a replay run's own scored outcomes into a memo the next run is told
* before it predicts anything.
*
* This is the piece that was missing. The generator has never once seen its own
* results: nothing in coordinatorWorker, replayWorker, llm.js or graphContext
* reads autonomy_outcomes. Calibration reads them, but calibration only decides
* whether to ACT on a prediction, it never changes what gets predicted. So the
* only thing that has ever altered this system's output is a human editing the
* prompt. A memo generated from the data is not a human editing the prompt.
*
* Everything here is computed from a TRAIN slice bounded by --until. Nothing
* from the evaluation window may appear in the text or the comparison is just
* fitting to the answer sheet.
*
* node scripts/build-feedback-brief.js --run 1 --until "2026-09-04 19:30"
*/
const Database = require("better-sqlite3");
const INTELLIGENCE = process.env.INTELLIGENCE_DB || "/data/intelligence.sqlite";
// When duriin-api-replay-1 restarted onto the prompt it runs today. Not the
// commit timestamp, which is five minutes later and would have been wrong.
// Replay was between daily budgets across the restart, so there is an eight
// hour hole in predictions around it and every candidate split inside that
// hole partitions the data identically.
const SPLIT = "2026-09-04 19:17:43";
function pct(x, digits = 1) { return `${(x * 100).toFixed(digits)}%`; }
function erf(x) {
const sign = x < 0 ? -1 : 1;
const z = Math.abs(x);
const t = 1 / (1 + 0.3275911 * z);
const y = 1 - ((((1.061405429 * t - 1.453152027) * t + 1.421413741) * t - 0.284496736) * t + 0.254829592) * t * Math.exp(-z * z);
return sign * y;
}
function twoSided(z) { return 2 * (1 - 0.5 * (1 + erf(Math.abs(z) / Math.SQRT2))); }
function loadTrain(db, { runId, createdBefore }) {
return db.prepare(`
SELECT p.direction, p.event_type, p.horizon_days, p.instrument,
o.direction_correct, o.excess_return
FROM autonomy_predictions p
JOIN autonomy_outcomes o ON o.prediction_id = p.id
WHERE p.origin = 'replay' AND p.replay_run_id = ? AND p.created_at < ?
`).all(runId, createdBefore);
}
// families and horizons that sit far enough below the constant to be worth
// naming. n floor keeps a handful of unlucky calls out of the memo.
function weakSlices(rows, key, { minimum = 40, factor }) {
const groups = new Map();
for (const row of rows) {
const k = String(row[key]);
if (!groups.has(k)) groups.set(k, []);
groups.get(k).push(row);
}
const kept = [...groups.entries()].filter(([, v]) => v.length >= minimum);
const adjust = factor || kept.length || 1;
return kept.map(([k, v]) => {
const hits = v.filter((r) => r.direction_correct).length;
const acc = hits / v.length;
const down = v.filter((r) => r.excess_return <= 0).length / v.length;
const bar = Math.max(down, 1 - down);
const se = Math.sqrt(bar * (1 - bar) / v.length);
const z = se > 0 ? (acc - bar) / se : 0;
return { key: k, n: v.length, acc, bar, p: Math.min(1, twoSided(z) * adjust) };
}).sort((a, b) => a.acc - b.acc);
}
function buildFeedbackBrief(db, { runId, createdBefore }) {
const rows = loadTrain(db, { runId, createdBefore });
if (rows.length < 200) {
throw new Error(`only ${rows.length} scored training predictions for run ${runId}, refusing to write a brief off that`);
}
const n = rows.length;
const positives = rows.filter((r) => r.direction === "positive");
const positiveShare = positives.length / n;
const actuallyUp = rows.filter((r) => r.excess_return > 0).length / n;
const acc = rows.filter((r) => r.direction_correct).length / n;
const alwaysNegative = 1 - actuallyUp;
const upGivenPositive = positives.filter((r) => r.excess_return > 0).length / (positives.length || 1);
const negatives = rows.filter((r) => r.direction === "negative");
const upGivenNegative = negatives.filter((r) => r.excess_return > 0).length / (negatives.length || 1);
const families = weakSlices(rows, "event_type", { minimum: 40 });
const horizons = weakSlices(rows, "horizon_days", { minimum: 40 });
const weakFamilies = families.filter((f) => f.acc < f.bar - 0.08).slice(0, 5);
const strongFamilies = [...families].reverse().filter((f) => f.acc > f.bar + 0.05).slice(0, 4);
const weakHorizons = horizons.filter((h) => h.acc < h.bar - 0.08).slice(0, 3);
const lines = [];
lines.push(`CALIBRATION FEEDBACK. The following comes from ${n} of your own earlier predictions on this`);
lines.push(`archive, every one of them scored on realised excess return against SPY. It describes how you`);
lines.push(`have actually performed, not how you think you perform. Use it.`);
lines.push("");
lines.push(`1. Your directional prior is wrong. You said "positive" on ${pct(positiveShare)} of those predictions.`);
lines.push(` Only ${pct(actuallyUp)} of the same bars actually beat SPY. Most individual names underperform a`);
lines.push(` cap weighted index over any horizon, so "positive" is the minority answer, not the default.`);
lines.push(` You were right ${pct(acc)} of the time. Answering "negative" to every single one of those bars`);
lines.push(` would have scored ${pct(alwaysNegative)}. You are currently below a constant.`);
lines.push("");
lines.push(`2. Beating SPY is the bar, not the company doing well. Good news that the index already had`);
lines.push(` priced, or that lifts the whole sector, is not a positive excess return. Ask whether this name`);
lines.push(` outperforms the market, never whether the story is upbeat.`);
lines.push("");
// a spread under two points is not worth telling it to keep, it would just be
// flattering noise back at itself
if (upGivenPositive - upGivenNegative >= 0.02) {
lines.push(`3. Your instinct on WHICH way is weakly right: bars you called positive beat SPY ${pct(upGivenPositive)}`);
lines.push(` of the time versus ${pct(upGivenNegative)} for the ones you called negative. That separation is small`);
lines.push(` enough that it could still be luck, and it is buried by how often you default to positive.`);
} else {
lines.push(`3. Your choice of direction carries no information yet: bars you called positive beat SPY`);
lines.push(` ${pct(upGivenPositive)} of the time versus ${pct(upGivenNegative)} for the ones you called negative. Only predict when`);
lines.push(` the evidence gives you a genuine mechanism, and return nothing otherwise.`);
}
lines.push("");
if (weakFamilies.length) {
lines.push(`4. Event families you read worst, accuracy against the constant on the same bars:`);
for (const f of weakFamilies) {
lines.push(` ${f.key}: you ${pct(f.acc)}, constant ${pct(f.bar)}, n=${f.n}`);
}
lines.push(` On these, prefer returning an empty predictions array over a weak call.`);
lines.push("");
}
if (strongFamilies.length) {
lines.push(`5. Event families you read best, where a confident call is warranted:`);
for (const f of strongFamilies) {
lines.push(` ${f.key}: you ${pct(f.acc)}, constant ${pct(f.bar)}, n=${f.n}`);
}
lines.push("");
}
if (weakHorizons.length) {
lines.push(`6. Horizons that went worst for you: ${weakHorizons.map((h) => `${h.key}d (${pct(h.acc)}, n=${h.n})`).join(", ")}.`);
lines.push(` The longer the horizon the more of the move is market and sector rather than the event.`);
lines.push("");
}
lines.push(`None of this tells you what to answer for the evidence below. It tells you which of your habits`);
lines.push(`have already cost you. An empty predictions array is always available and costs nothing.`);
return { text: lines.join("\n"), stats: { n, positiveShare, actuallyUp, acc, alwaysNegative,
upGivenPositive, upGivenNegative, weakFamilies, strongFamilies, weakHorizons } };
}
function main() {
const argv = process.argv.slice(2);
const opts = {};
for (let i = 0; i < argv.length; i += 1) {
if (!argv[i].startsWith("--")) continue;
const key = argv[i].slice(2);
const next = argv[i + 1];
opts[key] = (next && !next.startsWith("--")) ? (i += 1, next) : true;
}
const db = new Database(INTELLIGENCE, { readonly: true });
db.pragma("busy_timeout = 20000");
const { text, stats } = buildFeedbackBrief(db, {
runId: Number(opts.run || 1),
createdBefore: String(opts.until || SPLIT),
});
console.log(text);
console.log(`\n--- derived from ${stats.n} scored training predictions ---`);
db.close();
}
if (require.main === module) main();
module.exports = { buildFeedbackBrief };