feat: let the generator read its own results, and measure it honestly

Nothing in the pipeline has ever fed an outcome back to the thing that makes
predictions. Calibration reads autonomy_outcomes, but calibration only gates
whether to act on a prediction, never what the prediction is. So the only thing
that has ever changed this system's output is a human editing the prompt.

score-replay-runs.js asks "compared to what". The answer is not flattering:
over 2,114 scored replay predictions the system is right 50.05% of the time
while answering "negative" to every one of the same bars scores 54.45%. It is
4.4 points below a constant, z=-4.06. The whole deficit is the prior. It says
positive on 65% of calls when 45.5% of bars beat SPY, a 19 point skew. Its
discrimination, P(up|positive) minus P(up|negative), is +3.2 points with
p=0.15, so the direction it picks is weakly informative and completely buried
by how often it defaults to positive.

The first version of that script compared each direction group's accuracy to
"always that direction" on the same rows, which is an identity and tests
nothing. T3 replaces it with the two proportion test that actually asks whether
the choice of direction carries information.

build-feedback-brief.js turns a run's scored outcomes into a memo the next run
reads before predicting. Generated from the data, not written by hand, or it is
just me editing the prompt again with extra steps.

Replay can now be pinned to an explicit article set, which is what makes two
runs comparable at all. Comparing two calendar windows of one run compares two
market regimes: the epochs in run 1 line up exactly with article vintage, E0 is
late 2024 and E2 is 2026, so nothing could be attributed. A new run also
inherits its parent's watermark instead of recomputing it from today, which
silently guaranteed a different archive slice every time.

prompt_version never moved across four material prompt changes, so every
proposal on record claims to come from the first prompt. coordinator-2 and
replay-coordinator-2.

docs/replay-run-2-preregistration.md fixes the bar before the run exists,
including which result counts as learning and which is only calibration.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-09-08 00:56:43 +01:00
co-authored by Claude Opus 5
parent 028c07688d
commit ec29e64e96
8 changed files with 881 additions and 12 deletions
+6 -2
View File
@@ -14,6 +14,10 @@ function sleep(ms) { return new Promise((resolve) => setTimeout(resolve, ms)); }
// bumped when the prompt changes in a way that changes what a prediction means.
// autonomy-2 is the first version that can see the relationship graph.
const STRATEGY_VERSION = 'autonomy-2';
// This never moved across four prompt changes, so every proposal on record
// claims to come from the same prompt as the very first one. Attribution was
// impossible, which is why the only honest before/after we had was a timestamp.
const PROMPT_VERSION = 'coordinator-2';
function loadConfig() {
const configPath = path.resolve(process.env.DURIIN_CONFIG || path.join(__dirname, '..', 'config.json'));
@@ -72,7 +76,7 @@ async function runCoordinatorWorker({ archivePath, intelligencePath, workerId =
eventId: event.id,
informationCutoff,
model: config.openRouter.llmModel || 'unknown',
promptVersion: 'coordinator-1',
promptVersion: PROMPT_VERSION,
strategyVersion: STRATEGY_VERSION,
// only a genuine live lane job may ever feed learning
origin: historical ? 'historical' : 'live',
@@ -84,7 +88,7 @@ async function runCoordinatorWorker({ archivePath, intelligencePath, workerId =
eventId: event.id,
informationCutoff,
model: config.openRouter.llmModel || 'unknown',
promptVersion: 'coordinator-1',
promptVersion: PROMPT_VERSION,
origin: historical ? 'historical' : 'live',
learningEligible: !historical,
}, validationError.message);
+82 -9
View File
@@ -27,8 +27,13 @@ function articleTimeColumns(archiveDb) {
return { columns, effective: candidates.length === 1 ? candidates[0] : `COALESCE(${candidates.join(', ')})` };
}
function replayPrompt(article) {
// bumped when the prompt changes in a way that changes what a prediction means.
const STRATEGY_VERSION = 'autonomy-2';
const PROMPT_VERSION = 'replay-coordinator-2';
function replayPrompt(article, feedbackBrief = '') {
return `Historical evidence cutoff: ${article.effective_at}\n\n` +
(feedbackBrief ? `${feedbackBrief}\n\n` : '') +
`[Evidence 1] article_id=${article.id}\nTitle: ${article.title || ''}\n${String(article.content || article.description || '').slice(0, 6000)}\n\n` +
`Return JSON only in this shape:\n${JSON.stringify({ predictions: [{
instrument: '<ticker supported by the evidence>', direction: '<positive or negative>',
@@ -46,15 +51,44 @@ function replayPrompt(article) {
function activeRun(db, config) {
let run = db.prepare("SELECT * FROM autonomy_replay_runs WHERE status = 'running' ORDER BY id DESC LIMIT 1").get();
if (run) return run;
// A fresh run used to recompute its watermark from today, so the run after a
// pause covered a different archive slice and could never be compared with
// the one before it. Inherit instead, and only fall back to the env window
// when there is genuinely no predecessor.
const previous = db.prepare('SELECT * FROM autonomy_replay_runs ORDER BY id DESC LIMIT 1').get();
const watermarkDays = Math.max(1, Number(process.env.AUTONOMY_REPLAY_WATERMARK_DAYS) || 7);
const result = db.prepare(`
INSERT INTO autonomy_replay_runs (watermark_at, strategy_version, prompt_version, coordinator_model)
VALUES (datetime('now', ?), 'autonomy-1', 'replay-coordinator-1', ?)
`).run(`-${watermarkDays} days`, config.openRouter.llmModel || 'unknown');
const watermark = previous && previous.watermark_at ? previous.watermark_at : null;
const result = watermark
? db.prepare(`INSERT INTO autonomy_replay_runs (watermark_at, strategy_version, prompt_version, coordinator_model, parent_run_id)
VALUES (?, ?, ?, ?, ?)`).run(watermark, STRATEGY_VERSION, PROMPT_VERSION, config.openRouter.llmModel || 'unknown', previous.id)
: db.prepare(`INSERT INTO autonomy_replay_runs (watermark_at, strategy_version, prompt_version, coordinator_model)
VALUES (datetime('now', ?), ?, ?, ?)`).run(`-${watermarkDays} days`, STRATEGY_VERSION, PROMPT_VERSION, config.openRouter.llmModel || 'unknown');
return db.prepare('SELECT * FROM autonomy_replay_runs WHERE id = ?').get(result.lastInsertRowid);
}
// A run that has been handed an explicit article set walks only that set. This
// is what makes two runs comparable, they answer the same articles instead of
// two different windows of the archive.
function pinnedArticles(db, run, cursorEffectiveAt, cursorArticleId, limit) {
const cursorFilter = cursorEffectiveAt
? 'AND (effective_at > ? OR (effective_at = ? AND article_id > ?))'
: '';
const params = [run.id];
if (cursorEffectiveAt) params.push(cursorEffectiveAt, cursorEffectiveAt, cursorArticleId);
return db.prepare(`
SELECT article_id, effective_at FROM autonomy_replay_run_articles
WHERE run_id = ? ${cursorFilter}
ORDER BY effective_at ASC, article_id ASC LIMIT ${limit}
`).all(...params);
}
function hasPinnedSet(db, run) {
return !!db.prepare('SELECT 1 FROM autonomy_replay_run_articles WHERE run_id = ? LIMIT 1').get(run.id);
}
function scheduleNext(db, archiveDb, run) {
if (hasPinnedSet(db, run)) return schedulePinned(db, archiveDb, run);
const { columns, effective } = articleTimeColumns(archiveDb);
const content = columns.has('content') ? "content IS NOT NULL AND content != ''" : '1=1';
const indexFilter = columns.has('is_index_page') ? 'AND (is_index_page = 0 OR is_index_page IS NULL)' : '';
@@ -99,6 +133,43 @@ function scheduleNext(db, archiveDb, run) {
throw new Error('replay scheduler skipped too many terminal jobs in one pass');
}
function schedulePinned(db, archiveDb, run) {
const { effective } = articleTimeColumns(archiveDb);
let cursorEffectiveAt = run.cursor_effective_at;
let cursorArticleId = run.cursor_article_id;
for (let skipped = 0; skipped < 100; skipped += 1) {
const [next] = pinnedArticles(db, run, cursorEffectiveAt, cursorArticleId, 1);
if (!next) return null;
const article = archiveDb.prepare(
`SELECT id, title, description, content, ${effective} AS effective_at FROM articles WHERE id = ?`
).get(next.article_id);
const idempotencyKey = `replay:${run.id}:article:${next.article_id}`;
const existing = db.prepare('SELECT status, last_error FROM autonomy_jobs WHERE idempotency_key = ?').get(idempotencyKey);
const terminal = existing && ['complete', 'dead_letter'].includes(existing.status);
if (!article || terminal) {
db.prepare(`UPDATE autonomy_replay_runs SET cursor_article_id=?, cursor_effective_at=?, last_error=?, updated_at=datetime('now') WHERE id=?`)
.run(next.article_id, next.effective_at, !article
? `Pinned article ${next.article_id} is no longer in the archive`
: (existing.status === 'dead_letter'
? `Skipped dead-letter replay job for article ${next.article_id}: ${existing.last_error || 'unknown error'}`
: run.last_error), run.id);
if (!article) console.error(`[replay] run ${run.id} pinned article ${next.article_id} is missing from the archive, skipping`);
cursorEffectiveAt = next.effective_at;
cursorArticleId = next.article_id;
continue;
}
enqueueJob(db, {
jobType: 'replay_article', lane: 'historical', priority: 1, entityType: 'article', entityId: next.article_id,
idempotencyKey,
});
return article;
}
throw new Error('pinned replay scheduler skipped too many terminal jobs in one pass');
}
async function runReplayWorker({ archivePath, intelligencePath, workerId = `replay-${os.hostname()}-${process.pid}`, pollMs = 15000 } = {}) {
const archiveDb = openRuntimeDb(archivePath, { schema: 'archive', readonly: true });
const db = openRuntimeDb(intelligencePath, { schema: 'intelligence' });
@@ -119,16 +190,17 @@ async function runReplayWorker({ archivePath, intelligencePath, workerId = `repl
const { effective } = articleTimeColumns(archiveDb);
const article = archiveDb.prepare(`SELECT id, title, description, content, ${effective} AS effective_at FROM articles WHERE id=?`).get(job.entity_id);
if (!article || !article.effective_at) throw new Error(`replay article ${job.entity_id} is unavailable`);
const raw = await callCoordinator(config, replayPrompt(article));
const raw = await callCoordinator(config, replayPrompt(article, run.feedback_brief || ''));
try {
acceptProposal(db, archiveDb, raw, {
informationCutoff: article.effective_at, model: config.openRouter.llmModel || 'unknown',
promptVersion: 'replay-coordinator-1', strategyVersion: 'autonomy-1', learningEligible: false,
promptVersion: run.prompt_version || PROMPT_VERSION,
strategyVersion: run.strategy_version || STRATEGY_VERSION, learningEligible: false,
origin: 'replay', replayRunId: run.id,
});
} catch (validationError) {
console.error(`[${workerId}] replay proposal rejected for article ${article.id}:`, validationError.message);
recordRejectedProposal(db, raw, { informationCutoff: article.effective_at, model: config.openRouter.llmModel || 'unknown', promptVersion: 'replay-coordinator-1', origin: 'replay' }, validationError.message);
recordRejectedProposal(db, raw, { informationCutoff: article.effective_at, model: config.openRouter.llmModel || 'unknown', promptVersion: run.prompt_version || PROMPT_VERSION, origin: 'replay' }, validationError.message);
}
db.prepare(`UPDATE autonomy_replay_runs SET cursor_article_id=?, cursor_effective_at=?, processed_articles=processed_articles+1, updated_at=datetime('now') WHERE id=?`)
.run(article.id, article.effective_at, run.id);
@@ -139,4 +211,5 @@ async function runReplayWorker({ archivePath, intelligencePath, workerId = `repl
}
}
module.exports = { articleTimeColumns, replayPrompt, scheduleNext, runReplayWorker };
module.exports = { articleTimeColumns, replayPrompt, scheduleNext, schedulePinned, runReplayWorker,
STRATEGY_VERSION, PROMPT_VERSION };