feat: let the generator read its own results, and measure it honestly
Nothing in the pipeline has ever fed an outcome back to the thing that makes predictions. Calibration reads autonomy_outcomes, but calibration only gates whether to act on a prediction, never what the prediction is. So the only thing that has ever changed this system's output is a human editing the prompt. score-replay-runs.js asks "compared to what". The answer is not flattering: over 2,114 scored replay predictions the system is right 50.05% of the time while answering "negative" to every one of the same bars scores 54.45%. It is 4.4 points below a constant, z=-4.06. The whole deficit is the prior. It says positive on 65% of calls when 45.5% of bars beat SPY, a 19 point skew. Its discrimination, P(up|positive) minus P(up|negative), is +3.2 points with p=0.15, so the direction it picks is weakly informative and completely buried by how often it defaults to positive. The first version of that script compared each direction group's accuracy to "always that direction" on the same rows, which is an identity and tests nothing. T3 replaces it with the two proportion test that actually asks whether the choice of direction carries information. build-feedback-brief.js turns a run's scored outcomes into a memo the next run reads before predicting. Generated from the data, not written by hand, or it is just me editing the prompt again with extra steps. Replay can now be pinned to an explicit article set, which is what makes two runs comparable at all. Comparing two calendar windows of one run compares two market regimes: the epochs in run 1 line up exactly with article vintage, E0 is late 2024 and E2 is 2026, so nothing could be attributed. A new run also inherits its parent's watermark instead of recomputing it from today, which silently guaranteed a different archive slice every time. prompt_version never moved across four material prompt changes, so every proposal on record claims to come from the first prompt. coordinator-2 and replay-coordinator-2. docs/replay-run-2-preregistration.md fixes the bar before the run exists, including which result counts as learning and which is only calibration. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
+72
-1
@@ -14,7 +14,7 @@ const { createOrderIntent } = require('../src/autonomy/orderIntents');
|
||||
const { enqueueCoordinatorEvent, reconcileArchiveBatch, reconcileLiveBatch, isTransientCoordinatorFailure } = require('../workers/autonomyWorker');
|
||||
const { buildGraphContext } = require('../src/autonomy/graphContext');
|
||||
const { buildPrompt } = require('../workers/coordinatorWorker');
|
||||
const { scheduleNext } = require('../workers/replayWorker');
|
||||
const { scheduleNext, replayPrompt } = require('../workers/replayWorker');
|
||||
const { refreshHistoricalCalibration, createDecisions, ensureCalibrationColumns } = require('../workers/calibrationWorker');
|
||||
|
||||
test('autonomy schema and leased jobs are restart-safe', () => {
|
||||
@@ -112,6 +112,77 @@ test('replay scheduler skips terminal replay jobs instead of pinning the cursor'
|
||||
assert.equal(intelligence.prepare("SELECT COUNT(*) count FROM autonomy_jobs WHERE status='pending' AND entity_id='2'").get().count, 1);
|
||||
});
|
||||
|
||||
test('a run with a pinned article set walks only that set, in date order', () => {
|
||||
const archive = new Database(':memory:');
|
||||
archive.exec(`
|
||||
CREATE TABLE articles (
|
||||
id INTEGER PRIMARY KEY, title TEXT, description TEXT, content TEXT,
|
||||
pub_date_effective TEXT, is_index_page INTEGER
|
||||
)
|
||||
`);
|
||||
for (const id of [1, 2, 3, 4]) {
|
||||
archive.prepare('INSERT INTO articles VALUES (?, ?, ?, ?, ?, 0)')
|
||||
.run(id, `a${id}`, '', 'content', `2020-01-0${id}T00:00:00Z`);
|
||||
}
|
||||
|
||||
const intelligence = new Database(':memory:');
|
||||
initAutonomySchema(intelligence);
|
||||
const runId = intelligence.prepare(`
|
||||
INSERT INTO autonomy_replay_runs (watermark_at, strategy_version, prompt_version, coordinator_model)
|
||||
VALUES ('2020-01-09T00:00:00Z', 'test', 'test', 'test')
|
||||
`).run().lastInsertRowid;
|
||||
// deliberately out of order and deliberately not article 1, the whole point
|
||||
// is that the run ignores the archive walk and answers these
|
||||
for (const [articleId, at] of [[4, '2020-01-04T00:00:00Z'], [2, '2020-01-02T00:00:00Z']]) {
|
||||
intelligence.prepare('INSERT INTO autonomy_replay_run_articles (run_id, article_id, effective_at) VALUES (?, ?, ?)')
|
||||
.run(runId, articleId, at);
|
||||
}
|
||||
|
||||
const run = intelligence.prepare('SELECT * FROM autonomy_replay_runs WHERE id=?').get(runId);
|
||||
const first = scheduleNext(intelligence, archive, run);
|
||||
assert.equal(first.id, 2, 'earliest pinned article first, not article 1');
|
||||
|
||||
intelligence.prepare("UPDATE autonomy_jobs SET status='complete' WHERE entity_id='2'").run();
|
||||
intelligence.prepare('UPDATE autonomy_replay_runs SET cursor_article_id=2, cursor_effective_at=? WHERE id=?')
|
||||
.run('2020-01-02T00:00:00Z', runId);
|
||||
const second = scheduleNext(intelligence, archive, intelligence.prepare('SELECT * FROM autonomy_replay_runs WHERE id=?').get(runId));
|
||||
assert.equal(second.id, 4);
|
||||
|
||||
intelligence.prepare("UPDATE autonomy_jobs SET status='complete' WHERE entity_id='4'").run();
|
||||
intelligence.prepare('UPDATE autonomy_replay_runs SET cursor_article_id=4, cursor_effective_at=? WHERE id=?')
|
||||
.run('2020-01-04T00:00:00Z', runId);
|
||||
const exhausted = scheduleNext(intelligence, archive, intelligence.prepare('SELECT * FROM autonomy_replay_runs WHERE id=?').get(runId));
|
||||
assert.equal(exhausted, null, 'a pinned run stops when its set is done, it does not fall back to the archive');
|
||||
});
|
||||
|
||||
test('a run with no pinned set still walks the archive exactly as before', () => {
|
||||
const archive = new Database(':memory:');
|
||||
archive.exec(`
|
||||
CREATE TABLE articles (
|
||||
id INTEGER PRIMARY KEY, title TEXT, description TEXT, content TEXT,
|
||||
pub_date_effective TEXT, is_index_page INTEGER
|
||||
)
|
||||
`);
|
||||
archive.prepare("INSERT INTO articles VALUES (7, 'only', '', 'content', '2020-01-01T00:00:00Z', 0)").run();
|
||||
const intelligence = new Database(':memory:');
|
||||
initAutonomySchema(intelligence);
|
||||
const runId = intelligence.prepare(`
|
||||
INSERT INTO autonomy_replay_runs (watermark_at, strategy_version, prompt_version, coordinator_model)
|
||||
VALUES ('2020-01-09T00:00:00Z', 'test', 'test', 'test')
|
||||
`).run().lastInsertRowid;
|
||||
const run = intelligence.prepare('SELECT * FROM autonomy_replay_runs WHERE id=?').get(runId);
|
||||
assert.equal(scheduleNext(intelligence, archive, run).id, 7);
|
||||
});
|
||||
|
||||
test('the feedback brief reaches the prompt and stays out of it when empty', () => {
|
||||
const article = { id: 1, title: 't', content: 'body', effective_at: '2020-01-01T00:00:00Z' };
|
||||
const withBrief = replayPrompt(article, 'CALIBRATION FEEDBACK. you over-call positive.');
|
||||
assert.ok(withBrief.includes('you over-call positive'));
|
||||
assert.ok(withBrief.indexOf('CALIBRATION FEEDBACK') < withBrief.indexOf('[Evidence 1]'),
|
||||
'the brief has to land before the evidence, not after it');
|
||||
assert.ok(!replayPrompt(article).includes('CALIBRATION FEEDBACK'));
|
||||
});
|
||||
|
||||
test('calibration and policy abstain on insufficient evidence', () => {
|
||||
const calibration = calibrateOutcomes([
|
||||
{ excess_return: 0.02, direction_correct: 1 },
|
||||
|
||||
Reference in New Issue
Block a user