fix: a recovered replay job can no longer hijack the active run

leaseNextJob hands back any pending replay_article job, it has no idea about
runs, and the worker was attributing whatever came back to whichever run was
active. One recovered dead letter from run 1 would have been stamped with run
2's id, given run 2's feedback brief, and dragged run 2's cursor to wherever
that old article sits in the archive. A pinned run would then decide its set was
finished after a couple of articles. There are 139 dead letters and they are
built to recover, so this was not hypothetical.

The idempotency key already says which run enqueued the job. Ask it.

Also: refuse to inherit the parent's model label when starting a run. Inheriting
is exactly how run 1 came to be labelled qwen for predictions deepseek made.

The split moves to the replay container's actual restart time rather than the
commit timestamp five minutes later. Verified the running container really does
have the instrument rules, the de-anchoring and the enum before trusting it as
the boundary. It makes no difference to the partition, there are no replay
predictions at all between 15:57 and midnight that day, but the boundary should
be the thing that actually changed the prompt.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-09-08 01:01:36 +01:00
co-authored by Claude Opus 5
parent ec29e64e96
commit b246bd9d4b
5 changed files with 109 additions and 17 deletions
+32 -8
View File
@@ -170,6 +170,29 @@ function schedulePinned(db, archiveDb, run) {
throw new Error('pinned replay scheduler skipped too many terminal jobs in one pass');
}
// leaseNextJob hands back any pending replay_article job, it knows nothing
// about runs. Attributing whatever comes back to the currently active run means
// one recovered dead letter from an older run gets that run's article stamped
// with the new run's id, the new run's brief in its prompt, and worst of all
// drags the new run's cursor to wherever that old article sat in the archive.
// A pinned run would then find its set "finished" after a couple of articles.
// The job says which run it belongs to, so ask the job.
function runForJob(db, job, fallback) {
const match = /^replay:(\d+):article:/.exec(String(job.idempotency_key || ''));
if (!match) {
console.error(`[replay] job ${job.id} has no run in its idempotency key`
+ ` (${job.idempotency_key}), attributing it to run ${fallback.id}`);
return fallback;
}
const run = db.prepare('SELECT * FROM autonomy_replay_runs WHERE id = ?').get(Number(match[1]));
if (!run) {
console.error(`[replay] job ${job.id} points at run ${match[1]} which no longer exists,`
+ ` attributing it to run ${fallback.id}`);
return fallback;
}
return run;
}
async function runReplayWorker({ archivePath, intelligencePath, workerId = `replay-${os.hostname()}-${process.pid}`, pollMs = 15000 } = {}) {
const archiveDb = openRuntimeDb(archivePath, { schema: 'archive', readonly: true });
const db = openRuntimeDb(intelligencePath, { schema: 'intelligence' });
@@ -186,24 +209,25 @@ async function runReplayWorker({ archivePath, intelligencePath, workerId = `repl
scheduleNext(db, archiveDb, run);
const job = leaseNextJob(db, workerId, 300, ['replay_article']);
if (!job) { await sleep(pollMs); continue; }
const owner = runForJob(db, job, run);
try {
const { effective } = articleTimeColumns(archiveDb);
const article = archiveDb.prepare(`SELECT id, title, description, content, ${effective} AS effective_at FROM articles WHERE id=?`).get(job.entity_id);
if (!article || !article.effective_at) throw new Error(`replay article ${job.entity_id} is unavailable`);
const raw = await callCoordinator(config, replayPrompt(article, run.feedback_brief || ''));
const raw = await callCoordinator(config, replayPrompt(article, owner.feedback_brief || ''));
try {
acceptProposal(db, archiveDb, raw, {
informationCutoff: article.effective_at, model: config.openRouter.llmModel || 'unknown',
promptVersion: run.prompt_version || PROMPT_VERSION,
strategyVersion: run.strategy_version || STRATEGY_VERSION, learningEligible: false,
origin: 'replay', replayRunId: run.id,
promptVersion: owner.prompt_version || PROMPT_VERSION,
strategyVersion: owner.strategy_version || STRATEGY_VERSION, learningEligible: false,
origin: 'replay', replayRunId: owner.id,
});
} catch (validationError) {
console.error(`[${workerId}] replay proposal rejected for article ${article.id}:`, validationError.message);
recordRejectedProposal(db, raw, { informationCutoff: article.effective_at, model: config.openRouter.llmModel || 'unknown', promptVersion: run.prompt_version || PROMPT_VERSION, origin: 'replay' }, validationError.message);
recordRejectedProposal(db, raw, { informationCutoff: article.effective_at, model: config.openRouter.llmModel || 'unknown', promptVersion: owner.prompt_version || PROMPT_VERSION, origin: 'replay' }, validationError.message);
}
db.prepare(`UPDATE autonomy_replay_runs SET cursor_article_id=?, cursor_effective_at=?, processed_articles=processed_articles+1, updated_at=datetime('now') WHERE id=?`)
.run(article.id, article.effective_at, run.id);
.run(article.id, article.effective_at, owner.id);
completeJob(db, job.id, workerId);
} catch (error) { failJob(db, job.id, workerId, error); }
} catch (error) { console.error(`[${workerId}] replay error:`, error.message); }
@@ -211,5 +235,5 @@ async function runReplayWorker({ archivePath, intelligencePath, workerId = `repl
}
}
module.exports = { articleTimeColumns, replayPrompt, scheduleNext, schedulePinned, runReplayWorker,
STRATEGY_VERSION, PROMPT_VERSION };
module.exports = { articleTimeColumns, replayPrompt, scheduleNext, schedulePinned, runForJob,
runReplayWorker, STRATEGY_VERSION, PROMPT_VERSION };