fix: stop the coordinator copying its own prompt example

The JSON shape in both prompts used real values as placeholders, and the model
was reading them as the answer:

  instrument: 'NVDA'          -> 397 of 611 predictions are NVDA (65%),
                                 second place is LMT with 8
  horizon_days: 10            -> 606 of 611 are horizon 10 (99.2%), out of
                                 seven allowed horizons
  direction: 'positive|negative' -> 502 of 611 are positive (82.2%)
  event_type: 'stable_enum'   -> the enum was never listed, so the model
                                 invented one label per event, 201 distinct
                                 values across 611 predictions

replayWorker had its own copy of the same prompt with the same values, which is
why both lanes show the identical skew (replay is 147/147 horizon 10, 138/147
NVDA).

Every placeholder is now a description of the field rather than a usable value,
with an explicit line saying not to copy them. event_type is validated against
the same closed family list the cohort key uses, so a label cannot mean one
thing in the prompt and another in calibration. Off-enum labels are salvaged
through the existing mapper when they are placeable and rejected when they are
not, so 'other' does not quietly become the bin again.

This does not by itself create edge. It means the next batch of predictions
measures the model's judgement instead of its willingness to copy an example.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
This commit is contained in:
ImBenji
2026-08-29 23:29:11 +01:00
co-authored by Claude Opus 5
parent c4650fe45c
commit b8b3987e35
4 changed files with 61 additions and 8 deletions
+24 -1
View File
@@ -1,5 +1,28 @@
const { EVENT_FAMILY_NAMES, normalizeEventType } = require('./calibration');
const ALLOWED_DIRECTIONS = new Set(['positive', 'negative']); const ALLOWED_DIRECTIONS = new Set(['positive', 'negative']);
const ALLOWED_HORIZONS = new Set([1, 5, 10, 20, 30, 60, 90]); const ALLOWED_HORIZONS = new Set([1, 5, 10, 20, 30, 60, 90]);
// The prompt used to say event_type was a 'stable_enum' and then never listed the
// enum, so the model invented one label per event and production ended up with 201
// distinct values. Same closed set the cohort key uses, so a label can never mean
// one thing in the prompt and another in calibration.
const ALLOWED_EVENT_TYPES = new Set(EVENT_FAMILY_NAMES);
// An exact family is what we want. If the model ignores the list we try to salvage
// the label through the same mapper calibration uses, and only give up when it is
// unplaceable -- an explicit 'other' is a legitimate answer, unplaceable free text
// is not, and the difference is what stops 'other' quietly becoming the bin again.
function normalizeProposedEventType(raw, instrument) {
const value = String(raw || '').trim().toLowerCase().replace(/[\s-]+/g, '_');
if (ALLOWED_EVENT_TYPES.has(value)) return value;
const salvaged = normalizeEventType(raw);
if (salvaged !== 'other') {
console.warn(`[coordinator] ${instrument} event_type "${raw}" is not in the enum, mapped to "${salvaged}"`);
return salvaged;
}
throw new Error(`event_type must be one of ${EVENT_FAMILY_NAMES.join(', ')} (got "${raw}")`);
}
function normalizeProposal(raw, { informationCutoff, model = 'unknown', promptVersion = 'unknown' } = {}) { function normalizeProposal(raw, { informationCutoff, model = 'unknown', promptVersion = 'unknown' } = {}) {
if (!raw || typeof raw !== 'object') throw new Error('coordinator output must be an object'); if (!raw || typeof raw !== 'object') throw new Error('coordinator output must be an object');
@@ -18,7 +41,7 @@ function normalizeProposal(raw, { informationCutoff, model = 'unknown', promptVe
return { return {
instrument, instrument,
direction, direction,
eventType: String(item.event_type || 'unknown').trim().toLowerCase(), eventType: normalizeProposedEventType(item.event_type, instrument),
causalChannel: item.causal_channel ? String(item.causal_channel).trim() : null, causalChannel: item.causal_channel ? String(item.causal_channel).trim() : null,
horizonDays, horizonDays,
evidenceArticleIds: [...new Set(articleIds)], evidenceArticleIds: [...new Set(articleIds)],
+22
View File
@@ -272,3 +272,25 @@ test('order intents require the explicit instrument allowlist', () => {
assert.equal(intent.side, 'buy'); assert.equal(intent.side, 'buy');
assert.equal(db.prepare('SELECT status FROM autonomy_order_intents').get().status, 'shadow'); assert.equal(db.prepare('SELECT status FROM autonomy_order_intents').get().status, 'shadow');
}); });
test('event_type is held to the closed family enum', () => {
const base = { instrument: 'NVDA', direction: 'positive', horizon_days: 10, evidence_article_ids: [1] };
const typeOf = (eventType) => normalizeProposal({ predictions: [{ ...base, event_type: eventType }] }).predictions[0].eventType;
// the enum itself, in the shapes a model actually emits
assert.equal(typeOf('earnings'), 'earnings');
assert.equal(typeOf('Supply Chain'), 'supply_chain');
assert.equal(typeOf('M_AND_A'), 'm_and_a');
// "none of these fit" is a legitimate answer and has to survive
assert.equal(typeOf('other'), 'other');
// off-enum but placeable: salvaged onto the family calibration would have
// picked anyway, so we dont throw away a usable prediction over a label
assert.equal(typeOf('earnings_beat_q3'), 'earnings');
assert.equal(typeOf('ceo resignation'), 'leadership');
// unplaceable free text is the 201-distinct-values failure, and is rejected
assert.throws(() => typeOf('vibes_shifted'), /event_type must be one of/);
assert.throws(() => typeOf(''), /event_type must be one of/);
});
+7 -4
View File
@@ -6,6 +6,7 @@ const { initAutonomySchema } = require('../src/autonomy/schema');
const { leaseNextJob, completeJob, failJob } = require('../src/autonomy/jobs'); const { leaseNextJob, completeJob, failJob } = require('../src/autonomy/jobs');
const { callCoordinator } = require('../src/autonomy/llm'); const { callCoordinator } = require('../src/autonomy/llm');
const { acceptProposal, recordRejectedProposal } = require('../src/autonomy/coordinator'); const { acceptProposal, recordRejectedProposal } = require('../src/autonomy/coordinator');
const { EVENT_FAMILY_NAMES } = require('../src/autonomy/calibration');
function sleep(ms) { return new Promise((resolve) => setTimeout(resolve, ms)); } function sleep(ms) { return new Promise((resolve) => setTimeout(resolve, ms)); }
@@ -24,10 +25,12 @@ function buildPrompt(event, articles) {
`[Evidence ${index + 1}] article_id=${article.id}\nTitle: ${article.title}\n${String(article.content || article.description || '').slice(0, 4000)}` `[Evidence ${index + 1}] article_id=${article.id}\nTitle: ${article.title}\n${String(article.content || article.description || '').slice(0, 4000)}`
).join('\n\n---\n\n'); ).join('\n\n---\n\n');
return `Event title: ${event.title}\n\n${evidence}\n\nReturn JSON only in this shape:\n${JSON.stringify({ predictions: [{ return `Event title: ${event.title}\n\n${evidence}\n\nReturn JSON only in this shape:\n${JSON.stringify({ predictions: [{
instrument: 'NVDA', direction: 'positive|negative', event_type: 'stable_enum', instrument: '<ticker supported by the articles>', direction: '<positive or negative>',
causal_channel: 'short description', horizon_days: 10, event_type: '<one value from the event_type list below>',
evidence_article_ids: [123], invalidation_condition: 'condition', causal_channel: '<short description>', horizon_days: '<one of 1, 5, 10, 20, 30, 60, 90>',
}] }, null, 2)}\n\nUse only instruments and evidence directly supported by the articles. Return an empty predictions array when there is no clear, tradable hypothesis. Never include probabilities, returns, confidence, position sizes, or actions.`; evidence_article_ids: ['<article_id values from the evidence above>'],
invalidation_condition: '<what would falsify this>',
}] }, null, 2)}\n\nEvery value in that shape is a placeholder describing the field. Do not copy them. Choose instrument, direction and horizon_days from the evidence in front of you.\n\nevent_type must be exactly one of: ${EVENT_FAMILY_NAMES.join(', ')}. Pick the closest one. Use "other" only when none of them genuinely apply, and never invent a value outside this list.\n\nUse only instruments and evidence directly supported by the articles. Return an empty predictions array when there is no clear, tradable hypothesis. Never include probabilities, returns, confidence, position sizes, or actions.`;
} }
async function runCoordinatorWorker({ archivePath, intelligencePath, workerId = `coordinator-${os.hostname()}-${process.pid}`, pollMs = 1000 } = {}) { async function runCoordinatorWorker({ archivePath, intelligencePath, workerId = `coordinator-${os.hostname()}-${process.pid}`, pollMs = 1000 } = {}) {
+8 -3
View File
@@ -6,6 +6,7 @@ const { initAutonomySchema } = require('../src/autonomy/schema');
const { enqueueJob, leaseNextJob, completeJob, failJob } = require('../src/autonomy/jobs'); const { enqueueJob, leaseNextJob, completeJob, failJob } = require('../src/autonomy/jobs');
const { callCoordinator } = require('../src/autonomy/llm'); const { callCoordinator } = require('../src/autonomy/llm');
const { acceptProposal, recordRejectedProposal } = require('../src/autonomy/coordinator'); const { acceptProposal, recordRejectedProposal } = require('../src/autonomy/coordinator');
const { EVENT_FAMILY_NAMES } = require('../src/autonomy/calibration');
function sleep(ms) { return new Promise((resolve) => setTimeout(resolve, ms)); } function sleep(ms) { return new Promise((resolve) => setTimeout(resolve, ms)); }
@@ -30,10 +31,14 @@ function replayPrompt(article) {
return `Historical evidence cutoff: ${article.effective_at}\n\n` + return `Historical evidence cutoff: ${article.effective_at}\n\n` +
`[Evidence 1] article_id=${article.id}\nTitle: ${article.title || ''}\n${String(article.content || article.description || '').slice(0, 6000)}\n\n` + `[Evidence 1] article_id=${article.id}\nTitle: ${article.title || ''}\n${String(article.content || article.description || '').slice(0, 6000)}\n\n` +
`Return JSON only in this shape:\n${JSON.stringify({ predictions: [{ `Return JSON only in this shape:\n${JSON.stringify({ predictions: [{
instrument: 'NVDA', direction: 'positive|negative', event_type: 'stable_enum', instrument: '<ticker supported by the evidence>', direction: '<positive or negative>',
causal_channel: 'short description', horizon_days: 10, evidence_article_ids: [123], event_type: '<one value from the event_type list below>',
invalidation_condition: 'condition', causal_channel: '<short description>', horizon_days: '<one of 1, 5, 10, 20, 30, 60, 90>',
evidence_article_ids: ['<article_id values from the evidence above>'],
invalidation_condition: '<what would falsify this>',
}] }, null, 2)}\n\n` + }] }, null, 2)}\n\n` +
'Every value in that shape is a placeholder describing the field. Do not copy them. Choose instrument, direction and horizon_days from the evidence in front of you.\n\n' +
`event_type must be exactly one of: ${EVENT_FAMILY_NAMES.join(', ')}. Pick the closest one. Use "other" only when none of them genuinely apply, and never invent a value outside this list.\n\n` +
'Use only this dated evidence. Return an empty predictions array when there is no clear, tradable hypothesis. Never include probabilities, returns, confidence, position sizes, or actions.'; 'Use only this dated evidence. Return an empty predictions array when there is no clear, tradable hypothesis. Never include probabilities, returns, confidence, position sizes, or actions.';
} }