Nothing in the pipeline has ever fed an outcome back to the thing that makes predictions. Calibration reads autonomy_outcomes, but calibration only gates whether to act on a prediction, never what the prediction is. So the only thing that has ever changed this system's output is a human editing the prompt. score-replay-runs.js asks "compared to what". The answer is not flattering: over 2,114 scored replay predictions the system is right 50.05% of the time while answering "negative" to every one of the same bars scores 54.45%. It is 4.4 points below a constant, z=-4.06. The whole deficit is the prior. It says positive on 65% of calls when 45.5% of bars beat SPY, a 19 point skew. Its discrimination, P(up|positive) minus P(up|negative), is +3.2 points with p=0.15, so the direction it picks is weakly informative and completely buried by how often it defaults to positive. The first version of that script compared each direction group's accuracy to "always that direction" on the same rows, which is an identity and tests nothing. T3 replaces it with the two proportion test that actually asks whether the choice of direction carries information. build-feedback-brief.js turns a run's scored outcomes into a memo the next run reads before predicting. Generated from the data, not written by hand, or it is just me editing the prompt again with extra steps. Replay can now be pinned to an explicit article set, which is what makes two runs comparable at all. Comparing two calendar windows of one run compares two market regimes: the epochs in run 1 line up exactly with article vintage, E0 is late 2024 and E2 is 2026, so nothing could be attributed. A new run also inherits its parent's watermark instead of recomputing it from today, which silently guaranteed a different archive slice every time. prompt_version never moved across four material prompt changes, so every proposal on record claims to come from the first prompt. coordinator-2 and replay-coordinator-2. docs/replay-run-2-preregistration.md fixes the bar before the run exists, including which result counts as learning and which is only calibration. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
105 lines
6.0 KiB
JavaScript
105 lines
6.0 KiB
JavaScript
const os = require('os');
|
|
const fs = require('fs');
|
|
const path = require('path');
|
|
const { openRuntimeDb } = require('../src/db/runtime');
|
|
const { initAutonomySchema } = require('../src/autonomy/schema');
|
|
const { leaseNextJob, completeJob, failJob } = require('../src/autonomy/jobs');
|
|
const { callCoordinator } = require('../src/autonomy/llm');
|
|
const { acceptProposal, recordRejectedProposal, INSTRUMENT_RULES } = require('../src/autonomy/coordinator');
|
|
const { EVENT_FAMILY_NAMES } = require('../src/autonomy/calibration');
|
|
const { buildGraphContext } = require('../src/autonomy/graphContext');
|
|
|
|
function sleep(ms) { return new Promise((resolve) => setTimeout(resolve, ms)); }
|
|
|
|
// bumped when the prompt changes in a way that changes what a prediction means.
|
|
// autonomy-2 is the first version that can see the relationship graph.
|
|
const STRATEGY_VERSION = 'autonomy-2';
|
|
// This never moved across four prompt changes, so every proposal on record
|
|
// claims to come from the same prompt as the very first one. Attribution was
|
|
// impossible, which is why the only honest before/after we had was a timestamp.
|
|
const PROMPT_VERSION = 'coordinator-2';
|
|
|
|
function loadConfig() {
|
|
const configPath = path.resolve(process.env.DURIIN_CONFIG || path.join(__dirname, '..', 'config.json'));
|
|
const raw = JSON.parse(fs.readFileSync(configPath, 'utf8'));
|
|
require('dotenv').config({ path: path.resolve(path.dirname(configPath), '.env') });
|
|
raw.openRouter = { ...(raw.openRouter || {}) };
|
|
if (process.env.OPEN_ROUTER_API_KEY) raw.openRouter.apiKey = process.env.OPEN_ROUTER_API_KEY;
|
|
if (process.env.OPEN_ROUTER_LLM_MODEL) raw.openRouter.llmModel = process.env.OPEN_ROUTER_LLM_MODEL;
|
|
return raw;
|
|
}
|
|
|
|
function buildPrompt(event, articles, graphContext = '') {
|
|
const evidence = articles.map((article, index) =>
|
|
`[Evidence ${index + 1}] article_id=${article.id}\nTitle: ${article.title}\n${String(article.content || article.description || '').slice(0, 4000)}`
|
|
).join('\n\n---\n\n');
|
|
return `Event title: ${event.title}\n\n${evidence}\n\n${graphContext ? `${graphContext}\n\n` : ''}Return JSON only in this shape:\n${JSON.stringify({ predictions: [{
|
|
instrument: '<ticker supported by the articles>', direction: '<positive or negative>',
|
|
event_type: '<one value from the event_type list below>',
|
|
causal_channel: '<short description>', horizon_days: '<one of 1, 5, 10, 20, 30, 60, 90>',
|
|
evidence_article_ids: ['<article_id values from the evidence above>'],
|
|
invalidation_condition: '<what would falsify this>',
|
|
}] }, null, 2)}\n\nEvery value in that shape is a placeholder describing the field. Do not copy them. Choose instrument, direction and horizon_days from the evidence in front of you.\n\nevent_type must be exactly one of: ${EVENT_FAMILY_NAMES.join(', ')}. Pick the closest one. Use "other" only when none of them genuinely apply, and never invent a value outside this list.\n\n${INSTRUMENT_RULES}\n\nUse only instruments and evidence directly supported by the articles. Return an empty predictions array when there is no clear, tradable hypothesis. Never include probabilities, returns, confidence, position sizes, or actions.`;
|
|
}
|
|
|
|
async function runCoordinatorWorker({ archivePath, intelligencePath, workerId = `coordinator-${os.hostname()}-${process.pid}`, pollMs = 1000 } = {}) {
|
|
const archiveDb = openRuntimeDb(archivePath, { schema: 'archive', readonly: true });
|
|
const intelligenceDb = openRuntimeDb(intelligencePath, { schema: 'intelligence' });
|
|
intelligenceDb.pragma('journal_mode = WAL');
|
|
intelligenceDb.pragma('busy_timeout = 5000');
|
|
initAutonomySchema(intelligenceDb);
|
|
const config = loadConfig();
|
|
while (true) {
|
|
const job = leaseNextJob(intelligenceDb, workerId, 180, ['coordinator_event']);
|
|
if (!job) { await sleep(pollMs); continue; }
|
|
try {
|
|
const event = archiveDb.prepare('SELECT id, title FROM events WHERE id = ?').get(job.entity_id);
|
|
if (!event) throw new Error(`event ${job.entity_id} does not exist`);
|
|
const articles = archiveDb.prepare(`
|
|
SELECT id, title, description, content, pub_date_effective
|
|
FROM articles
|
|
WHERE event_id = ? AND content IS NOT NULL AND content != '' AND is_index_page = 0
|
|
ORDER BY pub_date_effective ASC, id ASC LIMIT 25
|
|
`).all(job.entity_id);
|
|
const allowlisted = intelligenceDb.prepare(
|
|
"SELECT 1 FROM autonomy_instruments WHERE active=1 AND tradable=1 LIMIT 1"
|
|
).get();
|
|
if (!allowlisted) throw new Error('no tradable instruments are allowlisted');
|
|
const historical = job.lane === 'historical';
|
|
const informationCutoff = historical
|
|
? (articles.map((article) => article.pub_date_effective).filter(Boolean).sort().pop() || new Date().toISOString())
|
|
: new Date().toISOString();
|
|
const graphContext = buildGraphContext(intelligenceDb, event.id, informationCutoff);
|
|
const raw = await callCoordinator(config, buildPrompt(event, articles, graphContext));
|
|
try {
|
|
acceptProposal(intelligenceDb, archiveDb, raw, {
|
|
eventId: event.id,
|
|
informationCutoff,
|
|
model: config.openRouter.llmModel || 'unknown',
|
|
promptVersion: PROMPT_VERSION,
|
|
strategyVersion: STRATEGY_VERSION,
|
|
// only a genuine live lane job may ever feed learning
|
|
origin: historical ? 'historical' : 'live',
|
|
learningEligible: !historical,
|
|
});
|
|
} catch (validationError) {
|
|
console.error(`[${workerId}] proposal rejected for event ${event.id}:`, validationError.message);
|
|
recordRejectedProposal(intelligenceDb, raw, {
|
|
eventId: event.id,
|
|
informationCutoff,
|
|
model: config.openRouter.llmModel || 'unknown',
|
|
promptVersion: PROMPT_VERSION,
|
|
origin: historical ? 'historical' : 'live',
|
|
learningEligible: !historical,
|
|
}, validationError.message);
|
|
}
|
|
completeJob(intelligenceDb, job.id, workerId);
|
|
} catch (error) {
|
|
failJob(intelligenceDb, job.id, workerId, error);
|
|
}
|
|
await sleep(pollMs);
|
|
}
|
|
}
|
|
|
|
module.exports = { buildPrompt, runCoordinatorWorker };
|