The event outcome worker re-requested PSTG and GROQ on every poll for as long as the process lived, because a fetch failure only logged and continued while the prediction stayed pending. Ten requests every five minutes, indefinitely. Per ticker backoff now doubles to an hour, so a symbol with no market data costs one request an hour instead of one a minute. It also translates dotted tickers the same way the autonomy worker does, which is why that helper moved into the shared price module rather than being copied. The gdelt loop had no pause on its error path at all, so once the api started refusing connections it spun through failures continuously, burning cpu and filling the log with the same stack. It has been doing that for days. Backs off to half an hour now and resets on success. isTransientCoordinatorFailure matched 408, 429 and 5xx but not a budget 402/403, so the 380 jobs that dead-lettered during the exhausted quota window could never come back on their own, including 55 live events. Budget failures are transient in a way an ordinary auth failure is not, and a wrong key still dies permanently because it says invalid or unauthorized rather than naming credits. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01WnNxwxfXSbeNtjvtz5gayb
145 lines
5.9 KiB
JavaScript
145 lines
5.9 KiB
JavaScript
const os = require('os');
|
|
const { openRuntimeDb } = require('../src/db/runtime');
|
|
const { initAutonomySchema } = require('../src/autonomy/schema');
|
|
const { enqueueJob, leaseNextJob, completeJob, failJob } = require('../src/autonomy/jobs');
|
|
|
|
function sleep(ms) { return new Promise((resolve) => setTimeout(resolve, ms)); }
|
|
|
|
function isTransientCoordinatorFailure(error) {
|
|
const value = String(error || '').toLowerCase();
|
|
return value.includes('fetch failed')
|
|
|| value.includes('network')
|
|
|| value.includes('timeout')
|
|
|| value.includes('before response')
|
|
|| /\b(408|429|5\d\d)\b/.test(value)
|
|
// A 402/403 for budget is temporary in a way an ordinary auth failure is not:
|
|
// monthly limits reset and credits get topped up. Without this, 380 jobs
|
|
// dead-lettered during one exhausted window and could never come back on
|
|
// their own, including 55 live events. A wrong key still fails permanently,
|
|
// because that says "invalid" or "unauthorized" rather than naming credits.
|
|
|| (/\b(402|403)\b/.test(value)
|
|
&& /credit|quota|key limit|afford|budget|exceeded/.test(value));
|
|
}
|
|
|
|
function enqueueCoordinatorEvent(intelligenceDb, row) {
|
|
const isLive = row.ingested_at && Date.now() - Date.parse(row.ingested_at) <= 48 * 60 * 60 * 1000;
|
|
const lane = isLive ? 'live' : 'historical';
|
|
const priority = isLive ? 100 : 10;
|
|
const idempotencyKey = `coordinator_event:${row.event_id}`;
|
|
const result = enqueueJob(intelligenceDb, {
|
|
jobType: 'coordinator_event',
|
|
lane,
|
|
priority,
|
|
entityType: 'event',
|
|
entityId: row.event_id,
|
|
idempotencyKey,
|
|
});
|
|
if (result.inserted) return { inserted: true, recovered: false };
|
|
|
|
const existing = intelligenceDb.prepare(`
|
|
SELECT id, status, last_error
|
|
FROM autonomy_jobs
|
|
WHERE idempotency_key = ? AND job_type = 'coordinator_event'
|
|
`).get(idempotencyKey);
|
|
if (!existing || existing.status !== 'dead_letter' || !isTransientCoordinatorFailure(existing.last_error)) {
|
|
return { inserted: false, recovered: false };
|
|
}
|
|
|
|
const recovered = intelligenceDb.prepare(`
|
|
UPDATE autonomy_jobs
|
|
SET status = 'pending',
|
|
lane = ?,
|
|
priority = ?,
|
|
attempts = 0,
|
|
available_at = datetime('now'),
|
|
leased_by = NULL,
|
|
lease_expires_at = NULL,
|
|
last_error = ?
|
|
WHERE id = ? AND status = 'dead_letter'
|
|
`).run(lane, priority, `Recovered transient coordinator failure: ${existing.last_error || 'unknown error'}`, existing.id);
|
|
return { inserted: false, recovered: recovered.changes > 0 };
|
|
}
|
|
|
|
function reconcileArchiveBatch(archiveDb, intelligenceDb, batchSize = 250) {
|
|
const cursor = intelligenceDb.prepare("SELECT value FROM autonomy_cursors WHERE key = 'archive_reconcile'").get();
|
|
const afterId = cursor ? cursor.value : 0;
|
|
const rows = archiveDb.prepare(`
|
|
SELECT id, event_id, ingested_at, content, has_embedding
|
|
FROM articles
|
|
WHERE id > ?
|
|
ORDER BY id ASC
|
|
LIMIT ?
|
|
`).all(afterId, batchSize);
|
|
if (!rows.length) {
|
|
intelligenceDb.prepare(`
|
|
INSERT INTO autonomy_cursors(key, value) VALUES ('archive_reconcile', 0)
|
|
ON CONFLICT(key) DO UPDATE SET value = 0, updated_at = datetime('now')
|
|
`).run();
|
|
return { scanned: 0, nextCursor: 0, reset: true };
|
|
}
|
|
const enqueue = intelligenceDb.transaction(() => {
|
|
for (const row of rows) {
|
|
const readyForIntelligence = row.event_id && row.content && row.has_embedding;
|
|
if (readyForIntelligence) {
|
|
enqueueCoordinatorEvent(intelligenceDb, row);
|
|
}
|
|
}
|
|
intelligenceDb.prepare(`
|
|
INSERT INTO autonomy_cursors(key, value) VALUES ('archive_reconcile', ?)
|
|
ON CONFLICT(key) DO UPDATE SET value = excluded.value, updated_at = datetime('now')
|
|
`).run(rows[rows.length - 1].id);
|
|
});
|
|
enqueue();
|
|
return { scanned: rows.length, nextCursor: rows[rows.length - 1].id, reset: false };
|
|
}
|
|
|
|
function reconcileLiveBatch(archiveDb, intelligenceDb, batchSize = 250) {
|
|
const rows = archiveDb.prepare(`
|
|
SELECT id, event_id, ingested_at, content, has_embedding
|
|
FROM articles
|
|
WHERE ingested_at >= datetime('now', '-48 hours')
|
|
ORDER BY ingested_at DESC, id DESC
|
|
LIMIT ?
|
|
`).all(batchSize);
|
|
let queued = 0;
|
|
for (const row of rows) {
|
|
if (!row.event_id || !row.content || !row.has_embedding) continue;
|
|
const result = enqueueCoordinatorEvent(intelligenceDb, row);
|
|
if (result.inserted || result.recovered) queued++;
|
|
}
|
|
return { scanned: rows.length, queued };
|
|
}
|
|
|
|
async function runAutonomyWorker({ archivePath, intelligencePath, workerId = `autonomy-${os.hostname()}-${process.pid}`, pollMs = 1000 } = {}) {
|
|
const archiveDb = openRuntimeDb(archivePath, { schema: 'archive', readonly: true });
|
|
const intelligenceDb = openRuntimeDb(intelligencePath, { schema: 'intelligence' });
|
|
intelligenceDb.pragma('journal_mode = WAL');
|
|
intelligenceDb.pragma('busy_timeout = 5000');
|
|
initAutonomySchema(intelligenceDb);
|
|
|
|
while (true) {
|
|
// This worker owns maintenance reconciliation only. Without the type filter it
|
|
// can lease coordinator_event jobs and complete them without analysis.
|
|
const job = leaseNextJob(intelligenceDb, workerId, 120, ['reconcile_archive']);
|
|
if (!job) { await sleep(pollMs); continue; }
|
|
try {
|
|
if (job.job_type === 'reconcile_archive') {
|
|
reconcileLiveBatch(archiveDb, intelligenceDb);
|
|
reconcileArchiveBatch(archiveDb, intelligenceDb);
|
|
// Keep the reconciler alive as a bounded maintenance loop.
|
|
enqueueJob(intelligenceDb, {
|
|
jobType: 'reconcile_archive', lane: 'maintenance', priority: 100,
|
|
entityType: 'archive', entityId: 'archive',
|
|
idempotencyKey: `reconcile_archive:${Date.now()}`,
|
|
});
|
|
}
|
|
completeJob(intelligenceDb, job.id, workerId);
|
|
} catch (error) {
|
|
failJob(intelligenceDb, job.id, workerId, error);
|
|
}
|
|
await sleep(pollMs);
|
|
}
|
|
}
|
|
|
|
module.exports = { enqueueCoordinatorEvent, reconcileArchiveBatch, reconcileLiveBatch, runAutonomyWorker, isTransientCoordinatorFailure };
|