Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 921048827a fix(remediation): pass LINK_EXTRACTOR_VERSION_TS to the extraction-lag gate
countExtractionLag omitted versionTs, so its predicate diverged from the
counter it claims to share with doctor's links_extraction_lag check and
the extract --stale walk: pages stamped before an extractor version bump
(links_extracted_at < LINK_EXTRACTOR_VERSION_TS) lagged for doctor and
extract but never tripped the sync.repo/extract.all gate. Pass the stamp;
pin the version-bump arm with a backdated-page test (fails without it).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 10:53:55 -07:00
0dff84b16a fix(remediation): gate sync/extract recs on real extraction lag, refresh at D7 recheck
Takeover of #2363. The sync.repo/extract.all recommendations gated on
health.stale_pages — a proxy (updated_at predates newest timeline entry)
that stopped meaning anything after migration v10 dropped the trigger
behind it. Gate them on the honest counter instead:
engine.countStalePagesForExtraction, the same staleness `gbrain extract
--stale` and doctor's links_extraction_lag use.

On top of the original PR, two repairs:

- runRemediation loads RecommendationContext once, but the D7 per-step
  recheck reused the frozen extractionLagPages — a completed sync/extract
  step could never clear the gate, so the pipeline re-fired every recheck
  until maxJobs. The recheck now refreshes the gate alongside getHealth
  via the shared countExtractionLag() helper (extracted into
  remediation/context.ts). Pinned by
  test/remediation-run-d7-refresh.serial.test.ts (serial: mock.module).
- autopilot builds its own RecommendationContext by hand; without wiring,
  it would silently never fire sync.repo/extract.all again. It now
  populates extractionLagPages from the same helper.

Co-authored-by: DarkNightForge <DarkNightForge@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 15:06:34 -07:00
21 changed files with 300 additions and 639 deletions
+4
View File
@@ -686,6 +686,7 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
try {
const { MinionQueue } = await import('../core/minions/queue.ts');
const { computeRecommendations, embeddingProviderConfigured, HOSTED_EMBED_KEY_CONFIG } = await import('../core/brain-score-recommendations.ts');
const { countExtractionLag } = await import('../core/remediation/context.ts');
const queue = new MinionQueue(engine);
const slotMs = Math.floor(Date.now() / (baseInterval * 1000)) * baseInterval * 1000;
const slot = new Date(slotMs).toISOString();
@@ -877,6 +878,9 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
return !!(process.env[envVar] || (cfgField ? embedKeyCfg[cfgField] : undefined));
}),
hasChatApiKey: !!(process.env.ANTHROPIC_API_KEY || await engine.getConfig('anthropic_api_key')),
// Real extraction-lag gate for sync.repo/extract.all — same counter
// loadRecommendationContext uses (replaces the health.stale_pages proxy).
extractionLagPages: await countExtractionLag(engine),
};
// v0.41.18.0 (A5 + A19 + A22, T15): consult onboard recommendations
// ALONGSIDE doctor's brain-score recommendations. Onboard's 4 new
-2
View File
@@ -2059,8 +2059,6 @@ export async function registerBuiltinHandlers(
sourceId,
windowSeconds,
brainDir: repoPath,
// #2750: worker cancel/timeout/lock-loss propagates into the drain.
abortSignal: job.signal,
});
} catch (e) {
if (e instanceof LockUnavailableError) {
+26 -8
View File
@@ -146,6 +146,16 @@ export interface RecommendationContext {
chatModel?: string;
/** Whether the chat provider has a usable API key. */
hasChatApiKey?: boolean;
/**
* Count of pages needing link/timeline extraction — the SAME staleness the
* `gbrain extract --stale` walk and doctor's `links_extraction_lag` check use
* (`engine.countStalePagesForExtraction`). Gates the sync→extract pipeline
* (sync.repo / extract.all). Replaces the old `health.stale_pages` gate, which
* counted "pages whose updated_at predates their newest timeline entry" — a
* proxy that broke when the updated_at-on-timeline-insert trigger was dropped
* (migration v10) and never reflected real extraction work.
*/
extractionLagPages?: number;
}
/** Triage result for one check. */
@@ -192,20 +202,28 @@ export function computeRecommendations(
const source = ctx.sourceId ?? 'default';
// ---------------------------------------------------------------------
// sync.repo — fires when sync hasn't run recently OR pages are stale
// sync.repo + extract.all — the materialization pipeline, gated on the REAL
// extraction lag (pages whose link/timeline edges are stale), NOT on the
// legacy `health.stale_pages` proxy. `extractionLagPages` comes from the same
// counter the `extract --stale` walk + doctor's `links_extraction_lag` use, so
// the recommendation can only fire when running extract will actually reduce
// it (and clear the rec). See RecommendationContext.extractionLagPages.
// sync.repo is the prerequisite: re-sync so pages are current before extract
// materializes their edges.
// ---------------------------------------------------------------------
if (ctx.repoPath && health.stale_pages > 0) {
const extractionLag = ctx.extractionLagPages ?? 0;
if (ctx.repoPath && extractionLag > 0) {
const params = { repoPath: ctx.repoPath, sourceId: ctx.sourceId, noEmbed: true };
out.push({
id: 'sync.repo',
job: 'sync',
params,
idempotency_key: idemKey(source, 'sync', params),
severity: health.stale_pages > 50 ? 'high' : 'medium',
est_seconds: Math.min(600, 30 + health.stale_pages * 0.5),
severity: extractionLag > 50 ? 'high' : 'medium',
est_seconds: Math.min(600, 30 + extractionLag * 0.5),
est_usd_cost: 0, // sync is fs+DB only
depends_on: [],
rationale: `${health.stale_pages} stale page${health.stale_pages === 1 ? '' : 's'} on disk`,
rationale: `Sync before extracting ${extractionLag} page${extractionLag === 1 ? '' : 's'} with stale link/timeline edges`,
status: 'remediable',
});
}
@@ -237,7 +255,7 @@ export function computeRecommendations(
est_seconds: Math.min(3600, 5 + health.missing_embeddings * 0.05),
est_usd_cost,
// sync should run first so embed sees fresh pages.
depends_on: ctx.repoPath && health.stale_pages > 0 ? ['sync.repo'] : [],
depends_on: ctx.repoPath && extractionLag > 0 ? ['sync.repo'] : [],
rationale: `${health.missing_embeddings} chunk${health.missing_embeddings === 1 ? '' : 's'} invisible to vector search`,
status: 'remediable',
});
@@ -267,7 +285,7 @@ export function computeRecommendations(
// Triggered when sync.repo fires (because sync was set to noEmbed:true,
// and noExtract:true after T5 lands → extract job is the materializer).
// ---------------------------------------------------------------------
if (ctx.repoPath && health.stale_pages > 0) {
if (ctx.repoPath && extractionLag > 0) {
const params = { mode: 'all', dir: ctx.repoPath };
out.push({
id: 'extract.all',
@@ -278,7 +296,7 @@ export function computeRecommendations(
est_seconds: Math.min(600, 30 + health.page_count * 0.01),
est_usd_cost: 0,
depends_on: ['sync.repo'],
rationale: 'Materialize link + timeline edges from fresh pages',
rationale: `Materialize link + timeline edges for ${extractionLag} page${extractionLag === 1 ? '' : 's'} with stale extraction`,
status: 'remediable',
});
}
+16 -77
View File
@@ -23,10 +23,6 @@
*/
import type { BrainEngine } from '../engine.ts';
import { anySignal } from '../abort-check.ts';
/** Fresh cleanup budget for the lock release after the window signal fires. */
const LOCK_RELEASE_GRACE_MS = 5_000;
export interface ExtractAtomsDrainDeps {
/**
@@ -34,13 +30,13 @@ export interface ExtractAtomsDrainDeps {
* via `withRefreshingLock`. MUST throw when the lock is held by another
* process (e.g. `LockUnavailableError`) — the drain lets that propagate so
* the caller can report `cycle_already_running` and exit, matching the
* routine cycle's skip contract. The signal bounds lock acquisition too.
* routine cycle's skip contract.
*/
withLock: <T>(work: () => Promise<T>, signal: AbortSignal) => Promise<T>;
/** Process one batch. The signal fires at the drain wallclock deadline. */
runBatch: (signal: AbortSignal) => Promise<{ extracted: number; skipped: number }>;
withLock: <T>(work: () => Promise<T>) => Promise<T>;
/** Process one bounded batch (rediscovers eligibility). Returns counts. */
runBatch: () => Promise<{ extracted: number; skipped: number }>;
/** Count remaining eligible-but-unextracted pages, or null on query error. */
countRemaining: (signal: AbortSignal) => Promise<number | null>;
countRemaining: () => Promise<number | null>;
/** Injectable clock. Production: Date.now. */
now: () => number;
/** Optional progress sink (one line per batch). */
@@ -52,8 +48,6 @@ export interface ExtractAtomsDrainOpts {
windowMs: number;
/** Hard cap on batches (belt-and-suspenders against a 0-progress loop). Default 1000. */
maxBatches?: number;
/** External caller cancellation (worker timeout / shutdown). */
abortSignal?: AbortSignal;
}
export interface ExtractAtomsDrainResult {
@@ -74,79 +68,35 @@ export async function runExtractAtomsDrain(
opts: ExtractAtomsDrainOpts,
): Promise<ExtractAtomsDrainResult> {
const maxBatches = opts.maxBatches ?? 1000;
const deadline = deps.now() + opts.windowMs;
// #2750: the window used to be checked only BETWEEN batches, so one slow
// batch (sequential LLM calls) or a hung lock/count/write overran it without
// bound (observed window=120s → 282.5s). A real-time deadline signal now
// cancels (Postgres) or abandons (PGLite, cooperative) whatever is in
// flight; the injected clock still drives loop-boundary checks so the pure
// loop stays unit-testable.
const signal = anySignal(
AbortSignal.timeout(Math.max(1, opts.windowMs)),
opts.abortSignal,
);
const result: ExtractAtomsDrainResult = await deps.withLock(async () => {
return deps.withLock(async () => {
const deadline = deps.now() + opts.windowMs;
let extracted = 0;
let skipped = 0;
let batches = 0;
let stopped: ExtractAtomsDrainResult['stopped'] = 'window';
while (deps.now() < deadline && !signal.aborted) {
while (deps.now() < deadline) {
if (batches >= maxBatches) { stopped = 'max_batches'; break; }
let before: number | null;
try {
before = await deps.countRemaining(signal);
} catch (err) {
if (signal.aborted) break;
throw err;
}
const before = await deps.countRemaining();
if (before === 0) { stopped = 'drained'; break; }
// The backlog count consumed the same wallclock budget — re-check so a
// slow count can't hand the batch a window that already expired.
if (deps.now() >= deadline || signal.aborted) break;
let r: { extracted: number; skipped: number };
try {
r = await deps.runBatch(signal);
} catch (err) {
if (signal.aborted) break;
throw err;
}
const r = await deps.runBatch();
extracted += r.extracted;
skipped += r.skipped;
batches++;
deps.onBatch?.({ batch: batches, extracted: r.extracted, remaining: before });
// A deadline abort inside the batch can surface as zero progress;
// window exhaustion wins over the generic no_progress label.
if (deps.now() >= deadline || signal.aborted) break;
// Stop if a batch made zero forward progress — extraction is failing or
// everything left is ineligible (e.g. all skipped). Prevents a hot loop
// that spends budget without draining.
if (r.extracted === 0 && r.skipped === 0) { stopped = 'no_progress'; break; }
}
// After the window elapsed, don't spend more unbounded time on a final
// count — report remaining as unknown instead of overrunning further.
const windowElapsed = signal.aborted || deps.now() >= deadline;
let remaining: number | null = null;
if (!windowElapsed) {
try {
remaining = await deps.countRemaining(signal);
} catch (err) {
if (!signal.aborted) throw err;
}
}
const remaining = await deps.countRemaining();
if (remaining === 0) stopped = 'drained';
return { phase: 'extract_atoms', status: 'ok', extracted, skipped, remaining, batches, stopped };
}, signal);
// Internal window expiry is a normal partial result. An EXTERNAL abort
// (worker cancel/timeout/shutdown) must reject so Minion records the abort.
if (opts.abortSignal?.aborted) throw opts.abortSignal.reason;
return result;
});
}
// ─── Shared wiring helper (v0.42.x #1685 DECISION 5A) ──────────────────────
@@ -184,8 +134,6 @@ export interface DrainForSourceOpts {
maxBatches?: number;
/** Optional per-batch progress sink (stderr line in dream; job progress in the handler). */
onBatch?: ExtractAtomsDrainDeps['onBatch'];
/** Worker cancellation / shutdown signal (Minion `job.signal`). */
abortSignal?: AbortSignal;
}
export async function runExtractAtomsDrainForSource(
@@ -201,17 +149,12 @@ export async function runExtractAtomsDrainForSource(
return runExtractAtomsDrain(
{
withLock: (work, signal) => withRefreshingLock(engine, lockId, work, {
ttlMinutes: 5,
signal,
releaseTimeoutMs: LOCK_RELEASE_GRACE_MS,
}),
runBatch: async (signal) => {
withLock: (work) => withRefreshingLock(engine, lockId, work, { ttlMinutes: 5 }),
runBatch: async () => {
const r = await runPhaseExtractAtoms(engine, {
sourceId: extractionSourceId,
dryRun: false,
brainDir: opts.brainDir,
abortSignal: signal,
});
const d = (r.details ?? {}) as Record<string, unknown>;
return {
@@ -219,14 +162,10 @@ export async function runExtractAtomsDrainForSource(
skipped: Number(d.duplicates_skipped ?? 0),
};
},
countRemaining: (signal) => countExtractAtomsBacklog(engine, extractionSourceId, signal),
countRemaining: () => countExtractAtomsBacklog(engine, extractionSourceId),
now: Date.now,
onBatch: opts.onBatch,
},
{
windowMs: opts.windowSeconds * 1000,
maxBatches: opts.maxBatches,
abortSignal: opts.abortSignal,
},
{ windowMs: opts.windowSeconds * 1000, maxBatches: opts.maxBatches },
);
}
+17 -68
View File
@@ -58,10 +58,6 @@ import { createHash } from 'crypto';
import { slugifySegment } from '../sync.ts';
const DEFAULT_BUDGET_USD = 0.3;
// #2750: fresh wallclock budget for the receipt/rollup bookkeeping writes when
// the caller's deadline already fired — committed atoms must not lose their
// cost/receipt trail, but the writes can't be unbounded either.
const BOOKKEEPING_GRACE_MS = 5_000;
// v0.42+ TODO: read atom_type enum from active pack manifest at runtime.
const ATOM_TYPES = [
@@ -159,13 +155,6 @@ export interface ExtractAtomsOpts {
* `heartbeat()` on the passed reporter.
*/
progress?: ProgressReporter;
/**
* #2750: caller deadline/cancellation. Forwarded to every gateway call and
* DB query/write so the drain window bounds real lifetime, plus a
* cooperative between-item check (the PGLite path, where query abort only
* abandons the waiter).
*/
abortSignal?: AbortSignal;
}
interface ExtractedAtom {
@@ -223,7 +212,6 @@ export async function discoverExtractablePages(
engine: BrainEngine,
sourceId: string,
affectedSlugs?: string[],
abortSignal?: AbortSignal,
): Promise<DiscoveredPage[]> {
const hasFilter = Array.isArray(affectedSlugs) && affectedSlugs.length > 0;
const sql = `
@@ -263,16 +251,13 @@ export async function discoverExtractablePages(
slug: string;
compiled_truth: string;
content_hash: string;
}>(sql, params, { signal: abortSignal });
}>(sql, params);
return rows.map((r) => ({
slug: r.slug,
content: r.compiled_truth,
contentHash: r.content_hash,
}));
} catch (err) {
// A deadline abort is not a fail-soft condition — propagate so the
// caller stops instead of proceeding with an empty page list.
if (abortSignal?.aborted) throw err;
const msg = err instanceof Error ? err.message : String(err);
console.error(`[extract_atoms] page-discovery query failed: ${msg}`);
return []; // fail-soft: transcript path still proceeds
@@ -297,7 +282,6 @@ export async function discoverExtractablePages(
export async function countExtractAtomsBacklog(
engine: BrainEngine,
sourceId?: string,
abortSignal?: AbortSignal,
): Promise<number | null> {
try {
// Two modes: scoped (the phase's per-source `remaining`) vs brain-wide
@@ -337,10 +321,9 @@ export async function countExtractAtomsBacklog(
const params = scoped
? [sourceId, extractableTypes, MIN_PAGE_CHARS_FOR_EXTRACTION]
: [extractableTypes, MIN_PAGE_CHARS_FOR_EXTRACTION];
const rows = await engine.executeRaw<{ cnt: string | number }>(sql, params, { signal: abortSignal });
const rows = await engine.executeRaw<{ cnt: string | number }>(sql, params);
return Number(rows[0]?.cnt ?? 0);
} catch (err) {
if (abortSignal?.aborted) throw err;
const msg = err instanceof Error ? err.message : String(err);
console.error(`[extract_atoms] backlog count failed: ${msg}`);
return null;
@@ -367,7 +350,6 @@ export async function atomsExistingForHashes(
engine: BrainEngine,
sourceId: string,
contentHash16s: string[],
abortSignal?: AbortSignal,
): Promise<Set<string>> {
if (contentHash16s.length === 0) return new Set();
try {
@@ -379,11 +361,9 @@ export async function atomsExistingForHashes(
AND deleted_at IS NULL
AND frontmatter->>'source_hash' = ANY($2::text[])`,
[sourceId, contentHash16s],
{ signal: abortSignal },
);
return new Set(rows.map(r => r.h));
} catch (err) {
if (abortSignal?.aborted) throw err;
const msg = err instanceof Error ? err.message : String(err);
console.error(`[extract_atoms] batch idempotency check failed (assuming none extracted): ${msg}`);
return new Set();
@@ -404,7 +384,6 @@ export async function runPhaseExtractAtoms(
): Promise<PhaseResult> {
const sourceId = opts.sourceId ?? 'default';
const chat = opts._chat ?? gatewayChat;
if (opts.abortSignal?.aborted) throw opts.abortSignal.reason;
// 1a. Get transcripts (test seam OR production discovery).
// v0.41.2.1: config loader switched to loadConfigWithEngine() so the
@@ -446,7 +425,7 @@ export async function runPhaseExtractAtoms(
if (opts._pages !== undefined) {
pages = opts._pages;
} else {
pages = await discoverExtractablePages(engine, sourceId, opts.affectedSlugs, opts.abortSignal);
pages = await discoverExtractablePages(engine, sourceId, opts.affectedSlugs);
}
// 2. Apply transcript-side source-hash idempotency in ONE batch query
@@ -458,7 +437,7 @@ export async function runPhaseExtractAtoms(
// Surface a heartbeat before the batch query so even an instant
// short-circuit shows a sign of life (closes Issue 2 silent-phase pain).
opts.progress?.heartbeat(`checking existing atoms for ${allHashes16.length} transcripts`);
const existingHashes = await atomsExistingForHashes(engine, sourceId, allHashes16, opts.abortSignal);
const existingHashes = await atomsExistingForHashes(engine, sourceId, allHashes16);
for (const t of transcripts) {
if (existingHashes.has(t.contentHash.slice(0, 16))) {
duplicatesSkipped++;
@@ -522,7 +501,6 @@ export async function runPhaseExtractAtoms(
const failures: Array<{ source: string; error: string }> = [];
let estimatedSpendUsd = 0;
const budgetCap = DEFAULT_BUDGET_USD;
let deadlineAborted = false;
// v0.41.19.0 (T3): throttled yield helper. Fires `opts.yieldDuringPhase`
// every 30s. Cycle.ts threads `buildYieldDuringPhase(lock, outer)` so
@@ -548,12 +526,6 @@ export async function runPhaseExtractAtoms(
}
for (const item of work) {
// #2750: cooperative between-item abort. Works on every engine — this is
// the primary bound on PGLite, where query abort only abandons the waiter.
if (opts.abortSignal?.aborted) {
deadlineAborted = true;
break;
}
await maybeYield();
if (estimatedSpendUsd >= budgetCap) {
if (item.kind === 'transcript') transcriptsSkipped++;
@@ -572,22 +544,16 @@ export async function runPhaseExtractAtoms(
},
],
maxTokens: 2000,
abortSignal: opts.abortSignal,
});
// Rough cost estimate — Haiku at ~$0.80/M input + $4/M output.
// A completed gateway call is billable even if the deadline fires
// immediately afterward, so record usage BEFORE the abort check.
estimatedSpendUsd +=
(result.usage.input_tokens * 0.8 + result.usage.output_tokens * 4.0) / 1_000_000;
if (opts.abortSignal?.aborted) {
deadlineAborted = true;
break;
}
// Post-await yield: closes the "long LLM call past TTL" hazard
// codex flagged. The 30s throttle inside maybeYield bounds the
// actual refresh rate so this is cheap when calls are fast.
await maybeYield();
// Rough cost estimate — Haiku at ~$0.80/M input + $4/M output
estimatedSpendUsd +=
(result.usage.input_tokens * 0.8 + result.usage.output_tokens * 4.0) / 1_000_000;
const atoms = parseAtomsResponse(result.text);
if (atoms.length === 0) {
if (item.kind === 'transcript') transcriptsProcessed++;
@@ -626,7 +592,7 @@ export async function runPhaseExtractAtoms(
},
timeline: '',
},
{ sourceId, signal: opts.abortSignal },
{ sourceId },
);
totalAtomsExtracted++;
}
@@ -639,11 +605,6 @@ export async function runPhaseExtractAtoms(
// Reporter rate-limits to ~1 line/sec; safe to tick every iter.
opts.progress?.tick(1, `${totalAtomsExtracted} atoms / ${duplicatesSkipped} skipped`);
} catch (err) {
// A deadline abort is a partial result, not a per-item failure.
if (opts.abortSignal?.aborted) {
deadlineAborted = true;
break;
}
failures.push({
source: originLabel,
error: err instanceof Error ? err.message : String(err),
@@ -654,12 +615,6 @@ export async function runPhaseExtractAtoms(
// v0.42 Wave B2: write extract receipt + rollup row when the phase
// actually extracted atoms. Both are best-effort per F-OUT-19 —
// audit-trail / search-visibility surfaces don't block the phase result.
//
// #2750: bookkeeping runs on a FRESH short grace signal, never the caller's
// work deadline — the deadline may have already fired (partial run) and
// committed atoms must not lose their receipt/cost trail; but the writes
// stay bounded so the overrun is capped at the grace window.
const bookkeepingSignal = opts.dryRun ? undefined : AbortSignal.timeout(BOOKKEEPING_GRACE_MS);
if (!opts.dryRun && totalAtomsExtracted > 0) {
const runId = `atoms-${Date.now().toString(36)}-${sourceId.slice(0, 4)}`;
try {
@@ -674,24 +629,19 @@ export async function runPhaseExtractAtoms(
summary:
`Extracted ${totalAtomsExtracted} atoms from ` +
`${transcriptsProcessed} transcripts + ${pagesProcessed} pages.`,
}, { signal: bookkeepingSignal });
});
} catch (err) {
console.error(`[extract_atoms] receipt write failed: ${(err as Error).message}`);
}
}
if (!opts.dryRun) {
try {
await upsertExtractRollup(engine, {
kind: 'atoms',
source_id: sourceId,
cost_delta: estimatedSpendUsd,
// A deadline-truncated run is not a completed round.
round_completed_delta: failures.length === 0 && !deadlineAborted ? 1 : 0,
halt_delta: failures.length > 0 ? 1 : 0,
}, { signal: bookkeepingSignal });
} catch (err) {
console.error(`[extract_atoms] rollup write failed: ${(err as Error).message}`);
}
await upsertExtractRollup(engine, {
kind: 'atoms',
source_id: sourceId,
cost_delta: estimatedSpendUsd,
round_completed_delta: failures.length === 0 ? 1 : 0,
halt_delta: failures.length > 0 ? 1 : 0,
});
}
return {
@@ -720,7 +670,6 @@ export async function runPhaseExtractAtoms(
budget_usd: budgetCap,
source_id: sourceId,
dry_run: opts.dryRun ?? false,
deadline_aborted: deadlineAborted,
},
};
}
+22 -50
View File
@@ -26,8 +26,7 @@ import type { BrainEngine } from './engine.ts';
export interface DbLockHandle {
id: string;
/** Optional signal bounds the release DELETE (deadline-bound callers). */
release: (signal?: AbortSignal) => Promise<void>;
release: () => Promise<void>;
refresh: () => Promise<void>;
}
@@ -174,7 +173,6 @@ export async function tryAcquireDbLock(
engine: BrainEngine,
lockId: string,
ttlMinutes: number = DEFAULT_TTL_MINUTES,
opts: { signal?: AbortSignal } = {},
): Promise<DbLockHandle | null> {
const pid = process.pid;
const host = hostname();
@@ -207,26 +205,20 @@ export async function tryAcquireDbLock(
// `gbrain sync --break-lock --max-age <s>` uses last_refreshed_at (not
// acquired_at) to identify wedged-but-alive holders without stealing
// healthy long-running holders that are actively refreshing.
// #2750: routed through executeRaw so a deadline-bound caller's signal
// can cancel a hung acquire (pool exhaustion). Cancellation is
// transactional; in the rare ambiguous-commit case the row's TTL is the
// backstop (drain locks use a short 5-minute TTL).
const rows = await engine.executeRaw<{ id: string }>(
`INSERT INTO gbrain_cycle_locks (id, holder_pid, holder_host, acquired_at, ttl_expires_at, last_refreshed_at)
VALUES ($1, $2, $3, NOW(), NOW() + $4::interval, NOW())
ON CONFLICT (id) DO UPDATE
SET holder_pid = $2,
holder_host = $3,
acquired_at = NOW(),
ttl_expires_at = NOW() + $4::interval,
last_refreshed_at = NOW()
WHERE gbrain_cycle_locks.ttl_expires_at < NOW()
AND (gbrain_cycle_locks.last_refreshed_at IS NULL
OR gbrain_cycle_locks.last_refreshed_at < NOW() - $5 * INTERVAL '1 second')
RETURNING id`,
[lockId, pid, host, ttl, stealGraceSeconds],
{ signal: opts.signal },
);
const rows: Array<{ id: string }> = await sql`
INSERT INTO gbrain_cycle_locks (id, holder_pid, holder_host, acquired_at, ttl_expires_at, last_refreshed_at)
VALUES (${lockId}, ${pid}, ${host}, NOW(), NOW() + ${ttl}::interval, NOW())
ON CONFLICT (id) DO UPDATE
SET holder_pid = ${pid},
holder_host = ${host},
acquired_at = NOW(),
ttl_expires_at = NOW() + ${ttl}::interval,
last_refreshed_at = NOW()
WHERE gbrain_cycle_locks.ttl_expires_at < NOW()
AND (gbrain_cycle_locks.last_refreshed_at IS NULL
OR gbrain_cycle_locks.last_refreshed_at < NOW() - ${stealGraceSeconds} * INTERVAL '1 second')
RETURNING id
`;
if (rows.length === 0) return null;
const deregister = registerCleanup(`db-lock:${lockId}`, async () => {
await sql`
@@ -249,17 +241,12 @@ export async function tryAcquireDbLock(
[ttl, lockId, pid],
);
},
release: async (signal?: AbortSignal) => {
release: async () => {
deregister();
// Direct session pool (same rationale as refresh, #1794) + optional
// signal so a deadline-bound caller's release can't hang forever on
// an exhausted pooler. TTL is the backstop if the DELETE is cancelled.
await engine.executeRawDirect(
`DELETE FROM gbrain_cycle_locks
WHERE id = $1 AND holder_pid = $2`,
[lockId, pid],
{ signal },
);
await sql`
DELETE FROM gbrain_cycle_locks
WHERE id = ${lockId} AND holder_pid = ${pid}
`;
},
};
}
@@ -316,11 +303,6 @@ export async function tryAcquireDbLock(
const first = await acquireOnce();
if (first) return first;
// #2750: deadline-bound callers prefer an honest busy result over the
// best-effort same-host takeover below, whose inspect/delete/retry calls
// are not signal-bounded. The initial upsert already reclaims expired locks.
if (opts.signal) return null;
// v0.42 (#1780 Gap 3): the lock is held and its TTL hasn't expired (the
// upsert's ON CONFLICT ... WHERE ttl_expires_at < NOW() returned no row).
// If the holder is on THIS host, provably dead, and past the grace window,
@@ -814,10 +796,6 @@ export interface WithRefreshingLockOpts {
ttlMinutes?: number;
/** Heartbeat-fail threshold in ms — abort if SELECT 1 takes longer. Default 30000. */
heartbeatTimeoutMs?: number;
/** #2750: bound lock acquisition with the caller's deadline signal. */
signal?: AbortSignal;
/** Fresh cleanup budget for the release DELETE when `signal` is set. Default 5000. */
releaseTimeoutMs?: number;
}
/**
@@ -837,7 +815,7 @@ export async function withRefreshingLock<T>(
// Refresh 6x per TTL window so a missed tick doesn't expire the lock.
const refreshIntervalMs = Math.max(15000, (ttlMinutes * 60 * 1000) / 6);
const handle = await tryAcquireDbLock(engine, lockId, ttlMinutes, { signal: opts.signal });
const handle = await tryAcquireDbLock(engine, lockId, ttlMinutes);
if (!handle) throw new LockUnavailableError(lockId);
let healthOk = true;
@@ -876,13 +854,7 @@ export async function withRefreshingLock<T>(
return await work();
} finally {
clearInterval(interval);
// #2750: when the caller is deadline-bound, its work signal may already
// have fired — release on a FRESH short grace signal so cleanup neither
// inherits the spent deadline nor hangs unbounded. TTL is the backstop.
const releaseSignal = opts.signal
? AbortSignal.timeout(opts.releaseTimeoutMs ?? 5_000)
: undefined;
try { await handle.release(releaseSignal); } catch { /* idempotent; TTL backstop */ }
try { await handle.release(); } catch { /* idempotent */ }
if (!healthOk) {
// Surface that the heartbeat detected backend trouble — caller can
// log to the connection-events audit if desired.
+1 -9
View File
@@ -696,16 +696,8 @@ export interface BrainEngine {
* is included in the INSERT column list so ON CONFLICT (source_id, slug)
* DO UPDATE actually targets the intended row instead of fabricating a
* duplicate at (default, slug). Multi-source brains MUST pass sourceId.
*
* `opts.signal` (#2750): optional cancellation for deadline-bound writers.
* Postgres cancels the in-flight statement; PGLite pre-checks only (query
* cancellation is not possible in-process — cooperative abort between calls).
*/
putPage(
slug: string,
page: PageInput,
opts?: { sourceId?: string; signal?: AbortSignal },
): Promise<Page>;
putPage(slug: string, page: PageInput, opts?: { sourceId?: string }): Promise<Page>;
/**
* v0.41.13 (#1309) — identity-based dedup pre-check for the import pipeline.
*
+1 -2
View File
@@ -187,7 +187,6 @@ function buildReceiptFrontmatter(input: ExtractReceiptInput): Record<string, unk
export async function writeReceipt(
engine: BrainEngine,
input: ExtractReceiptInput,
opts?: { signal?: AbortSignal },
): Promise<{ slug: string; page: Page }> {
const slug = receiptSlug(input);
const title = `${input.kind}${input.round}${input.source_id}`;
@@ -202,7 +201,7 @@ export async function writeReceipt(
compiled_truth,
frontmatter,
},
{ sourceId: input.source_id, signal: opts?.signal },
{ sourceId: input.source_id },
);
return { slug, page };
-4
View File
@@ -70,7 +70,6 @@ function today(): string {
export async function upsertExtractRollup(
engine: BrainEngine,
input: RollupUpsertInput,
opts?: { signal?: AbortSignal },
): Promise<{ ok: boolean; error?: string }> {
const day = input.day ?? today();
const cost = input.cost_delta ?? 0;
@@ -97,12 +96,9 @@ export async function upsertExtractRollup(
rollup_write_failures = extract_rollup_7d.rollup_write_failures + EXCLUDED.rollup_write_failures,
updated_at = now()`,
[input.kind, input.source_id, day, cost, halts, evalFails, evalPasses, completed, failures],
{ signal: opts?.signal },
);
return { ok: true };
} catch (err) {
// Signal-bounded callers get the abort surfaced, not a swallowed `ok:false`.
if (opts?.signal?.aborted) throw err;
const msg = (err as Error).message || String(err);
// Don't spam: log once per process per (kind, day) error class.
rollupErrorLogOnce(input.kind, day, msg);
+1 -9
View File
@@ -1003,15 +1003,7 @@ export class PGLiteEngine implements BrainEngine {
return { slug: r.slug, id: Number(r.id) };
}
async putPage(
slug: string,
page: PageInput,
opts?: { sourceId?: string; signal?: AbortSignal },
): Promise<Page> {
// #2750: PGLite is in-process WASM — no query cancellation. Pre-check so
// an already-fired deadline skips the write; abort is cooperative
// between calls (same posture as executeRaw's documented gap).
if (opts?.signal?.aborted) throw new DOMException('aborted', 'AbortError');
async putPage(slug: string, page: PageInput, opts?: { sourceId?: string }): Promise<Page> {
slug = validateSlug(slug);
const hash = page.content_hash || contentHash(page);
const frontmatter = page.frontmatter || {};
+3 -55
View File
@@ -72,32 +72,6 @@ function escapeSqlStringLiteral(value: string): string {
return value.replace(/'/g, "''");
}
/**
* #2750: race a promise against an AbortSignal, detaching the listener once
* settled (long-lived drain signals are reused across many calls, so a bare
* Promise.race would leak one listener per call). The abandoned promise keeps
* running; used only for pool-acquisition waits where that is harmless.
*/
function waitForSignal<T>(work: Promise<T>, signal?: AbortSignal): Promise<T> {
if (!signal) return work;
if (signal.aborted) return Promise.reject(new DOMException('aborted', 'AbortError'));
return new Promise<T>((resolve, reject) => {
let settled = false;
const finish = (fn: () => void) => {
if (settled) return;
settled = true;
signal.removeEventListener('abort', onAbort);
fn();
};
const onAbort = () => finish(() => reject(new DOMException('aborted', 'AbortError')));
signal.addEventListener('abort', onAbort, { once: true });
work.then(
(value) => finish(() => resolve(value)),
(err) => finish(() => reject(err)),
);
});
}
export function getPostgresSchema(
dims: number = DEFAULT_EMBEDDING_DIMENSIONS,
model: string = DEFAULT_EMBEDDING_MODEL,
@@ -1087,12 +1061,7 @@ export class PostgresEngine implements BrainEngine {
});
}
async putPage(
slug: string,
page: PageInput,
opts?: { sourceId?: string; signal?: AbortSignal },
): Promise<Page> {
if (opts?.signal?.aborted) throw new DOMException('aborted', 'AbortError');
async putPage(slug: string, page: PageInput, opts?: { sourceId?: string }): Promise<Page> {
slug = validateSlug(slug);
const sql = this.sql;
const hash = page.content_hash || contentHash(page);
@@ -1127,7 +1096,7 @@ export class PostgresEngine implements BrainEngine {
const sourceUri = page.source_uri ?? null;
const ingestedVia = page.ingested_via ?? null;
const ingestedAt = (sourceKind || sourceUri || ingestedVia) ? new Date() : null;
const pending = sql`
const rows = await sql`
INSERT INTO pages (source_id, slug, type, page_kind, title, compiled_truth, timeline, frontmatter, content_hash, updated_at, effective_date, effective_date_source, import_filename, chunker_version, source_path, source_kind, source_uri, ingested_via, ingested_at)
VALUES (${sourceId}, ${slug}, ${page.type}, ${pageKind}, ${page.title}, ${page.compiled_truth}, ${page.timeline || ''}, ${sql.json(frontmatter as Parameters<typeof sql.json>[0])}, ${hash}, now(), ${effectiveDate}, ${effectiveDateSource}, ${importFilename}, COALESCE(${chunkerVersion}::smallint, ${MARKDOWN_CHUNKER_VERSION}), ${sourcePath}, ${sourceKind}, ${sourceUri}, ${ingestedVia}, ${ingestedAt})
ON CONFLICT (source_id, slug) DO UPDATE SET
@@ -1150,22 +1119,6 @@ export class PostgresEngine implements BrainEngine {
ingested_at = COALESCE(EXCLUDED.ingested_at, pages.ingested_at)
RETURNING id, source_id, slug, type, title, compiled_truth, timeline, frontmatter, content_hash, created_at, updated_at, effective_date, effective_date_source, import_filename, source_kind, source_uri, ingested_via, ingested_at
`;
// #2750: cancel the in-flight statement when the caller's deadline fires,
// same .cancel() wiring as runUnsafe (postgres.js pending queries).
if (opts?.signal) {
const signal = opts.signal;
const onAbort = () => {
try { (pending as unknown as { cancel?: () => void }).cancel?.(); } catch { /* best-effort */ }
};
signal.addEventListener('abort', onAbort, { once: true });
try {
const rows = await pending;
return rowToPage(rows[0]);
} finally {
signal.removeEventListener('abort', onAbort);
}
}
const rows = await pending;
return rowToPage(rows[0]);
}
@@ -5854,16 +5807,11 @@ export class PostgresEngine implements BrainEngine {
params?: unknown[],
opts?: { signal?: AbortSignal },
): Promise<T[]> {
// #2750: an already-fired signal short-circuits BEFORE any pool routing,
// and the direct-pool acquisition itself is signal-bounded — under pooler
// exhaustion `ddl()` can stall indefinitely, which used to make even a
// "bounded" lock release hang past its caller's deadline.
if (opts?.signal?.aborted) throw new DOMException('aborted', 'AbortError');
// Inside an open transaction, _sql is the reserved tx connection (set via
// defineProperty in transaction()); never reroute off it.
const inTransaction = this._sql !== null && this.connectionManager?.peekReadPool() !== this._sql;
const conn = (!inTransaction && this.connectionManager?.isDualPoolActive())
? await waitForSignal(this.connectionManager.ddl(), opts?.signal)
? await this.connectionManager.ddl()
: this.sql;
return this.runUnsafe<T>(conn, sql, params, opts);
}
+25
View File
@@ -8,6 +8,7 @@
import type { BrainEngine } from '../engine.ts';
import type { RecommendationContext } from '../brain-score-recommendations.ts';
import { LINK_EXTRACTOR_VERSION_TS } from '../link-extraction.ts';
// Re-export so consumers can `import { RecommendationContext } from '../remediation'`
// — the canonical RecommendationContext type still lives in
@@ -68,5 +69,29 @@ export async function loadRecommendationContext(
embeddingDimensions,
embeddingProviderConfigured: embeddingConfigured,
hasChatApiKey: !!(process.env.ANTHROPIC_API_KEY || fileCfg?.anthropic_api_key),
extractionLagPages: await countExtractionLag(engine),
};
}
/**
* Real extraction-lag count — the SAME staleness `gbrain extract --stale`
* processes (engine.countStalePagesForExtraction with
* versionTs=LINK_EXTRACTOR_VERSION_TS, matching doctor's links_extraction_lag
* check — without versionTs, pages stamped before an extractor version bump
* would lag for doctor/extract but never trip this gate). Drives the
* sync→extract recommendation pipeline; replaces the legacy
* `health.stale_pages` proxy that no longer reflected real extraction work
* after the v10 trigger drop.
*
* Shared by loadRecommendationContext AND the D7 per-step recheck in
* runRemediation — the recheck MUST refresh this gate alongside getHealth,
* or a completed extract step keeps re-firing off the frozen initial count.
*/
export async function countExtractionLag(engine: BrainEngine): Promise<number> {
try {
return await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS });
} catch {
/* counter unavailable (very old brain / mid-migration) — treat as 0 */
return 0;
}
}
+7 -2
View File
@@ -16,7 +16,7 @@ import {
computeRecommendations,
} from '../brain-score-recommendations.ts';
import type { RemediationStep } from '../remediation-step.ts';
import { loadRecommendationContext } from './context.ts';
import { countExtractionLag, loadRecommendationContext } from './context.ts';
import { computeRemediationPlan } from './plan.ts';
import type {
RemediationHooks,
@@ -65,7 +65,7 @@ export async function runRemediation(
clearRemediationCheckpoint,
} = await import('../remediation-checkpoint.ts');
const ctx = await loadRecommendationContext(engine);
let ctx = await loadRecommendationContext(engine);
// Pre-flight ceiling check via the shared plan computation.
const initialPlan = await computeRemediationPlan(engine, { targetScore });
@@ -305,6 +305,11 @@ export async function runRemediation(
// steps with bumped retry suffix (D1).
if (recs.length === 0 || stepCount >= maxJobs) break;
const freshHealth = await engine.getHealth();
// Refresh the extraction-lag gate alongside health: ctx was loaded once
// before the loop, and a completed sync/extract step is exactly what
// drives the count down. Reusing the frozen initial count would re-fire
// sync.repo/extract.all every recheck until maxJobs.
ctx = { ...ctx, extractionLagPages: await countExtractionLag(engine) };
recs = computeRecommendations(freshHealth, ctx).filter((r) => r.status === 'remediable');
}
};
+9
View File
@@ -1423,6 +1423,15 @@ export interface BrainStats {
export interface BrainHealth {
page_count: number;
embed_coverage: number;
/**
* LEGACY proxy: count of pages whose `updated_at` predates their newest
* timeline entry. This bumped meaningfully only while a trigger updated
* `pages.updated_at` on timeline insert; that trigger was dropped in
* migration v10, so the metric no longer reflects real "needs work" state.
* NO LONGER gates remediations — the sync→extract pipeline now gates on
* `RecommendationContext.extractionLagPages` (the real extraction-lag from
* `countStalePagesForExtraction`). Retained for the CLI health line + back-compat.
*/
stale_pages: number;
/**
* Islanded pages — zero inbound AND zero outbound links. A hub page
+9 -12
View File
@@ -119,13 +119,12 @@ describe('computeRecommendations', () => {
expect(recs.find((r) => r.id === 'embed.stale')).toBeUndefined();
});
test('stale pages + dead links produce sync + backlinks + extract', () => {
test('extraction lag + dead links produce sync + backlinks + extract', () => {
const health = makeHealth({
stale_pages: 25,
dead_links: 8,
brain_score: 70,
});
const recs = computeRecommendations(health, { repoPath: '/brain', embeddingProviderConfigured: true });
const recs = computeRecommendations(health, { repoPath: '/brain', embeddingProviderConfigured: true, extractionLagPages: 25 });
const ids = recs.map((r) => r.id);
expect(ids).toContain('sync.repo');
expect(ids).toContain('backlinks.fix');
@@ -133,18 +132,17 @@ describe('computeRecommendations', () => {
});
test('extract.all depends on sync.repo (D14: stable ids)', () => {
const health = makeHealth({ stale_pages: 10 });
const recs = computeRecommendations(health, { repoPath: '/brain', embeddingProviderConfigured: true });
const health = makeHealth();
const recs = computeRecommendations(health, { repoPath: '/brain', embeddingProviderConfigured: true, extractionLagPages: 10 });
const extract = recs.find((r) => r.id === 'extract.all');
expect(extract?.depends_on).toContain('sync.repo');
});
test('embed.stale depends on sync.repo when sync also needed', () => {
test('embed.stale depends on sync.repo when extraction also needed', () => {
const health = makeHealth({
stale_pages: 10,
missing_embeddings: 100,
});
const recs = computeRecommendations(health, { repoPath: '/brain', embeddingProviderConfigured: true });
const recs = computeRecommendations(health, { repoPath: '/brain', embeddingProviderConfigured: true, extractionLagPages: 10 });
const embed = recs.find((r) => r.id === 'embed.stale');
expect(embed?.depends_on).toContain('sync.repo');
});
@@ -159,9 +157,9 @@ describe('computeRecommendations', () => {
test('severity ordering: critical before high before medium', () => {
const health = makeHealth({
missing_embeddings: 100, // critical
stale_pages: 80, // high
});
const recs = computeRecommendations(health, { repoPath: '/brain', embeddingProviderConfigured: true });
// extractionLagPages > 50 → sync.repo fires at 'high' severity.
const recs = computeRecommendations(health, { repoPath: '/brain', embeddingProviderConfigured: true, extractionLagPages: 80 });
const critIdx = recs.findIndex((r) => r.severity === 'critical');
const highIdx = recs.findIndex((r) => r.severity === 'high');
expect(critIdx).toBeLessThan(highIdx);
@@ -170,11 +168,10 @@ describe('computeRecommendations', () => {
// D6 #5 — THE critical regression test for the agent contract.
test('D6 #5: determinism — same input twice produces identical output', () => {
const health = makeHealth({
stale_pages: 10,
missing_embeddings: 50,
dead_links: 3,
});
const ctx = { repoPath: '/brain', embeddingProviderConfigured: true, sourceId: 'default' };
const ctx = { repoPath: '/brain', embeddingProviderConfigured: true, sourceId: 'default', extractionLagPages: 10 };
const run1 = computeRecommendations(health, ctx);
const run2 = computeRecommendations(health, ctx);
expect(JSON.stringify(run1)).toBe(JSON.stringify(run2));
@@ -17,7 +17,6 @@ import { runPhaseExtractAtoms, parseAtomsResponse } from '../../src/core/cycle/e
import { runPhaseSynthesizeConcepts } from '../../src/core/cycle/synthesize-concepts.ts';
import { resetPgliteState } from '../helpers/reset-pglite.ts';
import type { ChatResult, ChatOpts } from '../../src/core/ai/gateway.ts';
import type { BrainEngine } from '../../src/core/engine.ts';
let engine: PGLiteEngine;
@@ -178,180 +177,6 @@ describe('v0.41 T5: runPhaseExtractAtoms via stubbed chat', () => {
expect((result.details?.failures as unknown[]).length).toBe(1);
});
// ── #2750: caller deadline bounds the phase ────────────────────────────
test('caller deadline aborts a hung chat before processing the next item', async () => {
let calls = 0;
const chat = async (opts: ChatOpts) => {
calls++;
return await new Promise<never>((_resolve, reject) => {
const signal = opts.abortSignal;
if (!signal) return reject(new Error('missing abort signal'));
if (signal.aborted) return reject(signal.reason);
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
});
};
const started = Date.now();
const result = await runPhaseExtractAtoms(engine, {
_transcripts: [
{ filePath: '/hung.txt', content: 'a', contentHash: 'hung-a' },
{ filePath: '/never.txt', content: 'b', contentHash: 'hung-b' },
],
_pages: [],
_chat: chat as typeof import('../../src/core/ai/gateway.ts').chat,
abortSignal: AbortSignal.timeout(25),
});
expect(Date.now() - started).toBeLessThan(2_000);
expect(calls).toBe(1);
expect(result.status).toBe('ok');
expect(result.details?.deadline_aborted).toBe(true);
expect(result.details?.atoms_extracted).toBe(0);
expect(result.details?.failures).toEqual([]);
});
test('billable chat usage is counted when the deadline fires as the response resolves', async () => {
const controller = new AbortController();
const chat = async (opts: ChatOpts): Promise<ChatResult> => {
controller.abort(new DOMException('deadline', 'TimeoutError'));
return stubChat(`[{"title":"late","atom_type":"insight","body":"b"}]`, {
input_tokens: 1_000,
output_tokens: 500,
})(opts);
};
const result = await runPhaseExtractAtoms(engine, {
_transcripts: [{ filePath: '/late.txt', content: 'a', contentHash: 'late' }],
_pages: [],
_chat: chat,
abortSignal: controller.signal,
});
expect(result.details?.deadline_aborted).toBe(true);
expect(Number(result.details?.estimated_spend_usd)).toBeGreaterThan(0);
expect(result.details?.atoms_extracted).toBe(0);
});
test('deadline after partial progress still writes receipt and incomplete rollup', async () => {
const controller = new AbortController();
let calls = 0;
let notifySecondChat!: () => void;
const secondChatStarted = new Promise<void>((resolve) => { notifySecondChat = resolve; });
const chat = async (opts: ChatOpts): Promise<ChatResult> => {
calls++;
if (calls === 1) {
return stubChat(`[{"title":"committed","atom_type":"insight","body":"b"}]`)(opts);
}
notifySecondChat();
return await new Promise<never>((_resolve, reject) => {
const signal = opts.abortSignal;
if (!signal) return reject(new Error('missing abort signal'));
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
});
};
const pending = runPhaseExtractAtoms(engine, {
_transcripts: [
{ filePath: '/committed.txt', content: 'a', contentHash: 'committed-a' },
{ filePath: '/hung.txt', content: 'b', contentHash: 'hung-b' },
],
_pages: [],
_chat: chat,
abortSignal: controller.signal,
});
await secondChatStarted;
controller.abort(new DOMException('deadline', 'TimeoutError'));
const result = await pending;
const atoms = await engine.executeRaw<{ n: number }>(
`SELECT COUNT(*)::int AS n FROM pages WHERE type = 'atom'`,
);
const receipts = await engine.executeRaw<{ n: number }>(
`SELECT COUNT(*)::int AS n FROM pages WHERE type = 'extract_receipt'`,
);
const rollups = await engine.executeRaw<{
cost_usd: string | number;
round_completed_count: string | number;
}>(
`SELECT cost_usd, round_completed_count
FROM extract_rollup_7d
WHERE kind = 'atoms' AND source_id = 'default'`,
);
expect(result.details?.deadline_aborted).toBe(true);
expect(atoms[0].n).toBe(1);
expect(receipts[0].n).toBe(1);
expect(Number(rollups[0].cost_usd)).toBeGreaterThan(0);
expect(Number(rollups[0].round_completed_count)).toBe(0);
});
test('bookkeeping runs on a fresh grace signal, not the fired work deadline', async () => {
const controller = new AbortController();
let putCalls = 0;
let receiptSignal: AbortSignal | undefined;
let rollupSignal: AbortSignal | undefined;
const signalAwareEngine = {
executeRaw: async (sql: string, _params?: unknown[], opts?: { signal?: AbortSignal }) => {
if (sql.includes('INSERT INTO extract_rollup_7d')) rollupSignal = opts?.signal;
return [];
},
putPage: async (_slug: string, _page: unknown, opts?: { signal?: AbortSignal }) => {
putCalls++;
if (putCalls === 1) {
// Atom write in flight; the work deadline fires before bookkeeping.
controller.abort(new DOMException('work deadline', 'TimeoutError'));
} else {
receiptSignal = opts?.signal;
}
return {};
},
} as unknown as BrainEngine;
const result = await runPhaseExtractAtoms(signalAwareEngine, {
_transcripts: [{ filePath: '/one.txt', content: 'a', contentHash: 'one' }],
_pages: [],
_chat: stubChat(`[{"title":"one","atom_type":"insight","body":"b"}]`),
abortSignal: controller.signal,
});
expect(result.details?.atoms_extracted).toBe(1);
expect(putCalls).toBe(2); // atom write + receipt write
expect(receiptSignal).toBeDefined();
expect(receiptSignal).not.toBe(controller.signal);
expect(receiptSignal?.aborted).toBe(false);
expect(rollupSignal).toBe(receiptSignal);
});
test('caller deadline cancels a hung atom write and stops the phase', async () => {
const controller = new AbortController();
let notifyWriteStarted!: () => void;
const writeStarted = new Promise<void>((resolve) => { notifyWriteStarted = resolve; });
let writeCalls = 0;
const signalAwareEngine = {
executeRaw: async () => [],
putPage: async (_slug: string, _page: unknown, opts?: { signal?: AbortSignal }) => {
writeCalls++;
notifyWriteStarted();
return await new Promise<never>((_resolve, reject) => {
const signal = opts?.signal;
if (!signal) return reject(new Error('missing abort signal'));
if (signal.aborted) return reject(signal.reason);
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
});
},
} as unknown as BrainEngine;
const pending = runPhaseExtractAtoms(signalAwareEngine, {
_transcripts: [{ filePath: '/hung-write.txt', content: 'a', contentHash: 'hung-write' }],
_pages: [],
_chat: stubChat(`[{"title":"hung write","atom_type":"insight","body":"b"}]`),
abortSignal: controller.signal,
});
await writeStarted;
controller.abort(new DOMException('deadline', 'TimeoutError'));
const result = await pending;
expect(writeCalls).toBe(1);
expect(result.details?.deadline_aborted).toBe(true);
expect(result.details?.atoms_extracted).toBe(0);
});
// v0.41.2.1 regression case (D9 #14 wording): with _pages:[] and same
// _transcripts, all PRE-EXISTING PhaseResult.details fields match
// pre-fix values byte-for-byte. The new fields (pages_processed,
+9 -131
View File
@@ -43,135 +43,26 @@ describe('runExtractAtomsDrain (issue #1678)', () => {
expect(batches).toBe(3);
});
it('stops at the wallclock window; remaining is unknown (no post-window count)', async () => {
// Each batch consumes 60ms of the 100ms window: two batches fit, the
// third boundary check sees 120 ≥ 100 and stops. #2750: after the window
// elapses the final countRemaining is SKIPPED (it would overrun the
// window), so remaining reports null.
let now = 0;
it('stops at the wallclock window with remaining > 0', async () => {
// SYNC stepping clock: now() #1 sets deadline (0+100=100); the while-check
// then sees 50, 50 (two batches), then 999999 → past deadline → stop.
const times = [0, 50, 50, 999_999];
let ti = 0;
const now = () => times[Math.min(ti++, times.length - 1)];
const result = await runExtractAtomsDrain(
{
withLock: passThroughLock,
countRemaining: async () => 5, // never drains
runBatch: async () => {
now += 60;
return { extracted: 1, skipped: 0 };
},
now: () => now,
runBatch: async () => ({ extracted: 1, skipped: 0 }),
now,
},
{ windowMs: 100 },
);
expect(result.stopped).toBe('window');
expect(result.remaining).toBeNull();
expect(result.remaining).toBe(5);
expect(result.batches).toBe(2);
});
it('passes one drain-level deadline signal into count and batch', async () => {
const seen: AbortSignal[] = [];
const controller = new AbortController();
let now = 0;
const result = await runExtractAtomsDrain(
{
withLock: passThroughLock,
countRemaining: async (signal) => {
seen.push(signal);
return 5;
},
runBatch: async (signal) => {
seen.push(signal);
now = 100;
return { extracted: 1, skipped: 0 };
},
now: () => now,
},
{ windowMs: 100, abortSignal: controller.signal },
);
expect(result.stopped).toBe('window');
expect(result.batches).toBe(1);
expect(seen.length).toBe(2);
expect(seen[0]).toBe(seen[1]);
// Combined (timeout + external) signal, not the raw external one.
expect(seen[0]).not.toBe(controller.signal);
});
it('aborts a hung backlog count at the window deadline and releases the lock', async () => {
let released = false;
const result = await runExtractAtomsDrain(
{
withLock: async (work) => {
try { return await work(); }
finally { released = true; }
},
// Hangs until the drain's real-time deadline signal fires (10ms).
countRemaining: (signal) => new Promise((_resolve, reject) => {
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
}),
runBatch: async () => ({ extracted: 0, skipped: 0 }),
now: () => 0, // injected clock never advances — the SIGNAL must save us
},
{ windowMs: 10 },
);
expect(result.stopped).toBe('window');
expect(result.remaining).toBeNull();
expect(released).toBe(true);
});
it('rethrows external cancellation after releasing the lock', async () => {
const controller = new AbortController();
let released = false;
const pending = runExtractAtomsDrain(
{
withLock: async (work) => {
try { return await work(); }
finally { released = true; }
},
countRemaining: (signal) => new Promise((_resolve, reject) => {
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
}),
runBatch: async () => ({ extracted: 0, skipped: 0 }),
now: () => 0,
},
{ windowMs: 1_000_000, abortSignal: controller.signal },
);
controller.abort(new DOMException('worker timeout', 'AbortError'));
await expect(pending).rejects.toThrow('worker timeout');
expect(released).toBe(true);
});
it('classifies a deadline-exhausted zero-progress batch as window, not no_progress', async () => {
let now = 0;
const result = await runExtractAtomsDrain(
{
withLock: passThroughLock,
countRemaining: async () => 5,
runBatch: async () => {
now = 100; // batch consumed the whole window and returned nothing
return { extracted: 0, skipped: 0 };
},
now: () => now,
},
{ windowMs: 100 },
);
expect(result.stopped).toBe('window');
expect(result.batches).toBe(1);
});
it('bounds a hung lock acquisition with the drain deadline signal', async () => {
const started = Date.now();
await expect(runExtractAtomsDrain(
{
withLock: (_work, signal) => new Promise((_resolve, reject) => {
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
}),
countRemaining: async () => 1,
runBatch: async () => ({ extracted: 0, skipped: 0 }),
now: Date.now,
},
{ windowMs: 10 },
)).rejects.toThrow();
expect(Date.now() - started).toBeLessThan(1_000);
});
it('stops on a zero-progress batch (no hot loop)', async () => {
let batches = 0;
const result = await runExtractAtomsDrain(
@@ -242,17 +133,4 @@ describe('shared wiring helper holds the cycle lock (5A)', () => {
expect(src).toContain('cycleLockIdFor(opts.sourceId)');
expect(src).toContain('withRefreshingLock(engine, lockId');
});
// #2750: the deadline signal must reach the phase, the backlog count, AND
// the lock wrapper — and the transcript path (brainDir) must stay wired
// exactly as the routine callers expect (PR #2752 takeover reverted its
// unsanctioned transcript-suppression scope change).
it('threads the drain deadline signal through phase, count, and lock', () => {
const jobsSrc = readFileSync(join(import.meta.dir, '../src/commands/jobs.ts'), 'utf8');
expect(src).toContain('abortSignal: signal');
expect(src).toContain('countExtractAtomsBacklog(engine, extractionSourceId, signal)');
expect(src).toContain('brainDir: opts.brainDir');
expect(src).not.toContain('_transcripts');
expect(jobsSrc).toContain('abortSignal: job.signal');
});
});
+6 -9
View File
@@ -24,18 +24,15 @@ import type { BrainEngine } from '../../src/core/engine.ts';
// Mock engine: healthCheck() calls engine.executeRaw; return empty rows so
// the query path exercises without needing Postgres.
//
// #1849: start() acquires the queue-scoped DB singleton lock via
// tryAcquireDbLock. #2750 routed the acquire upsert through engine.executeRaw
// (signal-boundable) and release through engine.executeRawDirect, so the
// stub returns a single row from the lock upsert (length 1 → acquired) and
// empty rows everywhere else. Each spawned runner is a fresh process, so
// there's no cross-test lock state to clean up.
// #1849: start() now acquires the queue-scoped DB singleton lock via
// tryAcquireDbLock, which uses the postgres `sql` tagged-template escape hatch.
// The stub returns a single row from every call so acquire succeeds (length 1
// → acquired) and refresh/release are no-ops. Each spawned runner is a fresh
// process, so there's no cross-test lock state to clean up.
const sqlStub = (..._args: unknown[]) => Promise.resolve([{ id: 'supervisor-lock' }]);
const mockEngine: Partial<BrainEngine> = {
kind: 'postgres' as const,
executeRaw: async (query: string) =>
query.includes('gbrain_cycle_locks') ? [{ id: 'supervisor-lock' }] : [],
executeRawDirect: async () => [],
executeRaw: async () => [],
sql: sqlStub,
} as unknown as BrainEngine;
+1 -26
View File
@@ -98,39 +98,14 @@ describe('PostgresEngine.executeRawDirect — routing decision (PR #1816)', () =
});
test('already-aborted signal short-circuits with AbortError before routing the query', async () => {
let unsafeCalls = 0;
let ddlCalls = 0;
const readConn: FakeSql = { unsafe: async () => { unsafeCalls++; return []; } };
const readConn = fakeSql('read');
const directConn = fakeSql('direct');
const engine = makeEngine({ dualPoolActive: true, readConn, directConn });
const e = engine as unknown as { connectionManager: { ddl: () => Promise<FakeSql> } };
e.connectionManager.ddl = async () => { ddlCalls++; return directConn; };
const ac = new AbortController();
ac.abort();
await expect(
engine.executeRawDirect('UPDATE minion_jobs SET x=1', [], { signal: ac.signal }),
).rejects.toThrow(/abort/i);
// #2750: short-circuits BEFORE pool routing — no ddl(), no unsafe().
expect(ddlCalls).toBe(0);
expect(unsafeCalls).toBe(0);
});
test('#2750: signal bounds a stalled direct-pool acquisition before unsafe starts', async () => {
let unsafeCalls = 0;
const readConn: FakeSql = { unsafe: async () => { unsafeCalls++; return []; } };
const directConn = fakeSql('direct');
const engine = makeEngine({ dualPoolActive: true, readConn, directConn });
const e = engine as unknown as { connectionManager: { ddl: () => Promise<FakeSql> } };
e.connectionManager.ddl = () => new Promise<FakeSql>(() => {}); // pooler exhausted: never resolves
const started = Date.now();
await expect(engine.executeRawDirect(
'DELETE FROM gbrain_cycle_locks',
[],
{ signal: AbortSignal.timeout(10) },
)).rejects.toThrow(/abort/i);
expect(Date.now() - started).toBeLessThan(1_000);
expect(unsafeCalls).toBe(0);
});
});
@@ -0,0 +1,62 @@
// test/remediation-context-extraction-lag.test.ts
//
// Pins the v-next fix: the sync→extract remediation pipeline gates on REAL
// extraction lag, not the legacy `health.stale_pages` proxy (which counted
// "updated_at predates newest timeline entry" — meaningless after the v10
// trigger drop). loadRecommendationContext now populates `extractionLagPages`
// from `engine.countStalePagesForExtraction` — the SAME counter the
// `gbrain extract --stale` walk and doctor's `links_extraction_lag` use — so a
// recommendation can only fire when running extract will actually reduce it.
import { afterAll, beforeAll, describe, expect, it } from 'bun:test';
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
import { loadRecommendationContext } from '../src/core/remediation/context.ts';
let engine: PGLiteEngine;
beforeAll(async () => {
engine = new PGLiteEngine();
await engine.connect({});
await engine.initSchema();
});
afterAll(async () => {
await engine.disconnect();
});
describe('loadRecommendationContext — extractionLagPages wiring', () => {
it('is 0 on an empty brain (nothing to extract)', async () => {
const ctx = await loadRecommendationContext(engine);
expect(ctx.extractionLagPages).toBe(0);
});
it('reflects the real extraction-lag count once a page needs extraction', async () => {
// A freshly-imported page has links_extracted_at = NULL, which the canonical
// countStalePagesForExtraction predicate counts as stale-for-extraction.
await engine.putPage('p0', {
title: 'p0',
type: 'note' as never,
compiled_truth: 'body that is long enough to pass any minimum-length guards in the codebase',
timeline: '',
frontmatter: {},
source_path: 'p0.md',
});
const ctx = await loadRecommendationContext(engine);
expect(ctx.extractionLagPages).toBeGreaterThan(0);
});
it('counts pages stamped before LINK_EXTRACTOR_VERSION_TS (version-bump arm)', async () => {
// Backdate p0 so BOTH the NULL arm and the updated_at arm are quiet:
// updated_at < links_extracted_at, but links_extracted_at predates the
// extractor version stamp. doctor's links_extraction_lag and
// `extract --stale` both count this page; the remediation gate must too.
await engine.executeRaw(
`UPDATE pages SET updated_at = '2020-01-01T00:00:00Z'::timestamptz,
links_extracted_at = '2020-01-02T00:00:00Z'::timestamptz
WHERE slug = 'p0'`,
[],
);
const ctx = await loadRecommendationContext(engine);
expect(ctx.extractionLagPages).toBeGreaterThan(0);
});
});
@@ -0,0 +1,81 @@
// test/remediation-run-d7-refresh.serial.test.ts
//
// Pins the D7-recheck half of the extraction-lag gate fix: runRemediation
// loads RecommendationContext ONCE before the step loop, and the per-step
// recheck (D7) must REFRESH ctx.extractionLagPages alongside getHealth.
// Without the refresh, a completed sync/extract step keeps re-firing off
// the frozen initial count — the plan never converges and the loop burns
// steps until maxJobs.
//
// SERIAL (R2): uses top-level mock.module for the minion queue +
// wait-for-completion so no real worker is needed — mocks leak across
// files in a shard process, so this file must run in its own process.
import { describe, expect, mock, test } from 'bun:test';
// The fake brain: sync.repo clears the extraction lag when it "runs"
// (today's sync materializes link/timeline edges; extract.all is the
// explicit re-materializer). The frozen-ctx bug makes runRemediation
// ignore that and resubmit sync.repo on every D7 recheck.
let extractionLag = 25;
const submittedJobs: string[] = [];
mock.module('../src/core/minions/queue.ts', () => ({
MinionQueue: class {
constructor(_engine: unknown) {}
async add(job: string): Promise<{ id: number }> {
submittedJobs.push(job);
if (job === 'sync' || job === 'extract') extractionLag = 0;
return { id: submittedJobs.length };
}
},
}));
mock.module('../src/core/minions/wait-for-completion.ts', () => ({
waitForCompletion: async () => ({ status: 'completed' }),
}));
const health = () => ({
page_count: 100,
embed_coverage: 1.0,
stale_pages: 0, // legacy proxy stays 0 — the real counter drives the gate
orphan_pages: 0,
missing_embeddings: 0,
brain_score: 70,
dead_links: 0,
link_coverage: 1.0,
timeline_coverage: 1.0,
most_connected: [],
embed_coverage_score: 35,
link_density_score: 25,
timeline_coverage_score: 15,
no_orphans_score: 15,
no_dead_links_score: 10,
});
const fakeEngine = {
kind: 'pglite' as const,
getHealth: async () => health(),
getConfig: async (key: string) =>
key === 'sync.repo_path' ? '/tmp/brain-example' : null,
countStalePagesForExtraction: async () => extractionLag,
};
describe('runRemediation D7 recheck — extraction-lag gate refresh', () => {
test('a completed materializer step clears the gate; the pipeline is not resubmitted', async () => {
const { runRemediation } = await import('../src/core/remediation/run.ts');
const result = await runRemediation(
// Only the methods the orchestrator touches are needed.
fakeEngine as never,
{ targetScore: 0, maxJobs: 6 },
);
// Frozen-ctx bug: extractionLagPages stays 25 forever, so every D7
// recheck re-introduces the sync/extract pipeline and the loop burns
// all 6 maxJobs. With the refresh, the plan converges after the first
// completed step: no step id is ever submitted twice.
const ids = result.submitted.map((s) => s.id);
expect(new Set(ids).size).toBe(ids.length);
expect(submittedJobs.length).toBeLessThan(3);
expect(extractionLag).toBe(0);
});
});