Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 9e044021ef test(supervisor): update runner fixture mock for #2750 lock path
tryAcquireDbLock now acquires via engine.executeRaw (signal-boundable)
and releases via engine.executeRawDirect instead of the sql tagged
template. The fixture's mock engine returned [] from executeRaw, so
every spawned supervisor failed lock acquisition and exited LOCK_HELD
(2), breaking all 8 supervisor integration tests. The mock now returns
the lock row for gbrain_cycle_locks queries and stubs executeRawDirect.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 10:53:10 -07:00
553515c82c fix(cycle): enforce the extract-atoms drain deadline end to end (#2750)
The drain window was only checked BETWEEN batches: one batch (sequential
LLM calls per page) plus lock acquire/release and the backlog count ran
unbounded, so window=120s runs were observed overrunning to 282.5s.

The drain now derives a real-time deadline signal (window timeout + the
Minion job.signal) and threads it through everything the loop awaits:

- extract-atoms-drain: signal-driven loop; hung count/batch/lock-acquire
  are cancelled at the deadline; no post-window final count (remaining
  reports null instead of overrunning); external worker abort rethrows
  after the lock is released.
- extract-atoms phase: abortSignal forwarded to every gateway chat call,
  discovery/count/idempotency queries, and putPage writes, plus a
  cooperative between-item check (the bound on PGLite, whose query abort
  only abandons the waiter — the default engine keeps a working drain).
  Billable usage is recorded before the post-chat abort check; receipt +
  rollup bookkeeping runs on a fresh 5s grace signal so committed atoms
  never lose their cost trail; deadline-truncated runs don't count as
  completed rounds (deadline_aborted detail).
- db-lock: tryAcquireDbLock/withRefreshingLock accept a deadline signal;
  release routes through the direct session pool with a fresh grace
  timeout; deadline-bound callers skip the unbounded same-host takeover
  path (honest busy result; TTL stays the backstop).
- engines: putPage accepts an optional signal (Postgres cancels the
  in-flight statement; PGLite pre-checks); executeRawDirect short-circuits
  an already-fired signal before pool routing and bounds a stalled
  direct-pool acquisition.

Takeover of PR #2752: keeps its signal-threading design, but retains the
putPage write path (so the chunker_version default from #2988 applies —
the PR's bulk INSERT bypassed it), drops the PGLite hard refusal in favor
of cooperative abort, and reverts the unsanctioned transcript-suppression
scope change (_transcripts: []).

Fixes #2750

Co-authored-by: panda850819 <panda850819@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:41:34 -07:00
26 changed files with 656 additions and 578 deletions
+5 -10
View File
@@ -170,14 +170,10 @@ export async function runImport(
// v0.22.13 (PR #490 Q2): shared parseWorkers helper rejects bad input
// (--workers 0, -3, "foo") with a loud error instead of silently falling
// through to 1. Mirrors sync.ts's flag handling.
const { parseWorkers, autoConcurrency } = await import('../core/sync-concurrency.ts');
// #1207: undefined (no --workers flag) defers to autoConcurrency below —
// the shared sync/import policy (PGLite → 1, >100 files → 4) — instead of
// hardcoding serial. Large Postgres imports stop paying one embedding
// round-trip per file in sequence.
let workerCount: number | undefined;
const { parseWorkers } = await import('../core/sync-concurrency.ts');
let workerCount: number;
try {
workerCount = parseWorkers(workersArg ?? undefined);
workerCount = parseWorkers(workersArg ?? undefined) ?? 1;
} catch (e) {
console.error(e instanceof Error ? e.message : String(e));
process.exit(1);
@@ -256,9 +252,8 @@ export async function runImport(
}
const files = resumeFilter(allFiles, dir, completed);
// Determine actual worker count. Explicit --workers wins; otherwise the
// shared autoConcurrency policy decides from engine kind + file count.
const actualWorkers = autoConcurrency(engine, files.length, workerCount);
// Determine actual worker count
const actualWorkers = workerCount > 1 ? workerCount : 1;
if (actualWorkers > 1) {
console.log(`Using ${actualWorkers} parallel workers`);
}
+2
View File
@@ -2059,6 +2059,8 @@ export async function registerBuiltinHandlers(
sourceId,
windowSeconds,
brainDir: repoPath,
// #2750: worker cancel/timeout/lock-loss propagates into the drain.
abortSignal: job.signal,
});
} catch (e) {
if (e instanceof LockUnavailableError) {
+6 -24
View File
@@ -1513,21 +1513,12 @@ export async function embed(texts: string[], opts?: EmbedOpts): Promise<Float32A
const embedding = recipe.touchpoints?.embedding;
const maxBatchTokens = embedding?.max_batch_tokens;
const maxBatchCount = embedding?.max_batch_count;
const charsPerToken = embedding?.chars_per_token ?? DEFAULT_CHARS_PER_TOKEN;
// Pre-split is gated on max_batch_tokens / max_batch_count. Recipes with
// neither (e.g. OpenAI) ride the fast path: one embedMany call, no
// recursion safety net.
const batches = (maxBatchTokens || maxBatchCount)
? splitByTokenBudget(
truncated,
maxBatchTokens
? Math.floor(maxBatchTokens * effectiveSafetyFactor(recipe))
: Number.MAX_SAFE_INTEGER,
charsPerToken,
maxBatchCount,
)
// Pre-split is gated on max_batch_tokens. Recipes without it (e.g. OpenAI)
// ride the fast path: one embedMany call, no recursion safety net.
const batches = maxBatchTokens
? splitByTokenBudget(truncated, Math.floor(maxBatchTokens * effectiveSafetyFactor(recipe)), charsPerToken)
: [truncated];
const allEmbeddings: Float32Array[] = [];
@@ -1577,9 +1568,6 @@ export async function embed(texts: string[], opts?: EmbedOpts): Promise<Float32A
* responsible for applying any safety-factor shrink before passing in.
* @param charsPerToken - Provider-specific character density. Defaults to
* `DEFAULT_CHARS_PER_TOKEN` (4) when omitted, matching OpenAI tiktoken.
* @param maxBatchCount - #1199: optional cap on INPUTS per sub-batch, for
* providers that reject batches by count (DashScope: 10). When omitted,
* only the token budget governs.
*
* @internal exported for tests; not part of the public gateway API.
*/
@@ -1587,17 +1575,15 @@ export function splitByTokenBudget(
texts: string[],
budgetTokens: number,
charsPerToken: number = DEFAULT_CHARS_PER_TOKEN,
maxBatchCount?: number,
): string[][] {
const ratio = charsPerToken > 0 ? charsPerToken : DEFAULT_CHARS_PER_TOKEN;
const maxCount = maxBatchCount !== undefined && maxBatchCount > 0 ? maxBatchCount : Infinity;
const batches: string[][] = [];
let current: string[] = [];
let currentTokens = 0;
for (const text of texts) {
const estTokens = Math.ceil(text.length / ratio);
if (current.length > 0 && (currentTokens + estTokens > budgetTokens || current.length >= maxCount)) {
if (current.length > 0 && currentTokens + estTokens > budgetTokens) {
batches.push(current);
current = [];
currentTokens = 0;
@@ -1623,11 +1609,7 @@ export function isTokenLimitError(err: unknown): boolean {
/token.*limit.*exceeded/i.test(msg) ||
// OpenAI embeddings: "Invalid 'input': maximum request size is 300000 tokens per request."
/maximum request size.*tokens/i.test(msg) ||
/max.*tokens.*per.*request/i.test(msg) ||
// DashScope: "batch size is invalid, it should not be larger than 10." (#1199)
// Count-cap error, but recursive halving shrinks count too, so the same
// safety net converges.
/batch size is invalid/i.test(msg)
/max.*tokens.*per.*request/i.test(msg)
);
}
-4
View File
@@ -31,10 +31,6 @@ export const dashscope: Recipe = {
// path. Conservative declaration so the gateway pre-splits before
// hitting whatever undocumented server-side limit exists.
max_batch_tokens: 8192,
// #1199: DashScope hard-caps embeddings at 10 inputs per request
// ("batch size is invalid, it should not be larger than 10"). The
// token budget alone admits far more than 10 short chunks per batch.
max_batch_count: 10,
// text-embedding-v3 mixes English + CJK heavily; the tokenizer is
// closer to Voyage density than OpenAI tiktoken for CJK-dominant
// content. Conservative chars_per_token=2 leaves headroom.
-9
View File
@@ -16,15 +16,6 @@ export const google: Recipe = {
dims_options: [768, 1536, 3072],
cost_per_1m_tokens_usd: 0.15,
price_last_verified: '2026-04-20',
// #970: Gemini's documented limits are per-INPUT (2048 tokens,
// silently truncated beyond) and per-REQUEST count (batchEmbedContents
// caps at 100 inputs). There is no separate per-request token cap, so
// the token budget is derived: 100 inputs × 2048 tokens. The count cap
// binds first for typical chunk sizes. Do NOT copy the 2048 per-input
// limit into max_batch_tokens — that would over-split 50×.
max_batch_tokens: 204_800,
chars_per_token: 4,
max_batch_count: 100,
},
expansion: {
models: ['gemini-2.0-flash', 'gemini-2.0-flash-lite'],
+1 -4
View File
@@ -58,8 +58,5 @@ export function getRecipe(id: string): Recipe | undefined {
}
export function listRecipes(): Recipe[] {
// Read the map (not ALL) so there is one source of truth — getRecipe,
// model-resolver, and listRecipes all see the same registry, and tests
// can inject a synthetic recipe via RECIPES to exercise registry walks.
return [...RECIPES.values()];
return [...ALL];
}
-10
View File
@@ -46,16 +46,6 @@ export interface EmbeddingTouchpoint {
* Only consulted when `max_batch_tokens` is also set.
*/
chars_per_token?: number;
/**
* #1199: maximum number of INPUTS per embedding request, for providers
* that hard-cap batch size by count rather than (or in addition to)
* tokens — DashScope text-embedding-v3 rejects batches > 10 with
* `InvalidParameter`, Gemini batchEmbedContents caps at 100 requests.
* When set, the gateway's pre-split flushes a sub-batch at this count
* even if the token budget still has room. Independent of
* `max_batch_tokens`; either alone triggers the pre-split.
*/
max_batch_count?: number;
/**
* Budget-utilization ceiling in (0, 1]. The gateway pre-splits at
* `safety_factor × max_batch_tokens` to leave headroom for tokenizer
+77 -16
View File
@@ -23,6 +23,10 @@
*/
import type { BrainEngine } from '../engine.ts';
import { anySignal } from '../abort-check.ts';
/** Fresh cleanup budget for the lock release after the window signal fires. */
const LOCK_RELEASE_GRACE_MS = 5_000;
export interface ExtractAtomsDrainDeps {
/**
@@ -30,13 +34,13 @@ export interface ExtractAtomsDrainDeps {
* via `withRefreshingLock`. MUST throw when the lock is held by another
* process (e.g. `LockUnavailableError`) — the drain lets that propagate so
* the caller can report `cycle_already_running` and exit, matching the
* routine cycle's skip contract.
* routine cycle's skip contract. The signal bounds lock acquisition too.
*/
withLock: <T>(work: () => Promise<T>) => Promise<T>;
/** Process one bounded batch (rediscovers eligibility). Returns counts. */
runBatch: () => Promise<{ extracted: number; skipped: number }>;
withLock: <T>(work: () => Promise<T>, signal: AbortSignal) => Promise<T>;
/** Process one batch. The signal fires at the drain wallclock deadline. */
runBatch: (signal: AbortSignal) => Promise<{ extracted: number; skipped: number }>;
/** Count remaining eligible-but-unextracted pages, or null on query error. */
countRemaining: () => Promise<number | null>;
countRemaining: (signal: AbortSignal) => Promise<number | null>;
/** Injectable clock. Production: Date.now. */
now: () => number;
/** Optional progress sink (one line per batch). */
@@ -48,6 +52,8 @@ export interface ExtractAtomsDrainOpts {
windowMs: number;
/** Hard cap on batches (belt-and-suspenders against a 0-progress loop). Default 1000. */
maxBatches?: number;
/** External caller cancellation (worker timeout / shutdown). */
abortSignal?: AbortSignal;
}
export interface ExtractAtomsDrainResult {
@@ -68,35 +74,79 @@ export async function runExtractAtomsDrain(
opts: ExtractAtomsDrainOpts,
): Promise<ExtractAtomsDrainResult> {
const maxBatches = opts.maxBatches ?? 1000;
return deps.withLock(async () => {
const deadline = deps.now() + opts.windowMs;
const deadline = deps.now() + opts.windowMs;
// #2750: the window used to be checked only BETWEEN batches, so one slow
// batch (sequential LLM calls) or a hung lock/count/write overran it without
// bound (observed window=120s → 282.5s). A real-time deadline signal now
// cancels (Postgres) or abandons (PGLite, cooperative) whatever is in
// flight; the injected clock still drives loop-boundary checks so the pure
// loop stays unit-testable.
const signal = anySignal(
AbortSignal.timeout(Math.max(1, opts.windowMs)),
opts.abortSignal,
);
const result: ExtractAtomsDrainResult = await deps.withLock(async () => {
let extracted = 0;
let skipped = 0;
let batches = 0;
let stopped: ExtractAtomsDrainResult['stopped'] = 'window';
while (deps.now() < deadline) {
while (deps.now() < deadline && !signal.aborted) {
if (batches >= maxBatches) { stopped = 'max_batches'; break; }
const before = await deps.countRemaining();
let before: number | null;
try {
before = await deps.countRemaining(signal);
} catch (err) {
if (signal.aborted) break;
throw err;
}
if (before === 0) { stopped = 'drained'; break; }
const r = await deps.runBatch();
// The backlog count consumed the same wallclock budget — re-check so a
// slow count can't hand the batch a window that already expired.
if (deps.now() >= deadline || signal.aborted) break;
let r: { extracted: number; skipped: number };
try {
r = await deps.runBatch(signal);
} catch (err) {
if (signal.aborted) break;
throw err;
}
extracted += r.extracted;
skipped += r.skipped;
batches++;
deps.onBatch?.({ batch: batches, extracted: r.extracted, remaining: before });
// A deadline abort inside the batch can surface as zero progress;
// window exhaustion wins over the generic no_progress label.
if (deps.now() >= deadline || signal.aborted) break;
// Stop if a batch made zero forward progress — extraction is failing or
// everything left is ineligible (e.g. all skipped). Prevents a hot loop
// that spends budget without draining.
if (r.extracted === 0 && r.skipped === 0) { stopped = 'no_progress'; break; }
}
const remaining = await deps.countRemaining();
// After the window elapsed, don't spend more unbounded time on a final
// count — report remaining as unknown instead of overrunning further.
const windowElapsed = signal.aborted || deps.now() >= deadline;
let remaining: number | null = null;
if (!windowElapsed) {
try {
remaining = await deps.countRemaining(signal);
} catch (err) {
if (!signal.aborted) throw err;
}
}
if (remaining === 0) stopped = 'drained';
return { phase: 'extract_atoms', status: 'ok', extracted, skipped, remaining, batches, stopped };
});
}, signal);
// Internal window expiry is a normal partial result. An EXTERNAL abort
// (worker cancel/timeout/shutdown) must reject so Minion records the abort.
if (opts.abortSignal?.aborted) throw opts.abortSignal.reason;
return result;
}
// ─── Shared wiring helper (v0.42.x #1685 DECISION 5A) ──────────────────────
@@ -134,6 +184,8 @@ export interface DrainForSourceOpts {
maxBatches?: number;
/** Optional per-batch progress sink (stderr line in dream; job progress in the handler). */
onBatch?: ExtractAtomsDrainDeps['onBatch'];
/** Worker cancellation / shutdown signal (Minion `job.signal`). */
abortSignal?: AbortSignal;
}
export async function runExtractAtomsDrainForSource(
@@ -149,12 +201,17 @@ export async function runExtractAtomsDrainForSource(
return runExtractAtomsDrain(
{
withLock: (work) => withRefreshingLock(engine, lockId, work, { ttlMinutes: 5 }),
runBatch: async () => {
withLock: (work, signal) => withRefreshingLock(engine, lockId, work, {
ttlMinutes: 5,
signal,
releaseTimeoutMs: LOCK_RELEASE_GRACE_MS,
}),
runBatch: async (signal) => {
const r = await runPhaseExtractAtoms(engine, {
sourceId: extractionSourceId,
dryRun: false,
brainDir: opts.brainDir,
abortSignal: signal,
});
const d = (r.details ?? {}) as Record<string, unknown>;
return {
@@ -162,10 +219,14 @@ export async function runExtractAtomsDrainForSource(
skipped: Number(d.duplicates_skipped ?? 0),
};
},
countRemaining: () => countExtractAtomsBacklog(engine, extractionSourceId),
countRemaining: (signal) => countExtractAtomsBacklog(engine, extractionSourceId, signal),
now: Date.now,
onBatch: opts.onBatch,
},
{ windowMs: opts.windowSeconds * 1000, maxBatches: opts.maxBatches },
{
windowMs: opts.windowSeconds * 1000,
maxBatches: opts.maxBatches,
abortSignal: opts.abortSignal,
},
);
}
+68 -17
View File
@@ -58,6 +58,10 @@ import { createHash } from 'crypto';
import { slugifySegment } from '../sync.ts';
const DEFAULT_BUDGET_USD = 0.3;
// #2750: fresh wallclock budget for the receipt/rollup bookkeeping writes when
// the caller's deadline already fired — committed atoms must not lose their
// cost/receipt trail, but the writes can't be unbounded either.
const BOOKKEEPING_GRACE_MS = 5_000;
// v0.42+ TODO: read atom_type enum from active pack manifest at runtime.
const ATOM_TYPES = [
@@ -155,6 +159,13 @@ export interface ExtractAtomsOpts {
* `heartbeat()` on the passed reporter.
*/
progress?: ProgressReporter;
/**
* #2750: caller deadline/cancellation. Forwarded to every gateway call and
* DB query/write so the drain window bounds real lifetime, plus a
* cooperative between-item check (the PGLite path, where query abort only
* abandons the waiter).
*/
abortSignal?: AbortSignal;
}
interface ExtractedAtom {
@@ -212,6 +223,7 @@ export async function discoverExtractablePages(
engine: BrainEngine,
sourceId: string,
affectedSlugs?: string[],
abortSignal?: AbortSignal,
): Promise<DiscoveredPage[]> {
const hasFilter = Array.isArray(affectedSlugs) && affectedSlugs.length > 0;
const sql = `
@@ -251,13 +263,16 @@ export async function discoverExtractablePages(
slug: string;
compiled_truth: string;
content_hash: string;
}>(sql, params);
}>(sql, params, { signal: abortSignal });
return rows.map((r) => ({
slug: r.slug,
content: r.compiled_truth,
contentHash: r.content_hash,
}));
} catch (err) {
// A deadline abort is not a fail-soft condition — propagate so the
// caller stops instead of proceeding with an empty page list.
if (abortSignal?.aborted) throw err;
const msg = err instanceof Error ? err.message : String(err);
console.error(`[extract_atoms] page-discovery query failed: ${msg}`);
return []; // fail-soft: transcript path still proceeds
@@ -282,6 +297,7 @@ export async function discoverExtractablePages(
export async function countExtractAtomsBacklog(
engine: BrainEngine,
sourceId?: string,
abortSignal?: AbortSignal,
): Promise<number | null> {
try {
// Two modes: scoped (the phase's per-source `remaining`) vs brain-wide
@@ -321,9 +337,10 @@ export async function countExtractAtomsBacklog(
const params = scoped
? [sourceId, extractableTypes, MIN_PAGE_CHARS_FOR_EXTRACTION]
: [extractableTypes, MIN_PAGE_CHARS_FOR_EXTRACTION];
const rows = await engine.executeRaw<{ cnt: string | number }>(sql, params);
const rows = await engine.executeRaw<{ cnt: string | number }>(sql, params, { signal: abortSignal });
return Number(rows[0]?.cnt ?? 0);
} catch (err) {
if (abortSignal?.aborted) throw err;
const msg = err instanceof Error ? err.message : String(err);
console.error(`[extract_atoms] backlog count failed: ${msg}`);
return null;
@@ -350,6 +367,7 @@ export async function atomsExistingForHashes(
engine: BrainEngine,
sourceId: string,
contentHash16s: string[],
abortSignal?: AbortSignal,
): Promise<Set<string>> {
if (contentHash16s.length === 0) return new Set();
try {
@@ -361,9 +379,11 @@ export async function atomsExistingForHashes(
AND deleted_at IS NULL
AND frontmatter->>'source_hash' = ANY($2::text[])`,
[sourceId, contentHash16s],
{ signal: abortSignal },
);
return new Set(rows.map(r => r.h));
} catch (err) {
if (abortSignal?.aborted) throw err;
const msg = err instanceof Error ? err.message : String(err);
console.error(`[extract_atoms] batch idempotency check failed (assuming none extracted): ${msg}`);
return new Set();
@@ -384,6 +404,7 @@ export async function runPhaseExtractAtoms(
): Promise<PhaseResult> {
const sourceId = opts.sourceId ?? 'default';
const chat = opts._chat ?? gatewayChat;
if (opts.abortSignal?.aborted) throw opts.abortSignal.reason;
// 1a. Get transcripts (test seam OR production discovery).
// v0.41.2.1: config loader switched to loadConfigWithEngine() so the
@@ -425,7 +446,7 @@ export async function runPhaseExtractAtoms(
if (opts._pages !== undefined) {
pages = opts._pages;
} else {
pages = await discoverExtractablePages(engine, sourceId, opts.affectedSlugs);
pages = await discoverExtractablePages(engine, sourceId, opts.affectedSlugs, opts.abortSignal);
}
// 2. Apply transcript-side source-hash idempotency in ONE batch query
@@ -437,7 +458,7 @@ export async function runPhaseExtractAtoms(
// Surface a heartbeat before the batch query so even an instant
// short-circuit shows a sign of life (closes Issue 2 silent-phase pain).
opts.progress?.heartbeat(`checking existing atoms for ${allHashes16.length} transcripts`);
const existingHashes = await atomsExistingForHashes(engine, sourceId, allHashes16);
const existingHashes = await atomsExistingForHashes(engine, sourceId, allHashes16, opts.abortSignal);
for (const t of transcripts) {
if (existingHashes.has(t.contentHash.slice(0, 16))) {
duplicatesSkipped++;
@@ -501,6 +522,7 @@ export async function runPhaseExtractAtoms(
const failures: Array<{ source: string; error: string }> = [];
let estimatedSpendUsd = 0;
const budgetCap = DEFAULT_BUDGET_USD;
let deadlineAborted = false;
// v0.41.19.0 (T3): throttled yield helper. Fires `opts.yieldDuringPhase`
// every 30s. Cycle.ts threads `buildYieldDuringPhase(lock, outer)` so
@@ -526,6 +548,12 @@ export async function runPhaseExtractAtoms(
}
for (const item of work) {
// #2750: cooperative between-item abort. Works on every engine — this is
// the primary bound on PGLite, where query abort only abandons the waiter.
if (opts.abortSignal?.aborted) {
deadlineAborted = true;
break;
}
await maybeYield();
if (estimatedSpendUsd >= budgetCap) {
if (item.kind === 'transcript') transcriptsSkipped++;
@@ -544,16 +572,22 @@ export async function runPhaseExtractAtoms(
},
],
maxTokens: 2000,
abortSignal: opts.abortSignal,
});
// Rough cost estimate — Haiku at ~$0.80/M input + $4/M output.
// A completed gateway call is billable even if the deadline fires
// immediately afterward, so record usage BEFORE the abort check.
estimatedSpendUsd +=
(result.usage.input_tokens * 0.8 + result.usage.output_tokens * 4.0) / 1_000_000;
if (opts.abortSignal?.aborted) {
deadlineAborted = true;
break;
}
// Post-await yield: closes the "long LLM call past TTL" hazard
// codex flagged. The 30s throttle inside maybeYield bounds the
// actual refresh rate so this is cheap when calls are fast.
await maybeYield();
// Rough cost estimate — Haiku at ~$0.80/M input + $4/M output
estimatedSpendUsd +=
(result.usage.input_tokens * 0.8 + result.usage.output_tokens * 4.0) / 1_000_000;
const atoms = parseAtomsResponse(result.text);
if (atoms.length === 0) {
if (item.kind === 'transcript') transcriptsProcessed++;
@@ -592,7 +626,7 @@ export async function runPhaseExtractAtoms(
},
timeline: '',
},
{ sourceId },
{ sourceId, signal: opts.abortSignal },
);
totalAtomsExtracted++;
}
@@ -605,6 +639,11 @@ export async function runPhaseExtractAtoms(
// Reporter rate-limits to ~1 line/sec; safe to tick every iter.
opts.progress?.tick(1, `${totalAtomsExtracted} atoms / ${duplicatesSkipped} skipped`);
} catch (err) {
// A deadline abort is a partial result, not a per-item failure.
if (opts.abortSignal?.aborted) {
deadlineAborted = true;
break;
}
failures.push({
source: originLabel,
error: err instanceof Error ? err.message : String(err),
@@ -615,6 +654,12 @@ export async function runPhaseExtractAtoms(
// v0.42 Wave B2: write extract receipt + rollup row when the phase
// actually extracted atoms. Both are best-effort per F-OUT-19 —
// audit-trail / search-visibility surfaces don't block the phase result.
//
// #2750: bookkeeping runs on a FRESH short grace signal, never the caller's
// work deadline — the deadline may have already fired (partial run) and
// committed atoms must not lose their receipt/cost trail; but the writes
// stay bounded so the overrun is capped at the grace window.
const bookkeepingSignal = opts.dryRun ? undefined : AbortSignal.timeout(BOOKKEEPING_GRACE_MS);
if (!opts.dryRun && totalAtomsExtracted > 0) {
const runId = `atoms-${Date.now().toString(36)}-${sourceId.slice(0, 4)}`;
try {
@@ -629,19 +674,24 @@ export async function runPhaseExtractAtoms(
summary:
`Extracted ${totalAtomsExtracted} atoms from ` +
`${transcriptsProcessed} transcripts + ${pagesProcessed} pages.`,
});
}, { signal: bookkeepingSignal });
} catch (err) {
console.error(`[extract_atoms] receipt write failed: ${(err as Error).message}`);
}
}
if (!opts.dryRun) {
await upsertExtractRollup(engine, {
kind: 'atoms',
source_id: sourceId,
cost_delta: estimatedSpendUsd,
round_completed_delta: failures.length === 0 ? 1 : 0,
halt_delta: failures.length > 0 ? 1 : 0,
});
try {
await upsertExtractRollup(engine, {
kind: 'atoms',
source_id: sourceId,
cost_delta: estimatedSpendUsd,
// A deadline-truncated run is not a completed round.
round_completed_delta: failures.length === 0 && !deadlineAborted ? 1 : 0,
halt_delta: failures.length > 0 ? 1 : 0,
}, { signal: bookkeepingSignal });
} catch (err) {
console.error(`[extract_atoms] rollup write failed: ${(err as Error).message}`);
}
}
return {
@@ -670,6 +720,7 @@ export async function runPhaseExtractAtoms(
budget_usd: budgetCap,
source_id: sourceId,
dry_run: opts.dryRun ?? false,
deadline_aborted: deadlineAborted,
},
};
}
+50 -22
View File
@@ -26,7 +26,8 @@ import type { BrainEngine } from './engine.ts';
export interface DbLockHandle {
id: string;
release: () => Promise<void>;
/** Optional signal bounds the release DELETE (deadline-bound callers). */
release: (signal?: AbortSignal) => Promise<void>;
refresh: () => Promise<void>;
}
@@ -173,6 +174,7 @@ export async function tryAcquireDbLock(
engine: BrainEngine,
lockId: string,
ttlMinutes: number = DEFAULT_TTL_MINUTES,
opts: { signal?: AbortSignal } = {},
): Promise<DbLockHandle | null> {
const pid = process.pid;
const host = hostname();
@@ -205,20 +207,26 @@ export async function tryAcquireDbLock(
// `gbrain sync --break-lock --max-age <s>` uses last_refreshed_at (not
// acquired_at) to identify wedged-but-alive holders without stealing
// healthy long-running holders that are actively refreshing.
const rows: Array<{ id: string }> = await sql`
INSERT INTO gbrain_cycle_locks (id, holder_pid, holder_host, acquired_at, ttl_expires_at, last_refreshed_at)
VALUES (${lockId}, ${pid}, ${host}, NOW(), NOW() + ${ttl}::interval, NOW())
ON CONFLICT (id) DO UPDATE
SET holder_pid = ${pid},
holder_host = ${host},
acquired_at = NOW(),
ttl_expires_at = NOW() + ${ttl}::interval,
last_refreshed_at = NOW()
WHERE gbrain_cycle_locks.ttl_expires_at < NOW()
AND (gbrain_cycle_locks.last_refreshed_at IS NULL
OR gbrain_cycle_locks.last_refreshed_at < NOW() - ${stealGraceSeconds} * INTERVAL '1 second')
RETURNING id
`;
// #2750: routed through executeRaw so a deadline-bound caller's signal
// can cancel a hung acquire (pool exhaustion). Cancellation is
// transactional; in the rare ambiguous-commit case the row's TTL is the
// backstop (drain locks use a short 5-minute TTL).
const rows = await engine.executeRaw<{ id: string }>(
`INSERT INTO gbrain_cycle_locks (id, holder_pid, holder_host, acquired_at, ttl_expires_at, last_refreshed_at)
VALUES ($1, $2, $3, NOW(), NOW() + $4::interval, NOW())
ON CONFLICT (id) DO UPDATE
SET holder_pid = $2,
holder_host = $3,
acquired_at = NOW(),
ttl_expires_at = NOW() + $4::interval,
last_refreshed_at = NOW()
WHERE gbrain_cycle_locks.ttl_expires_at < NOW()
AND (gbrain_cycle_locks.last_refreshed_at IS NULL
OR gbrain_cycle_locks.last_refreshed_at < NOW() - $5 * INTERVAL '1 second')
RETURNING id`,
[lockId, pid, host, ttl, stealGraceSeconds],
{ signal: opts.signal },
);
if (rows.length === 0) return null;
const deregister = registerCleanup(`db-lock:${lockId}`, async () => {
await sql`
@@ -241,12 +249,17 @@ export async function tryAcquireDbLock(
[ttl, lockId, pid],
);
},
release: async () => {
release: async (signal?: AbortSignal) => {
deregister();
await sql`
DELETE FROM gbrain_cycle_locks
WHERE id = ${lockId} AND holder_pid = ${pid}
`;
// Direct session pool (same rationale as refresh, #1794) + optional
// signal so a deadline-bound caller's release can't hang forever on
// an exhausted pooler. TTL is the backstop if the DELETE is cancelled.
await engine.executeRawDirect(
`DELETE FROM gbrain_cycle_locks
WHERE id = $1 AND holder_pid = $2`,
[lockId, pid],
{ signal },
);
},
};
}
@@ -303,6 +316,11 @@ export async function tryAcquireDbLock(
const first = await acquireOnce();
if (first) return first;
// #2750: deadline-bound callers prefer an honest busy result over the
// best-effort same-host takeover below, whose inspect/delete/retry calls
// are not signal-bounded. The initial upsert already reclaims expired locks.
if (opts.signal) return null;
// v0.42 (#1780 Gap 3): the lock is held and its TTL hasn't expired (the
// upsert's ON CONFLICT ... WHERE ttl_expires_at < NOW() returned no row).
// If the holder is on THIS host, provably dead, and past the grace window,
@@ -796,6 +814,10 @@ export interface WithRefreshingLockOpts {
ttlMinutes?: number;
/** Heartbeat-fail threshold in ms — abort if SELECT 1 takes longer. Default 30000. */
heartbeatTimeoutMs?: number;
/** #2750: bound lock acquisition with the caller's deadline signal. */
signal?: AbortSignal;
/** Fresh cleanup budget for the release DELETE when `signal` is set. Default 5000. */
releaseTimeoutMs?: number;
}
/**
@@ -815,7 +837,7 @@ export async function withRefreshingLock<T>(
// Refresh 6x per TTL window so a missed tick doesn't expire the lock.
const refreshIntervalMs = Math.max(15000, (ttlMinutes * 60 * 1000) / 6);
const handle = await tryAcquireDbLock(engine, lockId, ttlMinutes);
const handle = await tryAcquireDbLock(engine, lockId, ttlMinutes, { signal: opts.signal });
if (!handle) throw new LockUnavailableError(lockId);
let healthOk = true;
@@ -854,7 +876,13 @@ export async function withRefreshingLock<T>(
return await work();
} finally {
clearInterval(interval);
try { await handle.release(); } catch { /* idempotent */ }
// #2750: when the caller is deadline-bound, its work signal may already
// have fired — release on a FRESH short grace signal so cleanup neither
// inherits the spent deadline nor hangs unbounded. TTL is the backstop.
const releaseSignal = opts.signal
? AbortSignal.timeout(opts.releaseTimeoutMs ?? 5_000)
: undefined;
try { await handle.release(releaseSignal); } catch { /* idempotent; TTL backstop */ }
if (!healthOk) {
// Surface that the heartbeat detected backend trouble — caller can
// log to the connection-events audit if desired.
+6 -56
View File
@@ -79,34 +79,15 @@ export interface EmbedBatchOptions {
* and amplify rate-limit pressure.
*/
maxRetries?: number;
/**
* #1818: bounded parallelism across BATCH_SIZE sub-batches. Defaults to
* `GBRAIN_EMBED_BATCH_CONCURRENCY` env, else 4. Results are
* index-addressed so output order always matches input order. Set 1 to
* force the pre-v0.42 serial dispatch.
*/
concurrency?: number;
}
/**
* Embed a batch of texts via the gateway. Sub-batches of 100 so upstream
* progress callbacks fire incrementally on large imports. The gateway owns
* adaptive batch splitting and per-recipe token-budget logic; this paginator
* owns progress-callback granularity and (#1818) bounded parallel dispatch
* of the sub-batches — the embed-stale.ts worker-pool pattern, scoped down.
* is purely about progress-callback granularity.
*/
const BATCH_SIZE = 100;
const DEFAULT_EMBED_BATCH_CONCURRENCY = 4;
function resolveEmbedBatchConcurrency(options: EmbedBatchOptions): number {
if (options.concurrency !== undefined) {
return Math.max(1, Math.floor(options.concurrency));
}
const env = Number(process.env.GBRAIN_EMBED_BATCH_CONCURRENCY);
if (Number.isFinite(env) && env >= 1) return Math.floor(env);
return DEFAULT_EMBED_BATCH_CONCURRENCY;
}
export async function embedBatch(
texts: string[],
options: EmbedBatchOptions = {},
@@ -122,44 +103,13 @@ export async function embedBatch(
if (texts.length <= BATCH_SIZE && !options.onBatchComplete) {
return gatewayEmbed(texts, gwOpts);
}
// #1818: dispatch sub-batches through a bounded worker pool instead of a
// serial loop. Results are written into a preallocated index-addressed
// array so output order matches input order regardless of completion
// order; onBatchComplete reports a monotonic completed-embedding count.
const slices: Array<{ start: number; texts: string[] }> = [];
const results: Float32Array[] = [];
for (let i = 0; i < texts.length; i += BATCH_SIZE) {
slices.push({ start: i, texts: texts.slice(i, i + BATCH_SIZE) });
const slice = texts.slice(i, i + BATCH_SIZE);
const out = await gatewayEmbed(slice, gwOpts);
results.push(...out);
options.onBatchComplete?.(results.length, texts.length);
}
const results = new Array<Float32Array>(texts.length);
let next = 0;
let done = 0;
const numWorkers = Math.min(resolveEmbedBatchConcurrency(options), slices.length);
// Once any sub-batch fails, `failed` stops the surviving workers from
// dispatching FURTHER slices — the whole call is rejecting anyway, so
// continuing would burn real provider spend in the background and fire
// onBatchComplete after the caller already saw the failure (worst with
// embedBatchWithBackoff, whose 429 backoff assumes nothing is in flight).
// In-flight sibling calls still run to completion (bounded by numWorkers-1).
let failed = false;
const worker = async (): Promise<void> => {
while (!failed && next < slices.length) {
// NOTE: no local aborted-check here — an aborted signal makes the next
// gatewayEmbed call throw (SDK-side), which rejects the pool. Returning
// silently instead would resolve with holes in `results`.
const slice = slices[next++];
let out: Float32Array[];
try {
out = await gatewayEmbed(slice.texts, gwOpts);
} catch (err) {
failed = true;
throw err;
}
for (let j = 0; j < out.length; j++) results[slice.start + j] = out[j];
done += out.length;
if (!failed) options.onBatchComplete?.(done, texts.length);
}
};
await Promise.all(Array.from({ length: numWorkers }, () => worker()));
return results;
}
+9 -1
View File
@@ -696,8 +696,16 @@ export interface BrainEngine {
* is included in the INSERT column list so ON CONFLICT (source_id, slug)
* DO UPDATE actually targets the intended row instead of fabricating a
* duplicate at (default, slug). Multi-source brains MUST pass sourceId.
*
* `opts.signal` (#2750): optional cancellation for deadline-bound writers.
* Postgres cancels the in-flight statement; PGLite pre-checks only (query
* cancellation is not possible in-process — cooperative abort between calls).
*/
putPage(slug: string, page: PageInput, opts?: { sourceId?: string }): Promise<Page>;
putPage(
slug: string,
page: PageInput,
opts?: { sourceId?: string; signal?: AbortSignal },
): Promise<Page>;
/**
* v0.41.13 (#1309) — identity-based dedup pre-check for the import pipeline.
*
+2 -1
View File
@@ -187,6 +187,7 @@ function buildReceiptFrontmatter(input: ExtractReceiptInput): Record<string, unk
export async function writeReceipt(
engine: BrainEngine,
input: ExtractReceiptInput,
opts?: { signal?: AbortSignal },
): Promise<{ slug: string; page: Page }> {
const slug = receiptSlug(input);
const title = `${input.kind}${input.round}${input.source_id}`;
@@ -201,7 +202,7 @@ export async function writeReceipt(
compiled_truth,
frontmatter,
},
{ sourceId: input.source_id },
{ sourceId: input.source_id, signal: opts?.signal },
);
return { slug, page };
+4
View File
@@ -70,6 +70,7 @@ function today(): string {
export async function upsertExtractRollup(
engine: BrainEngine,
input: RollupUpsertInput,
opts?: { signal?: AbortSignal },
): Promise<{ ok: boolean; error?: string }> {
const day = input.day ?? today();
const cost = input.cost_delta ?? 0;
@@ -96,9 +97,12 @@ export async function upsertExtractRollup(
rollup_write_failures = extract_rollup_7d.rollup_write_failures + EXCLUDED.rollup_write_failures,
updated_at = now()`,
[input.kind, input.source_id, day, cost, halts, evalFails, evalPasses, completed, failures],
{ signal: opts?.signal },
);
return { ok: true };
} catch (err) {
// Signal-bounded callers get the abort surfaced, not a swallowed `ok:false`.
if (opts?.signal?.aborted) throw err;
const msg = (err as Error).message || String(err);
// Don't spam: log once per process per (kind, day) error class.
rollupErrorLogOnce(input.kind, day, msg);
+9 -1
View File
@@ -1003,7 +1003,15 @@ export class PGLiteEngine implements BrainEngine {
return { slug: r.slug, id: Number(r.id) };
}
async putPage(slug: string, page: PageInput, opts?: { sourceId?: string }): Promise<Page> {
async putPage(
slug: string,
page: PageInput,
opts?: { sourceId?: string; signal?: AbortSignal },
): Promise<Page> {
// #2750: PGLite is in-process WASM — no query cancellation. Pre-check so
// an already-fired deadline skips the write; abort is cooperative
// between calls (same posture as executeRaw's documented gap).
if (opts?.signal?.aborted) throw new DOMException('aborted', 'AbortError');
slug = validateSlug(slug);
const hash = page.content_hash || contentHash(page);
const frontmatter = page.frontmatter || {};
+55 -3
View File
@@ -72,6 +72,32 @@ function escapeSqlStringLiteral(value: string): string {
return value.replace(/'/g, "''");
}
/**
* #2750: race a promise against an AbortSignal, detaching the listener once
* settled (long-lived drain signals are reused across many calls, so a bare
* Promise.race would leak one listener per call). The abandoned promise keeps
* running; used only for pool-acquisition waits where that is harmless.
*/
function waitForSignal<T>(work: Promise<T>, signal?: AbortSignal): Promise<T> {
if (!signal) return work;
if (signal.aborted) return Promise.reject(new DOMException('aborted', 'AbortError'));
return new Promise<T>((resolve, reject) => {
let settled = false;
const finish = (fn: () => void) => {
if (settled) return;
settled = true;
signal.removeEventListener('abort', onAbort);
fn();
};
const onAbort = () => finish(() => reject(new DOMException('aborted', 'AbortError')));
signal.addEventListener('abort', onAbort, { once: true });
work.then(
(value) => finish(() => resolve(value)),
(err) => finish(() => reject(err)),
);
});
}
export function getPostgresSchema(
dims: number = DEFAULT_EMBEDDING_DIMENSIONS,
model: string = DEFAULT_EMBEDDING_MODEL,
@@ -1061,7 +1087,12 @@ export class PostgresEngine implements BrainEngine {
});
}
async putPage(slug: string, page: PageInput, opts?: { sourceId?: string }): Promise<Page> {
async putPage(
slug: string,
page: PageInput,
opts?: { sourceId?: string; signal?: AbortSignal },
): Promise<Page> {
if (opts?.signal?.aborted) throw new DOMException('aborted', 'AbortError');
slug = validateSlug(slug);
const sql = this.sql;
const hash = page.content_hash || contentHash(page);
@@ -1096,7 +1127,7 @@ export class PostgresEngine implements BrainEngine {
const sourceUri = page.source_uri ?? null;
const ingestedVia = page.ingested_via ?? null;
const ingestedAt = (sourceKind || sourceUri || ingestedVia) ? new Date() : null;
const rows = await sql`
const pending = sql`
INSERT INTO pages (source_id, slug, type, page_kind, title, compiled_truth, timeline, frontmatter, content_hash, updated_at, effective_date, effective_date_source, import_filename, chunker_version, source_path, source_kind, source_uri, ingested_via, ingested_at)
VALUES (${sourceId}, ${slug}, ${page.type}, ${pageKind}, ${page.title}, ${page.compiled_truth}, ${page.timeline || ''}, ${sql.json(frontmatter as Parameters<typeof sql.json>[0])}, ${hash}, now(), ${effectiveDate}, ${effectiveDateSource}, ${importFilename}, COALESCE(${chunkerVersion}::smallint, ${MARKDOWN_CHUNKER_VERSION}), ${sourcePath}, ${sourceKind}, ${sourceUri}, ${ingestedVia}, ${ingestedAt})
ON CONFLICT (source_id, slug) DO UPDATE SET
@@ -1119,6 +1150,22 @@ export class PostgresEngine implements BrainEngine {
ingested_at = COALESCE(EXCLUDED.ingested_at, pages.ingested_at)
RETURNING id, source_id, slug, type, title, compiled_truth, timeline, frontmatter, content_hash, created_at, updated_at, effective_date, effective_date_source, import_filename, source_kind, source_uri, ingested_via, ingested_at
`;
// #2750: cancel the in-flight statement when the caller's deadline fires,
// same .cancel() wiring as runUnsafe (postgres.js pending queries).
if (opts?.signal) {
const signal = opts.signal;
const onAbort = () => {
try { (pending as unknown as { cancel?: () => void }).cancel?.(); } catch { /* best-effort */ }
};
signal.addEventListener('abort', onAbort, { once: true });
try {
const rows = await pending;
return rowToPage(rows[0]);
} finally {
signal.removeEventListener('abort', onAbort);
}
}
const rows = await pending;
return rowToPage(rows[0]);
}
@@ -5807,11 +5854,16 @@ export class PostgresEngine implements BrainEngine {
params?: unknown[],
opts?: { signal?: AbortSignal },
): Promise<T[]> {
// #2750: an already-fired signal short-circuits BEFORE any pool routing,
// and the direct-pool acquisition itself is signal-bounded — under pooler
// exhaustion `ddl()` can stall indefinitely, which used to make even a
// "bounded" lock release hang past its caller's deadline.
if (opts?.signal?.aborted) throw new DOMException('aborted', 'AbortError');
// Inside an open transaction, _sql is the reserved tx connection (set via
// defineProperty in transaction()); never reroute off it.
const inTransaction = this._sql !== null && this.connectionManager?.peekReadPool() !== this._sql;
const conn = (!inTransaction && this.connectionManager?.isDualPoolActive())
? await this.connectionManager.ddl()
? await waitForSignal(this.connectionManager.ddl(), opts?.signal)
: this.sql;
return this.runUnsafe<T>(conn, sql, params, opts);
}
+5 -109
View File
@@ -39,8 +39,6 @@ import {
__getShrinkStateForTests,
} from '../../src/core/ai/gateway.ts';
import { AIConfigError, AITransientError } from '../../src/core/ai/errors.ts';
import { RECIPES } from '../../src/core/ai/recipes/index.ts';
import type { Recipe } from '../../src/core/ai/types.ts';
// The last test in this file leaves the gateway configured with a remote
// provider + fake key and a REAL embed transport. Without a final reset,
@@ -95,14 +93,6 @@ function configureGoogle(): void {
});
}
function configureDashscope(): void {
configureGateway({
embedding_model: 'dashscope:text-embedding-v3',
embedding_dimensions: 1024,
env: { DASHSCOPE_API_KEY: 'sk-fake' },
});
}
// --------- 1. Pure helpers ---------
describe('splitByTokenBudget (pure helper)', () => {
@@ -159,27 +149,6 @@ describe('splitByTokenBudget (pure helper)', () => {
expect(splitByTokenBudget(texts, 96_000, 0)).toEqual(splitByTokenBudget(texts, 96_000, 4));
expect(splitByTokenBudget(texts, 96_000, -1)).toEqual(splitByTokenBudget(texts, 96_000, 4));
});
// #1199: count cap for providers that reject batches by input count.
test('max_batch_count flushes even when token budget has room', () => {
const texts = Array.from({ length: 25 }, (_, i) => `t${i}`);
const result = splitByTokenBudget(texts, 1_000_000, 4, 10);
expect(result.map(b => b.length)).toEqual([10, 10, 5]);
expect(result.flat()).toEqual(texts);
});
test('token budget still governs alongside max_batch_count', () => {
const texts = ['a'.repeat(50_000), 'b'.repeat(50_000), 'c'.repeat(50_000)];
const result = splitByTokenBudget(texts, 96_000, 1, 10);
expect(result).toHaveLength(3);
});
test('undefined / zero / negative max_batch_count is ignored', () => {
const texts = Array.from({ length: 25 }, () => 'x');
expect(splitByTokenBudget(texts, 1_000_000, 4, undefined)).toHaveLength(1);
expect(splitByTokenBudget(texts, 1_000_000, 4, 0)).toHaveLength(1);
expect(splitByTokenBudget(texts, 1_000_000, 4, -5)).toHaveLength(1);
});
});
describe('isTokenLimitError (pure helper)', () => {
@@ -210,12 +179,6 @@ describe('isTokenLimitError (pure helper)', () => {
expect(isTokenLimitError(new Error('Exceeded 300000 max tokens per request'))).toBe(true);
});
test('matches DashScope batch-count error (#1199)', () => {
expect(isTokenLimitError(new Error(
'InvalidParameter: batch size is invalid, it should not be larger than 10.',
))).toBe(true);
});
test('does not match unrelated errors', () => {
expect(isTokenLimitError(new Error('Connection refused'))).toBe(false);
expect(isTokenLimitError(new Error('Invalid API key'))).toBe(false);
@@ -424,92 +387,26 @@ describe('shrink-on-miss adaptive cache', () => {
});
});
// --------- 8. Pre-split count cap through public embed() (#1199 / #970) ---------
describe('embed() pre-split honors max_batch_count', () => {
beforeEach(() => resetGateway());
afterEach(() => __setEmbedTransportForTests(null));
test('dashscope never dispatches more than 10 inputs per call (#1199)', async () => {
configureDashscope();
const stub = mock(async ({ values }: { values: string[] }) => fakeEmbeddings(values, 1024));
__setEmbedTransportForTests(stub as any);
// 25 short texts fit trivially in the 8192-token budget; without the
// count cap they'd ship as ONE batch and DashScope would reject it.
const texts = Array.from({ length: 25 }, (_, i) => `short-${i}`);
const result = await embed(texts);
expect(result).toHaveLength(25);
const callLengths = stub.mock.calls.map(([arg]) => (arg as { values: string[] }).values.length);
expect(Math.max(...callLengths)).toBeLessThanOrEqual(10);
expect(callLengths.reduce((a, b) => a + b, 0)).toBe(25);
// Order preserved across sub-batches.
expect((stub.mock.calls[0][0] as { values: string[] }).values[0]).toBe('short-0');
});
test('google pre-splits at 100 inputs per batchEmbedContents call (#970)', async () => {
configureGoogle();
const stub = mock(async ({ values }: { values: string[] }) => fakeEmbeddings(values, 768));
__setEmbedTransportForTests(stub as any);
const texts = Array.from({ length: 250 }, (_, i) => `g${i}`);
const result = await embed(texts);
expect(result).toHaveLength(250);
const callLengths = stub.mock.calls.map(([arg]) => (arg as { values: string[] }).values.length);
expect(callLengths).toEqual([100, 100, 50]);
});
});
// --------- 7. Startup warning (D9-B) ---------
describe('startup warning for recipes missing max_batch_tokens', () => {
beforeEach(() => resetGateway());
// #970 closed google's missing cap, so no registered recipe is capless
// anymore. Inject a synthetic capless recipe to keep the warning path
// covered for the NEXT recipe that forgets the field.
const caplessRecipe: Recipe = {
id: 'capless-test',
name: 'Capless Test Provider',
tier: 'openai-compat',
implementation: 'openai-compatible',
base_url_default: 'https://example.invalid/v1',
auth_env: { required: [] },
touchpoints: {
embedding: { models: ['capless-embed-1'], default_dims: 768 },
},
};
function configureCapless(): void {
configureGateway({
embedding_model: 'capless-test:capless-embed-1',
embedding_dimensions: 768,
env: {},
});
}
test('configured missing-cap recipe warns once; unrelated recipes stay quiet', () => {
const warnings: string[] = [];
const original = console.warn;
console.warn = (msg: string) => warnings.push(String(msg));
RECIPES.set(caplessRecipe.id, caplessRecipe);
try {
configureOpenAI();
expect(warnings.length).toBe(0);
// #970 regression: google now declares max_batch_tokens → quiet.
configureGoogle();
expect(warnings.length).toBe(0);
configureCapless();
const firstCallCount = warnings.length;
// Reconfigure: the warning should NOT re-fire for the same recipes
// within one process (we already told the operator).
configureCapless();
configureGoogle();
expect(warnings.length).toBe(firstCallCount);
} finally {
console.warn = original;
RECIPES.delete(caplessRecipe.id);
}
// The warning text should match the documented contract.
@@ -518,12 +415,11 @@ describe('startup warning for recipes missing max_batch_tokens', () => {
);
expect(contractMatch.length).toBe(1);
// Voyage + google declare max_batch_tokens → suppressed. OpenAI is the
// canonical fast-path recipe → also suppressed by id. All must be
// absent from the warnings; only the synthetic capless recipe fires.
// Voyage declares max_batch_tokens → suppressed. OpenAI is the
// canonical fast-path recipe → also suppressed by id. Both must be
// absent from the warnings.
expect(warnings.find(w => w.includes('"voyage"'))).toBeUndefined();
expect(warnings.find(w => w.includes('"openai"'))).toBeUndefined();
expect(warnings.find(w => w.includes('"google"'))).toBeUndefined();
expect(warnings.find(w => w.includes('"capless-test"'))).toBeDefined();
expect(warnings.find(w => w.includes('"google"'))).toBeDefined();
});
});
+13 -13
View File
@@ -52,7 +52,16 @@ describe('v0.32 #779: no_batch_cap suppresses the missing-max_batch_tokens warni
}
});
test('configureGateway does NOT warn for google now that it declares batch caps (#970)', () => {
test('configureGateway warns for google only when google embedding is configured', () => {
warnSpy.mockClear();
resetGateway();
configureGateway({ env: {} });
let messages = warnSpy.mock.calls.map(c => String(c[0] ?? ''));
expect(
messages.some(m => m.includes('"google"') && m.includes('without max_batch_tokens')),
'google should not warn while OpenAI default is configured',
).toBe(false);
warnSpy.mockClear();
resetGateway();
configureGateway({
@@ -60,20 +69,11 @@ describe('v0.32 #779: no_batch_cap suppresses the missing-max_batch_tokens warni
embedding_dimensions: 768,
env: { GOOGLE_GENERATIVE_AI_API_KEY: 'fake' },
});
const messages = warnSpy.mock.calls.map(c => String(c[0] ?? ''));
messages = warnSpy.mock.calls.map(c => String(c[0] ?? ''));
expect(
messages.some(m => m.includes('"google"') && m.includes('without max_batch_tokens')),
'google declares max_batch_tokens/max_batch_count since #970 — no warning',
).toBe(false);
});
test('google recipe declares its derived batch caps (#970)', () => {
const e = getRecipe('google')!.touchpoints.embedding!;
// Count cap is the REAL Gemini limit (batchEmbedContents: 100 inputs);
// the token budget is derived (100 × 2048 per-input tokens), NOT the
// 2048 per-input limit — copying that verbatim would over-split 50×.
expect(e.max_batch_count).toBe(100);
expect(e.max_batch_tokens).toBe(204_800);
'google should warn when configured because it has fixed-cap models',
).toBe(true);
});
test('every recipe with empty models[] declares user_provided_models OR has openai-fast-path', () => {
-5
View File
@@ -55,11 +55,6 @@ describe('recipe: dashscope', () => {
expect(r.touchpoints.embedding!.chars_per_token).toBeGreaterThan(0);
});
test('declares max_batch_count: 10 — DashScope rejects larger batches (#1199)', () => {
const r = getRecipe('dashscope')!;
expect(r.touchpoints.embedding!.max_batch_count).toBe(10);
});
test('dimsProviderOptions threads dimensions for text-embedding-v3 (Matryoshka)', async () => {
// Codex finding #1: DashScope text-embedding-v3 is Matryoshka 64-1024.
// Without `dimensions` on the wire, user-selected non-default dims are
@@ -17,6 +17,7 @@ import { runPhaseExtractAtoms, parseAtomsResponse } from '../../src/core/cycle/e
import { runPhaseSynthesizeConcepts } from '../../src/core/cycle/synthesize-concepts.ts';
import { resetPgliteState } from '../helpers/reset-pglite.ts';
import type { ChatResult, ChatOpts } from '../../src/core/ai/gateway.ts';
import type { BrainEngine } from '../../src/core/engine.ts';
let engine: PGLiteEngine;
@@ -177,6 +178,180 @@ describe('v0.41 T5: runPhaseExtractAtoms via stubbed chat', () => {
expect((result.details?.failures as unknown[]).length).toBe(1);
});
// ── #2750: caller deadline bounds the phase ────────────────────────────
test('caller deadline aborts a hung chat before processing the next item', async () => {
let calls = 0;
const chat = async (opts: ChatOpts) => {
calls++;
return await new Promise<never>((_resolve, reject) => {
const signal = opts.abortSignal;
if (!signal) return reject(new Error('missing abort signal'));
if (signal.aborted) return reject(signal.reason);
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
});
};
const started = Date.now();
const result = await runPhaseExtractAtoms(engine, {
_transcripts: [
{ filePath: '/hung.txt', content: 'a', contentHash: 'hung-a' },
{ filePath: '/never.txt', content: 'b', contentHash: 'hung-b' },
],
_pages: [],
_chat: chat as typeof import('../../src/core/ai/gateway.ts').chat,
abortSignal: AbortSignal.timeout(25),
});
expect(Date.now() - started).toBeLessThan(2_000);
expect(calls).toBe(1);
expect(result.status).toBe('ok');
expect(result.details?.deadline_aborted).toBe(true);
expect(result.details?.atoms_extracted).toBe(0);
expect(result.details?.failures).toEqual([]);
});
test('billable chat usage is counted when the deadline fires as the response resolves', async () => {
const controller = new AbortController();
const chat = async (opts: ChatOpts): Promise<ChatResult> => {
controller.abort(new DOMException('deadline', 'TimeoutError'));
return stubChat(`[{"title":"late","atom_type":"insight","body":"b"}]`, {
input_tokens: 1_000,
output_tokens: 500,
})(opts);
};
const result = await runPhaseExtractAtoms(engine, {
_transcripts: [{ filePath: '/late.txt', content: 'a', contentHash: 'late' }],
_pages: [],
_chat: chat,
abortSignal: controller.signal,
});
expect(result.details?.deadline_aborted).toBe(true);
expect(Number(result.details?.estimated_spend_usd)).toBeGreaterThan(0);
expect(result.details?.atoms_extracted).toBe(0);
});
test('deadline after partial progress still writes receipt and incomplete rollup', async () => {
const controller = new AbortController();
let calls = 0;
let notifySecondChat!: () => void;
const secondChatStarted = new Promise<void>((resolve) => { notifySecondChat = resolve; });
const chat = async (opts: ChatOpts): Promise<ChatResult> => {
calls++;
if (calls === 1) {
return stubChat(`[{"title":"committed","atom_type":"insight","body":"b"}]`)(opts);
}
notifySecondChat();
return await new Promise<never>((_resolve, reject) => {
const signal = opts.abortSignal;
if (!signal) return reject(new Error('missing abort signal'));
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
});
};
const pending = runPhaseExtractAtoms(engine, {
_transcripts: [
{ filePath: '/committed.txt', content: 'a', contentHash: 'committed-a' },
{ filePath: '/hung.txt', content: 'b', contentHash: 'hung-b' },
],
_pages: [],
_chat: chat,
abortSignal: controller.signal,
});
await secondChatStarted;
controller.abort(new DOMException('deadline', 'TimeoutError'));
const result = await pending;
const atoms = await engine.executeRaw<{ n: number }>(
`SELECT COUNT(*)::int AS n FROM pages WHERE type = 'atom'`,
);
const receipts = await engine.executeRaw<{ n: number }>(
`SELECT COUNT(*)::int AS n FROM pages WHERE type = 'extract_receipt'`,
);
const rollups = await engine.executeRaw<{
cost_usd: string | number;
round_completed_count: string | number;
}>(
`SELECT cost_usd, round_completed_count
FROM extract_rollup_7d
WHERE kind = 'atoms' AND source_id = 'default'`,
);
expect(result.details?.deadline_aborted).toBe(true);
expect(atoms[0].n).toBe(1);
expect(receipts[0].n).toBe(1);
expect(Number(rollups[0].cost_usd)).toBeGreaterThan(0);
expect(Number(rollups[0].round_completed_count)).toBe(0);
});
test('bookkeeping runs on a fresh grace signal, not the fired work deadline', async () => {
const controller = new AbortController();
let putCalls = 0;
let receiptSignal: AbortSignal | undefined;
let rollupSignal: AbortSignal | undefined;
const signalAwareEngine = {
executeRaw: async (sql: string, _params?: unknown[], opts?: { signal?: AbortSignal }) => {
if (sql.includes('INSERT INTO extract_rollup_7d')) rollupSignal = opts?.signal;
return [];
},
putPage: async (_slug: string, _page: unknown, opts?: { signal?: AbortSignal }) => {
putCalls++;
if (putCalls === 1) {
// Atom write in flight; the work deadline fires before bookkeeping.
controller.abort(new DOMException('work deadline', 'TimeoutError'));
} else {
receiptSignal = opts?.signal;
}
return {};
},
} as unknown as BrainEngine;
const result = await runPhaseExtractAtoms(signalAwareEngine, {
_transcripts: [{ filePath: '/one.txt', content: 'a', contentHash: 'one' }],
_pages: [],
_chat: stubChat(`[{"title":"one","atom_type":"insight","body":"b"}]`),
abortSignal: controller.signal,
});
expect(result.details?.atoms_extracted).toBe(1);
expect(putCalls).toBe(2); // atom write + receipt write
expect(receiptSignal).toBeDefined();
expect(receiptSignal).not.toBe(controller.signal);
expect(receiptSignal?.aborted).toBe(false);
expect(rollupSignal).toBe(receiptSignal);
});
test('caller deadline cancels a hung atom write and stops the phase', async () => {
const controller = new AbortController();
let notifyWriteStarted!: () => void;
const writeStarted = new Promise<void>((resolve) => { notifyWriteStarted = resolve; });
let writeCalls = 0;
const signalAwareEngine = {
executeRaw: async () => [],
putPage: async (_slug: string, _page: unknown, opts?: { signal?: AbortSignal }) => {
writeCalls++;
notifyWriteStarted();
return await new Promise<never>((_resolve, reject) => {
const signal = opts?.signal;
if (!signal) return reject(new Error('missing abort signal'));
if (signal.aborted) return reject(signal.reason);
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
});
},
} as unknown as BrainEngine;
const pending = runPhaseExtractAtoms(signalAwareEngine, {
_transcripts: [{ filePath: '/hung-write.txt', content: 'a', contentHash: 'hung-write' }],
_pages: [],
_chat: stubChat(`[{"title":"hung write","atom_type":"insight","body":"b"}]`),
abortSignal: controller.signal,
});
await writeStarted;
controller.abort(new DOMException('deadline', 'TimeoutError'));
const result = await pending;
expect(writeCalls).toBe(1);
expect(result.details?.deadline_aborted).toBe(true);
expect(result.details?.atoms_extracted).toBe(0);
});
// v0.41.2.1 regression case (D9 #14 wording): with _pages:[] and same
// _transcripts, all PRE-EXISTING PhaseResult.details fields match
// pre-fix values byte-for-byte. The new fields (pages_processed,
-161
View File
@@ -1,161 +0,0 @@
/**
* #1818: embedBatch dispatches its 100-input sub-batches through a bounded
* worker pool (the embed-stale.ts concurrency pattern) instead of a serial
* `for` loop. This file pins:
*
* - output order matches input order regardless of completion order
* (index-addressed results)
* - parallelism actually happens (max in-flight > 1) and stays bounded
* (max in-flight <= configured concurrency)
* - concurrency: 1 restores the serial pre-#1818 dispatch
* - GBRAIN_EMBED_BATCH_CONCURRENCY env is honored when the option is unset
* - onBatchComplete reports a monotonic completed count ending at total
*
* Transport is stubbed via the gateway's __setEmbedTransportForTests seam
* (same pattern as test/ai/adaptive-embed-batch.test.ts). OpenAI recipe =
* fast path (no pre-split), so each embedBatch sub-batch is exactly one
* transport call.
*/
import { afterAll, afterEach, beforeEach, describe, expect, test } from 'bun:test';
import {
configureGateway,
resetGateway,
__setEmbedTransportForTests,
} from '../src/core/ai/gateway.ts';
import { embedBatch } from '../src/core/embedding.ts';
import { withEnv } from './helpers/with-env.ts';
const DIMS = 1536;
function configureOpenAI(): void {
configureGateway({
embedding_model: 'openai:text-embedding-3-large',
embedding_dimensions: DIMS,
env: { OPENAI_API_KEY: 'sk-fake' },
});
}
/**
* Install a transport whose returned embedding encodes the GLOBAL input
* index in dim 0 (texts are `t<N>`), so order can be asserted end-to-end.
* Tracks the max number of concurrently in-flight transport calls.
*/
function installTrackingTransport(delayMs = 5): { maxInFlight: () => number } {
let inFlight = 0;
let maxInFlight = 0;
__setEmbedTransportForTests((async ({ values }: { values: string[] }) => {
inFlight++;
maxInFlight = Math.max(maxInFlight, inFlight);
await new Promise(r => setTimeout(r, delayMs));
inFlight--;
return {
embeddings: values.map(v => {
const idx = Number(v.slice(1));
return Array.from({ length: DIMS }, (_, j) => (j === 0 ? idx : 0.1));
}),
};
}) as any);
return { maxInFlight: () => maxInFlight };
}
const texts = Array.from({ length: 250 }, (_, i) => `t${i}`);
afterAll(() => resetGateway());
describe('embedBatch bounded parallelism (#1818)', () => {
beforeEach(() => {
resetGateway();
configureOpenAI();
});
afterEach(() => {
__setEmbedTransportForTests(null);
});
test('default pool dispatches sub-batches in parallel, order preserved', async () => {
const tracker = installTrackingTransport();
const result = await embedBatch(texts, { onBatchComplete: () => {} });
expect(result).toHaveLength(250);
for (let i = 0; i < 250; i++) {
expect(result[i][0]).toBe(i);
}
// 250 texts → 3 sub-batches; default concurrency 4 → all 3 in flight.
expect(tracker.maxInFlight()).toBeGreaterThan(1);
expect(tracker.maxInFlight()).toBeLessThanOrEqual(4);
});
test('concurrency: 1 keeps the serial dispatch', async () => {
const tracker = installTrackingTransport();
const result = await embedBatch(texts, { concurrency: 1, onBatchComplete: () => {} });
expect(result).toHaveLength(250);
expect(tracker.maxInFlight()).toBe(1);
});
test('GBRAIN_EMBED_BATCH_CONCURRENCY env bounds the pool when option unset', async () => {
const tracker = installTrackingTransport();
await withEnv({ GBRAIN_EMBED_BATCH_CONCURRENCY: '2' }, async () => {
await embedBatch(texts, { onBatchComplete: () => {} });
});
expect(tracker.maxInFlight()).toBeGreaterThan(1);
expect(tracker.maxInFlight()).toBeLessThanOrEqual(2);
});
test('onBatchComplete reports a monotonic count ending at total', async () => {
installTrackingTransport();
const seen: number[] = [];
await embedBatch(texts, {
onBatchComplete: (done, total) => {
expect(total).toBe(250);
seen.push(done);
},
});
expect(seen).toHaveLength(3); // 100 + 100 + 50 sub-batches
for (let i = 1; i < seen.length; i++) {
expect(seen[i]).toBeGreaterThan(seen[i - 1]);
}
expect(seen[seen.length - 1]).toBe(250);
});
test('a failing sub-batch rejects the whole call', async () => {
let call = 0;
__setEmbedTransportForTests((async ({ values }: { values: string[] }) => {
call++;
if (call === 2) throw new Error('boom');
await new Promise(r => setTimeout(r, 2));
return { embeddings: values.map(() => Array.from({ length: DIMS }, () => 0.1)) };
}) as any);
await expect(embedBatch(texts, { onBatchComplete: () => {} })).rejects.toThrow();
});
test('after a failure, surviving workers stop dispatching new slices', async () => {
// 1000 texts → 10 slices, concurrency 2. First call fails immediately;
// without the `failed` flag the second worker would keep draining all
// 10 slices in the background AFTER embedBatch already rejected —
// burning provider spend and firing onBatchComplete post-rejection.
let calls = 0;
const completions: number[] = [];
__setEmbedTransportForTests((async ({ values }: { values: string[] }) => {
calls++;
if (calls === 1) throw new Error('boom');
await new Promise(r => setTimeout(r, 5));
return { embeddings: values.map(() => Array.from({ length: DIMS }, () => 0.1)) };
}) as any);
const many = Array.from({ length: 1000 }, (_, i) => `t${i}`);
await expect(
embedBatch(many, { concurrency: 2, onBatchComplete: d => completions.push(d) }),
).rejects.toThrow('boom');
const callsAtRejection = calls;
await new Promise(r => setTimeout(r, 50)); // would-be background drain window
expect(calls).toBe(callsAtRejection); // no new dispatch after rejection
expect(calls).toBeLessThanOrEqual(2); // only the in-flight sibling ran
expect(completions).toHaveLength(0); // no progress reported after failure
});
test('single small batch without callback stays on the one-call fast path', async () => {
const tracker = installTrackingTransport(1);
const result = await embedBatch(['t0', 't1', 't2']);
expect(result).toHaveLength(3);
expect(result[1][0]).toBe(1);
expect(tracker.maxInFlight()).toBe(1);
});
});
+131 -9
View File
@@ -43,26 +43,135 @@ describe('runExtractAtomsDrain (issue #1678)', () => {
expect(batches).toBe(3);
});
it('stops at the wallclock window with remaining > 0', async () => {
// SYNC stepping clock: now() #1 sets deadline (0+100=100); the while-check
// then sees 50, 50 (two batches), then 999999 → past deadline → stop.
const times = [0, 50, 50, 999_999];
let ti = 0;
const now = () => times[Math.min(ti++, times.length - 1)];
it('stops at the wallclock window; remaining is unknown (no post-window count)', async () => {
// Each batch consumes 60ms of the 100ms window: two batches fit, the
// third boundary check sees 120 ≥ 100 and stops. #2750: after the window
// elapses the final countRemaining is SKIPPED (it would overrun the
// window), so remaining reports null.
let now = 0;
const result = await runExtractAtomsDrain(
{
withLock: passThroughLock,
countRemaining: async () => 5, // never drains
runBatch: async () => ({ extracted: 1, skipped: 0 }),
now,
runBatch: async () => {
now += 60;
return { extracted: 1, skipped: 0 };
},
now: () => now,
},
{ windowMs: 100 },
);
expect(result.stopped).toBe('window');
expect(result.remaining).toBe(5);
expect(result.remaining).toBeNull();
expect(result.batches).toBe(2);
});
it('passes one drain-level deadline signal into count and batch', async () => {
const seen: AbortSignal[] = [];
const controller = new AbortController();
let now = 0;
const result = await runExtractAtomsDrain(
{
withLock: passThroughLock,
countRemaining: async (signal) => {
seen.push(signal);
return 5;
},
runBatch: async (signal) => {
seen.push(signal);
now = 100;
return { extracted: 1, skipped: 0 };
},
now: () => now,
},
{ windowMs: 100, abortSignal: controller.signal },
);
expect(result.stopped).toBe('window');
expect(result.batches).toBe(1);
expect(seen.length).toBe(2);
expect(seen[0]).toBe(seen[1]);
// Combined (timeout + external) signal, not the raw external one.
expect(seen[0]).not.toBe(controller.signal);
});
it('aborts a hung backlog count at the window deadline and releases the lock', async () => {
let released = false;
const result = await runExtractAtomsDrain(
{
withLock: async (work) => {
try { return await work(); }
finally { released = true; }
},
// Hangs until the drain's real-time deadline signal fires (10ms).
countRemaining: (signal) => new Promise((_resolve, reject) => {
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
}),
runBatch: async () => ({ extracted: 0, skipped: 0 }),
now: () => 0, // injected clock never advances — the SIGNAL must save us
},
{ windowMs: 10 },
);
expect(result.stopped).toBe('window');
expect(result.remaining).toBeNull();
expect(released).toBe(true);
});
it('rethrows external cancellation after releasing the lock', async () => {
const controller = new AbortController();
let released = false;
const pending = runExtractAtomsDrain(
{
withLock: async (work) => {
try { return await work(); }
finally { released = true; }
},
countRemaining: (signal) => new Promise((_resolve, reject) => {
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
}),
runBatch: async () => ({ extracted: 0, skipped: 0 }),
now: () => 0,
},
{ windowMs: 1_000_000, abortSignal: controller.signal },
);
controller.abort(new DOMException('worker timeout', 'AbortError'));
await expect(pending).rejects.toThrow('worker timeout');
expect(released).toBe(true);
});
it('classifies a deadline-exhausted zero-progress batch as window, not no_progress', async () => {
let now = 0;
const result = await runExtractAtomsDrain(
{
withLock: passThroughLock,
countRemaining: async () => 5,
runBatch: async () => {
now = 100; // batch consumed the whole window and returned nothing
return { extracted: 0, skipped: 0 };
},
now: () => now,
},
{ windowMs: 100 },
);
expect(result.stopped).toBe('window');
expect(result.batches).toBe(1);
});
it('bounds a hung lock acquisition with the drain deadline signal', async () => {
const started = Date.now();
await expect(runExtractAtomsDrain(
{
withLock: (_work, signal) => new Promise((_resolve, reject) => {
signal.addEventListener('abort', () => reject(signal.reason), { once: true });
}),
countRemaining: async () => 1,
runBatch: async () => ({ extracted: 0, skipped: 0 }),
now: Date.now,
},
{ windowMs: 10 },
)).rejects.toThrow();
expect(Date.now() - started).toBeLessThan(1_000);
});
it('stops on a zero-progress batch (no hot loop)', async () => {
let batches = 0;
const result = await runExtractAtomsDrain(
@@ -133,4 +242,17 @@ describe('shared wiring helper holds the cycle lock (5A)', () => {
expect(src).toContain('cycleLockIdFor(opts.sourceId)');
expect(src).toContain('withRefreshingLock(engine, lockId');
});
// #2750: the deadline signal must reach the phase, the backlog count, AND
// the lock wrapper — and the transcript path (brainDir) must stay wired
// exactly as the routine callers expect (PR #2752 takeover reverted its
// unsanctioned transcript-suppression scope change).
it('threads the drain deadline signal through phase, count, and lock', () => {
const jobsSrc = readFileSync(join(import.meta.dir, '../src/commands/jobs.ts'), 'utf8');
expect(src).toContain('abortSignal: signal');
expect(src).toContain('countExtractAtomsBacklog(engine, extractionSourceId, signal)');
expect(src).toContain('brainDir: opts.brainDir');
expect(src).not.toContain('_transcripts');
expect(jobsSrc).toContain('abortSignal: job.signal');
});
});
+9 -6
View File
@@ -24,15 +24,18 @@ import type { BrainEngine } from '../../src/core/engine.ts';
// Mock engine: healthCheck() calls engine.executeRaw; return empty rows so
// the query path exercises without needing Postgres.
//
// #1849: start() now acquires the queue-scoped DB singleton lock via
// tryAcquireDbLock, which uses the postgres `sql` tagged-template escape hatch.
// The stub returns a single row from every call so acquire succeeds (length 1
// → acquired) and refresh/release are no-ops. Each spawned runner is a fresh
// process, so there's no cross-test lock state to clean up.
// #1849: start() acquires the queue-scoped DB singleton lock via
// tryAcquireDbLock. #2750 routed the acquire upsert through engine.executeRaw
// (signal-boundable) and release through engine.executeRawDirect, so the
// stub returns a single row from the lock upsert (length 1 → acquired) and
// empty rows everywhere else. Each spawned runner is a fresh process, so
// there's no cross-test lock state to clean up.
const sqlStub = (..._args: unknown[]) => Promise.resolve([{ id: 'supervisor-lock' }]);
const mockEngine: Partial<BrainEngine> = {
kind: 'postgres' as const,
executeRaw: async () => [],
executeRaw: async (query: string) =>
query.includes('gbrain_cycle_locks') ? [{ id: 'supervisor-lock' }] : [],
executeRawDirect: async () => [],
sql: sqlStub,
} as unknown as BrainEngine;
+3 -27
View File
@@ -19,7 +19,7 @@
* overwrites this preload.
*/
import { configureGateway, getEmbeddingDimensions } from '../../src/core/ai/gateway.ts';
import { afterEach, beforeEach } from 'bun:test';
import { beforeEach } from 'bun:test';
const LEGACY_CONFIG = {
embedding_model: 'openai:text-embedding-3-large',
@@ -52,7 +52,7 @@ applyLegacy();
// 2. file-local beforeAll → may overwrite to ZE/1280
// Since beforeAll runs once per file BEFORE the first beforeEach,
// file-local beforeAll wins for that file's tests. ✓
function applyLegacyIfEmpty() {
beforeEach(() => {
try {
// Only re-apply if the gateway was reset (or never configured).
// Tests that explicitly configured a different model in their
@@ -62,28 +62,4 @@ function applyLegacyIfEmpty() {
} catch {
applyLegacy();
}
}
beforeEach(applyLegacyIfEmpty);
// PR #3130 shard-order fix: beforeEach alone leaves ONE window open — a file
// whose LAST afterEach calls resetGateway() poisons the NEXT file's
// beforeAll, which runs BEFORE any beforeEach fires. A beforeAll there that
// does engine.initSchema() then sizes the embedding column from the gateway
// DEFAULTS (zembed-1/1280d) instead of the pinned legacy 1536, and every
// 1536-d Float32Array fixture in that file dies with
// "expected 1280 dimensions, not 1536". Which file pair collides is a
// function of shard composition, so adding/removing ANY test file can
// surface it (that is exactly how it bit shard 9).
//
// Preload hooks are registered before any file-local hooks, and bun runs
// after-hooks inside-out (file-local afterEach first, then this one), so
// this repairs the empty slot immediately after the poisoning reset —
// before the next file's beforeAll can observe it.
//
// Known remaining window: a file whose afterAll() resets the gateway (no
// hook runs between its afterAll and the next file's beforeAll). Files
// that reset in afterAll and can precede a schema-creating file should
// re-apply their own config, or the victim file should configureGateway()
// explicitly in its beforeAll.
afterEach(applyLegacyIfEmpty);
});
-69
View File
@@ -1,69 +0,0 @@
/**
* #1207: `gbrain import` without `--workers` used to hardcode workerCount=1,
* so a large Postgres import paid one serial embedding round-trip per file.
* runImport now routes the default through the shared autoConcurrency policy
* (PGLite → 1, >100 files on Postgres → DEFAULT_PARALLEL_WORKERS), while an
* explicit `--workers N` still wins.
*
* The engine here is a minimal postgres-kind stub with no database_url in
* config — runImport's parallel branch then falls back to serial processing
* (its PR #490 guard) but the WORKER-COUNT DECISION (the thing #1207 fixes)
* is still observable via the "Using N parallel workers" log line. Per-file
* imports fail against the stub engine and are swallowed by runImport's
* per-file catch; that's fine — this test pins the policy, not the import.
*/
import { afterEach, beforeEach, describe, expect, test } from 'bun:test';
import { mkdtempSync, writeFileSync, mkdirSync, rmSync, realpathSync } from 'fs';
import { tmpdir } from 'os';
import { join } from 'path';
import { withEnv } from './helpers/with-env.ts';
import { runImport } from '../src/commands/import.ts';
const fakePostgresEngine = {
kind: 'postgres',
executeRaw: async () => [],
logIngest: async () => {},
setConfig: async () => {},
getConfig: async () => null,
} as any;
let workspace: string;
let brainDir: string;
let logs: string[];
const realLog = console.log;
beforeEach(() => {
workspace = mkdtempSync(join(tmpdir(), 'gbrain-import-workers-home-'));
mkdirSync(join(workspace, '.gbrain'), { recursive: true });
brainDir = realpathSync(mkdtempSync(join(tmpdir(), 'gbrain-import-workers-brain-')));
// 101 files: one past AUTO_CONCURRENCY_FILE_THRESHOLD (100).
for (let i = 0; i < 101; i++) {
writeFileSync(join(brainDir, `page-${i}.md`), `# Page ${i}\n\nbody ${i}\n`);
}
logs = [];
console.log = (msg?: unknown) => logs.push(String(msg));
});
afterEach(() => {
console.log = realLog;
rmSync(workspace, { recursive: true, force: true });
rmSync(brainDir, { recursive: true, force: true });
});
describe('import default worker count (#1207)', () => {
test('no --workers flag → autoConcurrency picks 4 for >100 files on Postgres', async () => {
await withEnv({ GBRAIN_HOME: join(workspace, '.gbrain'), GBRAIN_SOURCE: undefined }, async () => {
await runImport(fakePostgresEngine, [brainDir, '--no-embed'], { sourceId: 'default' });
});
expect(logs.some(l => l.includes('Using 4 parallel workers'))).toBe(true);
});
test('explicit --workers 2 still wins over the auto policy', async () => {
await withEnv({ GBRAIN_HOME: join(workspace, '.gbrain'), GBRAIN_SOURCE: undefined }, async () => {
await runImport(fakePostgresEngine, [brainDir, '--no-embed', '--workers', '2'], { sourceId: 'default' });
});
expect(logs.some(l => l.includes('Using 2 parallel workers'))).toBe(true);
expect(logs.some(l => l.includes('Using 4 parallel workers'))).toBe(false);
});
});
+26 -1
View File
@@ -98,14 +98,39 @@ describe('PostgresEngine.executeRawDirect — routing decision (PR #1816)', () =
});
test('already-aborted signal short-circuits with AbortError before routing the query', async () => {
const readConn = fakeSql('read');
let unsafeCalls = 0;
let ddlCalls = 0;
const readConn: FakeSql = { unsafe: async () => { unsafeCalls++; return []; } };
const directConn = fakeSql('direct');
const engine = makeEngine({ dualPoolActive: true, readConn, directConn });
const e = engine as unknown as { connectionManager: { ddl: () => Promise<FakeSql> } };
e.connectionManager.ddl = async () => { ddlCalls++; return directConn; };
const ac = new AbortController();
ac.abort();
await expect(
engine.executeRawDirect('UPDATE minion_jobs SET x=1', [], { signal: ac.signal }),
).rejects.toThrow(/abort/i);
// #2750: short-circuits BEFORE pool routing — no ddl(), no unsafe().
expect(ddlCalls).toBe(0);
expect(unsafeCalls).toBe(0);
});
test('#2750: signal bounds a stalled direct-pool acquisition before unsafe starts', async () => {
let unsafeCalls = 0;
const readConn: FakeSql = { unsafe: async () => { unsafeCalls++; return []; } };
const directConn = fakeSql('direct');
const engine = makeEngine({ dualPoolActive: true, readConn, directConn });
const e = engine as unknown as { connectionManager: { ddl: () => Promise<FakeSql> } };
e.connectionManager.ddl = () => new Promise<FakeSql>(() => {}); // pooler exhausted: never resolves
const started = Date.now();
await expect(engine.executeRawDirect(
'DELETE FROM gbrain_cycle_locks',
[],
{ signal: AbortSignal.timeout(10) },
)).rejects.toThrow(/abort/i);
expect(Date.now() - started).toBeLessThan(1_000);
expect(unsafeCalls).toBe(0);
});
});