Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 3fa01a4538 fix(cycle): ambiguous local_path resolves no source — don't stamp an arbitrary pick
resolveSourceForDir used LIMIT 1, so when two sources share a local_path
the dir-keyed freshness stamp (new in this PR) fired for a
nondeterministic one — exactly the 'freshness stamp that lies' the
cycleSourceId comment warns against, and the CI failure in
test/dream.test.ts ('gbrain dream (no --source) leaves all sources
untouched'). Require exactly one match; ambiguous or no match falls back
to the pre-v0.18 behavior (undefined), same as before this PR for the
no-match case.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 11:06:41 -07:00
ded4aeaeae fix(cycle): stamp last_full_cycle_at for the resolved source; stale freshness forces autopilot fanout (#1993, #2060)
Two fixes to the per-source cycle freshness loop:

1. runCycle's freshness stamp now keys off cycleSourceId (opts.sourceId ??
   the source resolved from brainDir) — the same id the cycle locked and
   scoped its phases to — instead of raw opts.sourceId. The autopilot's
   inline cycle passes brainDir with no explicit sourceId, so the stamp
   never fired and cycle_freshness stayed stale forever. Keeps master's
   !aborted guard and the last_source_cycle_at write. (takeover of #1993)

2. autopilot's dispatch decision now consults per-source cycle staleness:
   countStaleSources (new pure helper in autopilot-fanout.ts) over
   listAllSources({ localPathOnly: true }). A stale source forces the
   fanout path and blocks the healthy-sleep gate, so a brain sitting at
   score 70-94 with a small targeted plan can no longer starve per-source
   cycle dispatch indefinitely. Fail-open to 0 on read errors;
   dispatchPerSource's existing throttles (skipped_fresh / fanoutMax /
   failure cooldown) bound the work. (#2060)

Tests: cycle-last-full-cycle-at gains the brainDir-resolves-source and
brainDir-matches-nothing cases (first one fails without fix 1);
autopilot-fanout unit tests cover countStaleSources; the fanout wiring
guard pins the staleCycleSources terms in shouldFullCycle/shouldSleep.

Co-authored-by: 100menotu001 <100menotu001@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:24:01 -07:00
19 changed files with 156 additions and 521 deletions
+12
View File
@@ -186,6 +186,18 @@ export function isSourceStale(src: SourceRow, now = Date.now(), floorMin = FULL_
return ageMin >= floorMin;
}
/**
* #2060: count sources past the per-source cycle freshness floor. Consumed
* by autopilot's dispatch decision — a stale source forces the fanout path
* even when the doctor plan is small (score 7094, plan ≤ 3, est < 300s),
* so targeted mode can't leave cycle_freshness stale indefinitely.
* dispatchPerSource's own throttles (skipped_fresh / fanoutMax / failure
* cooldown) bound the resulting work.
*/
export function countStaleSources(sources: SourceRow[], now = Date.now(), floorMin = FULL_CYCLE_FLOOR_MIN): number {
return sources.filter((s) => isSourceStale(s, now, floorMin)).length;
}
/**
* Most recent SUCCESSFUL cycle for a source. Prefers `last_source_cycle_at`
* (per-source phases, written by the split cycle) and falls back to the legacy
+16 -2
View File
@@ -901,13 +901,27 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
const FULL_CYCLE_FLOOR_MIN = 60;
const minutesSinceLastFull = (Date.now() - lastFullCycleAt) / 60000;
// #2060: stale per-source cycle freshness is a dispatch input. Without
// it, a brain sitting at score 7094 with a small targeted plan (≤3
// steps, <300s) stays in targeted mode indefinitely and no per-source
// cycle is ever dispatched — cycle_freshness never advances. A stale
// source forces the fanout path; dispatchPerSource's throttles
// (skipped_fresh / fanoutMax / failure cooldown) bound the work.
// Fail-open to 0: a read failure must not block dispatch.
let staleCycleSources = 0;
try {
const { countStaleSources } = await import('./autopilot-fanout.ts');
staleCycleSources = countStaleSources(await engine.listAllSources({ localPathOnly: true }));
} catch { /* fail-open: freshness is a dispatch hint, not a gate */ }
const shouldFullCycle =
(score >= 95 && plan.length === 0 && minutesSinceLastFull >= FULL_CYCLE_FLOOR_MIN) ||
plan.length > 3 ||
estTotal >= 300 ||
score < 70;
score < 70 ||
staleCycleSources > 0;
const shouldSleep = score >= 95 && plan.length === 0 && minutesSinceLastFull < FULL_CYCLE_FLOOR_MIN;
const shouldSleep = score >= 95 && plan.length === 0 && minutesSinceLastFull < FULL_CYCLE_FLOOR_MIN && staleCycleSources === 0;
if (shouldSleep) {
if (jsonMode) {
+5 -10
View File
@@ -170,14 +170,10 @@ export async function runImport(
// v0.22.13 (PR #490 Q2): shared parseWorkers helper rejects bad input
// (--workers 0, -3, "foo") with a loud error instead of silently falling
// through to 1. Mirrors sync.ts's flag handling.
const { parseWorkers, autoConcurrency } = await import('../core/sync-concurrency.ts');
// #1207: undefined (no --workers flag) defers to autoConcurrency below —
// the shared sync/import policy (PGLite → 1, >100 files → 4) — instead of
// hardcoding serial. Large Postgres imports stop paying one embedding
// round-trip per file in sequence.
let workerCount: number | undefined;
const { parseWorkers } = await import('../core/sync-concurrency.ts');
let workerCount: number;
try {
workerCount = parseWorkers(workersArg ?? undefined);
workerCount = parseWorkers(workersArg ?? undefined) ?? 1;
} catch (e) {
console.error(e instanceof Error ? e.message : String(e));
process.exit(1);
@@ -256,9 +252,8 @@ export async function runImport(
}
const files = resumeFilter(allFiles, dir, completed);
// Determine actual worker count. Explicit --workers wins; otherwise the
// shared autoConcurrency policy decides from engine kind + file count.
const actualWorkers = autoConcurrency(engine, files.length, workerCount);
// Determine actual worker count
const actualWorkers = workerCount > 1 ? workerCount : 1;
if (actualWorkers > 1) {
console.log(`Using ${actualWorkers} parallel workers`);
}
+6 -24
View File
@@ -1513,21 +1513,12 @@ export async function embed(texts: string[], opts?: EmbedOpts): Promise<Float32A
const embedding = recipe.touchpoints?.embedding;
const maxBatchTokens = embedding?.max_batch_tokens;
const maxBatchCount = embedding?.max_batch_count;
const charsPerToken = embedding?.chars_per_token ?? DEFAULT_CHARS_PER_TOKEN;
// Pre-split is gated on max_batch_tokens / max_batch_count. Recipes with
// neither (e.g. OpenAI) ride the fast path: one embedMany call, no
// recursion safety net.
const batches = (maxBatchTokens || maxBatchCount)
? splitByTokenBudget(
truncated,
maxBatchTokens
? Math.floor(maxBatchTokens * effectiveSafetyFactor(recipe))
: Number.MAX_SAFE_INTEGER,
charsPerToken,
maxBatchCount,
)
// Pre-split is gated on max_batch_tokens. Recipes without it (e.g. OpenAI)
// ride the fast path: one embedMany call, no recursion safety net.
const batches = maxBatchTokens
? splitByTokenBudget(truncated, Math.floor(maxBatchTokens * effectiveSafetyFactor(recipe)), charsPerToken)
: [truncated];
const allEmbeddings: Float32Array[] = [];
@@ -1577,9 +1568,6 @@ export async function embed(texts: string[], opts?: EmbedOpts): Promise<Float32A
* responsible for applying any safety-factor shrink before passing in.
* @param charsPerToken - Provider-specific character density. Defaults to
* `DEFAULT_CHARS_PER_TOKEN` (4) when omitted, matching OpenAI tiktoken.
* @param maxBatchCount - #1199: optional cap on INPUTS per sub-batch, for
* providers that reject batches by count (DashScope: 10). When omitted,
* only the token budget governs.
*
* @internal exported for tests; not part of the public gateway API.
*/
@@ -1587,17 +1575,15 @@ export function splitByTokenBudget(
texts: string[],
budgetTokens: number,
charsPerToken: number = DEFAULT_CHARS_PER_TOKEN,
maxBatchCount?: number,
): string[][] {
const ratio = charsPerToken > 0 ? charsPerToken : DEFAULT_CHARS_PER_TOKEN;
const maxCount = maxBatchCount !== undefined && maxBatchCount > 0 ? maxBatchCount : Infinity;
const batches: string[][] = [];
let current: string[] = [];
let currentTokens = 0;
for (const text of texts) {
const estTokens = Math.ceil(text.length / ratio);
if (current.length > 0 && (currentTokens + estTokens > budgetTokens || current.length >= maxCount)) {
if (current.length > 0 && currentTokens + estTokens > budgetTokens) {
batches.push(current);
current = [];
currentTokens = 0;
@@ -1623,11 +1609,7 @@ export function isTokenLimitError(err: unknown): boolean {
/token.*limit.*exceeded/i.test(msg) ||
// OpenAI embeddings: "Invalid 'input': maximum request size is 300000 tokens per request."
/maximum request size.*tokens/i.test(msg) ||
/max.*tokens.*per.*request/i.test(msg) ||
// DashScope: "batch size is invalid, it should not be larger than 10." (#1199)
// Count-cap error, but recursive halving shrinks count too, so the same
// safety net converges.
/batch size is invalid/i.test(msg)
/max.*tokens.*per.*request/i.test(msg)
);
}
-4
View File
@@ -31,10 +31,6 @@ export const dashscope: Recipe = {
// path. Conservative declaration so the gateway pre-splits before
// hitting whatever undocumented server-side limit exists.
max_batch_tokens: 8192,
// #1199: DashScope hard-caps embeddings at 10 inputs per request
// ("batch size is invalid, it should not be larger than 10"). The
// token budget alone admits far more than 10 short chunks per batch.
max_batch_count: 10,
// text-embedding-v3 mixes English + CJK heavily; the tokenizer is
// closer to Voyage density than OpenAI tiktoken for CJK-dominant
// content. Conservative chars_per_token=2 leaves headroom.
-9
View File
@@ -16,15 +16,6 @@ export const google: Recipe = {
dims_options: [768, 1536, 3072],
cost_per_1m_tokens_usd: 0.15,
price_last_verified: '2026-04-20',
// #970: Gemini's documented limits are per-INPUT (2048 tokens,
// silently truncated beyond) and per-REQUEST count (batchEmbedContents
// caps at 100 inputs). There is no separate per-request token cap, so
// the token budget is derived: 100 inputs × 2048 tokens. The count cap
// binds first for typical chunk sizes. Do NOT copy the 2048 per-input
// limit into max_batch_tokens — that would over-split 50×.
max_batch_tokens: 204_800,
chars_per_token: 4,
max_batch_count: 100,
},
expansion: {
models: ['gemini-2.0-flash', 'gemini-2.0-flash-lite'],
+1 -4
View File
@@ -58,8 +58,5 @@ export function getRecipe(id: string): Recipe | undefined {
}
export function listRecipes(): Recipe[] {
// Read the map (not ALL) so there is one source of truth — getRecipe,
// model-resolver, and listRecipes all see the same registry, and tests
// can inject a synthetic recipe via RECIPES to exercise registry walks.
return [...RECIPES.values()];
return [...ALL];
}
-10
View File
@@ -46,16 +46,6 @@ export interface EmbeddingTouchpoint {
* Only consulted when `max_batch_tokens` is also set.
*/
chars_per_token?: number;
/**
* #1199: maximum number of INPUTS per embedding request, for providers
* that hard-cap batch size by count rather than (or in addition to)
* tokens — DashScope text-embedding-v3 rejects batches > 10 with
* `InvalidParameter`, Gemini batchEmbedContents caps at 100 requests.
* When set, the gateway's pre-split flushes a sub-batch at this count
* even if the token budget still has room. Independent of
* `max_batch_tokens`; either alone triggers the pre-split.
*/
max_batch_count?: number;
/**
* Budget-utilization ceiling in (0, 1]. The gateway pre-splits at
* `safety_factor × max_batch_tokens` to leave headroom for tokenizer
+19 -10
View File
@@ -854,7 +854,10 @@ interface SyncPhaseResult extends PhaseResult {
/**
* Resolve the source id for a brain directory by looking up the sources
* table. Returns undefined when no registered source matches (falls back
* to pre-v0.18 global config.sync.* keys).
* to pre-v0.18 global config.sync.* keys) OR when MORE than one source
* claims the path — an ambiguous match must not scope phases or stamp
* last_full_cycle_at for an arbitrarily-picked source (the "freshness
* stamp that lies" this resolution exists to prevent).
*/
async function resolveSourceForDir(
engine: BrainEngine,
@@ -865,10 +868,10 @@ async function resolveSourceForDir(
if (brainDir === null) return undefined;
try {
const rows = await engine.executeRaw<{ id: string }>(
`SELECT id FROM sources WHERE local_path = $1 LIMIT 1`,
`SELECT id FROM sources WHERE local_path = $1 LIMIT 2`,
[brainDir],
);
return rows[0]?.id;
return rows.length === 1 ? rows[0]!.id : undefined;
} catch {
// sources table might not exist on very old brains — fall through.
return undefined;
@@ -2365,17 +2368,23 @@ export async function runCycle(
}
// v0.38 (codex r1 P0-5): persist per-source cycle completion timestamp
// when the cycle ran successfully against an explicit source. Read by
// autopilot's per-source freshness gate next tick. Skipped when:
// - opts.sourceId is unset (legacy callers — autopilot still here)
// - engine is null (no-DB path)
// when the cycle ran successfully against a resolvable source. Read by
// autopilot's per-source freshness gate next tick.
//
// #1993: keyed off `cycleSourceId` (opts.sourceId ?? the source resolved
// from brainDir) — the SAME id the cycle locked + scoped its phases to —
// NOT raw opts.sourceId. The autopilot's inline cycle sets brainDir but
// passes no explicit sourceId, so keying off opts.sourceId alone never
// advanced last_full_cycle_at and cycle_freshness stayed stale even while
// the autopilot cycled every interval. Skipped when:
// - no source resolves (engine null, or no checkout AND no opts.sourceId)
// - status is 'failed' or 'skipped' (don't mark a non-run as fresh)
// - dryRun (writes are out of scope)
//
// Best-effort: a write failure does NOT change the CycleReport status.
// The cost of writing the wrong timestamp post-failure is higher than
// the cost of missing a successful write (next cycle will redo work).
if (opts.sourceId && engine && !dryRun && !aborted && (status === 'ok' || status === 'clean' || status === 'partial')) {
if (cycleSourceId && engine && !dryRun && !aborted && (status === 'ok' || status === 'clean' || status === 'partial')) {
try {
const nowIso = new Date().toISOString();
// #2194 fix #3 (the cycle split): `last_source_cycle_at` is the NEW gate
@@ -2385,13 +2394,13 @@ export async function runCycle(
// phases (those gate on autopilot.last_global_at), so writing it on a
// source-only cycle does not re-introduce the freshness poisoning codex
// flagged in the rejected skip-based design.
await engine.updateSourceConfig(opts.sourceId, {
await engine.updateSourceConfig(cycleSourceId, {
last_source_cycle_at: nowIso,
last_full_cycle_at: nowIso,
});
} catch (e) {
// Best-effort; cycle already succeeded by the time we get here.
console.warn(`[cycle] failed to write last_source_cycle_at for source ${opts.sourceId}: ${e instanceof Error ? e.message : String(e)}`);
console.warn(`[cycle] failed to write last_source_cycle_at for source ${cycleSourceId}: ${e instanceof Error ? e.message : String(e)}`);
}
}
+6 -56
View File
@@ -79,34 +79,15 @@ export interface EmbedBatchOptions {
* and amplify rate-limit pressure.
*/
maxRetries?: number;
/**
* #1818: bounded parallelism across BATCH_SIZE sub-batches. Defaults to
* `GBRAIN_EMBED_BATCH_CONCURRENCY` env, else 4. Results are
* index-addressed so output order always matches input order. Set 1 to
* force the pre-v0.42 serial dispatch.
*/
concurrency?: number;
}
/**
* Embed a batch of texts via the gateway. Sub-batches of 100 so upstream
* progress callbacks fire incrementally on large imports. The gateway owns
* adaptive batch splitting and per-recipe token-budget logic; this paginator
* owns progress-callback granularity and (#1818) bounded parallel dispatch
* of the sub-batches — the embed-stale.ts worker-pool pattern, scoped down.
* is purely about progress-callback granularity.
*/
const BATCH_SIZE = 100;
const DEFAULT_EMBED_BATCH_CONCURRENCY = 4;
function resolveEmbedBatchConcurrency(options: EmbedBatchOptions): number {
if (options.concurrency !== undefined) {
return Math.max(1, Math.floor(options.concurrency));
}
const env = Number(process.env.GBRAIN_EMBED_BATCH_CONCURRENCY);
if (Number.isFinite(env) && env >= 1) return Math.floor(env);
return DEFAULT_EMBED_BATCH_CONCURRENCY;
}
export async function embedBatch(
texts: string[],
options: EmbedBatchOptions = {},
@@ -122,44 +103,13 @@ export async function embedBatch(
if (texts.length <= BATCH_SIZE && !options.onBatchComplete) {
return gatewayEmbed(texts, gwOpts);
}
// #1818: dispatch sub-batches through a bounded worker pool instead of a
// serial loop. Results are written into a preallocated index-addressed
// array so output order matches input order regardless of completion
// order; onBatchComplete reports a monotonic completed-embedding count.
const slices: Array<{ start: number; texts: string[] }> = [];
const results: Float32Array[] = [];
for (let i = 0; i < texts.length; i += BATCH_SIZE) {
slices.push({ start: i, texts: texts.slice(i, i + BATCH_SIZE) });
const slice = texts.slice(i, i + BATCH_SIZE);
const out = await gatewayEmbed(slice, gwOpts);
results.push(...out);
options.onBatchComplete?.(results.length, texts.length);
}
const results = new Array<Float32Array>(texts.length);
let next = 0;
let done = 0;
const numWorkers = Math.min(resolveEmbedBatchConcurrency(options), slices.length);
// Once any sub-batch fails, `failed` stops the surviving workers from
// dispatching FURTHER slices — the whole call is rejecting anyway, so
// continuing would burn real provider spend in the background and fire
// onBatchComplete after the caller already saw the failure (worst with
// embedBatchWithBackoff, whose 429 backoff assumes nothing is in flight).
// In-flight sibling calls still run to completion (bounded by numWorkers-1).
let failed = false;
const worker = async (): Promise<void> => {
while (!failed && next < slices.length) {
// NOTE: no local aborted-check here — an aborted signal makes the next
// gatewayEmbed call throw (SDK-side), which rejects the pool. Returning
// silently instead would resolve with holes in `results`.
const slice = slices[next++];
let out: Float32Array[];
try {
out = await gatewayEmbed(slice.texts, gwOpts);
} catch (err) {
failed = true;
throw err;
}
for (let j = 0; j < out.length; j++) results[slice.start + j] = out[j];
done += out.length;
if (!failed) options.onBatchComplete?.(done, texts.length);
}
};
await Promise.all(Array.from({ length: numWorkers }, () => worker()));
return results;
}
+5 -109
View File
@@ -39,8 +39,6 @@ import {
__getShrinkStateForTests,
} from '../../src/core/ai/gateway.ts';
import { AIConfigError, AITransientError } from '../../src/core/ai/errors.ts';
import { RECIPES } from '../../src/core/ai/recipes/index.ts';
import type { Recipe } from '../../src/core/ai/types.ts';
// The last test in this file leaves the gateway configured with a remote
// provider + fake key and a REAL embed transport. Without a final reset,
@@ -95,14 +93,6 @@ function configureGoogle(): void {
});
}
function configureDashscope(): void {
configureGateway({
embedding_model: 'dashscope:text-embedding-v3',
embedding_dimensions: 1024,
env: { DASHSCOPE_API_KEY: 'sk-fake' },
});
}
// --------- 1. Pure helpers ---------
describe('splitByTokenBudget (pure helper)', () => {
@@ -159,27 +149,6 @@ describe('splitByTokenBudget (pure helper)', () => {
expect(splitByTokenBudget(texts, 96_000, 0)).toEqual(splitByTokenBudget(texts, 96_000, 4));
expect(splitByTokenBudget(texts, 96_000, -1)).toEqual(splitByTokenBudget(texts, 96_000, 4));
});
// #1199: count cap for providers that reject batches by input count.
test('max_batch_count flushes even when token budget has room', () => {
const texts = Array.from({ length: 25 }, (_, i) => `t${i}`);
const result = splitByTokenBudget(texts, 1_000_000, 4, 10);
expect(result.map(b => b.length)).toEqual([10, 10, 5]);
expect(result.flat()).toEqual(texts);
});
test('token budget still governs alongside max_batch_count', () => {
const texts = ['a'.repeat(50_000), 'b'.repeat(50_000), 'c'.repeat(50_000)];
const result = splitByTokenBudget(texts, 96_000, 1, 10);
expect(result).toHaveLength(3);
});
test('undefined / zero / negative max_batch_count is ignored', () => {
const texts = Array.from({ length: 25 }, () => 'x');
expect(splitByTokenBudget(texts, 1_000_000, 4, undefined)).toHaveLength(1);
expect(splitByTokenBudget(texts, 1_000_000, 4, 0)).toHaveLength(1);
expect(splitByTokenBudget(texts, 1_000_000, 4, -5)).toHaveLength(1);
});
});
describe('isTokenLimitError (pure helper)', () => {
@@ -210,12 +179,6 @@ describe('isTokenLimitError (pure helper)', () => {
expect(isTokenLimitError(new Error('Exceeded 300000 max tokens per request'))).toBe(true);
});
test('matches DashScope batch-count error (#1199)', () => {
expect(isTokenLimitError(new Error(
'InvalidParameter: batch size is invalid, it should not be larger than 10.',
))).toBe(true);
});
test('does not match unrelated errors', () => {
expect(isTokenLimitError(new Error('Connection refused'))).toBe(false);
expect(isTokenLimitError(new Error('Invalid API key'))).toBe(false);
@@ -424,92 +387,26 @@ describe('shrink-on-miss adaptive cache', () => {
});
});
// --------- 8. Pre-split count cap through public embed() (#1199 / #970) ---------
describe('embed() pre-split honors max_batch_count', () => {
beforeEach(() => resetGateway());
afterEach(() => __setEmbedTransportForTests(null));
test('dashscope never dispatches more than 10 inputs per call (#1199)', async () => {
configureDashscope();
const stub = mock(async ({ values }: { values: string[] }) => fakeEmbeddings(values, 1024));
__setEmbedTransportForTests(stub as any);
// 25 short texts fit trivially in the 8192-token budget; without the
// count cap they'd ship as ONE batch and DashScope would reject it.
const texts = Array.from({ length: 25 }, (_, i) => `short-${i}`);
const result = await embed(texts);
expect(result).toHaveLength(25);
const callLengths = stub.mock.calls.map(([arg]) => (arg as { values: string[] }).values.length);
expect(Math.max(...callLengths)).toBeLessThanOrEqual(10);
expect(callLengths.reduce((a, b) => a + b, 0)).toBe(25);
// Order preserved across sub-batches.
expect((stub.mock.calls[0][0] as { values: string[] }).values[0]).toBe('short-0');
});
test('google pre-splits at 100 inputs per batchEmbedContents call (#970)', async () => {
configureGoogle();
const stub = mock(async ({ values }: { values: string[] }) => fakeEmbeddings(values, 768));
__setEmbedTransportForTests(stub as any);
const texts = Array.from({ length: 250 }, (_, i) => `g${i}`);
const result = await embed(texts);
expect(result).toHaveLength(250);
const callLengths = stub.mock.calls.map(([arg]) => (arg as { values: string[] }).values.length);
expect(callLengths).toEqual([100, 100, 50]);
});
});
// --------- 7. Startup warning (D9-B) ---------
describe('startup warning for recipes missing max_batch_tokens', () => {
beforeEach(() => resetGateway());
// #970 closed google's missing cap, so no registered recipe is capless
// anymore. Inject a synthetic capless recipe to keep the warning path
// covered for the NEXT recipe that forgets the field.
const caplessRecipe: Recipe = {
id: 'capless-test',
name: 'Capless Test Provider',
tier: 'openai-compat',
implementation: 'openai-compatible',
base_url_default: 'https://example.invalid/v1',
auth_env: { required: [] },
touchpoints: {
embedding: { models: ['capless-embed-1'], default_dims: 768 },
},
};
function configureCapless(): void {
configureGateway({
embedding_model: 'capless-test:capless-embed-1',
embedding_dimensions: 768,
env: {},
});
}
test('configured missing-cap recipe warns once; unrelated recipes stay quiet', () => {
const warnings: string[] = [];
const original = console.warn;
console.warn = (msg: string) => warnings.push(String(msg));
RECIPES.set(caplessRecipe.id, caplessRecipe);
try {
configureOpenAI();
expect(warnings.length).toBe(0);
// #970 regression: google now declares max_batch_tokens → quiet.
configureGoogle();
expect(warnings.length).toBe(0);
configureCapless();
const firstCallCount = warnings.length;
// Reconfigure: the warning should NOT re-fire for the same recipes
// within one process (we already told the operator).
configureCapless();
configureGoogle();
expect(warnings.length).toBe(firstCallCount);
} finally {
console.warn = original;
RECIPES.delete(caplessRecipe.id);
}
// The warning text should match the documented contract.
@@ -518,12 +415,11 @@ describe('startup warning for recipes missing max_batch_tokens', () => {
);
expect(contractMatch.length).toBe(1);
// Voyage + google declare max_batch_tokens → suppressed. OpenAI is the
// canonical fast-path recipe → also suppressed by id. All must be
// absent from the warnings; only the synthetic capless recipe fires.
// Voyage declares max_batch_tokens → suppressed. OpenAI is the
// canonical fast-path recipe → also suppressed by id. Both must be
// absent from the warnings.
expect(warnings.find(w => w.includes('"voyage"'))).toBeUndefined();
expect(warnings.find(w => w.includes('"openai"'))).toBeUndefined();
expect(warnings.find(w => w.includes('"google"'))).toBeUndefined();
expect(warnings.find(w => w.includes('"capless-test"'))).toBeDefined();
expect(warnings.find(w => w.includes('"google"'))).toBeDefined();
});
});
+13 -13
View File
@@ -52,7 +52,16 @@ describe('v0.32 #779: no_batch_cap suppresses the missing-max_batch_tokens warni
}
});
test('configureGateway does NOT warn for google now that it declares batch caps (#970)', () => {
test('configureGateway warns for google only when google embedding is configured', () => {
warnSpy.mockClear();
resetGateway();
configureGateway({ env: {} });
let messages = warnSpy.mock.calls.map(c => String(c[0] ?? ''));
expect(
messages.some(m => m.includes('"google"') && m.includes('without max_batch_tokens')),
'google should not warn while OpenAI default is configured',
).toBe(false);
warnSpy.mockClear();
resetGateway();
configureGateway({
@@ -60,20 +69,11 @@ describe('v0.32 #779: no_batch_cap suppresses the missing-max_batch_tokens warni
embedding_dimensions: 768,
env: { GOOGLE_GENERATIVE_AI_API_KEY: 'fake' },
});
const messages = warnSpy.mock.calls.map(c => String(c[0] ?? ''));
messages = warnSpy.mock.calls.map(c => String(c[0] ?? ''));
expect(
messages.some(m => m.includes('"google"') && m.includes('without max_batch_tokens')),
'google declares max_batch_tokens/max_batch_count since #970 — no warning',
).toBe(false);
});
test('google recipe declares its derived batch caps (#970)', () => {
const e = getRecipe('google')!.touchpoints.embedding!;
// Count cap is the REAL Gemini limit (batchEmbedContents: 100 inputs);
// the token budget is derived (100 × 2048 per-input tokens), NOT the
// 2048 per-input limit — copying that verbatim would over-split 50×.
expect(e.max_batch_count).toBe(100);
expect(e.max_batch_tokens).toBe(204_800);
'google should warn when configured because it has fixed-cap models',
).toBe(true);
});
test('every recipe with empty models[] declares user_provided_models OR has openai-fast-path', () => {
-5
View File
@@ -55,11 +55,6 @@ describe('recipe: dashscope', () => {
expect(r.touchpoints.embedding!.chars_per_token).toBeGreaterThan(0);
});
test('declares max_batch_count: 10 — DashScope rejects larger batches (#1199)', () => {
const r = getRecipe('dashscope')!;
expect(r.touchpoints.embedding!.max_batch_count).toBe(10);
});
test('dimsProviderOptions threads dimensions for text-embedding-v3 (Matryoshka)', async () => {
// Codex finding #1: DashScope text-embedding-v3 is Matryoshka 64-1024.
// Without `dimensions` on the wire, user-selected non-default dims are
+14
View File
@@ -54,6 +54,20 @@ describe('autopilot.ts ↔ dispatchPerSource wiring', () => {
expect(AUTOPILOT_SRC).toMatch(/lastFullCycleAt\s*=\s*Date\.now\(\)/);
});
test('stale per-source cycle freshness is a shouldFullCycle input (#2060)', () => {
// Targeted mode (score 7094, plan ≤3, est <300s) must not be able to
// starve per-source cycle dispatch: a stale source (per countStaleSources
// over listAllSources) forces the fanout path, and the sleep gate must
// not fire while stale sources exist. Without these terms, cycle
// freshness never advances for a brain that always lands in targeted mode.
expect(AUTOPILOT_SRC).toMatch(/countStaleSources/);
const fullCycleDeclIdx = AUTOPILOT_SRC.indexOf('const shouldFullCycle');
expect(fullCycleDeclIdx).toBeGreaterThan(-1);
const decl = AUTOPILOT_SRC.slice(fullCycleDeclIdx, fullCycleDeclIdx + 700);
expect(decl).toMatch(/staleCycleSources\s*>\s*0/);
expect(decl).toMatch(/const shouldSleep[^;]*staleCycleSources\s*===\s*0/);
});
test('does NOT regress to the single-job dispatch on the full-cycle path', () => {
// Pre-PR: the shouldFullCycle branch did:
// const job = await queue.add('autopilot-cycle', { repoPath }, {
+18
View File
@@ -14,6 +14,7 @@ import { describe, test, expect } from 'bun:test';
import {
readLastFullCycleAt,
isSourceStale,
countStaleSources,
selectSourcesForDispatch,
resolveFanoutMax,
dispatchPerSource,
@@ -74,6 +75,23 @@ describe('isSourceStale', () => {
});
});
describe('countStaleSources (#2060 dispatch-decision input)', () => {
const NOW = Date.parse('2026-05-22T12:00:00.000Z');
test('counts never-cycled + past-floor sources, ignores fresh', () => {
const sources = [
src('never-cycled'), // stale (null)
src('old', new Date(NOW - 2 * 60 * 60_000).toISOString()), // stale (2h)
src('fresh', new Date(NOW - 30 * 60_000).toISOString()), // fresh (30min)
];
expect(countStaleSources(sources, NOW)).toBe(2);
});
test('returns 0 for all-fresh and for empty list', () => {
const fresh = src('a', new Date(NOW - 10 * 60_000).toISOString());
expect(countStaleSources([fresh], NOW)).toBe(0);
expect(countStaleSources([], NOW)).toBe(0);
});
});
describe('selectSourcesForDispatch', () => {
const NOW = Date.parse('2026-05-22T12:00:00.000Z');
const fresh = (id: string, agoMin: number) =>
+38 -8
View File
@@ -3,8 +3,10 @@
* cycles. Closes codex round-1 P0-5 (write site for last_full_cycle_at
* was unspecified pre-PR).
*
* Conditions for write:
* - opts.sourceId is set (legacy callers without sourceId skip the write)
* Conditions for write (keyed off `cycleSourceId` = opts.sourceId ?? the
* source resolved from brainDir, so the autopilot's inline cycle — brainDir
* set, no explicit sourceId — also advances the timestamp, #1993):
* - a source resolves (explicit sourceId, or brainDir matches a source)
* - engine is non-null (no-DB path skips)
* - status is 'ok' | 'clean' | 'partial' (failed/skipped don't mark fresh)
* - dryRun is false
@@ -90,17 +92,45 @@ describe('runCycle last_full_cycle_at exit hook', () => {
});
});
test('legacy caller (no sourceId) does NOT write any source timestamp', async () => {
test('no explicit sourceId but brainDir resolves a source → writes the resolved source timestamp', async () => {
await withEnv({ GBRAIN_HOME: gbrainHome }, async () => {
await seedSource('default-like');
// No sourceId passed; should remain untouched.
// The autopilot's inline cycle sets brainDir but passes no sourceId.
// runCycle resolves the source from brainDir (local_path match) into
// cycleSourceId and stamps last_full_cycle_at for it — otherwise
// cycle_freshness reports the brain stale even while the autopilot
// cycles every interval (#1993).
await seedSource('resolved-from-dir'); // local_path = brainDir
expect(await readLastFullCycleAt('resolved-from-dir')).toBeNull();
const t0 = Date.now();
const report = await runCycle(engine, {
brainDir,
phases: ['lint'],
});
expect(['ok', 'clean']).toContain(report.status);
const after = await readLastFullCycleAt('resolved-from-dir');
expect(after).not.toBeNull();
expect(new Date(after!).getTime()).toBeGreaterThanOrEqual(t0);
});
});
test('no sourceId and brainDir matches no source → does not write', async () => {
await withEnv({ GBRAIN_HOME: gbrainHome }, async () => {
// A source exists but its local_path does NOT match brainDir, so
// resolveSourceForDir returns undefined, cycleSourceId is undefined,
// and no per-source timestamp is written.
await engine.executeRaw(
`INSERT INTO sources (id, name, local_path, config, archived, created_at)
VALUES ('unmatched', 'unmatched', '/no/such/repo', '{}'::jsonb, false, NOW())
ON CONFLICT (id) DO UPDATE SET local_path = EXCLUDED.local_path`,
[],
);
await runCycle(engine, {
brainDir,
phases: ['lint'],
});
// No per-source write happens; default source's config stays empty.
const after = await readLastFullCycleAt('default-like');
expect(after).toBeNull();
expect(await readLastFullCycleAt('unmatched')).toBeNull();
});
});
-161
View File
@@ -1,161 +0,0 @@
/**
* #1818: embedBatch dispatches its 100-input sub-batches through a bounded
* worker pool (the embed-stale.ts concurrency pattern) instead of a serial
* `for` loop. This file pins:
*
* - output order matches input order regardless of completion order
* (index-addressed results)
* - parallelism actually happens (max in-flight > 1) and stays bounded
* (max in-flight <= configured concurrency)
* - concurrency: 1 restores the serial pre-#1818 dispatch
* - GBRAIN_EMBED_BATCH_CONCURRENCY env is honored when the option is unset
* - onBatchComplete reports a monotonic completed count ending at total
*
* Transport is stubbed via the gateway's __setEmbedTransportForTests seam
* (same pattern as test/ai/adaptive-embed-batch.test.ts). OpenAI recipe =
* fast path (no pre-split), so each embedBatch sub-batch is exactly one
* transport call.
*/
import { afterAll, afterEach, beforeEach, describe, expect, test } from 'bun:test';
import {
configureGateway,
resetGateway,
__setEmbedTransportForTests,
} from '../src/core/ai/gateway.ts';
import { embedBatch } from '../src/core/embedding.ts';
import { withEnv } from './helpers/with-env.ts';
const DIMS = 1536;
function configureOpenAI(): void {
configureGateway({
embedding_model: 'openai:text-embedding-3-large',
embedding_dimensions: DIMS,
env: { OPENAI_API_KEY: 'sk-fake' },
});
}
/**
* Install a transport whose returned embedding encodes the GLOBAL input
* index in dim 0 (texts are `t<N>`), so order can be asserted end-to-end.
* Tracks the max number of concurrently in-flight transport calls.
*/
function installTrackingTransport(delayMs = 5): { maxInFlight: () => number } {
let inFlight = 0;
let maxInFlight = 0;
__setEmbedTransportForTests((async ({ values }: { values: string[] }) => {
inFlight++;
maxInFlight = Math.max(maxInFlight, inFlight);
await new Promise(r => setTimeout(r, delayMs));
inFlight--;
return {
embeddings: values.map(v => {
const idx = Number(v.slice(1));
return Array.from({ length: DIMS }, (_, j) => (j === 0 ? idx : 0.1));
}),
};
}) as any);
return { maxInFlight: () => maxInFlight };
}
const texts = Array.from({ length: 250 }, (_, i) => `t${i}`);
afterAll(() => resetGateway());
describe('embedBatch bounded parallelism (#1818)', () => {
beforeEach(() => {
resetGateway();
configureOpenAI();
});
afterEach(() => {
__setEmbedTransportForTests(null);
});
test('default pool dispatches sub-batches in parallel, order preserved', async () => {
const tracker = installTrackingTransport();
const result = await embedBatch(texts, { onBatchComplete: () => {} });
expect(result).toHaveLength(250);
for (let i = 0; i < 250; i++) {
expect(result[i][0]).toBe(i);
}
// 250 texts → 3 sub-batches; default concurrency 4 → all 3 in flight.
expect(tracker.maxInFlight()).toBeGreaterThan(1);
expect(tracker.maxInFlight()).toBeLessThanOrEqual(4);
});
test('concurrency: 1 keeps the serial dispatch', async () => {
const tracker = installTrackingTransport();
const result = await embedBatch(texts, { concurrency: 1, onBatchComplete: () => {} });
expect(result).toHaveLength(250);
expect(tracker.maxInFlight()).toBe(1);
});
test('GBRAIN_EMBED_BATCH_CONCURRENCY env bounds the pool when option unset', async () => {
const tracker = installTrackingTransport();
await withEnv({ GBRAIN_EMBED_BATCH_CONCURRENCY: '2' }, async () => {
await embedBatch(texts, { onBatchComplete: () => {} });
});
expect(tracker.maxInFlight()).toBeGreaterThan(1);
expect(tracker.maxInFlight()).toBeLessThanOrEqual(2);
});
test('onBatchComplete reports a monotonic count ending at total', async () => {
installTrackingTransport();
const seen: number[] = [];
await embedBatch(texts, {
onBatchComplete: (done, total) => {
expect(total).toBe(250);
seen.push(done);
},
});
expect(seen).toHaveLength(3); // 100 + 100 + 50 sub-batches
for (let i = 1; i < seen.length; i++) {
expect(seen[i]).toBeGreaterThan(seen[i - 1]);
}
expect(seen[seen.length - 1]).toBe(250);
});
test('a failing sub-batch rejects the whole call', async () => {
let call = 0;
__setEmbedTransportForTests((async ({ values }: { values: string[] }) => {
call++;
if (call === 2) throw new Error('boom');
await new Promise(r => setTimeout(r, 2));
return { embeddings: values.map(() => Array.from({ length: DIMS }, () => 0.1)) };
}) as any);
await expect(embedBatch(texts, { onBatchComplete: () => {} })).rejects.toThrow();
});
test('after a failure, surviving workers stop dispatching new slices', async () => {
// 1000 texts → 10 slices, concurrency 2. First call fails immediately;
// without the `failed` flag the second worker would keep draining all
// 10 slices in the background AFTER embedBatch already rejected —
// burning provider spend and firing onBatchComplete post-rejection.
let calls = 0;
const completions: number[] = [];
__setEmbedTransportForTests((async ({ values }: { values: string[] }) => {
calls++;
if (calls === 1) throw new Error('boom');
await new Promise(r => setTimeout(r, 5));
return { embeddings: values.map(() => Array.from({ length: DIMS }, () => 0.1)) };
}) as any);
const many = Array.from({ length: 1000 }, (_, i) => `t${i}`);
await expect(
embedBatch(many, { concurrency: 2, onBatchComplete: d => completions.push(d) }),
).rejects.toThrow('boom');
const callsAtRejection = calls;
await new Promise(r => setTimeout(r, 50)); // would-be background drain window
expect(calls).toBe(callsAtRejection); // no new dispatch after rejection
expect(calls).toBeLessThanOrEqual(2); // only the in-flight sibling ran
expect(completions).toHaveLength(0); // no progress reported after failure
});
test('single small batch without callback stays on the one-call fast path', async () => {
const tracker = installTrackingTransport(1);
const result = await embedBatch(['t0', 't1', 't2']);
expect(result).toHaveLength(3);
expect(result[1][0]).toBe(1);
expect(tracker.maxInFlight()).toBe(1);
});
});
+3 -27
View File
@@ -19,7 +19,7 @@
* overwrites this preload.
*/
import { configureGateway, getEmbeddingDimensions } from '../../src/core/ai/gateway.ts';
import { afterEach, beforeEach } from 'bun:test';
import { beforeEach } from 'bun:test';
const LEGACY_CONFIG = {
embedding_model: 'openai:text-embedding-3-large',
@@ -52,7 +52,7 @@ applyLegacy();
// 2. file-local beforeAll → may overwrite to ZE/1280
// Since beforeAll runs once per file BEFORE the first beforeEach,
// file-local beforeAll wins for that file's tests. ✓
function applyLegacyIfEmpty() {
beforeEach(() => {
try {
// Only re-apply if the gateway was reset (or never configured).
// Tests that explicitly configured a different model in their
@@ -62,28 +62,4 @@ function applyLegacyIfEmpty() {
} catch {
applyLegacy();
}
}
beforeEach(applyLegacyIfEmpty);
// PR #3130 shard-order fix: beforeEach alone leaves ONE window open — a file
// whose LAST afterEach calls resetGateway() poisons the NEXT file's
// beforeAll, which runs BEFORE any beforeEach fires. A beforeAll there that
// does engine.initSchema() then sizes the embedding column from the gateway
// DEFAULTS (zembed-1/1280d) instead of the pinned legacy 1536, and every
// 1536-d Float32Array fixture in that file dies with
// "expected 1280 dimensions, not 1536". Which file pair collides is a
// function of shard composition, so adding/removing ANY test file can
// surface it (that is exactly how it bit shard 9).
//
// Preload hooks are registered before any file-local hooks, and bun runs
// after-hooks inside-out (file-local afterEach first, then this one), so
// this repairs the empty slot immediately after the poisoning reset —
// before the next file's beforeAll can observe it.
//
// Known remaining window: a file whose afterAll() resets the gateway (no
// hook runs between its afterAll and the next file's beforeAll). Files
// that reset in afterAll and can precede a schema-creating file should
// re-apply their own config, or the victim file should configureGateway()
// explicitly in its beforeAll.
afterEach(applyLegacyIfEmpty);
});
-69
View File
@@ -1,69 +0,0 @@
/**
* #1207: `gbrain import` without `--workers` used to hardcode workerCount=1,
* so a large Postgres import paid one serial embedding round-trip per file.
* runImport now routes the default through the shared autoConcurrency policy
* (PGLite → 1, >100 files on Postgres → DEFAULT_PARALLEL_WORKERS), while an
* explicit `--workers N` still wins.
*
* The engine here is a minimal postgres-kind stub with no database_url in
* config — runImport's parallel branch then falls back to serial processing
* (its PR #490 guard) but the WORKER-COUNT DECISION (the thing #1207 fixes)
* is still observable via the "Using N parallel workers" log line. Per-file
* imports fail against the stub engine and are swallowed by runImport's
* per-file catch; that's fine — this test pins the policy, not the import.
*/
import { afterEach, beforeEach, describe, expect, test } from 'bun:test';
import { mkdtempSync, writeFileSync, mkdirSync, rmSync, realpathSync } from 'fs';
import { tmpdir } from 'os';
import { join } from 'path';
import { withEnv } from './helpers/with-env.ts';
import { runImport } from '../src/commands/import.ts';
const fakePostgresEngine = {
kind: 'postgres',
executeRaw: async () => [],
logIngest: async () => {},
setConfig: async () => {},
getConfig: async () => null,
} as any;
let workspace: string;
let brainDir: string;
let logs: string[];
const realLog = console.log;
beforeEach(() => {
workspace = mkdtempSync(join(tmpdir(), 'gbrain-import-workers-home-'));
mkdirSync(join(workspace, '.gbrain'), { recursive: true });
brainDir = realpathSync(mkdtempSync(join(tmpdir(), 'gbrain-import-workers-brain-')));
// 101 files: one past AUTO_CONCURRENCY_FILE_THRESHOLD (100).
for (let i = 0; i < 101; i++) {
writeFileSync(join(brainDir, `page-${i}.md`), `# Page ${i}\n\nbody ${i}\n`);
}
logs = [];
console.log = (msg?: unknown) => logs.push(String(msg));
});
afterEach(() => {
console.log = realLog;
rmSync(workspace, { recursive: true, force: true });
rmSync(brainDir, { recursive: true, force: true });
});
describe('import default worker count (#1207)', () => {
test('no --workers flag → autoConcurrency picks 4 for >100 files on Postgres', async () => {
await withEnv({ GBRAIN_HOME: join(workspace, '.gbrain'), GBRAIN_SOURCE: undefined }, async () => {
await runImport(fakePostgresEngine, [brainDir, '--no-embed'], { sourceId: 'default' });
});
expect(logs.some(l => l.includes('Using 4 parallel workers'))).toBe(true);
});
test('explicit --workers 2 still wins over the auto policy', async () => {
await withEnv({ GBRAIN_HOME: join(workspace, '.gbrain'), GBRAIN_SOURCE: undefined }, async () => {
await runImport(fakePostgresEngine, [brainDir, '--no-embed', '--workers', '2'], { sourceId: 'default' });
});
expect(logs.some(l => l.includes('Using 2 parallel workers'))).toBe(true);
expect(logs.some(l => l.includes('Using 4 parallel workers'))).toBe(false);
});
});