mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-17 02:12:40 +00:00
Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
60fb33c0d9 | ||
|
|
11ed0871c2 | ||
|
|
595eeb7d6f |
+13
-43
@@ -1,7 +1,6 @@
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { embedBatch, currentEmbeddingSignature } from '../core/embedding.ts';
|
||||
import type { ChunkInput, ResolvedColumn } from '../core/types.ts';
|
||||
import { resolveWriteColumnForEngine } from '../core/search/embedding-column.ts';
|
||||
import type { ChunkInput } from '../core/types.ts';
|
||||
import { chunkText } from '../core/chunkers/recursive.ts';
|
||||
import { createProgress, type ProgressReporter } from '../core/progress.ts';
|
||||
import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts';
|
||||
@@ -184,13 +183,8 @@ export class EmbeddingDimMismatchError extends Error {
|
||||
* fresh-install bug class at the very first invocation instead of letting
|
||||
* the worker pool hammer N pages with raw 22000 errors.
|
||||
*/
|
||||
async function preflightDimMismatch(engine: BrainEngine, dryRun: boolean, embeddingColumn?: ResolvedColumn): Promise<void> {
|
||||
async function preflightDimMismatch(engine: BrainEngine, dryRun: boolean): Promise<void> {
|
||||
if (dryRun) return; // dry-run never embeds, no risk
|
||||
// #1262: an alt-column brain writes to `embeddingColumn`, not the legacy
|
||||
// `embedding` column — the legacy column's dims are irrelevant, and the
|
||||
// registry entry (validated at resolve time) pins the target's dims. Only
|
||||
// the legacy default path needs the schema-vs-gateway dim comparison.
|
||||
if (embeddingColumn && embeddingColumn.name !== 'embedding') return;
|
||||
const { readContentChunksEmbeddingDim, embeddingMismatchMessage } = await import('../core/embedding-dim-check.ts');
|
||||
const { getEmbeddingDimensions, getEmbeddingModel } = await import('../core/ai/gateway.ts');
|
||||
let existing;
|
||||
@@ -244,12 +238,7 @@ export async function runEmbedCore(engine: BrainEngine, opts: EmbedOpts): Promis
|
||||
// v0.37.11.0 (Lane D.2): pre-flight dim-mismatch check. Catches the headline
|
||||
// fresh-install bug class before the worker pool spends 20 parallel calls
|
||||
// hitting raw Postgres dimension errors.
|
||||
// #1262: resolve the write-side embedding column ONCE at the boundary
|
||||
// (merged config + gateway model) and thread the descriptor through every
|
||||
// upsertChunks / stale-scan below. undefined => legacy `embedding` column.
|
||||
const embeddingColumn = await resolveWriteColumnForEngine(engine);
|
||||
|
||||
await preflightDimMismatch(engine, !!opts.dryRun, embeddingColumn);
|
||||
await preflightDimMismatch(engine, !!opts.dryRun);
|
||||
|
||||
const result: EmbedResult = {
|
||||
embedded: 0,
|
||||
@@ -264,7 +253,7 @@ export async function runEmbedCore(engine: BrainEngine, opts: EmbedOpts): Promis
|
||||
for (const s of opts.slugs) {
|
||||
if (isAborted(opts.signal)) break; // #1737: stop the per-slug loop on abort
|
||||
try {
|
||||
await embedPage(engine, s, !!opts.dryRun, result, opts.sourceId, opts.signal, embeddingColumn);
|
||||
await embedPage(engine, s, !!opts.dryRun, result, opts.sourceId, opts.signal);
|
||||
} catch (e: unknown) {
|
||||
serr(` Error embedding ${s}: ${e instanceof Error ? e.message : e}`);
|
||||
}
|
||||
@@ -358,7 +347,7 @@ export async function runEmbedCore(engine: BrainEngine, opts: EmbedOpts): Promis
|
||||
catchUp: opts.catchUp,
|
||||
pacer,
|
||||
paceMaxConcurrency,
|
||||
}, opts.signal, embeddingColumn);
|
||||
}, opts.signal);
|
||||
} finally {
|
||||
// E1: surface pacing telemetry (human + structured) when pacing was on.
|
||||
const snap = pacer.snapshot();
|
||||
@@ -387,7 +376,7 @@ export async function runEmbedCore(engine: BrainEngine, opts: EmbedOpts): Promis
|
||||
return result;
|
||||
}
|
||||
if (opts.slug) {
|
||||
await embedPage(engine, opts.slug, !!opts.dryRun, result, opts.sourceId, opts.signal, embeddingColumn);
|
||||
await embedPage(engine, opts.slug, !!opts.dryRun, result, opts.sourceId, opts.signal);
|
||||
return result;
|
||||
}
|
||||
throw new Error('No embed target specified. Pass { slug }, { slugs }, { all }, or { stale }.');
|
||||
@@ -532,13 +521,8 @@ async function embedPage(
|
||||
result: EmbedResult,
|
||||
sourceId?: string,
|
||||
signal?: AbortSignal,
|
||||
embeddingColumn?: ResolvedColumn,
|
||||
) {
|
||||
const opts = sourceId ? { sourceId } : undefined;
|
||||
// #1262: write-side descriptor rides only on WRITE calls (upsertChunks).
|
||||
const chunkOpts = (sourceId || embeddingColumn)
|
||||
? { ...(sourceId && { sourceId }), ...(embeddingColumn && { embeddingColumn }) }
|
||||
: undefined;
|
||||
const page = await engine.getPage(slug, opts);
|
||||
if (!page) {
|
||||
throw new Error(`Page not found: ${slug}`);
|
||||
@@ -570,7 +554,7 @@ async function embedPage(
|
||||
}
|
||||
|
||||
if (inputs.length > 0) {
|
||||
await engine.upsertChunks(slug, inputs, chunkOpts);
|
||||
await engine.upsertChunks(slug, inputs, opts);
|
||||
chunks = await engine.getChunks(slug, opts);
|
||||
}
|
||||
}
|
||||
@@ -605,7 +589,7 @@ async function embedPage(
|
||||
token_count: c.token_count || Math.ceil(c.chunk_text.length / 4),
|
||||
}));
|
||||
|
||||
await engine.upsertChunks(slug, updated, chunkOpts);
|
||||
await engine.upsertChunks(slug, updated, opts);
|
||||
// v0.41.31: stamp provenance so a later model/dims swap is detectable as
|
||||
// stale. embedPage is the per-slug path used by `gbrain embed <slug>` AND
|
||||
// by `gbrain sync`'s post-import embed step (runEmbedCore({slugs})).
|
||||
@@ -638,7 +622,6 @@ async function embedAll(
|
||||
paceMaxConcurrency?: number;
|
||||
},
|
||||
signal?: AbortSignal,
|
||||
embeddingColumn?: ResolvedColumn,
|
||||
) {
|
||||
// v0.41.31: current embedding provenance signature. Stamped onto pages
|
||||
// when their chunks are (re)embedded so a later model/dimension swap is
|
||||
@@ -661,7 +644,7 @@ async function embedAll(
|
||||
// D7: thread sourceId so `gbrain embed --stale --source X` actually scopes.
|
||||
// v0.41.18.0 (A13): thread batchSize/priority/catchUp into the stale path.
|
||||
// #1737: thread the external abort signal so the cycle embed phase bails.
|
||||
return await embedAllStale(engine, sourceId, dryRun, result, onProgress, staleOpts, signature, signal, embeddingColumn);
|
||||
return await embedAllStale(engine, sourceId, dryRun, result, onProgress, staleOpts, signature, signal);
|
||||
}
|
||||
|
||||
// --all path: pacer (no-op when off). E-1: lower the worker count to the
|
||||
@@ -742,10 +725,7 @@ async function embedAll(
|
||||
embedding: embeddingMap.get(c.chunk_index) ?? undefined,
|
||||
token_count: c.token_count || Math.ceil(c.chunk_text.length / 4),
|
||||
}));
|
||||
await observed(pacer, () => engine.upsertChunks(page.slug, updated, {
|
||||
...(pageSourceId && { sourceId: pageSourceId }),
|
||||
...(embeddingColumn && { embeddingColumn }),
|
||||
}));
|
||||
await observed(pacer, () => engine.upsertChunks(page.slug, updated, pageOpts));
|
||||
// v0.41.31: stamp embedding provenance so a later model swap is
|
||||
// detectable as stale.
|
||||
await observed(pacer, () =>
|
||||
@@ -825,16 +805,10 @@ async function embedAllStale(
|
||||
},
|
||||
signature?: string,
|
||||
externalSignal?: AbortSignal,
|
||||
embeddingColumn?: ResolvedColumn,
|
||||
) {
|
||||
// D7: thread sourceId so source-scoped runs only count + visit
|
||||
// that source's NULL embeddings.
|
||||
// #1262: the stale predicate follows the write-side column — without it an
|
||||
// alt-column brain would perpetually re-select (and re-pay for) chunks whose
|
||||
// target column is already populated.
|
||||
const sourceOpt = (sourceId || embeddingColumn)
|
||||
? { ...(sourceId && { sourceId }), ...(embeddingColumn && { embeddingColumn }) }
|
||||
: undefined;
|
||||
const sourceOpt = sourceId ? { sourceId } : undefined;
|
||||
|
||||
// v0.41.31: re-embed pages whose embedding_signature drifted (model/dims
|
||||
// swap). dry-run must NOT mutate, so it counts signature-stale via the
|
||||
@@ -993,7 +967,6 @@ async function embedAllStale(
|
||||
afterUpdatedAt,
|
||||
}),
|
||||
...(sourceId && { sourceId }),
|
||||
...(embeddingColumn && { embeddingColumn }),
|
||||
}),
|
||||
);
|
||||
if (batch.length === 0) {
|
||||
@@ -1046,10 +1019,7 @@ async function embedAllStale(
|
||||
embedding: staleIdxToEmbedding.get(c.chunk_index) ?? undefined,
|
||||
token_count: c.token_count || Math.ceil(c.chunk_text.length / 4),
|
||||
}));
|
||||
await observed(pacer, () => engine.upsertChunks(slug, merged, {
|
||||
sourceId: keySourceId,
|
||||
...(embeddingColumn && { embeddingColumn }),
|
||||
}));
|
||||
await observed(pacer, () => engine.upsertChunks(slug, merged, { sourceId: keySourceId }));
|
||||
// v0.41.31: stamp provenance after the page's chunks are embedded —
|
||||
// but only when EVERY chunk was stale (fully re-embedded this pass).
|
||||
// A partially-stale page keeps preserved chunks of unknown/old
|
||||
@@ -1120,7 +1090,7 @@ async function embedAllStale(
|
||||
// as a clean run — re-running won't help until the underlying failure is fixed.
|
||||
if (staleOpts?.catchUp && !effectiveSignal.aborted && embedFailures > 0) {
|
||||
const remaining = await engine.countStaleChunks(
|
||||
signature ? { signature, ...sourceOpt } : sourceOpt,
|
||||
signature ? { signature, ...(sourceId ? { sourceId } : {}) } : (sourceId ? { sourceId } : undefined),
|
||||
);
|
||||
if (remaining > 0) {
|
||||
serr(`\n [embed] catch-up finished but ${remaining} chunk(s) remain stale after ${embedFailures} embed failure(s). These are not embeddable as-is; re-running won't clear them until the underlying error is resolved.`);
|
||||
|
||||
+10
-5
@@ -170,10 +170,14 @@ export async function runImport(
|
||||
// v0.22.13 (PR #490 Q2): shared parseWorkers helper rejects bad input
|
||||
// (--workers 0, -3, "foo") with a loud error instead of silently falling
|
||||
// through to 1. Mirrors sync.ts's flag handling.
|
||||
const { parseWorkers } = await import('../core/sync-concurrency.ts');
|
||||
let workerCount: number;
|
||||
const { parseWorkers, autoConcurrency } = await import('../core/sync-concurrency.ts');
|
||||
// #1207: undefined (no --workers flag) defers to autoConcurrency below —
|
||||
// the shared sync/import policy (PGLite → 1, >100 files → 4) — instead of
|
||||
// hardcoding serial. Large Postgres imports stop paying one embedding
|
||||
// round-trip per file in sequence.
|
||||
let workerCount: number | undefined;
|
||||
try {
|
||||
workerCount = parseWorkers(workersArg ?? undefined) ?? 1;
|
||||
workerCount = parseWorkers(workersArg ?? undefined);
|
||||
} catch (e) {
|
||||
console.error(e instanceof Error ? e.message : String(e));
|
||||
process.exit(1);
|
||||
@@ -252,8 +256,9 @@ export async function runImport(
|
||||
}
|
||||
const files = resumeFilter(allFiles, dir, completed);
|
||||
|
||||
// Determine actual worker count
|
||||
const actualWorkers = workerCount > 1 ? workerCount : 1;
|
||||
// Determine actual worker count. Explicit --workers wins; otherwise the
|
||||
// shared autoConcurrency policy decides from engine kind + file count.
|
||||
const actualWorkers = autoConcurrency(engine, files.length, workerCount);
|
||||
if (actualWorkers > 1) {
|
||||
console.log(`Using ${actualWorkers} parallel workers`);
|
||||
}
|
||||
|
||||
@@ -576,16 +576,9 @@ async function runInlineCostGate(
|
||||
|
||||
// Stale backlog: cheap single SQL; fail-open to 0 so a transient DB hiccup
|
||||
// never blocks the sync. Signature-aware (model/dims swap surfaces here).
|
||||
// #1262: follow the write-side embedding column — otherwise an alt-column
|
||||
// brain's fully-embedded corpus counts as phantom backlog on every gate.
|
||||
let staleChars = 0;
|
||||
try {
|
||||
const { resolveWriteColumnForEngine } = await import('../core/search/embedding-column.ts');
|
||||
const embeddingColumn = await resolveWriteColumnForEngine(engine);
|
||||
staleChars = await engine.sumStaleChunkChars({
|
||||
signature: currentEmbeddingSignature(),
|
||||
...(embeddingColumn && { embeddingColumn }),
|
||||
});
|
||||
staleChars = await engine.sumStaleChunkChars({ signature: currentEmbeddingSignature() });
|
||||
} catch {
|
||||
staleChars = 0;
|
||||
}
|
||||
|
||||
+24
-6
@@ -1513,12 +1513,21 @@ export async function embed(texts: string[], opts?: EmbedOpts): Promise<Float32A
|
||||
|
||||
const embedding = recipe.touchpoints?.embedding;
|
||||
const maxBatchTokens = embedding?.max_batch_tokens;
|
||||
const maxBatchCount = embedding?.max_batch_count;
|
||||
const charsPerToken = embedding?.chars_per_token ?? DEFAULT_CHARS_PER_TOKEN;
|
||||
|
||||
// Pre-split is gated on max_batch_tokens. Recipes without it (e.g. OpenAI)
|
||||
// ride the fast path: one embedMany call, no recursion safety net.
|
||||
const batches = maxBatchTokens
|
||||
? splitByTokenBudget(truncated, Math.floor(maxBatchTokens * effectiveSafetyFactor(recipe)), charsPerToken)
|
||||
// Pre-split is gated on max_batch_tokens / max_batch_count. Recipes with
|
||||
// neither (e.g. OpenAI) ride the fast path: one embedMany call, no
|
||||
// recursion safety net.
|
||||
const batches = (maxBatchTokens || maxBatchCount)
|
||||
? splitByTokenBudget(
|
||||
truncated,
|
||||
maxBatchTokens
|
||||
? Math.floor(maxBatchTokens * effectiveSafetyFactor(recipe))
|
||||
: Number.MAX_SAFE_INTEGER,
|
||||
charsPerToken,
|
||||
maxBatchCount,
|
||||
)
|
||||
: [truncated];
|
||||
|
||||
const allEmbeddings: Float32Array[] = [];
|
||||
@@ -1568,6 +1577,9 @@ export async function embed(texts: string[], opts?: EmbedOpts): Promise<Float32A
|
||||
* responsible for applying any safety-factor shrink before passing in.
|
||||
* @param charsPerToken - Provider-specific character density. Defaults to
|
||||
* `DEFAULT_CHARS_PER_TOKEN` (4) when omitted, matching OpenAI tiktoken.
|
||||
* @param maxBatchCount - #1199: optional cap on INPUTS per sub-batch, for
|
||||
* providers that reject batches by count (DashScope: 10). When omitted,
|
||||
* only the token budget governs.
|
||||
*
|
||||
* @internal exported for tests; not part of the public gateway API.
|
||||
*/
|
||||
@@ -1575,15 +1587,17 @@ export function splitByTokenBudget(
|
||||
texts: string[],
|
||||
budgetTokens: number,
|
||||
charsPerToken: number = DEFAULT_CHARS_PER_TOKEN,
|
||||
maxBatchCount?: number,
|
||||
): string[][] {
|
||||
const ratio = charsPerToken > 0 ? charsPerToken : DEFAULT_CHARS_PER_TOKEN;
|
||||
const maxCount = maxBatchCount !== undefined && maxBatchCount > 0 ? maxBatchCount : Infinity;
|
||||
const batches: string[][] = [];
|
||||
let current: string[] = [];
|
||||
let currentTokens = 0;
|
||||
|
||||
for (const text of texts) {
|
||||
const estTokens = Math.ceil(text.length / ratio);
|
||||
if (current.length > 0 && currentTokens + estTokens > budgetTokens) {
|
||||
if (current.length > 0 && (currentTokens + estTokens > budgetTokens || current.length >= maxCount)) {
|
||||
batches.push(current);
|
||||
current = [];
|
||||
currentTokens = 0;
|
||||
@@ -1609,7 +1623,11 @@ export function isTokenLimitError(err: unknown): boolean {
|
||||
/token.*limit.*exceeded/i.test(msg) ||
|
||||
// OpenAI embeddings: "Invalid 'input': maximum request size is 300000 tokens per request."
|
||||
/maximum request size.*tokens/i.test(msg) ||
|
||||
/max.*tokens.*per.*request/i.test(msg)
|
||||
/max.*tokens.*per.*request/i.test(msg) ||
|
||||
// DashScope: "batch size is invalid, it should not be larger than 10." (#1199)
|
||||
// Count-cap error, but recursive halving shrinks count too, so the same
|
||||
// safety net converges.
|
||||
/batch size is invalid/i.test(msg)
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -31,6 +31,10 @@ export const dashscope: Recipe = {
|
||||
// path. Conservative declaration so the gateway pre-splits before
|
||||
// hitting whatever undocumented server-side limit exists.
|
||||
max_batch_tokens: 8192,
|
||||
// #1199: DashScope hard-caps embeddings at 10 inputs per request
|
||||
// ("batch size is invalid, it should not be larger than 10"). The
|
||||
// token budget alone admits far more than 10 short chunks per batch.
|
||||
max_batch_count: 10,
|
||||
// text-embedding-v3 mixes English + CJK heavily; the tokenizer is
|
||||
// closer to Voyage density than OpenAI tiktoken for CJK-dominant
|
||||
// content. Conservative chars_per_token=2 leaves headroom.
|
||||
|
||||
@@ -16,6 +16,15 @@ export const google: Recipe = {
|
||||
dims_options: [768, 1536, 3072],
|
||||
cost_per_1m_tokens_usd: 0.15,
|
||||
price_last_verified: '2026-04-20',
|
||||
// #970: Gemini's documented limits are per-INPUT (2048 tokens,
|
||||
// silently truncated beyond) and per-REQUEST count (batchEmbedContents
|
||||
// caps at 100 inputs). There is no separate per-request token cap, so
|
||||
// the token budget is derived: 100 inputs × 2048 tokens. The count cap
|
||||
// binds first for typical chunk sizes. Do NOT copy the 2048 per-input
|
||||
// limit into max_batch_tokens — that would over-split 50×.
|
||||
max_batch_tokens: 204_800,
|
||||
chars_per_token: 4,
|
||||
max_batch_count: 100,
|
||||
},
|
||||
expansion: {
|
||||
models: ['gemini-2.0-flash', 'gemini-2.0-flash-lite'],
|
||||
|
||||
@@ -58,5 +58,8 @@ export function getRecipe(id: string): Recipe | undefined {
|
||||
}
|
||||
|
||||
export function listRecipes(): Recipe[] {
|
||||
return [...ALL];
|
||||
// Read the map (not ALL) so there is one source of truth — getRecipe,
|
||||
// model-resolver, and listRecipes all see the same registry, and tests
|
||||
// can inject a synthetic recipe via RECIPES to exercise registry walks.
|
||||
return [...RECIPES.values()];
|
||||
}
|
||||
|
||||
@@ -46,6 +46,16 @@ export interface EmbeddingTouchpoint {
|
||||
* Only consulted when `max_batch_tokens` is also set.
|
||||
*/
|
||||
chars_per_token?: number;
|
||||
/**
|
||||
* #1199: maximum number of INPUTS per embedding request, for providers
|
||||
* that hard-cap batch size by count rather than (or in addition to)
|
||||
* tokens — DashScope text-embedding-v3 rejects batches > 10 with
|
||||
* `InvalidParameter`, Gemini batchEmbedContents caps at 100 requests.
|
||||
* When set, the gateway's pre-split flushes a sub-batch at this count
|
||||
* even if the token budget still has room. Independent of
|
||||
* `max_batch_tokens`; either alone triggers the pre-split.
|
||||
*/
|
||||
max_batch_count?: number;
|
||||
/**
|
||||
* Budget-utilization ceiling in (0, 1]. The gateway pre-splits at
|
||||
* `safety_factor × max_batch_tokens` to leave headroom for tokenizer
|
||||
|
||||
@@ -61,7 +61,6 @@ import {
|
||||
type SynopsisFailureKind,
|
||||
} from './audit-synopsis.ts';
|
||||
import type { BrainEngine } from './engine.ts';
|
||||
import { resolveWriteColumnForEngine } from './search/embedding-column.ts';
|
||||
import type { ChunkInput, CRMode, Page } from './types.ts';
|
||||
import type { SourceRow } from './sources-ops.ts';
|
||||
|
||||
@@ -287,13 +286,9 @@ export async function reembedPageWithContextualRetrieval(
|
||||
|
||||
// ── PHASE 2: single DB transaction ───────────────────────────
|
||||
try {
|
||||
// #1262: contextual re-embeds write TEXT embeddings — thread the
|
||||
// caller-resolved write column like every other embed path.
|
||||
const embeddingColumn = await resolveWriteColumnForEngine(args.engine);
|
||||
await args.engine.transaction(async (tx) => {
|
||||
await tx.upsertChunks(args.pageSlug, phase1.embeddedChunks, {
|
||||
sourceId: args.sourceId,
|
||||
...(embeddingColumn && { embeddingColumn }),
|
||||
});
|
||||
await tx.updatePageContextualRetrievalState(
|
||||
args.pageSlug,
|
||||
|
||||
+2
-13
@@ -18,7 +18,7 @@
|
||||
*/
|
||||
|
||||
import type { BrainEngine } from './engine.ts';
|
||||
import type { ChunkInput, ResolvedColumn } from './types.ts';
|
||||
import type { ChunkInput } from './types.ts';
|
||||
import { embedBatchWithBackoff } from '../commands/embed.ts';
|
||||
import { type DbPacer, createNoopPacer, observed } from './db-pacer.ts';
|
||||
import { AbortError } from './abort-check.ts';
|
||||
@@ -61,13 +61,6 @@ export interface EmbedStaleOpts {
|
||||
* Omit to keep the legacy `embedding IS NULL`-only behavior.
|
||||
*/
|
||||
embeddingSignature?: string;
|
||||
/**
|
||||
* #1262: caller-resolved write-side embedding column. Threaded into BOTH
|
||||
* listStaleChunks (staleness predicate) and upsertChunks (write target) so
|
||||
* an alt-column brain converges instead of re-selecting embedded rows.
|
||||
* Resolve at the boundary via `resolveWriteColumnForEngine()`.
|
||||
*/
|
||||
embeddingColumn?: ResolvedColumn;
|
||||
/**
|
||||
* DB-contention pacer (paced-backfill). When enabled it (a) supplies the
|
||||
* worker count via the caller passing `concurrency = bundle.maxConcurrency`
|
||||
@@ -163,7 +156,6 @@ export async function embedStaleForSource(
|
||||
afterPageId,
|
||||
afterChunkIndex,
|
||||
sourceId,
|
||||
...(opts.embeddingColumn && { embeddingColumn: opts.embeddingColumn }),
|
||||
}),
|
||||
);
|
||||
if (batch.length === 0) {
|
||||
@@ -231,10 +223,7 @@ export async function embedStaleForSource(
|
||||
doc_comment: c.doc_comment ?? undefined,
|
||||
symbol_name_qualified: c.symbol_name_qualified ?? undefined,
|
||||
}));
|
||||
await observed(pacer, () => engine.upsertChunks(slug, merged, {
|
||||
sourceId: keySourceId,
|
||||
...(opts.embeddingColumn && { embeddingColumn: opts.embeddingColumn }),
|
||||
}));
|
||||
await observed(pacer, () => engine.upsertChunks(slug, merged, { sourceId: keySourceId }));
|
||||
// v0.41.31: stamp provenance only when EVERY chunk was stale (fully
|
||||
// re-embedded this pass) — a partially-stale page keeps preserved
|
||||
// chunks of unknown provenance, so don't claim current. After the
|
||||
|
||||
+56
-6
@@ -79,15 +79,34 @@ export interface EmbedBatchOptions {
|
||||
* and amplify rate-limit pressure.
|
||||
*/
|
||||
maxRetries?: number;
|
||||
/**
|
||||
* #1818: bounded parallelism across BATCH_SIZE sub-batches. Defaults to
|
||||
* `GBRAIN_EMBED_BATCH_CONCURRENCY` env, else 4. Results are
|
||||
* index-addressed so output order always matches input order. Set 1 to
|
||||
* force the pre-v0.42 serial dispatch.
|
||||
*/
|
||||
concurrency?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Embed a batch of texts via the gateway. Sub-batches of 100 so upstream
|
||||
* progress callbacks fire incrementally on large imports. The gateway owns
|
||||
* adaptive batch splitting and per-recipe token-budget logic; this paginator
|
||||
* is purely about progress-callback granularity.
|
||||
* owns progress-callback granularity and (#1818) bounded parallel dispatch
|
||||
* of the sub-batches — the embed-stale.ts worker-pool pattern, scoped down.
|
||||
*/
|
||||
const BATCH_SIZE = 100;
|
||||
const DEFAULT_EMBED_BATCH_CONCURRENCY = 4;
|
||||
|
||||
function resolveEmbedBatchConcurrency(options: EmbedBatchOptions): number {
|
||||
if (options.concurrency !== undefined) {
|
||||
return Math.max(1, Math.floor(options.concurrency));
|
||||
}
|
||||
const env = Number(process.env.GBRAIN_EMBED_BATCH_CONCURRENCY);
|
||||
if (Number.isFinite(env) && env >= 1) return Math.floor(env);
|
||||
return DEFAULT_EMBED_BATCH_CONCURRENCY;
|
||||
}
|
||||
|
||||
export async function embedBatch(
|
||||
texts: string[],
|
||||
options: EmbedBatchOptions = {},
|
||||
@@ -103,13 +122,44 @@ export async function embedBatch(
|
||||
if (texts.length <= BATCH_SIZE && !options.onBatchComplete) {
|
||||
return gatewayEmbed(texts, gwOpts);
|
||||
}
|
||||
const results: Float32Array[] = [];
|
||||
// #1818: dispatch sub-batches through a bounded worker pool instead of a
|
||||
// serial loop. Results are written into a preallocated index-addressed
|
||||
// array so output order matches input order regardless of completion
|
||||
// order; onBatchComplete reports a monotonic completed-embedding count.
|
||||
const slices: Array<{ start: number; texts: string[] }> = [];
|
||||
for (let i = 0; i < texts.length; i += BATCH_SIZE) {
|
||||
const slice = texts.slice(i, i + BATCH_SIZE);
|
||||
const out = await gatewayEmbed(slice, gwOpts);
|
||||
results.push(...out);
|
||||
options.onBatchComplete?.(results.length, texts.length);
|
||||
slices.push({ start: i, texts: texts.slice(i, i + BATCH_SIZE) });
|
||||
}
|
||||
const results = new Array<Float32Array>(texts.length);
|
||||
let next = 0;
|
||||
let done = 0;
|
||||
const numWorkers = Math.min(resolveEmbedBatchConcurrency(options), slices.length);
|
||||
// Once any sub-batch fails, `failed` stops the surviving workers from
|
||||
// dispatching FURTHER slices — the whole call is rejecting anyway, so
|
||||
// continuing would burn real provider spend in the background and fire
|
||||
// onBatchComplete after the caller already saw the failure (worst with
|
||||
// embedBatchWithBackoff, whose 429 backoff assumes nothing is in flight).
|
||||
// In-flight sibling calls still run to completion (bounded by numWorkers-1).
|
||||
let failed = false;
|
||||
const worker = async (): Promise<void> => {
|
||||
while (!failed && next < slices.length) {
|
||||
// NOTE: no local aborted-check here — an aborted signal makes the next
|
||||
// gatewayEmbed call throw (SDK-side), which rejects the pool. Returning
|
||||
// silently instead would resolve with holes in `results`.
|
||||
const slice = slices[next++];
|
||||
let out: Float32Array[];
|
||||
try {
|
||||
out = await gatewayEmbed(slice.texts, gwOpts);
|
||||
} catch (err) {
|
||||
failed = true;
|
||||
throw err;
|
||||
}
|
||||
for (let j = 0; j < out.length; j++) results[slice.start + j] = out[j];
|
||||
done += out.length;
|
||||
if (!failed) options.onBatchComplete?.(done, texts.length);
|
||||
}
|
||||
};
|
||||
await Promise.all(Array.from({ length: numWorkers }, () => worker()));
|
||||
return results;
|
||||
}
|
||||
|
||||
|
||||
+3
-22
@@ -12,7 +12,6 @@ import type {
|
||||
BrainStats, BrainHealth,
|
||||
IngestLogEntry, IngestLogInput,
|
||||
EngineConfig,
|
||||
ResolvedColumn,
|
||||
CodeEdgeInput, CodeEdgeResult,
|
||||
EvalCandidate, EvalCandidateInput,
|
||||
EvalCaptureFailure, EvalCaptureFailureReason,
|
||||
@@ -988,13 +987,8 @@ export interface BrainEngine {
|
||||
* — Postgres rolls back automatically on conn drop, so commit-ambiguous
|
||||
* failure replays to the same end state. Callers MUST NOT wrap externally;
|
||||
* see {@link BatchOpts} retry-contract block.
|
||||
*
|
||||
* `opts.embeddingColumn` (optional) selects the content_chunks column that
|
||||
* receives TEXT embeddings (#1262). The caller resolves the descriptor at
|
||||
* the import/embed boundary via `resolveWriteColumn()`; engines never read
|
||||
* config or choose columns themselves. Omitted => legacy `embedding`.
|
||||
*/
|
||||
upsertChunks(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string; embeddingColumn?: ResolvedColumn } & BatchOpts): Promise<void>;
|
||||
upsertChunks(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string } & BatchOpts): Promise<void>;
|
||||
/**
|
||||
* Read every chunk for a page. `opts.sourceId` source-scopes the page
|
||||
* lookup; without it, multi-source brains return chunks from every
|
||||
@@ -1011,13 +1005,8 @@ export interface BrainEngine {
|
||||
* counts across every source in the brain. Operators running
|
||||
* `gbrain embed --stale --source media-corpus` expect only that
|
||||
* source's NULLs touched; the caller threads `sourceId` here.
|
||||
*
|
||||
* `opts.embeddingColumn` switches the staleness predicate from the legacy
|
||||
* `embedding` column to the resolved write-side column, so alt-column
|
||||
* brains do not perpetually re-select rows whose target column is already
|
||||
* populated (#1262). Must match the eventual upsertChunks target.
|
||||
*/
|
||||
countStaleChunks(opts?: { sourceId?: string; signature?: string; embeddingColumn?: ResolvedColumn }): Promise<number>;
|
||||
countStaleChunks(opts?: { sourceId?: string; signature?: string }): Promise<number>;
|
||||
/**
|
||||
* Sum of LENGTH(chunk_text) over stale chunks — the character-count
|
||||
* backlog the embed phase / embed-backfill will process. Sibling of
|
||||
@@ -1031,13 +1020,8 @@ export interface BrainEngine {
|
||||
* model signature (a model/dims swap). NULL signature is GRANDFATHERED
|
||||
* (never counted) so the post-migration corpus isn't flagged en masse.
|
||||
* Omit `signature` for the legacy `embedding IS NULL`-only count.
|
||||
*
|
||||
* `opts.embeddingColumn` switches the staleness predicate to the resolved
|
||||
* write-side column (#1262) — same contract as countStaleChunks — so the
|
||||
* sync cost gate doesn't count an alt-column brain's fully-embedded corpus
|
||||
* as phantom backlog.
|
||||
*/
|
||||
sumStaleChunkChars(opts?: { sourceId?: string; signature?: string; embeddingColumn?: ResolvedColumn }): Promise<number>;
|
||||
sumStaleChunkChars(opts?: { sourceId?: string; signature?: string }): Promise<number>;
|
||||
/**
|
||||
* Stamp `pages.embedding_signature = signature` for one page. Called after
|
||||
* a page's chunks are (re)embedded so a later model swap can detect it as
|
||||
@@ -1085,9 +1069,6 @@ export interface BrainEngine {
|
||||
// both round-trip TIMESTAMPTZ as Date | string; ISO string is the
|
||||
// common denominator on the wire).
|
||||
afterUpdatedAt?: string | null;
|
||||
// #1262: staleness predicate targets this column when set (must match
|
||||
// countStaleChunks and the eventual upsertChunks write target).
|
||||
embeddingColumn?: ResolvedColumn;
|
||||
}): Promise<StaleChunkRow[]>;
|
||||
/**
|
||||
* Delete every chunk for a page. Internal page-id lookup is sourceId-scoped
|
||||
|
||||
+4
-25
@@ -10,8 +10,7 @@ import { findChunkForOffset } from './chunkers/edge-extractor.ts';
|
||||
import { extractCodeRefs, imageOfCandidates } from './link-extraction.ts';
|
||||
import { embedBatch, embedMultimodal, currentEmbeddingSignature } from './embedding.ts';
|
||||
import { slugifyPath, slugifyCodePath, isCodeFilePath } from './sync.ts';
|
||||
import type { ChunkInput, PageInput, PageType, ResolvedColumn } from './types.ts';
|
||||
import { resolveWriteColumnForEngine } from './search/embedding-column.ts';
|
||||
import type { ChunkInput, PageInput, PageType } from './types.ts';
|
||||
import { computeEffectiveDate } from './effective-date.ts';
|
||||
import { MARKDOWN_CHUNKER_VERSION } from './chunkers/recursive.ts';
|
||||
import { logSlugFallback } from './audit-slug-fallback.ts';
|
||||
@@ -741,14 +740,6 @@ export async function importFromContent(
|
||||
// schema DEFAULT — required for multi-source brains; harmless ('default')
|
||||
// for single-source callers.
|
||||
const txOpts = sourceId ? { sourceId } : undefined;
|
||||
// #1262: resolve the write-side embedding column once (merged config +
|
||||
// gateway model) BEFORE the transaction; the descriptor rides only on
|
||||
// upsertChunks so text embeddings land in the registered column.
|
||||
const chunkWriteColumn = await resolveWriteColumnForEngine(engine);
|
||||
const chunkOpts: { sourceId?: string; embeddingColumn?: ResolvedColumn } | undefined =
|
||||
(sourceId || chunkWriteColumn)
|
||||
? { ...(sourceId && { sourceId }), ...(chunkWriteColumn && { embeddingColumn: chunkWriteColumn }) }
|
||||
: undefined;
|
||||
await engine.transaction(async (tx) => {
|
||||
if (existing) await tx.createVersion(slug, txOpts);
|
||||
|
||||
@@ -833,7 +824,7 @@ export async function importFromContent(
|
||||
}
|
||||
|
||||
if (chunks.length > 0) {
|
||||
await tx.upsertChunks(slug, chunks, chunkOpts);
|
||||
await tx.upsertChunks(slug, chunks, txOpts);
|
||||
// v0.41.31: stamp embedding provenance when this import actually
|
||||
// embedded (not --no-embed), so a later model/dims swap is detectable
|
||||
// as stale via embed --stale. The deferred/backfill + per-slug embed
|
||||
@@ -1073,12 +1064,6 @@ export async function importCodeFile(
|
||||
const title = `${relativePath} (${lang})`;
|
||||
const sourceId = opts.sourceId;
|
||||
const txOpts = sourceId ? { sourceId } : undefined;
|
||||
// #1262: write-side embedding column descriptor (rides only on upsertChunks).
|
||||
const chunkWriteColumn = await resolveWriteColumnForEngine(engine);
|
||||
const chunkOpts: { sourceId?: string; embeddingColumn?: ResolvedColumn } | undefined =
|
||||
(sourceId || chunkWriteColumn)
|
||||
? { ...(sourceId && { sourceId }), ...(chunkWriteColumn && { embeddingColumn: chunkWriteColumn }) }
|
||||
: undefined;
|
||||
|
||||
const byteLength = Buffer.byteLength(content, 'utf-8');
|
||||
if (byteLength > MAX_FILE_SIZE) {
|
||||
@@ -1198,7 +1183,7 @@ export async function importCodeFile(
|
||||
await tx.addTag(slug, lang, txOpts);
|
||||
|
||||
if (chunks.length > 0) {
|
||||
await tx.upsertChunks(slug, chunks, chunkOpts);
|
||||
await tx.upsertChunks(slug, chunks, txOpts);
|
||||
// v0.41.31: stamp embedding provenance ONLY when every chunk was
|
||||
// freshly embedded with the current model this call (no reuse-by-hash
|
||||
// carrying old-model vectors). Mixed pages stay unstamped rather than
|
||||
@@ -1347,12 +1332,6 @@ export async function withImportTransaction(
|
||||
): Promise<void> {
|
||||
const sourceId = spec.sourceId ?? 'default';
|
||||
const txOpts = spec.sourceId ? { sourceId: spec.sourceId } : undefined;
|
||||
// #1262: write-side embedding column descriptor (rides only on upsertChunks).
|
||||
const chunkWriteColumn = await resolveWriteColumnForEngine(engine);
|
||||
const chunkOpts: { sourceId?: string; embeddingColumn?: ResolvedColumn } | undefined =
|
||||
(spec.sourceId || chunkWriteColumn)
|
||||
? { ...(spec.sourceId && { sourceId: spec.sourceId }), ...(chunkWriteColumn && { embeddingColumn: chunkWriteColumn }) }
|
||||
: undefined;
|
||||
await engine.transaction(async (tx) => {
|
||||
if (spec.hadExisting) await tx.createVersion(spec.slug, txOpts);
|
||||
await tx.putPage(spec.slug, spec.page, txOpts);
|
||||
@@ -1368,7 +1347,7 @@ export async function withImportTransaction(
|
||||
}
|
||||
if (spec.chunks !== undefined) {
|
||||
if (spec.chunks.length > 0) {
|
||||
await tx.upsertChunks(spec.slug, spec.chunks, chunkOpts);
|
||||
await tx.upsertChunks(spec.slug, spec.chunks, txOpts);
|
||||
} else {
|
||||
await tx.deleteChunks(spec.slug, txOpts);
|
||||
}
|
||||
|
||||
@@ -35,7 +35,6 @@ import { tryAcquireDbLock } from '../../db-lock.ts';
|
||||
import { BudgetTracker, BudgetExhausted } from '../../budget/budget-tracker.ts';
|
||||
import { withBudgetTracker } from '../../ai/gateway.ts';
|
||||
import { embedStaleForSource } from '../../embed-stale.ts';
|
||||
import { resolveWriteColumnForEngine } from '../../search/embedding-column.ts';
|
||||
import { currentEmbeddingSignature } from '../../embedding.ts';
|
||||
import { type DbPacer, createDbPacer, createNoopPacer } from '../../db-pacer.ts';
|
||||
import { resolvePaceMode, loadPaceModeConfig, readPaceEnv } from '../../pace-mode.ts';
|
||||
@@ -165,16 +164,12 @@ export function makeEmbedBackfillHandler(engine: BrainEngine) {
|
||||
// the supervisor, so pacing it is the headline win.
|
||||
const { pacer, concurrency } = await resolveBackfillPacer(engine, job.data);
|
||||
|
||||
// #1262: resolve the write-side embedding column once at the job boundary.
|
||||
const embeddingColumn = await resolveWriteColumnForEngine(engine);
|
||||
|
||||
try {
|
||||
const result = await withBudgetTracker(tracker, async () =>
|
||||
embedStaleForSource(engine, sourceId, {
|
||||
batchSize,
|
||||
signal: job.signal,
|
||||
pacer,
|
||||
...(embeddingColumn && { embeddingColumn }),
|
||||
...(concurrency !== undefined && { concurrency }),
|
||||
// v0.41.31: re-embed pages whose model signature drifted + stamp
|
||||
// provenance as chunks land.
|
||||
|
||||
+22
-42
@@ -40,7 +40,6 @@ import type {
|
||||
BrainStats, BrainHealth,
|
||||
IngestLogEntry, IngestLogInput,
|
||||
EngineConfig,
|
||||
ResolvedColumn,
|
||||
EvalCandidate, EvalCandidateInput,
|
||||
EvalCaptureFailure, EvalCaptureFailureReason,
|
||||
SalienceOpts, SalienceResult, AnomaliesOpts, AnomalyResult,
|
||||
@@ -2231,20 +2230,12 @@ export class PGLiteEngine implements BrainEngine {
|
||||
}
|
||||
|
||||
// Chunks
|
||||
async upsertChunks(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string; embeddingColumn?: ResolvedColumn } & BatchOpts): Promise<void> {
|
||||
async upsertChunks(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string } & BatchOpts): Promise<void> {
|
||||
return this.batchRetry(opts?.auditSite ?? 'upsertChunks', opts?.signal, () => this._upsertChunksOnce(slug, chunks, opts), chunks.length);
|
||||
}
|
||||
|
||||
private async _upsertChunksOnce(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string; embeddingColumn?: ResolvedColumn }): Promise<void> {
|
||||
private async _upsertChunksOnce(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string }): Promise<void> {
|
||||
const sourceId = opts?.sourceId ?? 'default';
|
||||
// #1262: caller-resolved write target for TEXT embeddings. Descriptor
|
||||
// names are identifier-validated + quoted by buildVectorCastFragment;
|
||||
// omitted => legacy `embedding vector`. Mirrors postgres-engine.ts.
|
||||
const targetFragment = opts?.embeddingColumn
|
||||
? buildVectorCastFragment(opts.embeddingColumn)
|
||||
: undefined;
|
||||
const targetCol = targetFragment?.col ?? 'embedding';
|
||||
const embeddingCast = targetFragment?.castSql.replace('$1::', '') ?? 'vector';
|
||||
|
||||
// Source-scope the page-id lookup so duplicate slugs in different sources
|
||||
// do not return multiple rows or target the wrong page.
|
||||
@@ -2279,7 +2270,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
// list. Image chunks pass embedding=null + embedding_image=Float32Array
|
||||
// (1024-dim Voyage). Text/code chunks pass embedding=Float32Array +
|
||||
// embedding_image=null. Default modality='text' when omitted.
|
||||
const cols = `(page_id, chunk_index, chunk_text, chunk_source, ${targetCol}, model, token_count, embedded_at, language, symbol_name, symbol_type, start_line, end_line, parent_symbol_path, doc_comment, symbol_name_qualified, modality, embedding_image)`;
|
||||
const cols = '(page_id, chunk_index, chunk_text, chunk_source, embedding, model, token_count, embedded_at, language, symbol_name, symbol_type, start_line, end_line, parent_symbol_path, doc_comment, symbol_name_qualified, modality, embedding_image)';
|
||||
const rowParts: string[] = [];
|
||||
const params: unknown[] = [];
|
||||
let paramIdx = 1;
|
||||
@@ -2297,7 +2288,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
const modality = chunk.modality ?? 'text';
|
||||
|
||||
// Inline ::vector NULL literals to avoid a per-branch placeholder.
|
||||
const embeddingPh = embeddingStr ? `$${paramIdx++}::${embeddingCast}` : 'NULL';
|
||||
const embeddingPh = embeddingStr ? `$${paramIdx++}::vector` : 'NULL';
|
||||
const embeddedAtPh = embeddingStr ? 'now()' : 'NULL';
|
||||
const embeddingImagePh = embeddingImageStr ? `$${paramIdx++}::vector` : 'NULL';
|
||||
|
||||
@@ -2336,19 +2327,19 @@ export class PGLiteEngine implements BrainEngine {
|
||||
ON CONFLICT (page_id, chunk_index) DO UPDATE SET
|
||||
chunk_text = EXCLUDED.chunk_text,
|
||||
chunk_source = EXCLUDED.chunk_source,
|
||||
${targetCol} = CASE
|
||||
WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.${targetCol}
|
||||
WHEN content_chunks.${targetCol} IS NULL THEN EXCLUDED.${targetCol}
|
||||
embedding = CASE
|
||||
WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.embedding
|
||||
WHEN content_chunks.embedding IS NULL THEN EXCLUDED.embedding
|
||||
WHEN EXCLUDED.embedded_at IS NOT NULL
|
||||
AND (content_chunks.embedded_at IS NULL OR EXCLUDED.embedded_at > content_chunks.embedded_at)
|
||||
THEN EXCLUDED.${targetCol}
|
||||
ELSE content_chunks.${targetCol}
|
||||
THEN EXCLUDED.embedding
|
||||
ELSE content_chunks.embedding
|
||||
END,
|
||||
model = COALESCE(EXCLUDED.model, content_chunks.model),
|
||||
token_count = EXCLUDED.token_count,
|
||||
embedded_at = CASE
|
||||
WHEN EXCLUDED.chunk_text != content_chunks.chunk_text AND EXCLUDED.${targetCol} IS NULL THEN NULL
|
||||
WHEN content_chunks.${targetCol} IS NULL AND EXCLUDED.${targetCol} IS NOT NULL THEN EXCLUDED.embedded_at
|
||||
WHEN EXCLUDED.chunk_text != content_chunks.chunk_text AND EXCLUDED.embedding IS NULL THEN NULL
|
||||
WHEN content_chunks.embedding IS NULL AND EXCLUDED.embedding IS NOT NULL THEN EXCLUDED.embedded_at
|
||||
WHEN EXCLUDED.embedded_at IS NOT NULL
|
||||
AND (content_chunks.embedded_at IS NULL OR EXCLUDED.embedded_at > content_chunks.embedded_at)
|
||||
THEN EXCLUDED.embedded_at
|
||||
@@ -2386,19 +2377,14 @@ export class PGLiteEngine implements BrainEngine {
|
||||
* drift (NULL grandfathered → never stale). Shared by countStaleChunks +
|
||||
* sumStaleChunkChars so they can't drift.
|
||||
*/
|
||||
private buildStaleChunkWhere(opts?: { sourceId?: string; signature?: string; embeddingColumn?: ResolvedColumn }): { where: string; params: unknown[] } {
|
||||
// #1262: staleness targets the caller-resolved write column when set
|
||||
// (identifier-validated + quoted); legacy `embedding` otherwise.
|
||||
const staleCol = opts?.embeddingColumn
|
||||
? buildVectorCastFragment(opts.embeddingColumn).col
|
||||
: 'embedding';
|
||||
private buildStaleChunkWhere(opts?: { sourceId?: string; signature?: string }): { where: string; params: unknown[] } {
|
||||
const params: unknown[] = [];
|
||||
const conds: string[] = [];
|
||||
if (opts?.signature !== undefined) {
|
||||
params.push(opts.signature);
|
||||
conds.push(`(cc.${staleCol} IS NULL OR (p.embedding_signature IS NOT NULL AND p.embedding_signature <> $${params.length}))`);
|
||||
conds.push(`(cc.embedding IS NULL OR (p.embedding_signature IS NOT NULL AND p.embedding_signature <> $${params.length}))`);
|
||||
} else {
|
||||
conds.push(`cc.${staleCol} IS NULL`);
|
||||
conds.push(`cc.embedding IS NULL`);
|
||||
}
|
||||
conds.push(`NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')`);
|
||||
if (opts?.sourceId !== undefined) {
|
||||
@@ -2408,7 +2394,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
return { where: conds.join(' AND '), params };
|
||||
}
|
||||
|
||||
async countStaleChunks(opts?: { sourceId?: string; signature?: string; embeddingColumn?: ResolvedColumn }): Promise<number> {
|
||||
async countStaleChunks(opts?: { sourceId?: string; signature?: string }): Promise<number> {
|
||||
// D7: source-scoped count for `gbrain embed --stale --source X`. Always
|
||||
// JOIN pages so embed-skip + signature predicates apply. PGLite is
|
||||
// PostgreSQL 17.5 in WASM and supports the full JSONB operator set.
|
||||
@@ -2424,7 +2410,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
return Number(count);
|
||||
}
|
||||
|
||||
async sumStaleChunkChars(opts?: { sourceId?: string; signature?: string; embeddingColumn?: ResolvedColumn }): Promise<number> {
|
||||
async sumStaleChunkChars(opts?: { sourceId?: string; signature?: string }): Promise<number> {
|
||||
// Sibling of countStaleChunks: same stale predicate, summing chunk_text
|
||||
// length for the sync cost preview. ::bigint guards int4 overflow.
|
||||
const { where, params } = this.buildStaleChunkWhere(opts);
|
||||
@@ -2477,17 +2463,11 @@ export class PGLiteEngine implements BrainEngine {
|
||||
sourceId?: string;
|
||||
orderBy?: 'page_id' | 'updated_desc';
|
||||
afterUpdatedAt?: string | null;
|
||||
embeddingColumn?: ResolvedColumn;
|
||||
}): Promise<StaleChunkRow[]> {
|
||||
const limit = opts?.batchSize ?? 2000;
|
||||
const afterPid = opts?.afterPageId ?? 0;
|
||||
const afterIdx = opts?.afterChunkIndex ?? -1;
|
||||
const orderBy = opts?.orderBy ?? 'page_id';
|
||||
// #1262: staleness follows the caller-resolved write column (validated +
|
||||
// quoted identifier); legacy `embedding` otherwise.
|
||||
const staleCol = opts?.embeddingColumn
|
||||
? buildVectorCastFragment(opts.embeddingColumn).col
|
||||
: 'embedding';
|
||||
|
||||
// v0.41.18.0 (A13, codex #9): --priority recent path. See postgres-engine
|
||||
// sibling for full rationale. Same composite cursor + ORDER BY.
|
||||
@@ -2501,7 +2481,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
p.updated_at
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE cc.${staleCol} IS NULL
|
||||
WHERE cc.embedding IS NULL
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
ORDER BY p.updated_at DESC NULLS LAST, p.id ASC, cc.chunk_index ASC
|
||||
LIMIT $1`,
|
||||
@@ -2512,7 +2492,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
p.updated_at
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE cc.${staleCol} IS NULL
|
||||
WHERE cc.embedding IS NULL
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
AND (
|
||||
p.updated_at < $1::timestamptz
|
||||
@@ -2531,7 +2511,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
p.updated_at
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE cc.${staleCol} IS NULL
|
||||
WHERE cc.embedding IS NULL
|
||||
AND p.source_id = $1
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
ORDER BY p.updated_at DESC NULLS LAST, p.id ASC, cc.chunk_index ASC
|
||||
@@ -2543,7 +2523,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
p.updated_at
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE cc.${staleCol} IS NULL
|
||||
WHERE cc.embedding IS NULL
|
||||
AND p.source_id = $1
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
AND (
|
||||
@@ -2568,7 +2548,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
cc.model, cc.token_count, p.source_id, cc.page_id
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE cc.${staleCol} IS NULL
|
||||
WHERE cc.embedding IS NULL
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
AND (cc.page_id, cc.chunk_index) > ($1, $2)
|
||||
ORDER BY cc.page_id, cc.chunk_index
|
||||
@@ -2582,7 +2562,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
cc.model, cc.token_count, p.source_id, cc.page_id
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE cc.${staleCol} IS NULL
|
||||
WHERE cc.embedding IS NULL
|
||||
AND p.source_id = $1
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
AND (cc.page_id, cc.chunk_index) > ($2, $3)
|
||||
|
||||
+22
-43
@@ -50,7 +50,6 @@ import type {
|
||||
BrainStats, BrainHealth,
|
||||
IngestLogEntry, IngestLogInput,
|
||||
EngineConfig,
|
||||
ResolvedColumn,
|
||||
EvalCandidate, EvalCandidateInput,
|
||||
EvalCaptureFailure, EvalCaptureFailureReason,
|
||||
SalienceOpts, SalienceResult, AnomaliesOpts, AnomalyResult,
|
||||
@@ -2381,21 +2380,13 @@ export class PostgresEngine implements BrainEngine {
|
||||
}
|
||||
|
||||
// Chunks
|
||||
async upsertChunks(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string; embeddingColumn?: ResolvedColumn } & BatchOpts): Promise<void> {
|
||||
async upsertChunks(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string } & BatchOpts): Promise<void> {
|
||||
return this.batchRetry(opts?.auditSite ?? 'upsertChunks', opts?.signal, () => this._upsertChunksOnce(slug, chunks, opts), chunks.length);
|
||||
}
|
||||
|
||||
private async _upsertChunksOnce(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string; embeddingColumn?: ResolvedColumn }): Promise<void> {
|
||||
private async _upsertChunksOnce(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string }): Promise<void> {
|
||||
const sql = this.sql;
|
||||
const sourceId = opts?.sourceId ?? 'default';
|
||||
// #1262: caller-resolved write target for TEXT embeddings. Descriptor
|
||||
// names are identifier-validated + quoted by buildVectorCastFragment;
|
||||
// omitted => legacy `embedding vector`.
|
||||
const targetFragment = opts?.embeddingColumn
|
||||
? buildVectorCastFragment(opts.embeddingColumn)
|
||||
: undefined;
|
||||
const targetCol = targetFragment?.col ?? 'embedding';
|
||||
const embeddingCast = targetFragment?.castSql.replace('$1::', '') ?? 'vector';
|
||||
|
||||
// Source-scope the page-id lookup. Without this filter, multi-source
|
||||
// brains where the slug exists in 2+ sources return >1 row and the
|
||||
@@ -2422,7 +2413,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
// scope metadata through upserts.
|
||||
// v0.27.1 (Phase 8): added `modality` + `embedding_image` to the column
|
||||
// list. Image chunks pass embedding=null + embedding_image=Float32Array.
|
||||
const cols = `(page_id, chunk_index, chunk_text, chunk_source, ${targetCol}, model, token_count, embedded_at, language, symbol_name, symbol_type, start_line, end_line, parent_symbol_path, doc_comment, symbol_name_qualified, modality, embedding_image)`;
|
||||
const cols = '(page_id, chunk_index, chunk_text, chunk_source, embedding, model, token_count, embedded_at, language, symbol_name, symbol_type, start_line, end_line, parent_symbol_path, doc_comment, symbol_name_qualified, modality, embedding_image)';
|
||||
const rows: string[] = [];
|
||||
const params: unknown[] = [];
|
||||
let paramIdx = 1;
|
||||
@@ -2439,7 +2430,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
: null;
|
||||
const modality = chunk.modality ?? 'text';
|
||||
|
||||
const embeddingPh = embeddingStr ? `$${paramIdx++}::${embeddingCast}` : 'NULL';
|
||||
const embeddingPh = embeddingStr ? `$${paramIdx++}::vector` : 'NULL';
|
||||
const embeddedAtPh = embeddingStr ? 'now()' : 'NULL';
|
||||
const embeddingImagePh = embeddingImageStr ? `$${paramIdx++}::vector` : 'NULL';
|
||||
|
||||
@@ -2487,19 +2478,19 @@ export class PostgresEngine implements BrainEngine {
|
||||
ON CONFLICT (page_id, chunk_index) DO UPDATE SET
|
||||
chunk_text = EXCLUDED.chunk_text,
|
||||
chunk_source = EXCLUDED.chunk_source,
|
||||
${targetCol} = CASE
|
||||
WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.${targetCol}
|
||||
WHEN content_chunks.${targetCol} IS NULL THEN EXCLUDED.${targetCol}
|
||||
embedding = CASE
|
||||
WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.embedding
|
||||
WHEN content_chunks.embedding IS NULL THEN EXCLUDED.embedding
|
||||
WHEN EXCLUDED.embedded_at IS NOT NULL
|
||||
AND (content_chunks.embedded_at IS NULL OR EXCLUDED.embedded_at > content_chunks.embedded_at)
|
||||
THEN EXCLUDED.${targetCol}
|
||||
ELSE content_chunks.${targetCol}
|
||||
THEN EXCLUDED.embedding
|
||||
ELSE content_chunks.embedding
|
||||
END,
|
||||
model = COALESCE(EXCLUDED.model, content_chunks.model),
|
||||
token_count = EXCLUDED.token_count,
|
||||
embedded_at = CASE
|
||||
WHEN EXCLUDED.chunk_text != content_chunks.chunk_text AND EXCLUDED.${targetCol} IS NULL THEN NULL
|
||||
WHEN content_chunks.${targetCol} IS NULL AND EXCLUDED.${targetCol} IS NOT NULL THEN EXCLUDED.embedded_at
|
||||
WHEN EXCLUDED.chunk_text != content_chunks.chunk_text AND EXCLUDED.embedding IS NULL THEN NULL
|
||||
WHEN content_chunks.embedding IS NULL AND EXCLUDED.embedding IS NOT NULL THEN EXCLUDED.embedded_at
|
||||
WHEN EXCLUDED.embedded_at IS NOT NULL
|
||||
AND (content_chunks.embedded_at IS NULL OR EXCLUDED.embedded_at > content_chunks.embedded_at)
|
||||
THEN EXCLUDED.embedded_at
|
||||
@@ -2539,19 +2530,14 @@ export class PostgresEngine implements BrainEngine {
|
||||
* embedding_signature drift (NULL grandfathered). Shared by
|
||||
* countStaleChunks + sumStaleChunkChars (parity with the PGLite sibling).
|
||||
*/
|
||||
private buildStaleChunkWhere(opts?: { sourceId?: string; signature?: string; embeddingColumn?: ResolvedColumn }): { where: string; params: unknown[] } {
|
||||
// #1262: staleness targets the caller-resolved write column when set
|
||||
// (identifier-validated + quoted); legacy `embedding` otherwise.
|
||||
const staleCol = opts?.embeddingColumn
|
||||
? buildVectorCastFragment(opts.embeddingColumn).col
|
||||
: 'embedding';
|
||||
private buildStaleChunkWhere(opts?: { sourceId?: string; signature?: string }): { where: string; params: unknown[] } {
|
||||
const params: unknown[] = [];
|
||||
const conds: string[] = [];
|
||||
if (opts?.signature !== undefined) {
|
||||
params.push(opts.signature);
|
||||
conds.push(`(cc.${staleCol} IS NULL OR (p.embedding_signature IS NOT NULL AND p.embedding_signature <> $${params.length}))`);
|
||||
conds.push(`(cc.embedding IS NULL OR (p.embedding_signature IS NOT NULL AND p.embedding_signature <> $${params.length}))`);
|
||||
} else {
|
||||
conds.push(`cc.${staleCol} IS NULL`);
|
||||
conds.push(`cc.embedding IS NULL`);
|
||||
}
|
||||
conds.push(`NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')`);
|
||||
if (opts?.sourceId !== undefined) {
|
||||
@@ -2561,7 +2547,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
return { where: conds.join(' AND '), params };
|
||||
}
|
||||
|
||||
async countStaleChunks(opts?: { sourceId?: string; signature?: string; embeddingColumn?: ResolvedColumn }): Promise<number> {
|
||||
async countStaleChunks(opts?: { sourceId?: string; signature?: string }): Promise<number> {
|
||||
// Always JOIN pages so the embed_skip + signature predicates apply.
|
||||
// D7: source_id scoping. v0.41.31: optional signature widens staleness
|
||||
// to embedding_signature drift (NULL grandfathered).
|
||||
@@ -2579,7 +2565,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
});
|
||||
}
|
||||
|
||||
async sumStaleChunkChars(opts?: { sourceId?: string; signature?: string; embeddingColumn?: ResolvedColumn }): Promise<number> {
|
||||
async sumStaleChunkChars(opts?: { sourceId?: string; signature?: string }): Promise<number> {
|
||||
// Sibling of countStaleChunks: same stale predicate, summing chunk_text
|
||||
// length for the sync cost preview. ::bigint guards int4 overflow.
|
||||
const { where, params } = this.buildStaleChunkWhere(opts);
|
||||
@@ -2632,18 +2618,11 @@ export class PostgresEngine implements BrainEngine {
|
||||
sourceId?: string;
|
||||
orderBy?: 'page_id' | 'updated_desc';
|
||||
afterUpdatedAt?: string | null;
|
||||
embeddingColumn?: ResolvedColumn;
|
||||
}): Promise<StaleChunkRow[]> {
|
||||
const limit = opts?.batchSize ?? 2000;
|
||||
const afterPid = opts?.afterPageId ?? 0;
|
||||
const afterIdx = opts?.afterChunkIndex ?? -1;
|
||||
const orderBy = opts?.orderBy ?? 'page_id';
|
||||
// #1262: staleness follows the caller-resolved write column (validated +
|
||||
// quoted identifier); legacy `embedding` otherwise. Interpolated below as
|
||||
// an unsafe FRAGMENT (identifiers can't be bound parameters).
|
||||
const staleCol = opts?.embeddingColumn
|
||||
? buildVectorCastFragment(opts.embeddingColumn).col
|
||||
: 'embedding';
|
||||
|
||||
// RLS scope binding (opt-in via GBRAIN_RLS_SCOPE_BINDING).
|
||||
return await this.withScopedReadTransaction(undefined, opts?.sourceId, async (tx) => {
|
||||
@@ -2660,7 +2639,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
p.updated_at
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE ${tx.unsafe(`cc.${staleCol} IS NULL`)}
|
||||
WHERE cc.embedding IS NULL
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
ORDER BY p.updated_at DESC NULLS LAST, p.id ASC, cc.chunk_index ASC
|
||||
LIMIT ${limit}
|
||||
@@ -2670,7 +2649,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
p.updated_at
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE ${tx.unsafe(`cc.${staleCol} IS NULL`)}
|
||||
WHERE cc.embedding IS NULL
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
AND (
|
||||
p.updated_at < ${afterUpdated}::timestamptz
|
||||
@@ -2688,7 +2667,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
p.updated_at
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE ${tx.unsafe(`cc.${staleCol} IS NULL`)}
|
||||
WHERE cc.embedding IS NULL
|
||||
AND p.source_id = ${opts.sourceId}
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
ORDER BY p.updated_at DESC NULLS LAST, p.id ASC, cc.chunk_index ASC
|
||||
@@ -2699,7 +2678,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
p.updated_at
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE ${tx.unsafe(`cc.${staleCol} IS NULL`)}
|
||||
WHERE cc.embedding IS NULL
|
||||
AND p.source_id = ${opts.sourceId}
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
AND (
|
||||
@@ -2719,7 +2698,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
cc.model, cc.token_count, p.source_id, cc.page_id
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE ${tx.unsafe(`cc.${staleCol} IS NULL`)}
|
||||
WHERE cc.embedding IS NULL
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
AND (cc.page_id, cc.chunk_index) > (${afterPid}, ${afterIdx})
|
||||
ORDER BY cc.page_id, cc.chunk_index
|
||||
@@ -2732,7 +2711,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
cc.model, cc.token_count, p.source_id, cc.page_id
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE ${tx.unsafe(`cc.${staleCol} IS NULL`)}
|
||||
WHERE cc.embedding IS NULL
|
||||
AND p.source_id = ${opts.sourceId}
|
||||
AND NOT (COALESCE(p.frontmatter, '{}'::jsonb) ? 'embed_skip')
|
||||
AND (cc.page_id, cc.chunk_index) > (${afterPid}, ${afterIdx})
|
||||
|
||||
@@ -443,80 +443,6 @@ export function resolveEmbeddingColumn(
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolves the WRITE-side embedding column for the currently configured
|
||||
* embedding model (#1262). The read-side resolver above answers "which
|
||||
* column does this query search?"; this one answers "which column should
|
||||
* newly produced text embeddings land in?".
|
||||
*
|
||||
* Unlike read-side search, writes take no per-call column override. The
|
||||
* import/embed boundary resolves once from merged config + gateway state
|
||||
* and passes the descriptor into `engine.upsertChunks`; engines stay
|
||||
* config-free (same contract as the read-side descriptor).
|
||||
*
|
||||
* Behavior:
|
||||
* - no user-declared `embedding_columns` => undefined (legacy brain,
|
||||
* writes keep targeting the default `embedding` column)
|
||||
* - a user-declared entry whose `provider` matches the current
|
||||
* embedding model => that entry's descriptor
|
||||
* - no provider match => undefined (fall back to legacy `embedding`)
|
||||
*
|
||||
* Only USER-declared entries are consulted — never the cfg-derived
|
||||
* builtins. The `embedding_image` builtin's provider is the multimodal
|
||||
* model; matching it here would misroute text embeddings into the image
|
||||
* column. The no-match fallback is intentional: switching models before
|
||||
* registering a matching column must not silently write vectors into an
|
||||
* arbitrary column.
|
||||
*/
|
||||
export function resolveWriteColumn(cfg: GBrainConfig): ResolvedColumn | undefined {
|
||||
const userColumns = cfg.embedding_columns;
|
||||
if (
|
||||
!userColumns ||
|
||||
typeof userColumns !== 'object' ||
|
||||
Array.isArray(userColumns) ||
|
||||
Object.keys(userColumns).length === 0
|
||||
) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
// Same model-resolution chain as the registry builtin: cfg > gateway > default.
|
||||
let gwModel: string | undefined;
|
||||
try {
|
||||
const gw = require('../ai/gateway.ts') as typeof import('../ai/gateway.ts');
|
||||
gwModel = gw.getEmbeddingModel();
|
||||
} catch {
|
||||
// Gateway unconfigured — fall through to the canonical default.
|
||||
}
|
||||
const currentModel = cfg.embedding_model ?? gwModel ?? DEFAULT_EMBEDDING_MODEL;
|
||||
|
||||
for (const [name, entry] of Object.entries(userColumns)) {
|
||||
if (!entry) continue;
|
||||
validateColumnKey(name);
|
||||
validateColumnConfig(name, entry);
|
||||
if (entry.provider !== currentModel) continue;
|
||||
return {
|
||||
name,
|
||||
type: entry.type,
|
||||
dimensions: entry.dimensions,
|
||||
embeddingModel: entry.provider,
|
||||
};
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Engine-boundary convenience: merged config (file/env + DB plane) →
|
||||
* resolveWriteColumn. Dynamic import keeps config.ts out of this module's
|
||||
* static graph (mirrors the gateway require above).
|
||||
*/
|
||||
export async function resolveWriteColumnForEngine(
|
||||
engine: { getConfig(key: string): Promise<string | null | undefined> },
|
||||
): Promise<ResolvedColumn | undefined> {
|
||||
const { loadConfigWithEngine } = await import('../config.ts');
|
||||
const cfg = await loadConfigWithEngine(engine);
|
||||
return cfg ? resolveWriteColumn(cfg) : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the resolved column is the default `embedding` name.
|
||||
* Name-based check; does not compare embedding space.
|
||||
|
||||
@@ -39,6 +39,8 @@ import {
|
||||
__getShrinkStateForTests,
|
||||
} from '../../src/core/ai/gateway.ts';
|
||||
import { AIConfigError, AITransientError } from '../../src/core/ai/errors.ts';
|
||||
import { RECIPES } from '../../src/core/ai/recipes/index.ts';
|
||||
import type { Recipe } from '../../src/core/ai/types.ts';
|
||||
|
||||
// The last test in this file leaves the gateway configured with a remote
|
||||
// provider + fake key and a REAL embed transport. Without a final reset,
|
||||
@@ -93,6 +95,14 @@ function configureGoogle(): void {
|
||||
});
|
||||
}
|
||||
|
||||
function configureDashscope(): void {
|
||||
configureGateway({
|
||||
embedding_model: 'dashscope:text-embedding-v3',
|
||||
embedding_dimensions: 1024,
|
||||
env: { DASHSCOPE_API_KEY: 'sk-fake' },
|
||||
});
|
||||
}
|
||||
|
||||
// --------- 1. Pure helpers ---------
|
||||
|
||||
describe('splitByTokenBudget (pure helper)', () => {
|
||||
@@ -149,6 +159,27 @@ describe('splitByTokenBudget (pure helper)', () => {
|
||||
expect(splitByTokenBudget(texts, 96_000, 0)).toEqual(splitByTokenBudget(texts, 96_000, 4));
|
||||
expect(splitByTokenBudget(texts, 96_000, -1)).toEqual(splitByTokenBudget(texts, 96_000, 4));
|
||||
});
|
||||
|
||||
// #1199: count cap for providers that reject batches by input count.
|
||||
test('max_batch_count flushes even when token budget has room', () => {
|
||||
const texts = Array.from({ length: 25 }, (_, i) => `t${i}`);
|
||||
const result = splitByTokenBudget(texts, 1_000_000, 4, 10);
|
||||
expect(result.map(b => b.length)).toEqual([10, 10, 5]);
|
||||
expect(result.flat()).toEqual(texts);
|
||||
});
|
||||
|
||||
test('token budget still governs alongside max_batch_count', () => {
|
||||
const texts = ['a'.repeat(50_000), 'b'.repeat(50_000), 'c'.repeat(50_000)];
|
||||
const result = splitByTokenBudget(texts, 96_000, 1, 10);
|
||||
expect(result).toHaveLength(3);
|
||||
});
|
||||
|
||||
test('undefined / zero / negative max_batch_count is ignored', () => {
|
||||
const texts = Array.from({ length: 25 }, () => 'x');
|
||||
expect(splitByTokenBudget(texts, 1_000_000, 4, undefined)).toHaveLength(1);
|
||||
expect(splitByTokenBudget(texts, 1_000_000, 4, 0)).toHaveLength(1);
|
||||
expect(splitByTokenBudget(texts, 1_000_000, 4, -5)).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe('isTokenLimitError (pure helper)', () => {
|
||||
@@ -179,6 +210,12 @@ describe('isTokenLimitError (pure helper)', () => {
|
||||
expect(isTokenLimitError(new Error('Exceeded 300000 max tokens per request'))).toBe(true);
|
||||
});
|
||||
|
||||
test('matches DashScope batch-count error (#1199)', () => {
|
||||
expect(isTokenLimitError(new Error(
|
||||
'InvalidParameter: batch size is invalid, it should not be larger than 10.',
|
||||
))).toBe(true);
|
||||
});
|
||||
|
||||
test('does not match unrelated errors', () => {
|
||||
expect(isTokenLimitError(new Error('Connection refused'))).toBe(false);
|
||||
expect(isTokenLimitError(new Error('Invalid API key'))).toBe(false);
|
||||
@@ -387,26 +424,92 @@ describe('shrink-on-miss adaptive cache', () => {
|
||||
});
|
||||
});
|
||||
|
||||
// --------- 8. Pre-split count cap through public embed() (#1199 / #970) ---------
|
||||
|
||||
describe('embed() pre-split honors max_batch_count', () => {
|
||||
beforeEach(() => resetGateway());
|
||||
afterEach(() => __setEmbedTransportForTests(null));
|
||||
|
||||
test('dashscope never dispatches more than 10 inputs per call (#1199)', async () => {
|
||||
configureDashscope();
|
||||
const stub = mock(async ({ values }: { values: string[] }) => fakeEmbeddings(values, 1024));
|
||||
__setEmbedTransportForTests(stub as any);
|
||||
|
||||
// 25 short texts fit trivially in the 8192-token budget; without the
|
||||
// count cap they'd ship as ONE batch and DashScope would reject it.
|
||||
const texts = Array.from({ length: 25 }, (_, i) => `short-${i}`);
|
||||
const result = await embed(texts);
|
||||
|
||||
expect(result).toHaveLength(25);
|
||||
const callLengths = stub.mock.calls.map(([arg]) => (arg as { values: string[] }).values.length);
|
||||
expect(Math.max(...callLengths)).toBeLessThanOrEqual(10);
|
||||
expect(callLengths.reduce((a, b) => a + b, 0)).toBe(25);
|
||||
// Order preserved across sub-batches.
|
||||
expect((stub.mock.calls[0][0] as { values: string[] }).values[0]).toBe('short-0');
|
||||
});
|
||||
|
||||
test('google pre-splits at 100 inputs per batchEmbedContents call (#970)', async () => {
|
||||
configureGoogle();
|
||||
const stub = mock(async ({ values }: { values: string[] }) => fakeEmbeddings(values, 768));
|
||||
__setEmbedTransportForTests(stub as any);
|
||||
|
||||
const texts = Array.from({ length: 250 }, (_, i) => `g${i}`);
|
||||
const result = await embed(texts);
|
||||
|
||||
expect(result).toHaveLength(250);
|
||||
const callLengths = stub.mock.calls.map(([arg]) => (arg as { values: string[] }).values.length);
|
||||
expect(callLengths).toEqual([100, 100, 50]);
|
||||
});
|
||||
});
|
||||
|
||||
// --------- 7. Startup warning (D9-B) ---------
|
||||
|
||||
describe('startup warning for recipes missing max_batch_tokens', () => {
|
||||
beforeEach(() => resetGateway());
|
||||
|
||||
// #970 closed google's missing cap, so no registered recipe is capless
|
||||
// anymore. Inject a synthetic capless recipe to keep the warning path
|
||||
// covered for the NEXT recipe that forgets the field.
|
||||
const caplessRecipe: Recipe = {
|
||||
id: 'capless-test',
|
||||
name: 'Capless Test Provider',
|
||||
tier: 'openai-compat',
|
||||
implementation: 'openai-compatible',
|
||||
base_url_default: 'https://example.invalid/v1',
|
||||
auth_env: { required: [] },
|
||||
touchpoints: {
|
||||
embedding: { models: ['capless-embed-1'], default_dims: 768 },
|
||||
},
|
||||
};
|
||||
|
||||
function configureCapless(): void {
|
||||
configureGateway({
|
||||
embedding_model: 'capless-test:capless-embed-1',
|
||||
embedding_dimensions: 768,
|
||||
env: {},
|
||||
});
|
||||
}
|
||||
|
||||
test('configured missing-cap recipe warns once; unrelated recipes stay quiet', () => {
|
||||
const warnings: string[] = [];
|
||||
const original = console.warn;
|
||||
console.warn = (msg: string) => warnings.push(String(msg));
|
||||
RECIPES.set(caplessRecipe.id, caplessRecipe);
|
||||
try {
|
||||
configureOpenAI();
|
||||
expect(warnings.length).toBe(0);
|
||||
// #970 regression: google now declares max_batch_tokens → quiet.
|
||||
configureGoogle();
|
||||
expect(warnings.length).toBe(0);
|
||||
configureCapless();
|
||||
const firstCallCount = warnings.length;
|
||||
// Reconfigure: the warning should NOT re-fire for the same recipes
|
||||
// within one process (we already told the operator).
|
||||
configureGoogle();
|
||||
configureCapless();
|
||||
expect(warnings.length).toBe(firstCallCount);
|
||||
} finally {
|
||||
console.warn = original;
|
||||
RECIPES.delete(caplessRecipe.id);
|
||||
}
|
||||
|
||||
// The warning text should match the documented contract.
|
||||
@@ -415,11 +518,12 @@ describe('startup warning for recipes missing max_batch_tokens', () => {
|
||||
);
|
||||
expect(contractMatch.length).toBe(1);
|
||||
|
||||
// Voyage declares max_batch_tokens → suppressed. OpenAI is the
|
||||
// canonical fast-path recipe → also suppressed by id. Both must be
|
||||
// absent from the warnings.
|
||||
// Voyage + google declare max_batch_tokens → suppressed. OpenAI is the
|
||||
// canonical fast-path recipe → also suppressed by id. All must be
|
||||
// absent from the warnings; only the synthetic capless recipe fires.
|
||||
expect(warnings.find(w => w.includes('"voyage"'))).toBeUndefined();
|
||||
expect(warnings.find(w => w.includes('"openai"'))).toBeUndefined();
|
||||
expect(warnings.find(w => w.includes('"google"'))).toBeDefined();
|
||||
expect(warnings.find(w => w.includes('"google"'))).toBeUndefined();
|
||||
expect(warnings.find(w => w.includes('"capless-test"'))).toBeDefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -52,16 +52,7 @@ describe('v0.32 #779: no_batch_cap suppresses the missing-max_batch_tokens warni
|
||||
}
|
||||
});
|
||||
|
||||
test('configureGateway warns for google only when google embedding is configured', () => {
|
||||
warnSpy.mockClear();
|
||||
resetGateway();
|
||||
configureGateway({ env: {} });
|
||||
let messages = warnSpy.mock.calls.map(c => String(c[0] ?? ''));
|
||||
expect(
|
||||
messages.some(m => m.includes('"google"') && m.includes('without max_batch_tokens')),
|
||||
'google should not warn while OpenAI default is configured',
|
||||
).toBe(false);
|
||||
|
||||
test('configureGateway does NOT warn for google now that it declares batch caps (#970)', () => {
|
||||
warnSpy.mockClear();
|
||||
resetGateway();
|
||||
configureGateway({
|
||||
@@ -69,11 +60,20 @@ describe('v0.32 #779: no_batch_cap suppresses the missing-max_batch_tokens warni
|
||||
embedding_dimensions: 768,
|
||||
env: { GOOGLE_GENERATIVE_AI_API_KEY: 'fake' },
|
||||
});
|
||||
messages = warnSpy.mock.calls.map(c => String(c[0] ?? ''));
|
||||
const messages = warnSpy.mock.calls.map(c => String(c[0] ?? ''));
|
||||
expect(
|
||||
messages.some(m => m.includes('"google"') && m.includes('without max_batch_tokens')),
|
||||
'google should warn when configured because it has fixed-cap models',
|
||||
).toBe(true);
|
||||
'google declares max_batch_tokens/max_batch_count since #970 — no warning',
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
test('google recipe declares its derived batch caps (#970)', () => {
|
||||
const e = getRecipe('google')!.touchpoints.embedding!;
|
||||
// Count cap is the REAL Gemini limit (batchEmbedContents: 100 inputs);
|
||||
// the token budget is derived (100 × 2048 per-input tokens), NOT the
|
||||
// 2048 per-input limit — copying that verbatim would over-split 50×.
|
||||
expect(e.max_batch_count).toBe(100);
|
||||
expect(e.max_batch_tokens).toBe(204_800);
|
||||
});
|
||||
|
||||
test('every recipe with empty models[] declares user_provided_models OR has openai-fast-path', () => {
|
||||
|
||||
@@ -55,6 +55,11 @@ describe('recipe: dashscope', () => {
|
||||
expect(r.touchpoints.embedding!.chars_per_token).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
test('declares max_batch_count: 10 — DashScope rejects larger batches (#1199)', () => {
|
||||
const r = getRecipe('dashscope')!;
|
||||
expect(r.touchpoints.embedding!.max_batch_count).toBe(10);
|
||||
});
|
||||
|
||||
test('dimsProviderOptions threads dimensions for text-embedding-v3 (Matryoshka)', async () => {
|
||||
// Codex finding #1: DashScope text-embedding-v3 is Matryoshka 64-1024.
|
||||
// Without `dimensions` on the wire, user-selected non-default dims are
|
||||
|
||||
@@ -241,136 +241,3 @@ describe('buildVectorCastFragment — engine SQL composer (D3)', () => {
|
||||
expect(castSql).toBe('$1::halfvec(2560)');
|
||||
});
|
||||
});
|
||||
|
||||
describe('PGLite engine: upsertChunks write-side ResolvedColumn descriptor (#1262)', () => {
|
||||
test('halfvec descriptor writes the text embedding to the alternate column, not legacy embedding', async () => {
|
||||
await engine.putPage('docs/write-alt-pglite', {
|
||||
type: 'concept',
|
||||
title: 'Write alt column PGLite',
|
||||
compiled_truth: 'PGLite write-side alternate embedding column test.',
|
||||
});
|
||||
|
||||
const descriptor: ResolvedColumn = {
|
||||
name: 'embedding_ze',
|
||||
type: 'halfvec',
|
||||
dimensions: 2560,
|
||||
embeddingModel: 'zeroentropyai:zembed-1',
|
||||
};
|
||||
await engine.upsertChunks('docs/write-alt-pglite', [
|
||||
{
|
||||
chunk_index: 0,
|
||||
chunk_text: 'PGLite write-side alternate embedding column test.',
|
||||
chunk_source: 'compiled_truth',
|
||||
embedding: new Float32Array(2560).fill(0.25),
|
||||
},
|
||||
], { embeddingColumn: descriptor });
|
||||
|
||||
const rows = await engine.executeRaw<{
|
||||
has_default: boolean;
|
||||
has_ze: boolean;
|
||||
has_embedded_at: boolean;
|
||||
}>(
|
||||
`SELECT embedding IS NOT NULL AS has_default,
|
||||
embedding_ze IS NOT NULL AS has_ze,
|
||||
embedded_at IS NOT NULL AS has_embedded_at
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE p.slug = 'docs/write-alt-pglite'`,
|
||||
);
|
||||
expect(rows.length).toBe(1);
|
||||
expect(rows[0].has_default).toBe(false);
|
||||
expect(rows[0].has_ze).toBe(true);
|
||||
expect(rows[0].has_embedded_at).toBe(true);
|
||||
});
|
||||
|
||||
test('text-unchanged re-upsert without a vector preserves the alternate-column embedding', async () => {
|
||||
const descriptor: ResolvedColumn = {
|
||||
name: 'embedding_ze',
|
||||
type: 'halfvec',
|
||||
dimensions: 2560,
|
||||
embeddingModel: 'zeroentropyai:zembed-1',
|
||||
};
|
||||
// Same chunk_text, no embedding: the ON CONFLICT CASE must keep the
|
||||
// existing alternate-column vector (D24 semantics follow the column).
|
||||
await engine.upsertChunks('docs/write-alt-pglite', [
|
||||
{
|
||||
chunk_index: 0,
|
||||
chunk_text: 'PGLite write-side alternate embedding column test.',
|
||||
chunk_source: 'compiled_truth',
|
||||
},
|
||||
], { embeddingColumn: descriptor });
|
||||
const rows = await engine.executeRaw<{ has_ze: boolean }>(
|
||||
`SELECT embedding_ze IS NOT NULL AS has_ze
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE p.slug = 'docs/write-alt-pglite'`,
|
||||
);
|
||||
expect(rows).toEqual([{ has_ze: true }]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('PGLite: embed --stale converges on an alt-column brain (#1262)', () => {
|
||||
test('boundary resolves the write column; stale scan does not re-select embedded rows', async () => {
|
||||
const { runEmbedCore } = await import('../../src/commands/embed.ts');
|
||||
const local = new PGLiteEngine();
|
||||
const previousHome = process.env.GBRAIN_HOME;
|
||||
process.env.GBRAIN_HOME = `/tmp/gbrain-write-col-stale-${Date.now()}`;
|
||||
try {
|
||||
await local.connect({});
|
||||
await local.initSchema();
|
||||
await (local as any).db.exec(
|
||||
`ALTER TABLE content_chunks ADD COLUMN IF NOT EXISTS embedding_ze halfvec(2560)`,
|
||||
);
|
||||
|
||||
const descriptor: ResolvedColumn = {
|
||||
name: 'embedding_ze',
|
||||
type: 'halfvec',
|
||||
dimensions: 2560,
|
||||
embeddingModel: 'zeroentropyai:zembed-1',
|
||||
};
|
||||
await local.setConfig('embedding_columns', JSON.stringify({
|
||||
embedding_ze: { provider: 'zeroentropyai:zembed-1', dimensions: 2560, type: 'halfvec' },
|
||||
}));
|
||||
configureGateway({
|
||||
embedding_model: 'zeroentropyai:zembed-1',
|
||||
embedding_dimensions: 2560,
|
||||
env: {},
|
||||
});
|
||||
|
||||
await local.putPage('docs/stale-alt-pglite', {
|
||||
type: 'concept',
|
||||
title: 'Dynamic stale column',
|
||||
compiled_truth: 'A chunk that is embedded only in the dynamic column.',
|
||||
});
|
||||
await local.upsertChunks('docs/stale-alt-pglite', [
|
||||
{
|
||||
chunk_index: 0,
|
||||
chunk_text: 'A chunk that is embedded only in the dynamic column.',
|
||||
chunk_source: 'compiled_truth',
|
||||
embedding: new Float32Array(2560).fill(0.25),
|
||||
},
|
||||
], { embeddingColumn: descriptor });
|
||||
|
||||
// Engine-level contrast: legacy predicate still sees the row as stale;
|
||||
// the alt-column predicate does not.
|
||||
expect(await local.countStaleChunks()).toBe(1);
|
||||
expect(await local.countStaleChunks({ embeddingColumn: descriptor })).toBe(0);
|
||||
// sumStaleChunkChars feeds the sync cost gate — same predicate contract.
|
||||
expect(await local.sumStaleChunkChars()).toBeGreaterThan(0);
|
||||
expect(await local.sumStaleChunkChars({ embeddingColumn: descriptor })).toBe(0);
|
||||
expect(await local.listStaleChunks({ embeddingColumn: descriptor, batchSize: 100 })).toHaveLength(0);
|
||||
expect(await local.listStaleChunks({ batchSize: 100 })).toHaveLength(1);
|
||||
|
||||
// Boundary-level: `embed --stale --dry-run` resolves the write column
|
||||
// from merged config + gateway and reports NOTHING to embed. Without
|
||||
// the fix this reports 1 (perpetual re-embed loop).
|
||||
const result = await runEmbedCore(local, { stale: true, dryRun: true });
|
||||
expect(result.would_embed).toBe(0);
|
||||
} finally {
|
||||
await local.disconnect();
|
||||
if (previousHome === undefined) delete process.env.GBRAIN_HOME;
|
||||
else process.env.GBRAIN_HOME = previousHome;
|
||||
resetGateway();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
@@ -224,54 +224,4 @@ if (!dbUrl) {
|
||||
await engine.executeRaw(`UPDATE content_chunks SET embedding_voyage = '${v}'::vector WHERE id = ${dogId}`);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Postgres: upsertChunks write-side ResolvedColumn descriptor (#1262)', () => {
|
||||
const descriptor: ResolvedColumn = {
|
||||
name: 'embedding_ze',
|
||||
type: 'halfvec',
|
||||
dimensions: 2560,
|
||||
embeddingModel: 'zeroentropyai:zembed-1',
|
||||
};
|
||||
|
||||
test('halfvec descriptor writes the text embedding to the alternate column, not legacy embedding', async () => {
|
||||
await engine.putPage('docs/write-alt-postgres', {
|
||||
type: 'concept',
|
||||
title: 'Write alt column Postgres',
|
||||
compiled_truth: 'Postgres write-side alternate embedding column test.',
|
||||
});
|
||||
await engine.upsertChunks('docs/write-alt-postgres', [
|
||||
{
|
||||
chunk_index: 0,
|
||||
chunk_text: 'Postgres write-side alternate embedding column test.',
|
||||
chunk_source: 'compiled_truth',
|
||||
embedding: new Float32Array(2560).fill(0.25),
|
||||
},
|
||||
], { embeddingColumn: descriptor });
|
||||
|
||||
const rows = await engine.executeRaw<{
|
||||
has_default: boolean;
|
||||
has_ze: boolean;
|
||||
}>(
|
||||
`SELECT embedding IS NOT NULL AS has_default,
|
||||
embedding_ze IS NOT NULL AS has_ze
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE p.slug = 'docs/write-alt-postgres'`,
|
||||
);
|
||||
expect(rows.length).toBe(1);
|
||||
expect(rows[0].has_default).toBe(false);
|
||||
expect(rows[0].has_ze).toBe(true);
|
||||
}, 30_000);
|
||||
|
||||
test('stale scan follows the write-side column (count + list parity with the write target)', async () => {
|
||||
// Legacy predicate: cat/dog/write-alt rows all have embedding NULL.
|
||||
expect(await engine.countStaleChunks()).toBeGreaterThan(0);
|
||||
// Alt-column predicate: every chunk has embedding_ze populated.
|
||||
expect(await engine.countStaleChunks({ embeddingColumn: descriptor })).toBe(0);
|
||||
expect(await engine.listStaleChunks({ embeddingColumn: descriptor, batchSize: 100 })).toHaveLength(0);
|
||||
expect((await engine.listStaleChunks({ batchSize: 100 })).length).toBeGreaterThan(0);
|
||||
// updated_desc arm uses the same predicate.
|
||||
expect(await engine.listStaleChunks({ embeddingColumn: descriptor, orderBy: 'updated_desc', batchSize: 100 })).toHaveLength(0);
|
||||
}, 30_000);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
/**
|
||||
* #1818: embedBatch dispatches its 100-input sub-batches through a bounded
|
||||
* worker pool (the embed-stale.ts concurrency pattern) instead of a serial
|
||||
* `for` loop. This file pins:
|
||||
*
|
||||
* - output order matches input order regardless of completion order
|
||||
* (index-addressed results)
|
||||
* - parallelism actually happens (max in-flight > 1) and stays bounded
|
||||
* (max in-flight <= configured concurrency)
|
||||
* - concurrency: 1 restores the serial pre-#1818 dispatch
|
||||
* - GBRAIN_EMBED_BATCH_CONCURRENCY env is honored when the option is unset
|
||||
* - onBatchComplete reports a monotonic completed count ending at total
|
||||
*
|
||||
* Transport is stubbed via the gateway's __setEmbedTransportForTests seam
|
||||
* (same pattern as test/ai/adaptive-embed-batch.test.ts). OpenAI recipe =
|
||||
* fast path (no pre-split), so each embedBatch sub-batch is exactly one
|
||||
* transport call.
|
||||
*/
|
||||
|
||||
import { afterAll, afterEach, beforeEach, describe, expect, test } from 'bun:test';
|
||||
import {
|
||||
configureGateway,
|
||||
resetGateway,
|
||||
__setEmbedTransportForTests,
|
||||
} from '../src/core/ai/gateway.ts';
|
||||
import { embedBatch } from '../src/core/embedding.ts';
|
||||
import { withEnv } from './helpers/with-env.ts';
|
||||
|
||||
const DIMS = 1536;
|
||||
|
||||
function configureOpenAI(): void {
|
||||
configureGateway({
|
||||
embedding_model: 'openai:text-embedding-3-large',
|
||||
embedding_dimensions: DIMS,
|
||||
env: { OPENAI_API_KEY: 'sk-fake' },
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Install a transport whose returned embedding encodes the GLOBAL input
|
||||
* index in dim 0 (texts are `t<N>`), so order can be asserted end-to-end.
|
||||
* Tracks the max number of concurrently in-flight transport calls.
|
||||
*/
|
||||
function installTrackingTransport(delayMs = 5): { maxInFlight: () => number } {
|
||||
let inFlight = 0;
|
||||
let maxInFlight = 0;
|
||||
__setEmbedTransportForTests((async ({ values }: { values: string[] }) => {
|
||||
inFlight++;
|
||||
maxInFlight = Math.max(maxInFlight, inFlight);
|
||||
await new Promise(r => setTimeout(r, delayMs));
|
||||
inFlight--;
|
||||
return {
|
||||
embeddings: values.map(v => {
|
||||
const idx = Number(v.slice(1));
|
||||
return Array.from({ length: DIMS }, (_, j) => (j === 0 ? idx : 0.1));
|
||||
}),
|
||||
};
|
||||
}) as any);
|
||||
return { maxInFlight: () => maxInFlight };
|
||||
}
|
||||
|
||||
const texts = Array.from({ length: 250 }, (_, i) => `t${i}`);
|
||||
|
||||
afterAll(() => resetGateway());
|
||||
|
||||
describe('embedBatch bounded parallelism (#1818)', () => {
|
||||
beforeEach(() => {
|
||||
resetGateway();
|
||||
configureOpenAI();
|
||||
});
|
||||
afterEach(() => {
|
||||
__setEmbedTransportForTests(null);
|
||||
});
|
||||
|
||||
test('default pool dispatches sub-batches in parallel, order preserved', async () => {
|
||||
const tracker = installTrackingTransport();
|
||||
const result = await embedBatch(texts, { onBatchComplete: () => {} });
|
||||
expect(result).toHaveLength(250);
|
||||
for (let i = 0; i < 250; i++) {
|
||||
expect(result[i][0]).toBe(i);
|
||||
}
|
||||
// 250 texts → 3 sub-batches; default concurrency 4 → all 3 in flight.
|
||||
expect(tracker.maxInFlight()).toBeGreaterThan(1);
|
||||
expect(tracker.maxInFlight()).toBeLessThanOrEqual(4);
|
||||
});
|
||||
|
||||
test('concurrency: 1 keeps the serial dispatch', async () => {
|
||||
const tracker = installTrackingTransport();
|
||||
const result = await embedBatch(texts, { concurrency: 1, onBatchComplete: () => {} });
|
||||
expect(result).toHaveLength(250);
|
||||
expect(tracker.maxInFlight()).toBe(1);
|
||||
});
|
||||
|
||||
test('GBRAIN_EMBED_BATCH_CONCURRENCY env bounds the pool when option unset', async () => {
|
||||
const tracker = installTrackingTransport();
|
||||
await withEnv({ GBRAIN_EMBED_BATCH_CONCURRENCY: '2' }, async () => {
|
||||
await embedBatch(texts, { onBatchComplete: () => {} });
|
||||
});
|
||||
expect(tracker.maxInFlight()).toBeGreaterThan(1);
|
||||
expect(tracker.maxInFlight()).toBeLessThanOrEqual(2);
|
||||
});
|
||||
|
||||
test('onBatchComplete reports a monotonic count ending at total', async () => {
|
||||
installTrackingTransport();
|
||||
const seen: number[] = [];
|
||||
await embedBatch(texts, {
|
||||
onBatchComplete: (done, total) => {
|
||||
expect(total).toBe(250);
|
||||
seen.push(done);
|
||||
},
|
||||
});
|
||||
expect(seen).toHaveLength(3); // 100 + 100 + 50 sub-batches
|
||||
for (let i = 1; i < seen.length; i++) {
|
||||
expect(seen[i]).toBeGreaterThan(seen[i - 1]);
|
||||
}
|
||||
expect(seen[seen.length - 1]).toBe(250);
|
||||
});
|
||||
|
||||
test('a failing sub-batch rejects the whole call', async () => {
|
||||
let call = 0;
|
||||
__setEmbedTransportForTests((async ({ values }: { values: string[] }) => {
|
||||
call++;
|
||||
if (call === 2) throw new Error('boom');
|
||||
await new Promise(r => setTimeout(r, 2));
|
||||
return { embeddings: values.map(() => Array.from({ length: DIMS }, () => 0.1)) };
|
||||
}) as any);
|
||||
await expect(embedBatch(texts, { onBatchComplete: () => {} })).rejects.toThrow();
|
||||
});
|
||||
|
||||
test('after a failure, surviving workers stop dispatching new slices', async () => {
|
||||
// 1000 texts → 10 slices, concurrency 2. First call fails immediately;
|
||||
// without the `failed` flag the second worker would keep draining all
|
||||
// 10 slices in the background AFTER embedBatch already rejected —
|
||||
// burning provider spend and firing onBatchComplete post-rejection.
|
||||
let calls = 0;
|
||||
const completions: number[] = [];
|
||||
__setEmbedTransportForTests((async ({ values }: { values: string[] }) => {
|
||||
calls++;
|
||||
if (calls === 1) throw new Error('boom');
|
||||
await new Promise(r => setTimeout(r, 5));
|
||||
return { embeddings: values.map(() => Array.from({ length: DIMS }, () => 0.1)) };
|
||||
}) as any);
|
||||
const many = Array.from({ length: 1000 }, (_, i) => `t${i}`);
|
||||
await expect(
|
||||
embedBatch(many, { concurrency: 2, onBatchComplete: d => completions.push(d) }),
|
||||
).rejects.toThrow('boom');
|
||||
const callsAtRejection = calls;
|
||||
await new Promise(r => setTimeout(r, 50)); // would-be background drain window
|
||||
expect(calls).toBe(callsAtRejection); // no new dispatch after rejection
|
||||
expect(calls).toBeLessThanOrEqual(2); // only the in-flight sibling ran
|
||||
expect(completions).toHaveLength(0); // no progress reported after failure
|
||||
});
|
||||
|
||||
test('single small batch without callback stays on the one-call fast path', async () => {
|
||||
const tracker = installTrackingTransport(1);
|
||||
const result = await embedBatch(['t0', 't1', 't2']);
|
||||
expect(result).toHaveLength(3);
|
||||
expect(result[1][0]).toBe(1);
|
||||
expect(tracker.maxInFlight()).toBe(1);
|
||||
});
|
||||
});
|
||||
@@ -19,7 +19,7 @@
|
||||
* overwrites this preload.
|
||||
*/
|
||||
import { configureGateway, getEmbeddingDimensions } from '../../src/core/ai/gateway.ts';
|
||||
import { beforeEach } from 'bun:test';
|
||||
import { afterEach, beforeEach } from 'bun:test';
|
||||
|
||||
const LEGACY_CONFIG = {
|
||||
embedding_model: 'openai:text-embedding-3-large',
|
||||
@@ -52,7 +52,7 @@ applyLegacy();
|
||||
// 2. file-local beforeAll → may overwrite to ZE/1280
|
||||
// Since beforeAll runs once per file BEFORE the first beforeEach,
|
||||
// file-local beforeAll wins for that file's tests. ✓
|
||||
beforeEach(() => {
|
||||
function applyLegacyIfEmpty() {
|
||||
try {
|
||||
// Only re-apply if the gateway was reset (or never configured).
|
||||
// Tests that explicitly configured a different model in their
|
||||
@@ -62,4 +62,28 @@ beforeEach(() => {
|
||||
} catch {
|
||||
applyLegacy();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
beforeEach(applyLegacyIfEmpty);
|
||||
|
||||
// PR #3130 shard-order fix: beforeEach alone leaves ONE window open — a file
|
||||
// whose LAST afterEach calls resetGateway() poisons the NEXT file's
|
||||
// beforeAll, which runs BEFORE any beforeEach fires. A beforeAll there that
|
||||
// does engine.initSchema() then sizes the embedding column from the gateway
|
||||
// DEFAULTS (zembed-1/1280d) instead of the pinned legacy 1536, and every
|
||||
// 1536-d Float32Array fixture in that file dies with
|
||||
// "expected 1280 dimensions, not 1536". Which file pair collides is a
|
||||
// function of shard composition, so adding/removing ANY test file can
|
||||
// surface it (that is exactly how it bit shard 9).
|
||||
//
|
||||
// Preload hooks are registered before any file-local hooks, and bun runs
|
||||
// after-hooks inside-out (file-local afterEach first, then this one), so
|
||||
// this repairs the empty slot immediately after the poisoning reset —
|
||||
// before the next file's beforeAll can observe it.
|
||||
//
|
||||
// Known remaining window: a file whose afterAll() resets the gateway (no
|
||||
// hook runs between its afterAll and the next file's beforeAll). Files
|
||||
// that reset in afterAll and can precede a schema-creating file should
|
||||
// re-apply their own config, or the victim file should configureGateway()
|
||||
// explicitly in its beforeAll.
|
||||
afterEach(applyLegacyIfEmpty);
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
/**
|
||||
* #1207: `gbrain import` without `--workers` used to hardcode workerCount=1,
|
||||
* so a large Postgres import paid one serial embedding round-trip per file.
|
||||
* runImport now routes the default through the shared autoConcurrency policy
|
||||
* (PGLite → 1, >100 files on Postgres → DEFAULT_PARALLEL_WORKERS), while an
|
||||
* explicit `--workers N` still wins.
|
||||
*
|
||||
* The engine here is a minimal postgres-kind stub with no database_url in
|
||||
* config — runImport's parallel branch then falls back to serial processing
|
||||
* (its PR #490 guard) but the WORKER-COUNT DECISION (the thing #1207 fixes)
|
||||
* is still observable via the "Using N parallel workers" log line. Per-file
|
||||
* imports fail against the stub engine and are swallowed by runImport's
|
||||
* per-file catch; that's fine — this test pins the policy, not the import.
|
||||
*/
|
||||
|
||||
import { afterEach, beforeEach, describe, expect, test } from 'bun:test';
|
||||
import { mkdtempSync, writeFileSync, mkdirSync, rmSync, realpathSync } from 'fs';
|
||||
import { tmpdir } from 'os';
|
||||
import { join } from 'path';
|
||||
import { withEnv } from './helpers/with-env.ts';
|
||||
import { runImport } from '../src/commands/import.ts';
|
||||
|
||||
const fakePostgresEngine = {
|
||||
kind: 'postgres',
|
||||
executeRaw: async () => [],
|
||||
logIngest: async () => {},
|
||||
setConfig: async () => {},
|
||||
getConfig: async () => null,
|
||||
} as any;
|
||||
|
||||
let workspace: string;
|
||||
let brainDir: string;
|
||||
let logs: string[];
|
||||
const realLog = console.log;
|
||||
|
||||
beforeEach(() => {
|
||||
workspace = mkdtempSync(join(tmpdir(), 'gbrain-import-workers-home-'));
|
||||
mkdirSync(join(workspace, '.gbrain'), { recursive: true });
|
||||
brainDir = realpathSync(mkdtempSync(join(tmpdir(), 'gbrain-import-workers-brain-')));
|
||||
// 101 files: one past AUTO_CONCURRENCY_FILE_THRESHOLD (100).
|
||||
for (let i = 0; i < 101; i++) {
|
||||
writeFileSync(join(brainDir, `page-${i}.md`), `# Page ${i}\n\nbody ${i}\n`);
|
||||
}
|
||||
logs = [];
|
||||
console.log = (msg?: unknown) => logs.push(String(msg));
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
console.log = realLog;
|
||||
rmSync(workspace, { recursive: true, force: true });
|
||||
rmSync(brainDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe('import default worker count (#1207)', () => {
|
||||
test('no --workers flag → autoConcurrency picks 4 for >100 files on Postgres', async () => {
|
||||
await withEnv({ GBRAIN_HOME: join(workspace, '.gbrain'), GBRAIN_SOURCE: undefined }, async () => {
|
||||
await runImport(fakePostgresEngine, [brainDir, '--no-embed'], { sourceId: 'default' });
|
||||
});
|
||||
expect(logs.some(l => l.includes('Using 4 parallel workers'))).toBe(true);
|
||||
});
|
||||
|
||||
test('explicit --workers 2 still wins over the auto policy', async () => {
|
||||
await withEnv({ GBRAIN_HOME: join(workspace, '.gbrain'), GBRAIN_SOURCE: undefined }, async () => {
|
||||
await runImport(fakePostgresEngine, [brainDir, '--no-embed', '--workers', '2'], { sourceId: 'default' });
|
||||
});
|
||||
expect(logs.some(l => l.includes('Using 2 parallel workers'))).toBe(true);
|
||||
expect(logs.some(l => l.includes('Using 4 parallel workers'))).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -13,10 +13,9 @@
|
||||
* throw on unknown string.
|
||||
*/
|
||||
|
||||
import { describe, test, expect, afterAll, afterEach } from 'bun:test';
|
||||
import { describe, test, expect } from 'bun:test';
|
||||
import {
|
||||
resolveEmbeddingColumn,
|
||||
resolveWriteColumn,
|
||||
getEmbeddingColumnRegistry,
|
||||
buildVectorCastFragment,
|
||||
quoteIdentifier,
|
||||
@@ -35,28 +34,6 @@ import {
|
||||
} from '../../src/core/search/embedding-column.ts';
|
||||
import type { GBrainConfig } from '../../src/core/config.ts';
|
||||
import type { ResolvedColumn } from '../../src/core/types.ts';
|
||||
import { configureGateway, resetGateway } from '../../src/core/ai/gateway.ts';
|
||||
|
||||
/**
|
||||
* Teardown: reset AND re-apply the legacy preload config
|
||||
* (test/helpers/legacy-embedding-preload.ts). A bare resetGateway() would
|
||||
* leave the slot empty for the NEXT file's beforeAll (the preload's
|
||||
* per-test beforeEach only fires before tests, not before beforeAll), which
|
||||
* would make sibling PGLite fixtures initSchema at the 1280 default instead
|
||||
* of the legacy 1536 their seed vectors assume.
|
||||
*/
|
||||
function restorePreloadGateway() {
|
||||
resetGateway();
|
||||
configureGateway({
|
||||
embedding_model: 'openai:text-embedding-3-large',
|
||||
embedding_dimensions: 1536,
|
||||
env: { ...process.env },
|
||||
});
|
||||
}
|
||||
|
||||
afterAll(() => {
|
||||
restorePreloadGateway();
|
||||
});
|
||||
|
||||
function cfg(overrides: Partial<GBrainConfig> = {}): GBrainConfig {
|
||||
return { engine: 'pglite', ...overrides };
|
||||
@@ -545,89 +522,3 @@ describe('codex /ship #4 — isCacheSafe (embedding-space-based skip)', () => {
|
||||
expect(isCacheSafe(r, cfg())).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe('resolveWriteColumn — write-side boundary resolution (#1262)', () => {
|
||||
afterEach(() => {
|
||||
restorePreloadGateway();
|
||||
});
|
||||
|
||||
test('no registry / empty registry returns undefined (legacy single-column brain)', () => {
|
||||
expect(resolveWriteColumn(cfg())).toBeUndefined();
|
||||
expect(resolveWriteColumn(cfg({ embedding_columns: {} }))).toBeUndefined();
|
||||
});
|
||||
|
||||
test('provider match via cfg.embedding_model returns the descriptor', () => {
|
||||
const r = resolveWriteColumn(cfg({
|
||||
embedding_model: 'voyage:voyage-3-large',
|
||||
embedding_dimensions: 1024,
|
||||
embedding_columns: {
|
||||
embedding_voyage: { provider: 'voyage:voyage-3-large', dimensions: 1024, type: 'vector' },
|
||||
},
|
||||
}));
|
||||
expect(r).toEqual({
|
||||
name: 'embedding_voyage',
|
||||
type: 'vector',
|
||||
dimensions: 1024,
|
||||
embeddingModel: 'voyage:voyage-3-large',
|
||||
});
|
||||
});
|
||||
|
||||
test('provider match via gateway state (cfg.embedding_model unset) returns descriptor', () => {
|
||||
configureGateway({
|
||||
embedding_model: 'zeroentropyai:zembed-1',
|
||||
embedding_dimensions: 2560,
|
||||
env: {},
|
||||
});
|
||||
const r = resolveWriteColumn(cfg({
|
||||
embedding_columns: {
|
||||
embedding_ze: { provider: 'zeroentropyai:zembed-1', dimensions: 2560, type: 'halfvec' },
|
||||
},
|
||||
}));
|
||||
expect(r).toEqual({
|
||||
name: 'embedding_ze',
|
||||
type: 'halfvec',
|
||||
dimensions: 2560,
|
||||
embeddingModel: 'zeroentropyai:zembed-1',
|
||||
});
|
||||
});
|
||||
|
||||
test('no provider match returns undefined instead of guessing a column', () => {
|
||||
configureGateway({
|
||||
embedding_model: 'zeroentropyai:zembed-1',
|
||||
embedding_dimensions: 2560,
|
||||
env: {},
|
||||
});
|
||||
const r = resolveWriteColumn(cfg({
|
||||
embedding_columns: {
|
||||
embedding_voyage: { provider: 'voyage:voyage-3-large', dimensions: 1024, type: 'vector' },
|
||||
},
|
||||
}));
|
||||
expect(r).toBeUndefined();
|
||||
});
|
||||
|
||||
test('only USER-declared columns are consulted — multimodal builtin never captures text writes', () => {
|
||||
// Current model equals the embedding_image BUILTIN's provider; a registry
|
||||
// walk that consulted builtins would misroute text writes into the image
|
||||
// column. resolveWriteColumn must return undefined here.
|
||||
configureGateway({
|
||||
embedding_model: 'voyage:voyage-multimodal-3',
|
||||
embedding_dimensions: 1024,
|
||||
env: {},
|
||||
});
|
||||
const r = resolveWriteColumn(cfg({
|
||||
embedding_columns: {
|
||||
embedding_other: { provider: 'openai:text-embedding-3-large', dimensions: 1536, type: 'vector' },
|
||||
},
|
||||
}));
|
||||
expect(r).toBeUndefined();
|
||||
});
|
||||
|
||||
test('malformed registry entry throws loud (same validation as the read side)', () => {
|
||||
expect(() => resolveWriteColumn(cfg({
|
||||
embedding_model: 'voyage:voyage-3-large',
|
||||
embedding_columns: {
|
||||
'bad"col': { provider: 'voyage:voyage-3-large', dimensions: 1024, type: 'vector' },
|
||||
} as never,
|
||||
}))).toThrow(EmbeddingColumnConfigError);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user