Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 d98e6b507f test(config): use withEnv() for TTL env mutation to satisfy check-test-isolation R1
CI verify failed: test/postgres-engine-config-cache.test.ts mutated
process.env.GBRAIN_CONFIG_CACHE_TTL_MS directly in beforeEach/afterEach.
Wrap each test body in withEnv() (test/helpers/with-env.ts) instead.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 10:58:53 -07:00
d4bbf6eeae perf(config): batch + cache engine.getConfig to kill ~85 round-trips per query
Takeover of #1694: a single search fires ~85 serial getConfig() reads
(loadConfigWithEngine x2 plus the mode/cache/intent/rerank/graph-signals
resolvers); on a remote pooler each read is a round-trip, dominating query
latency and risking cli.ts's 10s disconnect force-exit truncating stdout.

The first read now batch-loads the whole config table into a process-
lifetime Map (single-flight under concurrency); setConfig/unsetConfig
write through; a 30s TTL bounds multi-writer staleness and
GBRAIN_CONFIG_CACHE_TTL_MS=0 restores per-key reads.

Rebased onto current master: unlike the original diff, both the batch
load and the TTL=0 per-key fallback stay inside connRetry() so the
#1603/#1891 retry+reconnect posture (pooler-drop self-heal) is preserved.
PGLite stays uncached (in-process, zero round-trips; raw-SQL test
fixtures rely on fresh reads). E2E helpers pin the cache off since those
suites seed config via raw SQL.

Co-authored-by: Omerbahari <Omerbahari@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:26:28 -07:00
7 changed files with 177 additions and 241 deletions
-64
View File
@@ -24,7 +24,6 @@ import type { GBrainConfig } from './core/config.ts';
import type { AIGatewayConfig } from './core/ai/types.ts';
import type { BrainEngine } from './core/engine.ts';
import { operations, OperationError } from './core/operations.ts';
import { resolveSourceIdEngineFree } from './core/source-resolver.ts';
import { formatVolunteeredPage } from './core/context/volunteer.ts';
import type { Operation, OperationContext } from './core/operations.ts';
import { shouldForceExitAfterMain, finishCliTeardown, flushThenExit, currentExitCode, setCliExitVerdict } from './core/cli-force-exit.ts';
@@ -383,15 +382,6 @@ async function main() {
if (op.localOnly) {
refuseThinClient(command, cfgPre!.remote_mcp!.mcp_url);
}
// #2098: the local path resolves --source / GBRAIN_SOURCE / .gbrain-source
// inside makeContext (ctx.sourceId), which this route never reaches — so
// scope must be mapped onto the op's source_id wire param before the call.
try {
applyThinClientSourceScope(op, params);
} catch (e: unknown) {
console.error(e instanceof Error ? e.message : String(e));
process.exit(1);
}
await runThinClientRouted(op, params, cfgPre!, cliOpts);
return;
}
@@ -812,60 +802,6 @@ export function parseOpArgs(op: Operation, args: string[]): Record<string, unkno
return params;
}
/**
* #2098: thin-client source scoping. Locally, --source / GBRAIN_SOURCE /
* .gbrain-source resolve to ctx.sourceId in makeContext; the thin-client
* route short-circuits before that, so `gbrain query --source X` against a
* remote brain silently searched unscoped. This runs the engine-free tiers
* (flag → env → dotfile; the DB-backed tiers can't run without an engine —
* the server's grant scoping covers the rest) and maps the result onto the
* op's `source_id` wire param.
*
* Ops that declare their OWN `source` param (facts add, etc.) are left
* untouched — their --source is an op param, not scope. An explicit --source
* on an op with no source_id wire param throws (loud beats silent drop);
* ambient env/dotfile scope with nowhere to send it is ignored, matching the
* pre-fix behavior for non-scopeable ops. Exported for tests.
*/
// Ops whose `source_id` wire param is NOT read-scope semantics: get_skill's
// source_id flips the lookup from host catalog to brain-resident-pack
// (getResidentSkillDetail). Ambient env/dotfile scope must never leak into
// these; an explicit --source-id still passes through untouched above.
const NON_SCOPE_SOURCE_ID_OPS = new Set(['get_skill']);
export function applyThinClientSourceScope(
op: Operation,
params: Record<string, unknown>,
cwd?: string,
): void {
if ('source' in op.params) return; // the op owns --source; not a scope flag
const explicit = typeof params.source === 'string' && params.source.length > 0
? (params.source as string)
: null;
delete params.source; // never a wire param on these ops — don't leak it
// Explicit per-call scope already on the wire wins over ambient tiers.
if (params.source_id !== undefined || params.all_sources === true) {
if (explicit) {
throw new Error('Pass either --source or --source-id/--all-sources, not both.');
}
return;
}
const resolved = resolveSourceIdEngineFree(explicit, cwd);
if (!resolved) return;
if (!('source_id' in op.params) || NON_SCOPE_SOURCE_ID_OPS.has(op.name)) {
if (explicit) {
const hint = NON_SCOPE_SOURCE_ID_OPS.has(op.name)
? `(its source_id parameter is not a scope filter; pass --source-id explicitly if you mean it)`
: `(the remote op has no source_id parameter; the server scopes it to your grant)`;
throw new Error(
`gbrain ${op.cliHints?.name || op.name} does not accept --source on a thin-client install ${hint}.`,
);
}
return; // ambient env/dotfile scope with nowhere to send it
}
params.source_id = resolved;
}
async function makeContext(engine: BrainEngine, params: Record<string, unknown>): Promise<OperationContext> {
// v0.31.8 (D11): resolve sourceId via the canonical 6-tier chain. Honors
// --source / GBRAIN_SOURCE / .gbrain-source / path-match / brain default /
+55 -7
View File
@@ -5565,30 +5565,78 @@ export class PostgresEngine implements BrainEngine {
});
}
/**
* perf (#1694 by @Omerbahari): process-lifetime config cache. A single
* search fires ~85 getConfig() reads (loadConfigWithEngine x2, plus
* mode/cache/intent/rerank/graph-signals resolvers). On a remote pooler
* each read is a round-trip; serial they dominate query latency and can
* push the op handler past cli.ts's 10s disconnect force-exit, truncating
* stdout. First read batch-loads the whole `config` table into this Map
* (inside the same connRetry posture as the per-key read #1603/#1891);
* setConfig/unsetConfig write through. TTL bounds staleness for
* multi-writer processes; GBRAIN_CONFIG_CACHE_TTL_MS=0 disables.
* Only present keys are stored Map.has() distinguishes known-absent.
*/
private _configCache: Map<string, string> | null = null;
private _configCacheLoadedAt = 0;
private _configCacheLoad: Promise<void> | null = null;
private get _configCacheTtlMs(): number {
const raw = process.env.GBRAIN_CONFIG_CACHE_TTL_MS;
if (raw !== undefined) {
const n = parseInt(raw, 10);
if (Number.isFinite(n) && n >= 0) return n;
}
return 30_000;
}
async getConfig(key: string): Promise<string | null> {
// #1603: a transient pooler drop on this read used to throw / fall through
// to defaults silently — which on remote Postgres surfaces as the wrong
// search mode/knobs and empty-stdout queries.
return this.connRetry(async () => {
const rows = await this.sql`SELECT value FROM config WHERE key = ${key}`;
return rows.length > 0 ? (rows[0].value as string) : null;
});
// search mode/knobs and empty-stdout queries. Both the batch load and the
// cache-off per-key read keep the connRetry reconnect posture.
const ttl = this._configCacheTtlMs;
if (ttl === 0) {
return this.connRetry(async () => {
const rows = await this.sql`SELECT value FROM config WHERE key = ${key}`;
return rows.length > 0 ? (rows[0].value as string) : null;
});
}
if (this._configCache === null || Date.now() - this._configCacheLoadedAt >= ttl) {
// Single-flight: concurrent cold reads share one batch load.
this._configCacheLoad ??= this.connRetry(async () => {
const rows = await this.sql`SELECT key, value FROM config` as unknown as
Array<{ key: string; value: string | null }>;
const map = new Map<string, string>();
for (const r of rows) if (r.value != null) map.set(r.key, r.value);
this._configCache = map;
this._configCacheLoadedAt = Date.now();
}).finally(() => {
this._configCacheLoad = null;
});
await this._configCacheLoad;
}
return this._configCache!.has(key) ? this._configCache!.get(key)! : null;
}
async setConfig(key: string, value: string): Promise<void> {
return this.connRetry(async () => {
await this.connRetry(async () => {
await this.sql`
INSERT INTO config (key, value) VALUES (${key}, ${value})
ON CONFLICT (key) DO UPDATE SET value = EXCLUDED.value
`;
});
// Write-through so a long-lived process never serves stale config.
this._configCache?.set(key, value);
}
async unsetConfig(key: string): Promise<number> {
return this.connRetry(async () => {
const count = await this.connRetry(async () => {
const result = await this.sql`DELETE FROM config WHERE key = ${key}` as unknown as { count: number };
return result.count ?? 0;
});
// Write-through: known-absent, so the cache doesn't serve a stale value.
this._configCache?.delete(key);
return count;
}
async listConfigKeys(prefix: string): Promise<string[]> {
-27
View File
@@ -160,33 +160,6 @@ export async function resolveSourceId(
return 'default';
}
/**
* Engine-free tiers (1-3) of the resolution chain: explicit flag →
* GBRAIN_SOURCE env → .gbrain-source dotfile walk. Used by the thin-client
* CLI path (#2098), which has no local engine to run tiers 4-6 or
* assertSourceExists against — the remote server enforces existence + grant.
* Returns null when no engine-free tier fires.
*/
export function resolveSourceIdEngineFree(
explicit: string | null | undefined,
cwd: string = process.cwd(),
): string | null {
if (explicit) {
if (!SOURCE_ID_RE.test(explicit)) {
throw new Error(`Invalid --source value "${explicit}". Must match [a-z0-9-]{1,32}.`);
}
return explicit;
}
const env = process.env.GBRAIN_SOURCE;
if (env && env.length > 0) {
if (!SOURCE_ID_RE.test(env)) {
throw new Error(`Invalid GBRAIN_SOURCE value "${env}". Must match [a-z0-9-]{1,32}.`);
}
return env;
}
return readDotfileWalk(cwd);
}
/**
* Returns the id of the SINGLE registered non-default source with a
* local_path, when exactly one such row exists. Returns null when:
+7
View File
@@ -29,6 +29,13 @@ if (existsSync(envPath)) {
}
}
// E2E suites seed/rewrite the config table via raw SQL and expect engine
// reads to see it immediately; disable the process-lifetime config cache
// (#1694) so read semantics match pre-cache behavior. Spawned CLI
// subprocesses inherit this. Cache semantics are pinned by
// test/postgres-engine-config-cache.test.ts.
process.env.GBRAIN_CONFIG_CACHE_TTL_MS ??= '0';
const DATABASE_URL = process.env.DATABASE_URL;
const FIXTURES_DIR = resolve(import.meta.dir, 'fixtures');
+112
View File
@@ -0,0 +1,112 @@
/**
* Process-lifetime config cache (#1694 takeover, by @Omerbahari).
*
* A single search fires ~85 getConfig() reads; on a remote pooler each is a
* round-trip. The first read now batch-loads the whole `config` table into a
* Map; setConfig/unsetConfig write through; TTL bounds multi-writer
* staleness; GBRAIN_CONFIG_CACHE_TTL_MS=0 restores per-key reads.
*
* Pure: stubs `_sql` with a call-counting fake; no real DB.
*/
import { describe, it, expect } from 'bun:test';
import { PostgresEngine } from '../src/core/postgres-engine.ts';
import { withEnv } from './helpers/with-env.ts';
const FAST_RETRY = { maxRetries: 3, delayMs: 1, delayMaxMs: 1, jitter: 'none' as const };
/** Engine whose `sql` records every query's template strings and returns `rows`. */
function makeEngine(rows: unknown[]) {
const e = new PostgresEngine();
const calls: string[] = [];
(e as unknown as { _connectionStyle: string })._connectionStyle = 'instance';
(e as unknown as { _bulkRetryOptsCache: unknown })._bulkRetryOptsCache = FAST_RETRY;
(e as unknown as { _sql: unknown })._sql = (strings: TemplateStringsArray) => {
calls.push(strings.join('?'));
return Promise.resolve(rows);
};
return { engine: e, calls };
}
/** Run `fn` with GBRAIN_CONFIG_CACHE_TTL_MS set (or cleared when undefined). */
const withTtl = (ttl: string | undefined, fn: () => Promise<void>) =>
withEnv({ GBRAIN_CONFIG_CACHE_TTL_MS: ttl }, fn);
describe('PostgresEngine config cache (#1694)', () => {
it('batch-loads once and serves repeat reads from the cache', () => withTtl(undefined, async () => {
const { engine, calls } = makeEngine([
{ key: 'search.mode', value: 'balanced' },
{ key: 'embedding_multimodal', value: 'true' },
]);
expect(await engine.getConfig('search.mode')).toBe('balanced');
expect(await engine.getConfig('embedding_multimodal')).toBe('true');
expect(await engine.getConfig('search.mode')).toBe('balanced');
// One SELECT total — this is the whole point of the fix.
expect(calls.length).toBe(1);
expect(calls[0]).toContain('SELECT key, value FROM config');
}));
it('returns null for a known-absent key without an extra round-trip', () => withTtl(undefined, async () => {
const { engine, calls } = makeEngine([{ key: 'a', value: '1' }]);
expect(await engine.getConfig('missing.key')).toBeNull();
expect(await engine.getConfig('missing.key')).toBeNull();
expect(calls.length).toBe(1);
}));
it('setConfig writes through so subsequent reads see the new value', () => withTtl(undefined, async () => {
const { engine, calls } = makeEngine([{ key: 'k', value: 'old' }]);
expect(await engine.getConfig('k')).toBe('old');
await engine.setConfig('k', 'new');
expect(await engine.getConfig('k')).toBe('new');
expect(calls.length).toBe(2); // batch load + upsert; no re-read
}));
it('unsetConfig writes through so subsequent reads see absence', () => withTtl(undefined, async () => {
const { engine } = makeEngine([{ key: 'k', value: 'v' }]);
expect(await engine.getConfig('k')).toBe('v');
await engine.unsetConfig('k');
expect(await engine.getConfig('k')).toBeNull();
}));
it('concurrent cold reads share a single batch load (single-flight)', () => withTtl(undefined, async () => {
const { engine, calls } = makeEngine([{ key: 'k', value: 'v' }]);
const [a, b, c] = await Promise.all([
engine.getConfig('k'),
engine.getConfig('k'),
engine.getConfig('other'),
]);
expect([a, b, c]).toEqual(['v', 'v', null]);
expect(calls.length).toBe(1);
}));
it('GBRAIN_CONFIG_CACHE_TTL_MS=0 disables the cache (per-key reads)', () => withTtl('0', async () => {
const { engine, calls } = makeEngine([{ value: 'v' }]);
expect(await engine.getConfig('k')).toBe('v');
expect(await engine.getConfig('k')).toBe('v');
expect(calls.length).toBe(2);
expect(calls[0]).toContain('SELECT value FROM config WHERE key =');
}));
it('an expired TTL reloads from the database', () => withTtl('1', async () => {
const { engine, calls } = makeEngine([{ key: 'k', value: 'v' }]);
expect(await engine.getConfig('k')).toBe('v');
await new Promise((r) => setTimeout(r, 5));
expect(await engine.getConfig('k')).toBe('v');
expect(calls.length).toBe(2); // two batch loads
}));
it('the batch load keeps the connRetry reconnect posture (#1603/#1891)', () => withTtl(undefined, async () => {
const e = new PostgresEngine();
(e as unknown as { _connectionStyle: string })._connectionStyle = 'instance';
(e as unknown as { _sql: unknown })._sql = null; // torn-down pool → retryable
(e as unknown as { _bulkRetryOptsCache: unknown })._bulkRetryOptsCache = FAST_RETRY;
let reconnects = 0;
(e as unknown as { reconnect: () => Promise<void> }).reconnect = async () => {
reconnects++;
(e as unknown as { _sql: unknown })._sql = () =>
Promise.resolve([{ key: 'k', value: 'v' }]);
};
expect(await e.getConfig('k')).toBe('v');
expect(reconnects).toBe(1);
}));
});
@@ -43,7 +43,9 @@ function makeTornDownEngine(poolResult: unknown): { engine: PostgresEngine; reco
describe('PostgresEngine non-batch config accessors self-heal (PR #1891 takeover)', () => {
it('getConfig reconnects + retries a null instance pool, then returns the value', async () => {
const { engine, reconnects } = makeTornDownEngine([{ value: 'live-value' }]);
// Rows carry `key` too: getConfig's default cached path batch-loads
// `SELECT key, value FROM config` (#1694) through the same connRetry.
const { engine, reconnects } = makeTornDownEngine([{ key: 'some.key', value: 'live-value' }]);
expect(await engine.getConfig('some.key')).toBe('live-value');
expect(reconnects()).toBe(1); // exactly one reconnect closed the gap
});
-142
View File
@@ -1,142 +0,0 @@
/**
* #2098: thin-client routing dropped --source / GBRAIN_SOURCE / .gbrain-source.
*
* The local CLI path resolves source scope in makeContext (ctx.sourceId); the
* thin-client route short-circuits before that and sent params verbatim, so
* `gbrain query --source X` against a remote brain silently searched unscoped
* (the server op ignores the unknown `source` key).
*
* applyThinClientSourceScope runs the engine-free tiers (flag → env → dotfile)
* and maps the result onto the op's `source_id` wire param. These tests fail
* without the fix (params.source_id stays undefined / params.source leaks).
*/
import { describe, test, expect } from 'bun:test';
import { mkdtempSync, rmSync, writeFileSync } from 'fs';
import { join } from 'path';
import { tmpdir } from 'os';
import { applyThinClientSourceScope, parseOpArgs } from '../src/cli.ts';
import { operationsByName } from '../src/core/operations.ts';
import { withEnv } from './helpers/with-env.ts';
const queryOp = operationsByName.query;
describe('applyThinClientSourceScope (#2098)', () => {
test('--source maps onto the query op wire param source_id', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const params = parseOpArgs(queryOp, ['find things', '--source', 'wiki']);
expect(params.source).toBe('wiki'); // pre-fix state: wrong key
applyThinClientSourceScope(queryOp, params, '/');
expect(params.source_id).toBe('wiki');
expect('source' in params).toBe(false); // never leaks the unknown key
});
});
test('GBRAIN_SOURCE env tier fires when no flag is passed', async () => {
await withEnv({ GBRAIN_SOURCE: 'gstack' }, () => {
const params = parseOpArgs(queryOp, ['find things']);
applyThinClientSourceScope(queryOp, params, '/');
expect(params.source_id).toBe('gstack');
});
});
test('.gbrain-source dotfile tier fires when flag and env are absent', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const tmp = mkdtempSync(join(tmpdir(), 'gbrain-thin-scope-'));
try {
writeFileSync(join(tmp, '.gbrain-source'), 'essays\n');
const params = parseOpArgs(queryOp, ['find things']);
applyThinClientSourceScope(queryOp, params, tmp);
expect(params.source_id).toBe('essays');
} finally {
rmSync(tmp, { recursive: true, force: true });
}
});
});
test('explicit --source-id on the wire wins over ambient env scope', async () => {
await withEnv({ GBRAIN_SOURCE: 'gstack' }, () => {
const params = parseOpArgs(queryOp, ['find things', '--source-id', 'wiki']);
applyThinClientSourceScope(queryOp, params, '/');
expect(params.source_id).toBe('wiki');
});
});
test('--source together with --source-id is rejected loudly', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const params = parseOpArgs(queryOp, ['q', '--source', 'a', '--source-id', 'b']);
expect(() => applyThinClientSourceScope(queryOp, params, '/')).toThrow(/not both/);
});
});
test('invalid --source value is rejected loudly', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const params = parseOpArgs(queryOp, ['q', '--source', 'Bad_Value!']);
expect(() => applyThinClientSourceScope(queryOp, params, '/')).toThrow(/Invalid --source/);
});
});
test('--source on an op with no source_id wire param errors instead of silently dropping', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const op = operationsByName.add_tag;
expect('source_id' in op.params).toBe(false);
const params = { slug: 'x', tag: 'y', source: 'wiki' };
expect(() => applyThinClientSourceScope(op, params, '/')).toThrow(/--source/);
});
});
test('ambient env scope on an op with no source_id wire param is ignored (no throw)', async () => {
await withEnv({ GBRAIN_SOURCE: 'wiki' }, () => {
const op = operationsByName.add_tag;
const params: Record<string, unknown> = { slug: 'x', tag: 'y' };
applyThinClientSourceScope(op, params, '/');
expect(params.source_id).toBeUndefined();
});
});
test('ops that declare their OWN source param are left untouched', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const op = operationsByName.put_raw_data;
expect('source' in op.params).toBe(true);
const params: Record<string, unknown> = { slug: 'x', source: 'crustdata', data: {} };
applyThinClientSourceScope(op, params, '/');
expect(params.source).toBe('crustdata');
expect(params.source_id).toBeUndefined();
});
});
test('get_skill: ambient scope never leaks into its non-scope source_id param', async () => {
await withEnv({ GBRAIN_SOURCE: 'wiki' }, () => {
const op = operationsByName.get_skill;
expect('source_id' in op.params).toBe(true); // has the param, but it is a mode switch
const params: Record<string, unknown> = { name: 'ingest' };
applyThinClientSourceScope(op, params, '/');
expect(params.source_id).toBeUndefined(); // would flip host catalog → brain-pack lookup
});
});
test('get_skill: explicit --source errors instead of masquerading as --source-id', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const op = operationsByName.get_skill;
const params: Record<string, unknown> = { name: 'ingest', source: 'wiki' };
expect(() => applyThinClientSourceScope(op, params, '/')).toThrow(/--source-id/);
});
});
test('get_skill: explicit --source-id passes through untouched', async () => {
await withEnv({ GBRAIN_SOURCE: 'gstack' }, () => {
const op = operationsByName.get_skill;
const params: Record<string, unknown> = { name: 'ingest', source_id: 'wiki' };
applyThinClientSourceScope(op, params, '/');
expect(params.source_id).toBe('wiki');
});
});
test('no scope from any tier leaves params unchanged', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const params = parseOpArgs(queryOp, ['find things']);
applyThinClientSourceScope(queryOp, params, '/');
expect(params.source_id).toBeUndefined();
});
});
});