Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 55290e9088 docs(tutorial): drop dead --remote flag from company-brain search examples
With unknown op flags now a hard error (this PR), the tutorial's
`gbrain search ... --remote` invocations would fail: no code path ever
read a --remote flag — thin-client installs route shared ops through the
remote MCP server automatically. Update the three examples to the
flagless form and explain the automatic routing.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 11:10:47 -07:00
04a2a1f7bf fix(cli): put --file, reject unknown op flags, list pagination (#380, #856, #2876)
Three CLI ergonomics fixes on the shared-op dispatch path:

- parseOpArgs now hard-errors on undeclared flags instead of silently
  swallowing them into params (#380). A pass-through allowlist keeps the
  undeclared-but-honored flags working: --source (makeContext's source
  resolver), --brain (mount axis), --dry-run (ctx.dryRun), --json.
  Value-taking flags with no value also error instead of being dropped.

- `gbrain put SLUG --file PATH` reads content from a file (#380, takeover
  of #856). Driven by the op's cliHints.stdin declaration rather than
  put_page hard-coding, so volunteer-context gets it too. Mutually
  exclusive with --content/stdin in either flag order; 5MB cap shared
  with stdin; unreadable path is a loud error instead of an empty page.
  put --help documents the flag.

- `--json` on shared ops now actually emits raw JSON (docs already
  promised it); same seam on local-engine and thin-client routed paths.

- list_pages declares `offset` (both engines already supported it on
  PageFilters) and the limit description discloses the 100-row cap
  (#2876), so `gbrain list` can paginate past 100 instead of silently
  truncating.

Co-authored-by: Kage18 <Kage18@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:40:45 -07:00
8 changed files with 261 additions and 276 deletions
+5 -3
View File
@@ -223,14 +223,16 @@ export GBRAIN_REMOTE_CLIENT_ID=<Alice's client_id>
export GBRAIN_REMOTE_CLIENT_SECRET=<Alice's client_secret>
export GBRAIN_REMOTE_MCP_URL=https://brain.acme-co.com/mcp
gbrain search "performance review" --remote
gbrain search "performance review"
```
(On a thin-client install every shared op routes through the remote MCP server automatically — no flag needed. The env vars select whose credentials the call uses.)
Alice should see results only from `customers` and `shared`. The performance-review notes live in `internal`, which she's not scoped to read. She shouldn't see them.
```bash
# Terminal 2, as Bob (export his credentials similarly)
gbrain search "performance review" --remote
gbrain search "performance review"
```
Bob should see the performance-review notes from `internal`, plus anything related from `shared`. He shouldn't see anything that lives only in `customers`.
@@ -514,7 +516,7 @@ The first sync embeds every page, which takes time. Check `gbrain sources status
### "I see a page I shouldn't see"
This shouldn't happen, but if you suspect it, run `gbrain search <query> --remote --json` as the constrained client and inspect the `source_id` field on every returned result. Every row should be in the client's `--federated-read` set. If one isn't, file an issue with the exact slug and source IDs.
This shouldn't happen, but if you suspect it, run `gbrain search <query> --json` as the constrained client (thin-client install, with the client's `GBRAIN_REMOTE_*` env exported) and inspect the `source_id` field on every returned result. Every row should be in the client's `--federated-read` set. If one isn't, file an issue with the exact slug and source IDs.
### "The synthesized answer is wrong"
+86 -16
View File
@@ -464,7 +464,11 @@ async function main() {
// routed path. Date → ISO string; bigint → string (postgres.js shape);
// Buffer → object. Microsecond-cost; eliminates a whole drift bug class.
const result = JSON.parse(JSON.stringify(rawResult, bigintToStringReplacer));
const output = formatResult(op.name, result);
// #380 pass-through: `--json` (undeclared on most ops, promised by docs)
// emits the raw op result instead of the human formatter.
const output = params.json === true
? JSON.stringify(result, null, 2) + '\n'
: formatResult(op.name, result);
if (output) process.stdout.write(output);
} catch (e: unknown) {
// v0.42.20.0 (codex D4): on error, set exitCode + return so the `finally`
@@ -547,7 +551,10 @@ async function runThinClientRouted(
signal: sigintController.signal,
});
const result = unpackToolResult(raw);
const output = formatResult(op.name, result);
// #380: same --json seam as the local-engine path (renderer parity).
const output = params.json === true
? JSON.stringify(result, null, 2) + '\n'
: formatResult(op.name, result);
if (output) process.stdout.write(output);
} catch (e: unknown) {
if (e instanceof RemoteMcpError) {
@@ -757,10 +764,28 @@ export function resolveQueryImage(
return { path: imagePath, base64, mime };
}
/**
* #380: undeclared flags that are honored DOWNSTREAM of parseOpArgs and must
* keep passing through when unknown flags become hard errors:
* - source → makeContext's resolveSourceId (the --source axis)
* - brain → the mount/brain routing axis (docs promise the flag)
* - dry_run → makeContext's ctx.dryRun (ops without a declared dry_run)
* - json → raw-JSON output seam (local + thin-client paths)
*/
const PASSTHROUGH_VALUE_FLAGS = new Set(['source', 'brain']);
const PASSTHROUGH_BOOL_FLAGS = new Set(['dry_run', 'json']);
export function parseOpArgs(op: Operation, args: string[]): Record<string, unknown> {
const params: Record<string, unknown> = {};
const positional = op.cliHints?.positional || [];
let posIdx = 0;
const cliName = op.cliHints?.name || op.name;
const MAX_STDIN = 5_000_000; // 5MB cap, shared by stdin and --file
// #380: `--file <path>` fills the op's declared stdin param (put's `content`)
// from a file. Driven by cliHints.stdin — no per-op hard-coding — and
// disabled when the op declares a real `file` param of its own.
const fileParam = op.cliHints?.stdin && !op.params.file ? op.cliHints.stdin : undefined;
let filePath: string | undefined;
for (let i = 0; i < args.length; i++) {
const arg = args[i];
@@ -774,12 +799,42 @@ export function parseOpArgs(op: Operation, args: string[]): Record<string, unkno
}
}
const key = arg.slice(2).replace(/-/g, '_');
if (fileParam && key === 'file') {
if (i + 1 >= args.length) {
console.error(`Error: ${arg} requires a value.`);
process.exit(1);
}
filePath = args[++i];
continue;
}
const paramDef = op.params[key];
if (paramDef?.type === 'boolean') {
if (!paramDef) {
if (PASSTHROUGH_BOOL_FLAGS.has(key)) {
params[key] = true;
continue;
}
if (PASSTHROUGH_VALUE_FLAGS.has(key)) {
if (i + 1 >= args.length) {
console.error(`Error: ${arg} requires a value.`);
process.exit(1);
}
params[key] = args[++i];
continue;
}
// #380: unknown flags were silently swallowed into params, so typos
// like `put --file` created empty pages instead of erroring.
console.error(`Unknown option for gbrain ${cliName}: ${arg}`);
console.error(`Run 'gbrain ${cliName} --help' for valid flags.`);
process.exit(1);
}
if (paramDef.type === 'boolean') {
params[key] = true;
} else if (i + 1 < args.length) {
params[key] = args[++i];
if (paramDef?.type === 'number') params[key] = Number(params[key]);
if (paramDef.type === 'number') params[key] = Number(params[key]);
} else {
console.error(`Error: ${arg} requires a value.`);
process.exit(1);
}
} else if (posIdx < positional.length) {
const key = positional[posIdx++];
@@ -788,10 +843,30 @@ export function parseOpArgs(op: Operation, args: string[]): Record<string, unkno
}
}
// #380: resolve --file AFTER the loop so --file/--content conflicts are
// caught in either order.
if (filePath !== undefined && fileParam) {
if (params[fileParam] !== undefined) {
console.error(`Error: use only one of --file, --${fileParam}, or stdin for gbrain ${cliName}.`);
process.exit(1);
}
let fileContent: string;
try {
fileContent = readFileSync(filePath, 'utf-8');
} catch (e) {
console.error(`Error: cannot read --file ${filePath}: ${e instanceof Error ? e.message : String(e)}`);
process.exit(1);
}
if (Buffer.byteLength(fileContent, 'utf-8') > MAX_STDIN) {
console.error(`Error: file content exceeds ${MAX_STDIN} bytes. Split into smaller inputs.`);
process.exit(1);
}
params[fileParam] = fileContent;
}
// Read stdin for content params
if (op.cliHints?.stdin && !params[op.cliHints.stdin] && !process.stdin.isTTY) {
const stdinContent = readFileSync(0, 'utf-8');
const MAX_STDIN = 5_000_000; // 5MB
if (Buffer.byteLength(stdinContent, 'utf-8') > MAX_STDIN) {
console.error(`Error: stdin content exceeds ${MAX_STDIN} bytes. Split into smaller inputs.`);
process.exit(1);
@@ -808,20 +883,12 @@ async function makeContext(engine: BrainEngine, params: Record<string, unknown>)
// 'default'. Wrapped in try/catch so a doctor / single-source brain that
// never set up sources still returns 'default' silently.
let sourceId: string | undefined;
// #2561: when the source resolved via a NON-explicit tier (path-match /
// brain default / sole-non-default / seed default), unqualified search-shaped
// reads span every `config.federated = true` source. Computed here (the
// trusted local boundary) and consumed by federatedSearchScope in
// operations.ts, which additionally gates on ctx.remote === false.
let localFederated: string[] | undefined;
try {
const { resolveSourceWithTier, localFederatedSourceIds } = await import('./core/source-resolver.ts');
const { resolveSourceId } = await import('./core/source-resolver.ts');
// params.source is set when a CLI flag was parsed for the op (rare; most
// CLI ops don't take --source). Falls through to env/dotfile/path-match.
const explicit = (params.source as string | undefined) ?? null;
const resolved = await resolveSourceWithTier(engine, explicit);
sourceId = resolved.source_id;
localFederated = await localFederatedSourceIds(engine, resolved.source_id, resolved.tier);
sourceId = await resolveSourceId(engine, explicit);
} catch {
// Source resolution failed (e.g. sources table doesn't exist on a fresh
// pre-init brain). Leave sourceId unset; engine read methods fall through
@@ -842,7 +909,6 @@ async function makeContext(engine: BrainEngine, params: Record<string, unknown>)
// table). Matches dispatch.ts's auto-fill so the contract holds across
// every transport.
sourceId: sourceId ?? 'default',
...(localFederated ? { localFederatedSourceIds: localFederated } : {}),
};
}
@@ -2262,6 +2328,10 @@ export function printOpHelp(op: Operation, invokedName?: string) {
const prefix = isPos ? ` <${key}>` : ` --${key.replace(/_/g, '-')}`;
console.log(`${prefix.padEnd(28)} ${def.description || ''}${req}`);
}
// #380: ops that read stdin also accept --file <path> (parseOpArgs).
if (op.cliHints?.stdin && !op.params.file) {
console.log(`${' --file <path>'.padEnd(28)} Read ${op.cliHints.stdin} from a file (alternative to --${op.cliHints.stdin} or stdin)`);
}
}
}
+11 -63
View File
@@ -424,23 +424,6 @@ export interface OperationContext {
* satisfied even on single-source brains.
*/
sourceId: string;
/**
* #2561 — federated read scope for UNQUALIFIED local CLI reads.
*
* Set ONLY by the local CLI's context builder (src/cli.ts makeContext), and
* only when the source resolved via a non-explicit tier (local_path /
* brain_default / sole_non_default / seed_default — NOT --source, NOT
* GBRAIN_SOURCE, NOT a .gbrain-source dotfile). Contains the resolved
* source first, then every other `config.federated = true` source, so an
* unqualified `gbrain search "X"` spans federated sources as
* docs/guides/multi-source-brains.md promises.
*
* Consumed exclusively by `federatedSearchScope` and ONLY when
* `ctx.remote === false` — a remote caller's scope stays governed by
* `ctx.auth.allowedSources` / scalar `ctx.sourceId` (source-isolation
* invariant, fail-closed).
*/
localFederatedSourceIds?: string[];
}
/**
@@ -556,45 +539,6 @@ export function resolveRequestedScope(
return sourceScopeOpts(ctx);
}
/**
* #2561 — source scope for the search-shaped read ops (`search`, `query`).
*
* Delegates to `resolveRequestedScope` (the single trust+grant resolver), then
* widens an UNQUALIFIED trusted-local scalar scope to the CLI-computed
* federated set (`ctx.localFederatedSourceIds`, resolved source first). This is
* what makes `sources add --federated` mean something for local search: a
* federated source participates in unqualified `gbrain search "X"` results.
*
* The expansion NEVER applies when:
* - the caller is not strictly trusted-local (`ctx.remote !== false`) —
* remote scope stays grant-governed (fail-closed source isolation);
* - a per-call `source_id` was passed (explicit wins, including `__all__`);
* - the resolver already produced a federated array (OAuth grant);
* - the CLI resolved the source from an explicit signal (--source / env /
* dotfile) — makeContext leaves `localFederatedSourceIds` unset then.
*
* Deliberately NOT inside `sourceScopeOpts`: code-intel ops collapse a
* multi-element scope to an error (`resolveCodeIntelScope`), and non-search
* reads (get_page, get_links, …) keep their long-standing scalar behavior.
*/
export function federatedSearchScope(
ctx: OperationContext,
sourceIdParam?: string,
): { sourceId?: string; sourceIds?: string[] } {
const scope = resolveRequestedScope(ctx, sourceIdParam);
if (
ctx.remote === false &&
sourceIdParam === undefined &&
scope.sourceId !== undefined &&
scope.sourceIds === undefined &&
ctx.localFederatedSourceIds !== undefined &&
ctx.localFederatedSourceIds.length > 1
) {
return { sourceIds: ctx.localFederatedSourceIds };
}
return scope;
}
/**
* Code-intel adapter for `resolveRequestedScope`. Graph traversal
* (code_callers/code_callees/code_blast/code_flow) is single-source by design —
@@ -825,7 +769,7 @@ const get_page: Operation = {
const put_page: Operation = {
name: 'put_page',
description: 'Write/update a page (markdown with frontmatter). Chunks, embeds, reconciles tags, and (when auto_link/auto_timeline are enabled) extracts + reconciles graph links and timeline entries. For large content on Windows (pipe-buffer limit ~45KB) or any file-as-input workflow, use `gbrain capture --file PATH --slug SLUG` — capture reads the file as a Buffer with a binary-NUL guard and adds provenance write-through (v0.39.3.0).',
description: 'Write/update a page (markdown with frontmatter). Chunks, embeds, reconciles tags, and (when auto_link/auto_timeline are enabled) extracts + reconciles graph links and timeline entries. On the CLI, `gbrain put SLUG --file PATH` reads content from a file (also `--content` or stdin). For provenance write-through and a binary-NUL guard, prefer `gbrain capture --file PATH --slug SLUG` (v0.39.3.0).',
params: {
slug: { type: 'string', required: true, description: 'Page slug' },
content: { type: 'string', required: true, description: 'Full markdown content with YAML frontmatter' },
@@ -1440,7 +1384,10 @@ const list_pages: Operation = {
params: {
type: { type: 'string', description: 'Filter by page type' },
tag: { type: 'string', description: 'Filter by tag' },
limit: { type: 'number', description: 'Max results (default 50)' },
limit: { type: 'number', description: 'Max results (default 50, capped at 100 — use offset to paginate beyond)' },
// #2876: the 100-row cap was silent and there was no way past it even
// though both engines already support OFFSET on listPages.
offset: { type: 'number', description: 'Skip first N results (pagination; pair with limit)' },
// v0.29 — surface filter that already exists on PageFilters.
updated_after: {
type: 'string',
@@ -1471,6 +1418,10 @@ const list_pages: Operation = {
type: p.type as any,
tag: p.tag as string,
limit: clampSearchLimit(p.limit as number | undefined, 50, 100),
// #2876: thread pagination through (engines already honor offset).
offset: Number.isFinite(p.offset as number) && (p.offset as number) > 0
? Math.floor(p.offset as number)
: undefined,
includeDeleted: (p.include_deleted as boolean) === true,
updated_after: typeof p.updated_after === 'string' ? p.updated_after : undefined,
sort,
@@ -1504,8 +1455,7 @@ const search: Operation = {
const queryText = p.query as string;
const limit = (p.limit as number) || 20;
const offset = (p.offset as number) || 0;
// #2561: unqualified trusted-local search spans federated sources.
const scope = federatedSearchScope(ctx);
const scope = sourceScopeOpts(ctx);
// T4/D5 — per-call mode honored ONLY for trusted/local callers so a remote
// OAuth client can't escalate to the costly tokenmax bundle. Local + unknown
@@ -1667,9 +1617,7 @@ const query: Operation = {
// is spread into BOTH the image-similarity searchVector path and the text
// hybridSearch path below, so both honor the same grant.
const sourceIdParam = typeof p.source_id === 'string' ? p.source_id : undefined;
// #2561: unqualified trusted-local query spans federated sources (per-call
// source_id / remote grants still resolve through resolveRequestedScope).
const querySourceScope = federatedSearchScope(ctx, sourceIdParam);
const querySourceScope = resolveRequestedScope(ctx, sourceIdParam);
// v0.27.1: image-similarity branch. Bypasses hybridSearch (which is
// text-only); embeds the image via embedMultimodal and runs a direct
-39
View File
@@ -353,45 +353,6 @@ export async function resolveSourceWithTier(
return { source_id: 'default', tier: 'seed_default' };
}
/**
* #2561 — compute the federated read scope for an UNQUALIFIED local CLI call.
*
* `sources add --federated` promises that a `config.federated = true` source
* "participates in unqualified `gbrain search` results"
* (docs/guides/multi-source-brains.md). This helper turns that promise into a
* scope: given the resolved source and WHICH tier resolved it, return
* `[resolvedSource, ...other federated source ids]` — or `undefined` when the
* expansion must not apply:
*
* - explicit tiers (`flag` / `env` / `dotfile`): the user named a source;
* scalar scope stands (that IS the qualified case);
* - no other federated source exists: keep the scalar fast path unchanged.
*
* Archived sources are excluded (same rationale as pickSoleNonDefaultSource);
* the archived column is v34+, so fall back to the un-archived query on older
* brains. Callers put the result on `OperationContext.localFederatedSourceIds`
* — consumed only by `federatedSearchScope` and only when `remote === false`.
*/
export async function localFederatedSourceIds(
engine: BrainEngine,
sourceId: string,
tier: SourceTier,
): Promise<string[] | undefined> {
if (tier === 'flag' || tier === 'env' || tier === 'dotfile') return undefined;
let rows: Array<{ id: string }>;
try {
rows = await engine.executeRaw<{ id: string }>(
`SELECT id FROM sources WHERE config->>'federated' = 'true' AND archived = false ORDER BY id`,
);
} catch {
rows = await engine.executeRaw<{ id: string }>(
`SELECT id FROM sources WHERE config->>'federated' = 'true' ORDER BY id`,
);
}
const ids = [sourceId, ...rows.map((r) => r.id).filter((id) => id !== sourceId)];
return ids.length > 1 ? ids : undefined;
}
/** Exposed for tests. */
export const __testing = {
readDotfileWalk,
+33
View File
@@ -1,8 +1,41 @@
import { describe, expect, test } from 'bun:test';
import { mkdtempSync, rmSync, writeFileSync } from 'fs';
import { tmpdir } from 'os';
import { join } from 'path';
import { parseOpArgs } from '../src/cli.ts';
import { operationsByName } from '../src/core/operations.ts';
describe('parseOpArgs', () => {
// #380: `gbrain put SLUG --file PATH` reads content from the file instead
// of silently swallowing the flag and creating an empty page.
test('put --file reads the stdin param (content) from a file', () => {
const dir = mkdtempSync(join(tmpdir(), 'gbrain-put-file-'));
try {
const pagePath = join(dir, 'page.md');
writeFileSync(pagePath, '# From file\n\nBody loaded from --file.\n');
const params = parseOpArgs(operationsByName.put_page, ['concepts/from-file', '--file', pagePath]);
expect(params.slug).toBe('concepts/from-file');
expect(params.content).toBe('# From file\n\nBody loaded from --file.\n');
} finally {
rmSync(dir, { recursive: true, force: true });
}
});
// #380 regression guard: undeclared-but-honored flags must keep passing
// through when unknown flags become hard errors (--source is read by
// makeContext; --json by the output seam; --dry-run by ctx.dryRun).
test('pass-through allowlist flags survive on ops that do not declare them', () => {
const params = parseOpArgs(operationsByName.get_page, [
'people/alice-example', '--source', 'wiki', '--json', '--dry-run',
]);
expect(params).toEqual({
slug: 'people/alice-example',
source: 'wiki',
json: true,
dry_run: true,
});
});
test('--no-<boolean> maps to false without consuming the next flag', () => {
const params = parseOpArgs(operationsByName.query, [
'freshEmbedSourceScope code source',
+71 -1
View File
@@ -1,5 +1,5 @@
import { describe, test, expect } from 'bun:test';
import { existsSync, mkdtempSync, readFileSync, rmSync } from 'fs';
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'fs';
import { tmpdir } from 'os';
import { join } from 'path';
@@ -120,6 +120,76 @@ describe('CLI dispatch integration', () => {
expect(exitCode).toBe(0);
});
// #380 / PR #856: put --help documents the --file input path.
test('put --help documents --file input', async () => {
const proc = Bun.spawn(['bun', 'run', 'src/cli.ts', 'put', '--help'], {
cwd: repoRoot,
stdout: 'pipe',
stderr: 'pipe',
});
const stdout = await new Response(proc.stdout).text();
const exitCode = await proc.exited;
expect(stdout).toContain('Usage: gbrain put');
expect(stdout).toContain('--file <path>');
expect(exitCode).toBe(0);
});
// #380: unknown flags on shared ops are a hard error (previously silently
// swallowed into params — `put --file` created empty pages). parseOpArgs
// runs BEFORE engine connect, so the error must fire without a brain.
test('unknown shared-op flags fail before DB connection', async () => {
const home = mkdtempSync(join(tmpdir(), 'gbrain-cli-unknown-flag-'));
try {
const proc = Bun.spawn(['bun', 'run', 'src/cli.ts', 'get', 'people/alice', '--bogus'], {
cwd: repoRoot,
stdout: 'pipe',
stderr: 'pipe',
env: isolatedEnv(home),
});
const stderr = await new Response(proc.stderr).text();
const exitCode = await proc.exited;
expect(stderr).toContain('Unknown option for gbrain get: --bogus');
expect(stderr).not.toContain('No brain configured');
expect(exitCode).toBe(1);
} finally {
rmSync(home, { recursive: true, force: true });
}
});
test('put rejects combining --file and --content', async () => {
const home = mkdtempSync(join(tmpdir(), 'gbrain-cli-put-conflict-'));
try {
const pagePath = join(home, 'page.md');
writeFileSync(pagePath, 'file body\n');
const proc = Bun.spawn(
['bun', 'run', 'src/cli.ts', 'put', 'a/b', '--content', 'inline', '--file', pagePath],
{ cwd: repoRoot, stdout: 'pipe', stderr: 'pipe', env: isolatedEnv(home) },
);
const stderr = await new Response(proc.stderr).text();
const exitCode = await proc.exited;
expect(stderr).toContain('use only one of --file, --content, or stdin');
expect(exitCode).toBe(1);
} finally {
rmSync(home, { recursive: true, force: true });
}
});
test('put --file with a missing path errors instead of writing an empty page', async () => {
const home = mkdtempSync(join(tmpdir(), 'gbrain-cli-put-missing-file-'));
try {
const proc = Bun.spawn(
['bun', 'run', 'src/cli.ts', 'put', 'a/b', '--file', join(home, 'nope.md')],
{ cwd: repoRoot, stdout: 'pipe', stderr: 'pipe', env: isolatedEnv(home) },
);
const stderr = await new Response(proc.stderr).text();
const exitCode = await proc.exited;
expect(stderr).toContain('cannot read --file');
expect(exitCode).toBe(1);
} finally {
rmSync(home, { recursive: true, force: true });
}
});
test('upgrade --help prints usage without running upgrade', async () => {
const proc = Bun.spawn(['bun', 'run', 'src/cli.ts', 'upgrade', '--help'], {
cwd: repoRoot,
+55
View File
@@ -0,0 +1,55 @@
import { describe, test, expect } from 'bun:test';
import { operationsByName } from '../src/core/operations.ts';
/**
* #2876: `gbrain list --limit` silently clamped at 100 with no pagination.
* list_pages now declares `offset` (both engines already supported it on
* PageFilters) and the limit description discloses the 100-row cap.
*/
describe('list_pages pagination (#2876)', () => {
const listPagesOp = operationsByName.list_pages;
function makeCtx(captured: unknown[]) {
return {
engine: {
listPages: async (filters: unknown) => {
captured.push(filters);
return [];
},
},
config: { engine: 'pglite' },
logger: { info() {}, warn() {}, error() {} },
dryRun: false,
remote: false,
sourceId: 'default',
} as any;
}
test('declares offset param and discloses the 100-row cap on limit', () => {
expect(listPagesOp.params.offset).toBeDefined();
expect(listPagesOp.params.offset.type).toBe('number');
expect(listPagesOp.params.limit.description).toContain('100');
});
test('threads offset through to engine.listPages', async () => {
const captured: any[] = [];
await listPagesOp.handler(makeCtx(captured), { limit: 10, offset: 30 });
expect(captured[0].offset).toBe(30);
expect(captured[0].limit).toBe(10);
});
test('drops negative, non-finite, and zero offsets', async () => {
const captured: any[] = [];
const ctx = makeCtx(captured);
await listPagesOp.handler(ctx, { offset: -5 });
await listPagesOp.handler(ctx, { offset: Infinity });
await listPagesOp.handler(ctx, { offset: 0 });
for (const f of captured) expect(f.offset).toBeUndefined();
});
test('floors fractional offsets', async () => {
const captured: any[] = [];
await listPagesOp.handler(makeCtx(captured), { offset: 7.9 });
expect(captured[0].offset).toBe(7);
});
});
-154
View File
@@ -1,154 +0,0 @@
/**
* #2561 sources.config.federated participates in UNQUALIFIED local CLI
* search/query.
*
* Pre-fix: the local CLI always emitted a scalar `{sourceId}` scope (required
* field, auto-filled 'default'), so a source registered with
* `gbrain sources add --federated` was invisible to an unqualified
* `gbrain search "X"` contradicting docs/guides/multi-source-brains.md
* ("Source participates in unqualified `gbrain search` results").
*
* Fix: the CLI context builder computes `ctx.localFederatedSourceIds`
* (resolved source + every other federated source) whenever the source
* resolved via a NON-explicit tier; `federatedSearchScope` widens the scalar
* scope to that set for the `search` / `query` ops trusted-local only
* (`ctx.remote === false`), never for remote callers, never when a per-call
* `source_id` or an explicit --source/env/dotfile was given.
*/
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
import { localFederatedSourceIds } from '../src/core/source-resolver.ts';
import {
federatedSearchScope,
operations,
type OperationContext,
} from '../src/core/operations.ts';
let engine: PGLiteEngine;
const search = operations.find((o) => o.name === 'search')!;
function ctxOf(overrides: Partial<OperationContext> = {}): OperationContext {
return {
engine: engine as any,
config: {} as any,
logger: console as any,
dryRun: false,
remote: false,
sourceId: 'default',
...overrides,
};
}
beforeAll(async () => {
engine = new PGLiteEngine();
await engine.connect({});
await engine.initSchema();
// Seeded 'default' source is federated=true. Add:
// wiki — federated (must join unqualified search)
// private — NOT federated (must stay invisible unless explicitly named)
// oldnews — federated but archived (must stay excluded)
await engine.executeRaw(
`INSERT INTO sources (id, name, local_path, config) VALUES ('wiki', 'wiki', '/tmp/wiki', '{"federated": true}'::jsonb)`,
);
await engine.executeRaw(
`INSERT INTO sources (id, name, local_path, config) VALUES ('private', 'private', '/tmp/private', '{}'::jsonb)`,
);
await engine.executeRaw(
`INSERT INTO sources (id, name, local_path, config, archived) VALUES ('oldnews', 'oldnews', '/tmp/oldnews', '{"federated": true}'::jsonb, true)`,
);
const pages: Array<[slug: string, sourceId: string, where: string]> = [
['notes/home', 'default', 'default'],
['wiki/topic', 'wiki', 'wiki'],
['private/topic', 'private', 'private'],
['old/topic', 'oldnews', 'oldnews'],
];
for (const [slug, sourceId, where] of pages) {
await engine.putPage(slug, {
type: 'note', title: `Topic in ${where}`, compiled_truth: `the zebra telescope in ${where}`, frontmatter: {},
}, { sourceId });
await engine.upsertChunks(slug, [
{ chunk_index: 0, chunk_text: `the zebra telescope in ${where}`, chunk_source: 'compiled_truth' },
], { sourceId });
}
// Keyword-only search path: no embedding provider needed in tests.
await engine.setConfig('search.mcp_keyword_only', 'true');
}, 60_000);
afterAll(async () => {
if (engine) await engine.disconnect();
}, 60_000);
describe('localFederatedSourceIds — CLI-side scope computation', () => {
test('non-explicit tier: resolved source first, then other federated, archived excluded', async () => {
expect(await localFederatedSourceIds(engine, 'default', 'seed_default')).toEqual(['default', 'wiki']);
});
test('non-federated resolved source still joins its own scope', async () => {
expect(await localFederatedSourceIds(engine, 'private', 'brain_default')).toEqual(['private', 'default', 'wiki']);
});
test('explicit tiers (--source / env / dotfile) never expand', async () => {
expect(await localFederatedSourceIds(engine, 'default', 'flag')).toBeUndefined();
expect(await localFederatedSourceIds(engine, 'default', 'env')).toBeUndefined();
expect(await localFederatedSourceIds(engine, 'default', 'dotfile')).toBeUndefined();
});
test('single federated source (the resolved one) keeps the scalar fast path', async () => {
const solo = { executeRaw: async () => [{ id: 'default' }] } as any;
expect(await localFederatedSourceIds(solo, 'default', 'seed_default')).toBeUndefined();
});
});
describe('federatedSearchScope — trust + explicitness matrix', () => {
test('trusted local + unqualified widens to the federated set', () => {
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx)).toEqual({ sourceIds: ['default', 'wiki'] });
});
test('remote caller NEVER widens (fail-closed), even if the field is set', () => {
const ctx = ctxOf({ remote: true, localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx)).toEqual({ sourceId: 'default' });
});
test('per-call source_id wins over the federated set', () => {
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx, 'wiki')).toEqual({ sourceId: 'wiki' });
});
test('per-call __all__ keeps the whole-brain semantics for trusted local', () => {
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx, '__all__')).toEqual({});
});
test('a federated OAuth grant wins over the local set', () => {
const ctx = ctxOf({
localFederatedSourceIds: ['default', 'wiki'],
auth: { allowedSources: ['a', 'b'] } as OperationContext['auth'],
});
expect(federatedSearchScope(ctx)).toEqual({ sourceIds: ['a', 'b'] });
});
test('no local federated set → unchanged scalar scope', () => {
expect(federatedSearchScope(ctxOf())).toEqual({ sourceId: 'default' });
});
});
describe('search op — unqualified local search spans federated sources', () => {
test('federated source results appear; non-federated + archived stay invisible', async () => {
const ctx = ctxOf({
localFederatedSourceIds: await localFederatedSourceIds(engine, 'default', 'seed_default'),
});
const results = (await search.handler(ctx, { query: 'zebra telescope' })) as Array<{ slug: string }>;
const slugs = results.map((r) => r.slug);
expect(slugs).toContain('notes/home');
expect(slugs).toContain('wiki/topic'); // pre-#2561 this was missing
expect(slugs).not.toContain('private/topic');
expect(slugs).not.toContain('old/topic');
});
test('explicit source resolution (no federated set on ctx) stays single-source', async () => {
const results = (await search.handler(ctxOf(), { query: 'zebra telescope' })) as Array<{ slug: string }>;
const slugs = results.map((r) => r.slug);
expect(slugs).toEqual(['notes/home']);
});
});