Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 c80b8b6757 fix(search): honor sources.config.federated in unqualified local CLI search/query (#2561)
A source registered with `gbrain sources add --federated` was invisible to
an unqualified `gbrain search`/`gbrain query`: the local CLI always emitted
a scalar {sourceId} scope, and nothing on the read path ever consulted
sources.config.federated — contradicting docs/guides/multi-source-brains.md
('Source participates in unqualified gbrain search results').

Fix, at the trusted-local boundary only:
- src/cli.ts makeContext resolves the source WITH its tier and, when the
  tier is non-explicit (local_path / brain_default / sole_non_default /
  seed_default), computes ctx.localFederatedSourceIds = [resolved source,
  ...other config.federated=true sources] (archived excluded).
- New federatedSearchScope (operations.ts) delegates to
  resolveRequestedScope, then widens an unqualified trusted-local scalar
  scope to that set. Used by the search + query handlers only.
- Expansion NEVER applies when ctx.remote !== false (fail-closed source
  isolation), when a per-call source_id/__all__ is passed, when an OAuth
  grant (allowedSources) is present, or when --source/GBRAIN_SOURCE/dotfile
  named the source explicitly.

Deliberately NOT inside sourceScopeOpts: code-intel ops reject multi-source
scopes (resolveCodeIntelScope) and non-search reads keep their scalar
behavior. Cache contamination is already handled — cacheScopeKey folds
sourceIds sets into the query-cache key.

Fixes #2561

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:43:36 -07:00
11 changed files with 271 additions and 309 deletions
+11 -2
View File
@@ -808,12 +808,20 @@ async function makeContext(engine: BrainEngine, params: Record<string, unknown>)
// 'default'. Wrapped in try/catch so a doctor / single-source brain that
// never set up sources still returns 'default' silently.
let sourceId: string | undefined;
// #2561: when the source resolved via a NON-explicit tier (path-match /
// brain default / sole-non-default / seed default), unqualified search-shaped
// reads span every `config.federated = true` source. Computed here (the
// trusted local boundary) and consumed by federatedSearchScope in
// operations.ts, which additionally gates on ctx.remote === false.
let localFederated: string[] | undefined;
try {
const { resolveSourceId } = await import('./core/source-resolver.ts');
const { resolveSourceWithTier, localFederatedSourceIds } = await import('./core/source-resolver.ts');
// params.source is set when a CLI flag was parsed for the op (rare; most
// CLI ops don't take --source). Falls through to env/dotfile/path-match.
const explicit = (params.source as string | undefined) ?? null;
sourceId = await resolveSourceId(engine, explicit);
const resolved = await resolveSourceWithTier(engine, explicit);
sourceId = resolved.source_id;
localFederated = await localFederatedSourceIds(engine, resolved.source_id, resolved.tier);
} catch {
// Source resolution failed (e.g. sources table doesn't exist on a fresh
// pre-init brain). Leave sourceId unset; engine read methods fall through
@@ -834,6 +842,7 @@ async function makeContext(engine: BrainEngine, params: Record<string, unknown>)
// table). Matches dispatch.ts's auto-fill so the contract holds across
// every transport.
sourceId: sourceId ?? 'default',
...(localFederated ? { localFederatedSourceIds: localFederated } : {}),
};
}
+3 -22
View File
@@ -433,10 +433,7 @@ export async function extractLinksFromFile(
async resolve(name: string, dirHint?: string | string[]): Promise<string | null> {
if (!name) return null;
const trimmed = name.trim();
// Same broadened slug-shape as makeResolver step 1: accepts
// digit-leading folders (`90-people/nicolai`) and nested paths.
// Exact Set membership guards it — no false positives.
if (/\//.test(trimmed) && /^[a-z0-9][a-z0-9/_-]*$/.test(trimmed) && allSlugs.has(trimmed)) {
if (/^[a-z][a-z0-9-]*\/[a-z0-9][a-z0-9-]*$/.test(trimmed) && allSlugs.has(trimmed)) {
return trimmed;
}
const hints = Array.isArray(dirHint) ? dirHint : (dirHint ? [dirHint] : []);
@@ -585,17 +582,6 @@ export interface ExtractOpts {
* before (single-'default'-source brains unaffected).
*/
sourceId?: string;
/**
* v0.42 — also extract frontmatter links on the incremental (slugs) path.
* `extractForSlugs` extracts BODY links only by default; set this true to also
* parse each changed page's frontmatter so `sources:`/`related:` edges stay fresh
* when YAML is edited externally and synced in. Applied PER changed page, so the
* incremental walk stays bounded (no switch to a full DB scan). Only honored on
* the incremental path (`slugs` defined); the full-walk path already covers
* frontmatter via its own dispatch. Gated upstream by the config key
* `autopilot.incremental_extract_include_frontmatter` (default off).
*/
includeFrontmatter?: boolean;
}
/**
@@ -634,7 +620,7 @@ export async function runExtractCore(engine: BrainEngine, opts: ExtractOpts): Pr
// Nothing changed — skip entirely.
return result;
}
const r = await extractForSlugs(engine, opts.dir, opts.slugs, opts.mode, dryRun, jsonMode, workers, opts.signal, opts.sourceId, opts.includeFrontmatter);
const r = await extractForSlugs(engine, opts.dir, opts.slugs, opts.mode, dryRun, jsonMode, workers, opts.signal, opts.sourceId);
result.links_created = r.links_created;
result.timeline_entries_created = r.timeline_created;
result.pages_processed = r.pages;
@@ -1025,11 +1011,6 @@ async function extractForSlugs(
signal?: AbortSignal,
// #1747/#1503: stamp resolved brain source id on batch rows (see ExtractOpts.sourceId).
sourceId?: string,
// v0.42: when true, also extract frontmatter links per changed page so
// externally-edited YAML (`sources:`/`related:`) stays fresh on the cycle.
// Default false preserves the body-only incremental behavior. Gated upstream
// by `autopilot.incremental_extract_include_frontmatter`.
includeFrontmatter: boolean = false,
): Promise<{ links_created: number; timeline_created: number; pages: number }> {
// Build the full slug set for link resolution (fast: just readdir, no file reads)
const allFiles = walkMarkdownFiles(brainDir);
@@ -1104,7 +1085,7 @@ async function extractForSlugs(
const content = readFileSync(fullPath, 'utf-8');
if (doLinks) {
const links = await extractLinksFromFile(content, relPath, allSlugs, { globalBasename, includeFrontmatter });
const links = await extractLinksFromFile(content, relPath, allSlugs, { globalBasename });
for (const link of links) {
if (dryRun) {
if (!jsonMode) console.log(` ${link.from_slug}${link.to_slug} (${link.link_type})`);
-12
View File
@@ -121,18 +121,6 @@ export interface GBrainConfig {
/** Daily spend cap (USD); bounds drains/day = floor(cap / ~$0.30). Default 2.0. */
max_usd_per_day?: number;
};
/**
* v0.42 — keep frontmatter links fresh on the incremental cycle. The cycle's
* extract phase re-extracts only the slugs a sync changed, but `extractForSlugs`
* extracts BODY links only — frontmatter (`sources:`/`related:` etc.) link edges
* silently drift stale when a page's YAML is edited externally and synced in.
* Set true to also extract frontmatter links per changed page each cycle, keeping
* externally-edited YAML edges fresh without a full rescan. Default false
* (preserves current behavior). Read via the file/env/DB plane in the cycle's
* extract dispatch. Disable/enable with
* `gbrain config set autopilot.incremental_extract_include_frontmatter <bool>`.
*/
incremental_extract_include_frontmatter?: boolean;
};
eval?: {
/** false disables capture entirely. Defaults to true. */
-16
View File
@@ -996,21 +996,6 @@ async function runPhaseExtract(
): Promise<PhaseResult> {
try {
const { runExtractCore } = await import('../commands/extract.ts');
const { loadConfig } = await import('./config.ts');
// Default off: the incremental cycle extracts body links only unless the
// operator opts in to keeping externally-edited frontmatter links fresh too.
// Both planes, file wins (env > file > DB precedence, per loadConfigWithEngine):
// `gbrain config set autopilot.incremental_extract_include_frontmatter true`
// writes the DB plane (engine.setConfig), so a file-plane-only read here
// would make the documented enable command a silent no-op (#2120 class).
const fileVal = loadConfig()?.autopilot?.incremental_extract_include_frontmatter;
let includeFrontmatter = fileVal === true;
if (fileVal === undefined) {
try {
includeFrontmatter =
(await engine.getConfig('autopilot.incremental_extract_include_frontmatter')) === 'true';
} catch { /* config table unreadable → default off */ }
}
// Extract is read-mostly against the filesystem + write to links table.
// Honor dryRun by skipping with a 'skipped' entry: extract doesn't have
// a clean dry-run mode today and runCycle should be honest about it.
@@ -1031,7 +1016,6 @@ async function runPhaseExtract(
slugs: changedSlugs, // undefined = full walk (first run / manual)
signal,
sourceId,
includeFrontmatter, // honored on the incremental (slugs) path only
});
const linksCreated = result?.links_created ?? 0;
const timelineCreated = result?.timeline_entries_created ?? 0;
+3 -36
View File
@@ -942,17 +942,8 @@ export function makeResolver(
const hints = Array.isArray(dirHint) ? dirHint : (dirHint ? [dirHint] : []);
// Step 1: already a slug? Try an exact page lookup for any slug-shaped
// value (contains '/', slug charset). Broadened beyond the original
// single-segment lowercase-leading form (`^[a-z][a-z0-9-]*\/[a-z0-9]...`)
// to also accept digit-leading folders (`90-people/nicolai`,
// `01-trading/...`) and nested paths (`a/b/c`) — common in PARA-numbered
// vaults. This is an EXACT getPage match only — no fuzzy — so it never
// produces a false positive; a non-existent slug just falls through to
// the steps below. Fixes frontmatter `related: [[dir/slug]]` values
// (unwrapped by unwrapWikilink) that name a real page the strict regex
// could not reach and whose full-path fuzzy score is below threshold.
if (/\//.test(trimmed) && /^[a-z0-9][a-z0-9/_-]*$/.test(trimmed)) {
// Step 1: already a slug? (dir/name shape, lowercase, hyphenated)
if (/^[a-z][a-z0-9-]*\/[a-z0-9][a-z0-9-]*$/.test(trimmed)) {
const page = await engine.getPage(trimmed);
if (page) {
cache.set(cacheKey, trimmed);
@@ -1012,25 +1003,6 @@ export function makeResolver(
// ─── Frontmatter extractor ──────────────────────────────────────
/**
* Unwrap an Obsidian `[[wikilink]]` frontmatter value to its bare link
* target so the resolver (which expects bare titles / dir slugs) can match
* it. Mainstream Obsidian authors frontmatter links as `related: ["[[Page]]"]`;
* without this, the resolver treats the brackets as part of the value and a
* `[[90-people/nicolai]]` is normalized into `90peoplenicolai`, so it never
* resolves. Strips a trailing `|alias`, `#heading`, or `^block` suffix — the
* link target only. The regex is anchored to a wholly-wrapped value
* (`^\s*\[\[…\]\]\s*$`), so bare titles and any value not fully wrapped pass
* through unchanged and existing behavior is preserved exactly.
*/
export function unwrapWikilink(value: string): string {
const match = /^\s*\[\[(.+?)\]\]\s*$/.exec(value);
if (!match) return value;
// Take the link target: drop |alias, then #heading / ^block suffixes.
const target = match[1].split('|')[0].split('#')[0].split('^')[0];
return target.trim();
}
export interface UnresolvedFrontmatterRef {
/** The frontmatter field name. */
field: string;
@@ -1088,12 +1060,7 @@ export async function extractFrontmatterLinks(
}
if (!name) continue; // skip numbers, nulls, malformed objects
// Accept Obsidian `[[wikilink]]` values in frontmatter link fields by
// unwrapping to the bare target before resolution. Bare titles pass
// through unchanged; the original `name` is preserved for the
// unresolved report and edge context.
const linkTarget = unwrapWikilink(name);
const resolved = await resolver.resolve(linkTarget, mapping.dirHint);
const resolved = await resolver.resolve(name, mapping.dirHint);
if (!resolved) {
unresolved.push({ field, name });
continue;
+61 -2
View File
@@ -424,6 +424,23 @@ export interface OperationContext {
* satisfied even on single-source brains.
*/
sourceId: string;
/**
* #2561 — federated read scope for UNQUALIFIED local CLI reads.
*
* Set ONLY by the local CLI's context builder (src/cli.ts makeContext), and
* only when the source resolved via a non-explicit tier (local_path /
* brain_default / sole_non_default / seed_default — NOT --source, NOT
* GBRAIN_SOURCE, NOT a .gbrain-source dotfile). Contains the resolved
* source first, then every other `config.federated = true` source, so an
* unqualified `gbrain search "X"` spans federated sources as
* docs/guides/multi-source-brains.md promises.
*
* Consumed exclusively by `federatedSearchScope` and ONLY when
* `ctx.remote === false` — a remote caller's scope stays governed by
* `ctx.auth.allowedSources` / scalar `ctx.sourceId` (source-isolation
* invariant, fail-closed).
*/
localFederatedSourceIds?: string[];
}
/**
@@ -539,6 +556,45 @@ export function resolveRequestedScope(
return sourceScopeOpts(ctx);
}
/**
* #2561 — source scope for the search-shaped read ops (`search`, `query`).
*
* Delegates to `resolveRequestedScope` (the single trust+grant resolver), then
* widens an UNQUALIFIED trusted-local scalar scope to the CLI-computed
* federated set (`ctx.localFederatedSourceIds`, resolved source first). This is
* what makes `sources add --federated` mean something for local search: a
* federated source participates in unqualified `gbrain search "X"` results.
*
* The expansion NEVER applies when:
* - the caller is not strictly trusted-local (`ctx.remote !== false`) —
* remote scope stays grant-governed (fail-closed source isolation);
* - a per-call `source_id` was passed (explicit wins, including `__all__`);
* - the resolver already produced a federated array (OAuth grant);
* - the CLI resolved the source from an explicit signal (--source / env /
* dotfile) — makeContext leaves `localFederatedSourceIds` unset then.
*
* Deliberately NOT inside `sourceScopeOpts`: code-intel ops collapse a
* multi-element scope to an error (`resolveCodeIntelScope`), and non-search
* reads (get_page, get_links, …) keep their long-standing scalar behavior.
*/
export function federatedSearchScope(
ctx: OperationContext,
sourceIdParam?: string,
): { sourceId?: string; sourceIds?: string[] } {
const scope = resolveRequestedScope(ctx, sourceIdParam);
if (
ctx.remote === false &&
sourceIdParam === undefined &&
scope.sourceId !== undefined &&
scope.sourceIds === undefined &&
ctx.localFederatedSourceIds !== undefined &&
ctx.localFederatedSourceIds.length > 1
) {
return { sourceIds: ctx.localFederatedSourceIds };
}
return scope;
}
/**
* Code-intel adapter for `resolveRequestedScope`. Graph traversal
* (code_callers/code_callees/code_blast/code_flow) is single-source by design —
@@ -1448,7 +1504,8 @@ const search: Operation = {
const queryText = p.query as string;
const limit = (p.limit as number) || 20;
const offset = (p.offset as number) || 0;
const scope = sourceScopeOpts(ctx);
// #2561: unqualified trusted-local search spans federated sources.
const scope = federatedSearchScope(ctx);
// T4/D5 — per-call mode honored ONLY for trusted/local callers so a remote
// OAuth client can't escalate to the costly tokenmax bundle. Local + unknown
@@ -1610,7 +1667,9 @@ const query: Operation = {
// is spread into BOTH the image-similarity searchVector path and the text
// hybridSearch path below, so both honor the same grant.
const sourceIdParam = typeof p.source_id === 'string' ? p.source_id : undefined;
const querySourceScope = resolveRequestedScope(ctx, sourceIdParam);
// #2561: unqualified trusted-local query spans federated sources (per-call
// source_id / remote grants still resolve through resolveRequestedScope).
const querySourceScope = federatedSearchScope(ctx, sourceIdParam);
// v0.27.1: image-similarity branch. Bypasses hybridSearch (which is
// text-only); embeds the image via embedMultimodal and runs a direct
+39
View File
@@ -353,6 +353,45 @@ export async function resolveSourceWithTier(
return { source_id: 'default', tier: 'seed_default' };
}
/**
* #2561 — compute the federated read scope for an UNQUALIFIED local CLI call.
*
* `sources add --federated` promises that a `config.federated = true` source
* "participates in unqualified `gbrain search` results"
* (docs/guides/multi-source-brains.md). This helper turns that promise into a
* scope: given the resolved source and WHICH tier resolved it, return
* `[resolvedSource, ...other federated source ids]` — or `undefined` when the
* expansion must not apply:
*
* - explicit tiers (`flag` / `env` / `dotfile`): the user named a source;
* scalar scope stands (that IS the qualified case);
* - no other federated source exists: keep the scalar fast path unchanged.
*
* Archived sources are excluded (same rationale as pickSoleNonDefaultSource);
* the archived column is v34+, so fall back to the un-archived query on older
* brains. Callers put the result on `OperationContext.localFederatedSourceIds`
* — consumed only by `federatedSearchScope` and only when `remote === false`.
*/
export async function localFederatedSourceIds(
engine: BrainEngine,
sourceId: string,
tier: SourceTier,
): Promise<string[] | undefined> {
if (tier === 'flag' || tier === 'env' || tier === 'dotfile') return undefined;
let rows: Array<{ id: string }>;
try {
rows = await engine.executeRaw<{ id: string }>(
`SELECT id FROM sources WHERE config->>'federated' = 'true' AND archived = false ORDER BY id`,
);
} catch {
rows = await engine.executeRaw<{ id: string }>(
`SELECT id FROM sources WHERE config->>'federated' = 'true' ORDER BY id`,
);
}
const ids = [sourceId, ...rows.map((r) => r.id).filter((id) => id !== sourceId)];
return ids.length > 1 ? ids : undefined;
}
/** Exposed for tests. */
export const __testing = {
readDotfileWalk,
-34
View File
@@ -191,37 +191,3 @@ describe('runExtractCore — incremental cycle path (#417)', () => {
expect(result.links_created).toBeGreaterThan(0);
});
});
describe('runExtractCore — incremental frontmatter gate (includeFrontmatter)', () => {
// alice has a `source:` frontmatter edge but NO body links. The incremental
// path extracts body links only by default, so the frontmatter edge is the
// sole signal that distinguishes the gate off vs on.
const aliceFm = '---\nsource: companies/acme-example\n---\n# alice';
test('9. default (flag omitted) does NOT extract frontmatter links on the incremental path', async () => {
await seedPage('companies/acme-example', '# acme');
await seedPage('people/alice-example', aliceFm);
const result = await runExtractCore(engine as unknown as BrainEngine, {
mode: 'all',
dir: tempDir,
slugs: ['people/alice-example'],
});
// alice's only potential edge is her frontmatter `source:`; with the gate off
// it must not be extracted (preserves the body-only incremental behavior).
expect(result.pages_processed).toBe(1);
expect(result.links_created).toBe(0);
});
test('10. includeFrontmatter: true extracts the frontmatter link on the incremental path', async () => {
await seedPage('companies/acme-example', '# acme');
await seedPage('people/alice-example', aliceFm);
const result = await runExtractCore(engine as unknown as BrainEngine, {
mode: 'all',
dir: tempDir,
slugs: ['people/alice-example'],
includeFrontmatter: true,
});
// Same page, gate on → the `source:` frontmatter edge is now extracted.
expect(result.pages_processed).toBe(1);
expect(result.links_created).toBeGreaterThan(0);
});
});
-12
View File
@@ -76,18 +76,6 @@ describe('extractLinksFromFile', () => {
}
});
it('resolves wrapped [[wikilink]] digit-leading slug-path in frontmatter (fs resolver, broadened step 1)', async () => {
// Same bug class as makeResolver step 1 (#1983): the fs resolver's strict
// `^[a-z]…` slug regex rejected digit-leading / nested paths, so a PARA-vault
// `related: "[[90-people/nicolai]]"` never resolved even though the page exists.
const content = '---\nrelated: "[[90-people/nicolai]]"\ntype: concept\n---\nContent.';
const allSlugs = new Set(['wiki/note', '90-people/nicolai']);
const links = await extractLinksFromFile(content, 'wiki/note.md', allSlugs, { includeFrontmatter: true });
const related = links.filter(l => l.link_type === 'related_to');
expect(related).toHaveLength(1);
expect(related[0].to_slug).toBe('90-people/nicolai');
});
it('frontmatter extraction is default OFF (back-compat)', async () => {
// Without includeFrontmatter, fs-source no longer auto-extracts frontmatter.
// Matches db-source behavior. User opts in with --include-frontmatter flag.
-173
View File
@@ -9,7 +9,6 @@ import {
parseTimelineEntries,
isAutoLinkEnabled,
FRONTMATTER_LINK_MAP,
unwrapWikilink,
type SlugResolver,
} from '../src/core/link-extraction.ts';
import type { BrainEngine } from '../src/core/engine.ts';
@@ -1292,175 +1291,3 @@ describe('parseTimelineEntries — Format 3: inline [Source: ..., YYYY-MM-DD] ci
expect(parseTimelineEntries('[Source: import batch, 2025-07-01]')).toHaveLength(0);
});
});
// ─── Frontmatter [[wikilink]] + slug-path resolution ──────────────────────
// Mainstream Obsidian authors frontmatter links as `related: ["[[Page]]"]`,
// and PARA-numbered vaults use digit-leading / nested slug paths like
// `[[90-people/nicolai]]`. Both were silently dropped: brackets were treated
// as part of the value and the step-1 slug regex (`^[a-z]…`) rejected
// digit-leading / nested paths, while full-path fuzzy scored below threshold.
// Fix: unwrapWikilink() before resolution + an exact getPage() for any
// slug-shaped value (exact-match only → no false positives).
describe('unwrapWikilink', () => {
test('wrapped title → bare title', () => {
expect(unwrapWikilink('[[Monday Range]]')).toBe('Monday Range');
});
test('wrapped slug-path (digit-leading folder) → bare slug', () => {
expect(unwrapWikilink('[[90-people/nicolai]]')).toBe('90-people/nicolai');
});
test('wrapped nested slug-path → bare slug', () => {
expect(unwrapWikilink('[[01-trading/wiki/strategies/opening-range-breakout]]'))
.toBe('01-trading/wiki/strategies/opening-range-breakout');
});
test('strips |alias', () => {
expect(unwrapWikilink('[[90-people/nicolai|Nicolai]]')).toBe('90-people/nicolai');
});
test('strips #heading', () => {
expect(unwrapWikilink('[[Page#Section]]')).toBe('Page');
});
test('strips ^block', () => {
expect(unwrapWikilink('[[Page^abc123]]')).toBe('Page');
});
test('surrounding whitespace tolerated', () => {
expect(unwrapWikilink(' [[Page]] ')).toBe('Page');
});
test('bare title passes through unchanged', () => {
expect(unwrapWikilink('Monday Range')).toBe('Monday Range');
});
test('bare slug passes through unchanged', () => {
expect(unwrapWikilink('90-people/nicolai')).toBe('90-people/nicolai');
});
test('partially-wrapped value is NOT unwrapped (anchored)', () => {
// Not a wholly-wrapped value → left intact so existing behavior is exact.
expect(unwrapWikilink('see [[Page]] for detail')).toBe('see [[Page]] for detail');
});
});
describe('makeResolver — slug-path exact getPage (step 1 broadened)', () => {
function fakeEngine(
slugs: string[],
fuzzyMap: Map<string, { slug: string; similarity: number }> = new Map(),
): BrainEngine {
const lookup = new Set(slugs);
return {
async getPage(slug: string) { return lookup.has(slug) ? { slug } as any : null; },
async findByTitleFuzzy(name: string) { return fuzzyMap.get(name) ?? null; },
async searchKeyword() { return []; },
} as unknown as BrainEngine;
}
test('digit-leading folder slug resolves via exact getPage', async () => {
const r = makeResolver(fakeEngine(['90-people/nicolai']));
expect(await r.resolve('90-people/nicolai')).toBe('90-people/nicolai');
});
test('nested (>2 segment) slug resolves via exact getPage', async () => {
const r = makeResolver(fakeEngine(['01-trading/wiki/strategies/opening-range-breakout']));
expect(await r.resolve('01-trading/wiki/strategies/opening-range-breakout'))
.toBe('01-trading/wiki/strategies/opening-range-breakout');
});
test('regression: single-segment lowercase slug still resolves', async () => {
const r = makeResolver(fakeEngine(['people/pedro']));
expect(await r.resolve('people/pedro')).toBe('people/pedro');
});
test('exact-only: slug-shaped value with no matching page falls through (no false positive)', async () => {
// `90-people/ghost` is slug-shaped but absent → step-1 getPage misses,
// no fuzzy hit → null. Never invents an edge.
const r = makeResolver(fakeEngine(['90-people/nicolai']));
expect(await r.resolve('90-people/ghost')).toBeNull();
});
test('non-slug value still routes to fuzzy', async () => {
const r = makeResolver(fakeEngine(
['01-trading/monday-range'],
new Map([['Monday Range', { slug: '01-trading/monday-range', similarity: 1 }]]),
));
expect(await r.resolve('Monday Range')).toBe('01-trading/monday-range');
});
});
describe('extractFrontmatterLinks — [[wikilink]] related: values (end-to-end)', () => {
function fakeEngine(
slugs: string[],
fuzzyMap: Map<string, { slug: string; similarity: number }> = new Map(),
): BrainEngine {
const lookup = new Set(slugs);
return {
async getPage(slug: string) { return lookup.has(slug) ? { slug } as any : null; },
async findByTitleFuzzy(name: string) { return fuzzyMap.get(name) ?? null; },
async searchKeyword() { return []; },
} as unknown as BrainEngine;
}
test('wrapped slug-path related: resolves (the core win)', async () => {
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
const { candidates, unresolved } = await extractFrontmatterLinks(
'wiki/originals/ideas/note', 'note' as never,
{ related: '[[90-people/nicolai]]' }, resolver,
);
expect(unresolved).toHaveLength(0);
expect(candidates).toHaveLength(1);
expect(candidates[0]).toMatchObject({
fromSlug: 'wiki/originals/ideas/note',
targetSlug: '90-people/nicolai',
linkType: 'related_to',
linkSource: 'frontmatter',
});
});
test('wrapped nested slug-path related: resolves', async () => {
const resolver = makeResolver(fakeEngine(['01-trading/wiki/strategies/opening-range-breakout']));
const { candidates } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: ['[[01-trading/wiki/strategies/opening-range-breakout]]'] }, resolver,
);
expect(candidates).toHaveLength(1);
expect(candidates[0].targetSlug).toBe('01-trading/wiki/strategies/opening-range-breakout');
});
test('wrapped value with |alias resolves to the target', async () => {
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
const { candidates } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: '[[90-people/nicolai|Nicolai]]' }, resolver,
);
expect(candidates).toHaveLength(1);
expect(candidates[0].targetSlug).toBe('90-people/nicolai');
});
test('regression: bare slug related: still resolves', async () => {
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
const { candidates } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: '90-people/nicolai' }, resolver,
);
expect(candidates).toHaveLength(1);
expect(candidates[0].targetSlug).toBe('90-people/nicolai');
});
test('regression: wrapped title resolves via fuzzy (brackets harmless)', async () => {
const resolver = makeResolver(fakeEngine(
['01-trading/monday-range'],
new Map([['Monday Range', { slug: '01-trading/monday-range', similarity: 1 }]]),
));
const { candidates } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: '[[Monday Range]]' }, resolver,
);
expect(candidates).toHaveLength(1);
expect(candidates[0].targetSlug).toBe('01-trading/monday-range');
});
test('unknown wrapped slug → unresolved (no crash), original value preserved', async () => {
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
const { candidates, unresolved } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: '[[99-archive/does-not-exist]]' }, resolver,
);
expect(candidates).toHaveLength(0);
expect(unresolved).toHaveLength(1);
expect(unresolved[0]).toEqual({ field: 'related', name: '[[99-archive/does-not-exist]]' });
});
});
+154
View File
@@ -0,0 +1,154 @@
/**
* #2561 sources.config.federated participates in UNQUALIFIED local CLI
* search/query.
*
* Pre-fix: the local CLI always emitted a scalar `{sourceId}` scope (required
* field, auto-filled 'default'), so a source registered with
* `gbrain sources add --federated` was invisible to an unqualified
* `gbrain search "X"` contradicting docs/guides/multi-source-brains.md
* ("Source participates in unqualified `gbrain search` results").
*
* Fix: the CLI context builder computes `ctx.localFederatedSourceIds`
* (resolved source + every other federated source) whenever the source
* resolved via a NON-explicit tier; `federatedSearchScope` widens the scalar
* scope to that set for the `search` / `query` ops trusted-local only
* (`ctx.remote === false`), never for remote callers, never when a per-call
* `source_id` or an explicit --source/env/dotfile was given.
*/
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
import { localFederatedSourceIds } from '../src/core/source-resolver.ts';
import {
federatedSearchScope,
operations,
type OperationContext,
} from '../src/core/operations.ts';
let engine: PGLiteEngine;
const search = operations.find((o) => o.name === 'search')!;
function ctxOf(overrides: Partial<OperationContext> = {}): OperationContext {
return {
engine: engine as any,
config: {} as any,
logger: console as any,
dryRun: false,
remote: false,
sourceId: 'default',
...overrides,
};
}
beforeAll(async () => {
engine = new PGLiteEngine();
await engine.connect({});
await engine.initSchema();
// Seeded 'default' source is federated=true. Add:
// wiki — federated (must join unqualified search)
// private — NOT federated (must stay invisible unless explicitly named)
// oldnews — federated but archived (must stay excluded)
await engine.executeRaw(
`INSERT INTO sources (id, name, local_path, config) VALUES ('wiki', 'wiki', '/tmp/wiki', '{"federated": true}'::jsonb)`,
);
await engine.executeRaw(
`INSERT INTO sources (id, name, local_path, config) VALUES ('private', 'private', '/tmp/private', '{}'::jsonb)`,
);
await engine.executeRaw(
`INSERT INTO sources (id, name, local_path, config, archived) VALUES ('oldnews', 'oldnews', '/tmp/oldnews', '{"federated": true}'::jsonb, true)`,
);
const pages: Array<[slug: string, sourceId: string, where: string]> = [
['notes/home', 'default', 'default'],
['wiki/topic', 'wiki', 'wiki'],
['private/topic', 'private', 'private'],
['old/topic', 'oldnews', 'oldnews'],
];
for (const [slug, sourceId, where] of pages) {
await engine.putPage(slug, {
type: 'note', title: `Topic in ${where}`, compiled_truth: `the zebra telescope in ${where}`, frontmatter: {},
}, { sourceId });
await engine.upsertChunks(slug, [
{ chunk_index: 0, chunk_text: `the zebra telescope in ${where}`, chunk_source: 'compiled_truth' },
], { sourceId });
}
// Keyword-only search path: no embedding provider needed in tests.
await engine.setConfig('search.mcp_keyword_only', 'true');
}, 60_000);
afterAll(async () => {
if (engine) await engine.disconnect();
}, 60_000);
describe('localFederatedSourceIds — CLI-side scope computation', () => {
test('non-explicit tier: resolved source first, then other federated, archived excluded', async () => {
expect(await localFederatedSourceIds(engine, 'default', 'seed_default')).toEqual(['default', 'wiki']);
});
test('non-federated resolved source still joins its own scope', async () => {
expect(await localFederatedSourceIds(engine, 'private', 'brain_default')).toEqual(['private', 'default', 'wiki']);
});
test('explicit tiers (--source / env / dotfile) never expand', async () => {
expect(await localFederatedSourceIds(engine, 'default', 'flag')).toBeUndefined();
expect(await localFederatedSourceIds(engine, 'default', 'env')).toBeUndefined();
expect(await localFederatedSourceIds(engine, 'default', 'dotfile')).toBeUndefined();
});
test('single federated source (the resolved one) keeps the scalar fast path', async () => {
const solo = { executeRaw: async () => [{ id: 'default' }] } as any;
expect(await localFederatedSourceIds(solo, 'default', 'seed_default')).toBeUndefined();
});
});
describe('federatedSearchScope — trust + explicitness matrix', () => {
test('trusted local + unqualified widens to the federated set', () => {
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx)).toEqual({ sourceIds: ['default', 'wiki'] });
});
test('remote caller NEVER widens (fail-closed), even if the field is set', () => {
const ctx = ctxOf({ remote: true, localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx)).toEqual({ sourceId: 'default' });
});
test('per-call source_id wins over the federated set', () => {
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx, 'wiki')).toEqual({ sourceId: 'wiki' });
});
test('per-call __all__ keeps the whole-brain semantics for trusted local', () => {
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx, '__all__')).toEqual({});
});
test('a federated OAuth grant wins over the local set', () => {
const ctx = ctxOf({
localFederatedSourceIds: ['default', 'wiki'],
auth: { allowedSources: ['a', 'b'] } as OperationContext['auth'],
});
expect(federatedSearchScope(ctx)).toEqual({ sourceIds: ['a', 'b'] });
});
test('no local federated set → unchanged scalar scope', () => {
expect(federatedSearchScope(ctxOf())).toEqual({ sourceId: 'default' });
});
});
describe('search op — unqualified local search spans federated sources', () => {
test('federated source results appear; non-federated + archived stay invisible', async () => {
const ctx = ctxOf({
localFederatedSourceIds: await localFederatedSourceIds(engine, 'default', 'seed_default'),
});
const results = (await search.handler(ctx, { query: 'zebra telescope' })) as Array<{ slug: string }>;
const slugs = results.map((r) => r.slug);
expect(slugs).toContain('notes/home');
expect(slugs).toContain('wiki/topic'); // pre-#2561 this was missing
expect(slugs).not.toContain('private/topic');
expect(slugs).not.toContain('old/topic');
});
test('explicit source resolution (no federated set on ctx) stays single-source', async () => {
const results = (await search.handler(ctxOf(), { query: 'zebra telescope' })) as Array<{ slug: string }>;
const slugs = results.map((r) => r.slug);
expect(slugs).toEqual(['notes/home']);
});
});