mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-17 10:22:34 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c80b8b6757 |
+11
-2
@@ -808,12 +808,20 @@ async function makeContext(engine: BrainEngine, params: Record<string, unknown>)
|
||||
// 'default'. Wrapped in try/catch so a doctor / single-source brain that
|
||||
// never set up sources still returns 'default' silently.
|
||||
let sourceId: string | undefined;
|
||||
// #2561: when the source resolved via a NON-explicit tier (path-match /
|
||||
// brain default / sole-non-default / seed default), unqualified search-shaped
|
||||
// reads span every `config.federated = true` source. Computed here (the
|
||||
// trusted local boundary) and consumed by federatedSearchScope in
|
||||
// operations.ts, which additionally gates on ctx.remote === false.
|
||||
let localFederated: string[] | undefined;
|
||||
try {
|
||||
const { resolveSourceId } = await import('./core/source-resolver.ts');
|
||||
const { resolveSourceWithTier, localFederatedSourceIds } = await import('./core/source-resolver.ts');
|
||||
// params.source is set when a CLI flag was parsed for the op (rare; most
|
||||
// CLI ops don't take --source). Falls through to env/dotfile/path-match.
|
||||
const explicit = (params.source as string | undefined) ?? null;
|
||||
sourceId = await resolveSourceId(engine, explicit);
|
||||
const resolved = await resolveSourceWithTier(engine, explicit);
|
||||
sourceId = resolved.source_id;
|
||||
localFederated = await localFederatedSourceIds(engine, resolved.source_id, resolved.tier);
|
||||
} catch {
|
||||
// Source resolution failed (e.g. sources table doesn't exist on a fresh
|
||||
// pre-init brain). Leave sourceId unset; engine read methods fall through
|
||||
@@ -834,6 +842,7 @@ async function makeContext(engine: BrainEngine, params: Record<string, unknown>)
|
||||
// table). Matches dispatch.ts's auto-fill so the contract holds across
|
||||
// every transport.
|
||||
sourceId: sourceId ?? 'default',
|
||||
...(localFederated ? { localFederatedSourceIds: localFederated } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
+3
-22
@@ -433,10 +433,7 @@ export async function extractLinksFromFile(
|
||||
async resolve(name: string, dirHint?: string | string[]): Promise<string | null> {
|
||||
if (!name) return null;
|
||||
const trimmed = name.trim();
|
||||
// Same broadened slug-shape as makeResolver step 1: accepts
|
||||
// digit-leading folders (`90-people/nicolai`) and nested paths.
|
||||
// Exact Set membership guards it — no false positives.
|
||||
if (/\//.test(trimmed) && /^[a-z0-9][a-z0-9/_-]*$/.test(trimmed) && allSlugs.has(trimmed)) {
|
||||
if (/^[a-z][a-z0-9-]*\/[a-z0-9][a-z0-9-]*$/.test(trimmed) && allSlugs.has(trimmed)) {
|
||||
return trimmed;
|
||||
}
|
||||
const hints = Array.isArray(dirHint) ? dirHint : (dirHint ? [dirHint] : []);
|
||||
@@ -585,17 +582,6 @@ export interface ExtractOpts {
|
||||
* before (single-'default'-source brains unaffected).
|
||||
*/
|
||||
sourceId?: string;
|
||||
/**
|
||||
* v0.42 — also extract frontmatter links on the incremental (slugs) path.
|
||||
* `extractForSlugs` extracts BODY links only by default; set this true to also
|
||||
* parse each changed page's frontmatter so `sources:`/`related:` edges stay fresh
|
||||
* when YAML is edited externally and synced in. Applied PER changed page, so the
|
||||
* incremental walk stays bounded (no switch to a full DB scan). Only honored on
|
||||
* the incremental path (`slugs` defined); the full-walk path already covers
|
||||
* frontmatter via its own dispatch. Gated upstream by the config key
|
||||
* `autopilot.incremental_extract_include_frontmatter` (default off).
|
||||
*/
|
||||
includeFrontmatter?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -634,7 +620,7 @@ export async function runExtractCore(engine: BrainEngine, opts: ExtractOpts): Pr
|
||||
// Nothing changed — skip entirely.
|
||||
return result;
|
||||
}
|
||||
const r = await extractForSlugs(engine, opts.dir, opts.slugs, opts.mode, dryRun, jsonMode, workers, opts.signal, opts.sourceId, opts.includeFrontmatter);
|
||||
const r = await extractForSlugs(engine, opts.dir, opts.slugs, opts.mode, dryRun, jsonMode, workers, opts.signal, opts.sourceId);
|
||||
result.links_created = r.links_created;
|
||||
result.timeline_entries_created = r.timeline_created;
|
||||
result.pages_processed = r.pages;
|
||||
@@ -1025,11 +1011,6 @@ async function extractForSlugs(
|
||||
signal?: AbortSignal,
|
||||
// #1747/#1503: stamp resolved brain source id on batch rows (see ExtractOpts.sourceId).
|
||||
sourceId?: string,
|
||||
// v0.42: when true, also extract frontmatter links per changed page so
|
||||
// externally-edited YAML (`sources:`/`related:`) stays fresh on the cycle.
|
||||
// Default false preserves the body-only incremental behavior. Gated upstream
|
||||
// by `autopilot.incremental_extract_include_frontmatter`.
|
||||
includeFrontmatter: boolean = false,
|
||||
): Promise<{ links_created: number; timeline_created: number; pages: number }> {
|
||||
// Build the full slug set for link resolution (fast: just readdir, no file reads)
|
||||
const allFiles = walkMarkdownFiles(brainDir);
|
||||
@@ -1104,7 +1085,7 @@ async function extractForSlugs(
|
||||
const content = readFileSync(fullPath, 'utf-8');
|
||||
|
||||
if (doLinks) {
|
||||
const links = await extractLinksFromFile(content, relPath, allSlugs, { globalBasename, includeFrontmatter });
|
||||
const links = await extractLinksFromFile(content, relPath, allSlugs, { globalBasename });
|
||||
for (const link of links) {
|
||||
if (dryRun) {
|
||||
if (!jsonMode) console.log(` ${link.from_slug} → ${link.to_slug} (${link.link_type})`);
|
||||
|
||||
@@ -121,18 +121,6 @@ export interface GBrainConfig {
|
||||
/** Daily spend cap (USD); bounds drains/day = floor(cap / ~$0.30). Default 2.0. */
|
||||
max_usd_per_day?: number;
|
||||
};
|
||||
/**
|
||||
* v0.42 — keep frontmatter links fresh on the incremental cycle. The cycle's
|
||||
* extract phase re-extracts only the slugs a sync changed, but `extractForSlugs`
|
||||
* extracts BODY links only — frontmatter (`sources:`/`related:` etc.) link edges
|
||||
* silently drift stale when a page's YAML is edited externally and synced in.
|
||||
* Set true to also extract frontmatter links per changed page each cycle, keeping
|
||||
* externally-edited YAML edges fresh without a full rescan. Default false
|
||||
* (preserves current behavior). Read via the file/env/DB plane in the cycle's
|
||||
* extract dispatch. Disable/enable with
|
||||
* `gbrain config set autopilot.incremental_extract_include_frontmatter <bool>`.
|
||||
*/
|
||||
incremental_extract_include_frontmatter?: boolean;
|
||||
};
|
||||
eval?: {
|
||||
/** false disables capture entirely. Defaults to true. */
|
||||
|
||||
@@ -996,21 +996,6 @@ async function runPhaseExtract(
|
||||
): Promise<PhaseResult> {
|
||||
try {
|
||||
const { runExtractCore } = await import('../commands/extract.ts');
|
||||
const { loadConfig } = await import('./config.ts');
|
||||
// Default off: the incremental cycle extracts body links only unless the
|
||||
// operator opts in to keeping externally-edited frontmatter links fresh too.
|
||||
// Both planes, file wins (env > file > DB precedence, per loadConfigWithEngine):
|
||||
// `gbrain config set autopilot.incremental_extract_include_frontmatter true`
|
||||
// writes the DB plane (engine.setConfig), so a file-plane-only read here
|
||||
// would make the documented enable command a silent no-op (#2120 class).
|
||||
const fileVal = loadConfig()?.autopilot?.incremental_extract_include_frontmatter;
|
||||
let includeFrontmatter = fileVal === true;
|
||||
if (fileVal === undefined) {
|
||||
try {
|
||||
includeFrontmatter =
|
||||
(await engine.getConfig('autopilot.incremental_extract_include_frontmatter')) === 'true';
|
||||
} catch { /* config table unreadable → default off */ }
|
||||
}
|
||||
// Extract is read-mostly against the filesystem + write to links table.
|
||||
// Honor dryRun by skipping with a 'skipped' entry: extract doesn't have
|
||||
// a clean dry-run mode today and runCycle should be honest about it.
|
||||
@@ -1031,7 +1016,6 @@ async function runPhaseExtract(
|
||||
slugs: changedSlugs, // undefined = full walk (first run / manual)
|
||||
signal,
|
||||
sourceId,
|
||||
includeFrontmatter, // honored on the incremental (slugs) path only
|
||||
});
|
||||
const linksCreated = result?.links_created ?? 0;
|
||||
const timelineCreated = result?.timeline_entries_created ?? 0;
|
||||
|
||||
@@ -942,17 +942,8 @@ export function makeResolver(
|
||||
|
||||
const hints = Array.isArray(dirHint) ? dirHint : (dirHint ? [dirHint] : []);
|
||||
|
||||
// Step 1: already a slug? Try an exact page lookup for any slug-shaped
|
||||
// value (contains '/', slug charset). Broadened beyond the original
|
||||
// single-segment lowercase-leading form (`^[a-z][a-z0-9-]*\/[a-z0-9]...`)
|
||||
// to also accept digit-leading folders (`90-people/nicolai`,
|
||||
// `01-trading/...`) and nested paths (`a/b/c`) — common in PARA-numbered
|
||||
// vaults. This is an EXACT getPage match only — no fuzzy — so it never
|
||||
// produces a false positive; a non-existent slug just falls through to
|
||||
// the steps below. Fixes frontmatter `related: [[dir/slug]]` values
|
||||
// (unwrapped by unwrapWikilink) that name a real page the strict regex
|
||||
// could not reach and whose full-path fuzzy score is below threshold.
|
||||
if (/\//.test(trimmed) && /^[a-z0-9][a-z0-9/_-]*$/.test(trimmed)) {
|
||||
// Step 1: already a slug? (dir/name shape, lowercase, hyphenated)
|
||||
if (/^[a-z][a-z0-9-]*\/[a-z0-9][a-z0-9-]*$/.test(trimmed)) {
|
||||
const page = await engine.getPage(trimmed);
|
||||
if (page) {
|
||||
cache.set(cacheKey, trimmed);
|
||||
@@ -1012,25 +1003,6 @@ export function makeResolver(
|
||||
|
||||
// ─── Frontmatter extractor ──────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Unwrap an Obsidian `[[wikilink]]` frontmatter value to its bare link
|
||||
* target so the resolver (which expects bare titles / dir slugs) can match
|
||||
* it. Mainstream Obsidian authors frontmatter links as `related: ["[[Page]]"]`;
|
||||
* without this, the resolver treats the brackets as part of the value and a
|
||||
* `[[90-people/nicolai]]` is normalized into `90peoplenicolai`, so it never
|
||||
* resolves. Strips a trailing `|alias`, `#heading`, or `^block` suffix — the
|
||||
* link target only. The regex is anchored to a wholly-wrapped value
|
||||
* (`^\s*\[\[…\]\]\s*$`), so bare titles and any value not fully wrapped pass
|
||||
* through unchanged and existing behavior is preserved exactly.
|
||||
*/
|
||||
export function unwrapWikilink(value: string): string {
|
||||
const match = /^\s*\[\[(.+?)\]\]\s*$/.exec(value);
|
||||
if (!match) return value;
|
||||
// Take the link target: drop |alias, then #heading / ^block suffixes.
|
||||
const target = match[1].split('|')[0].split('#')[0].split('^')[0];
|
||||
return target.trim();
|
||||
}
|
||||
|
||||
export interface UnresolvedFrontmatterRef {
|
||||
/** The frontmatter field name. */
|
||||
field: string;
|
||||
@@ -1088,12 +1060,7 @@ export async function extractFrontmatterLinks(
|
||||
}
|
||||
if (!name) continue; // skip numbers, nulls, malformed objects
|
||||
|
||||
// Accept Obsidian `[[wikilink]]` values in frontmatter link fields by
|
||||
// unwrapping to the bare target before resolution. Bare titles pass
|
||||
// through unchanged; the original `name` is preserved for the
|
||||
// unresolved report and edge context.
|
||||
const linkTarget = unwrapWikilink(name);
|
||||
const resolved = await resolver.resolve(linkTarget, mapping.dirHint);
|
||||
const resolved = await resolver.resolve(name, mapping.dirHint);
|
||||
if (!resolved) {
|
||||
unresolved.push({ field, name });
|
||||
continue;
|
||||
|
||||
+61
-2
@@ -424,6 +424,23 @@ export interface OperationContext {
|
||||
* satisfied even on single-source brains.
|
||||
*/
|
||||
sourceId: string;
|
||||
/**
|
||||
* #2561 — federated read scope for UNQUALIFIED local CLI reads.
|
||||
*
|
||||
* Set ONLY by the local CLI's context builder (src/cli.ts makeContext), and
|
||||
* only when the source resolved via a non-explicit tier (local_path /
|
||||
* brain_default / sole_non_default / seed_default — NOT --source, NOT
|
||||
* GBRAIN_SOURCE, NOT a .gbrain-source dotfile). Contains the resolved
|
||||
* source first, then every other `config.federated = true` source, so an
|
||||
* unqualified `gbrain search "X"` spans federated sources as
|
||||
* docs/guides/multi-source-brains.md promises.
|
||||
*
|
||||
* Consumed exclusively by `federatedSearchScope` and ONLY when
|
||||
* `ctx.remote === false` — a remote caller's scope stays governed by
|
||||
* `ctx.auth.allowedSources` / scalar `ctx.sourceId` (source-isolation
|
||||
* invariant, fail-closed).
|
||||
*/
|
||||
localFederatedSourceIds?: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -539,6 +556,45 @@ export function resolveRequestedScope(
|
||||
return sourceScopeOpts(ctx);
|
||||
}
|
||||
|
||||
/**
|
||||
* #2561 — source scope for the search-shaped read ops (`search`, `query`).
|
||||
*
|
||||
* Delegates to `resolveRequestedScope` (the single trust+grant resolver), then
|
||||
* widens an UNQUALIFIED trusted-local scalar scope to the CLI-computed
|
||||
* federated set (`ctx.localFederatedSourceIds`, resolved source first). This is
|
||||
* what makes `sources add --federated` mean something for local search: a
|
||||
* federated source participates in unqualified `gbrain search "X"` results.
|
||||
*
|
||||
* The expansion NEVER applies when:
|
||||
* - the caller is not strictly trusted-local (`ctx.remote !== false`) —
|
||||
* remote scope stays grant-governed (fail-closed source isolation);
|
||||
* - a per-call `source_id` was passed (explicit wins, including `__all__`);
|
||||
* - the resolver already produced a federated array (OAuth grant);
|
||||
* - the CLI resolved the source from an explicit signal (--source / env /
|
||||
* dotfile) — makeContext leaves `localFederatedSourceIds` unset then.
|
||||
*
|
||||
* Deliberately NOT inside `sourceScopeOpts`: code-intel ops collapse a
|
||||
* multi-element scope to an error (`resolveCodeIntelScope`), and non-search
|
||||
* reads (get_page, get_links, …) keep their long-standing scalar behavior.
|
||||
*/
|
||||
export function federatedSearchScope(
|
||||
ctx: OperationContext,
|
||||
sourceIdParam?: string,
|
||||
): { sourceId?: string; sourceIds?: string[] } {
|
||||
const scope = resolveRequestedScope(ctx, sourceIdParam);
|
||||
if (
|
||||
ctx.remote === false &&
|
||||
sourceIdParam === undefined &&
|
||||
scope.sourceId !== undefined &&
|
||||
scope.sourceIds === undefined &&
|
||||
ctx.localFederatedSourceIds !== undefined &&
|
||||
ctx.localFederatedSourceIds.length > 1
|
||||
) {
|
||||
return { sourceIds: ctx.localFederatedSourceIds };
|
||||
}
|
||||
return scope;
|
||||
}
|
||||
|
||||
/**
|
||||
* Code-intel adapter for `resolveRequestedScope`. Graph traversal
|
||||
* (code_callers/code_callees/code_blast/code_flow) is single-source by design —
|
||||
@@ -1448,7 +1504,8 @@ const search: Operation = {
|
||||
const queryText = p.query as string;
|
||||
const limit = (p.limit as number) || 20;
|
||||
const offset = (p.offset as number) || 0;
|
||||
const scope = sourceScopeOpts(ctx);
|
||||
// #2561: unqualified trusted-local search spans federated sources.
|
||||
const scope = federatedSearchScope(ctx);
|
||||
|
||||
// T4/D5 — per-call mode honored ONLY for trusted/local callers so a remote
|
||||
// OAuth client can't escalate to the costly tokenmax bundle. Local + unknown
|
||||
@@ -1610,7 +1667,9 @@ const query: Operation = {
|
||||
// is spread into BOTH the image-similarity searchVector path and the text
|
||||
// hybridSearch path below, so both honor the same grant.
|
||||
const sourceIdParam = typeof p.source_id === 'string' ? p.source_id : undefined;
|
||||
const querySourceScope = resolveRequestedScope(ctx, sourceIdParam);
|
||||
// #2561: unqualified trusted-local query spans federated sources (per-call
|
||||
// source_id / remote grants still resolve through resolveRequestedScope).
|
||||
const querySourceScope = federatedSearchScope(ctx, sourceIdParam);
|
||||
|
||||
// v0.27.1: image-similarity branch. Bypasses hybridSearch (which is
|
||||
// text-only); embeds the image via embedMultimodal and runs a direct
|
||||
|
||||
@@ -353,6 +353,45 @@ export async function resolveSourceWithTier(
|
||||
return { source_id: 'default', tier: 'seed_default' };
|
||||
}
|
||||
|
||||
/**
|
||||
* #2561 — compute the federated read scope for an UNQUALIFIED local CLI call.
|
||||
*
|
||||
* `sources add --federated` promises that a `config.federated = true` source
|
||||
* "participates in unqualified `gbrain search` results"
|
||||
* (docs/guides/multi-source-brains.md). This helper turns that promise into a
|
||||
* scope: given the resolved source and WHICH tier resolved it, return
|
||||
* `[resolvedSource, ...other federated source ids]` — or `undefined` when the
|
||||
* expansion must not apply:
|
||||
*
|
||||
* - explicit tiers (`flag` / `env` / `dotfile`): the user named a source;
|
||||
* scalar scope stands (that IS the qualified case);
|
||||
* - no other federated source exists: keep the scalar fast path unchanged.
|
||||
*
|
||||
* Archived sources are excluded (same rationale as pickSoleNonDefaultSource);
|
||||
* the archived column is v34+, so fall back to the un-archived query on older
|
||||
* brains. Callers put the result on `OperationContext.localFederatedSourceIds`
|
||||
* — consumed only by `federatedSearchScope` and only when `remote === false`.
|
||||
*/
|
||||
export async function localFederatedSourceIds(
|
||||
engine: BrainEngine,
|
||||
sourceId: string,
|
||||
tier: SourceTier,
|
||||
): Promise<string[] | undefined> {
|
||||
if (tier === 'flag' || tier === 'env' || tier === 'dotfile') return undefined;
|
||||
let rows: Array<{ id: string }>;
|
||||
try {
|
||||
rows = await engine.executeRaw<{ id: string }>(
|
||||
`SELECT id FROM sources WHERE config->>'federated' = 'true' AND archived = false ORDER BY id`,
|
||||
);
|
||||
} catch {
|
||||
rows = await engine.executeRaw<{ id: string }>(
|
||||
`SELECT id FROM sources WHERE config->>'federated' = 'true' ORDER BY id`,
|
||||
);
|
||||
}
|
||||
const ids = [sourceId, ...rows.map((r) => r.id).filter((id) => id !== sourceId)];
|
||||
return ids.length > 1 ? ids : undefined;
|
||||
}
|
||||
|
||||
/** Exposed for tests. */
|
||||
export const __testing = {
|
||||
readDotfileWalk,
|
||||
|
||||
@@ -191,37 +191,3 @@ describe('runExtractCore — incremental cycle path (#417)', () => {
|
||||
expect(result.links_created).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
describe('runExtractCore — incremental frontmatter gate (includeFrontmatter)', () => {
|
||||
// alice has a `source:` frontmatter edge but NO body links. The incremental
|
||||
// path extracts body links only by default, so the frontmatter edge is the
|
||||
// sole signal that distinguishes the gate off vs on.
|
||||
const aliceFm = '---\nsource: companies/acme-example\n---\n# alice';
|
||||
|
||||
test('9. default (flag omitted) does NOT extract frontmatter links on the incremental path', async () => {
|
||||
await seedPage('companies/acme-example', '# acme');
|
||||
await seedPage('people/alice-example', aliceFm);
|
||||
const result = await runExtractCore(engine as unknown as BrainEngine, {
|
||||
mode: 'all',
|
||||
dir: tempDir,
|
||||
slugs: ['people/alice-example'],
|
||||
});
|
||||
// alice's only potential edge is her frontmatter `source:`; with the gate off
|
||||
// it must not be extracted (preserves the body-only incremental behavior).
|
||||
expect(result.pages_processed).toBe(1);
|
||||
expect(result.links_created).toBe(0);
|
||||
});
|
||||
|
||||
test('10. includeFrontmatter: true extracts the frontmatter link on the incremental path', async () => {
|
||||
await seedPage('companies/acme-example', '# acme');
|
||||
await seedPage('people/alice-example', aliceFm);
|
||||
const result = await runExtractCore(engine as unknown as BrainEngine, {
|
||||
mode: 'all',
|
||||
dir: tempDir,
|
||||
slugs: ['people/alice-example'],
|
||||
includeFrontmatter: true,
|
||||
});
|
||||
// Same page, gate on → the `source:` frontmatter edge is now extracted.
|
||||
expect(result.pages_processed).toBe(1);
|
||||
expect(result.links_created).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -76,18 +76,6 @@ describe('extractLinksFromFile', () => {
|
||||
}
|
||||
});
|
||||
|
||||
it('resolves wrapped [[wikilink]] digit-leading slug-path in frontmatter (fs resolver, broadened step 1)', async () => {
|
||||
// Same bug class as makeResolver step 1 (#1983): the fs resolver's strict
|
||||
// `^[a-z]…` slug regex rejected digit-leading / nested paths, so a PARA-vault
|
||||
// `related: "[[90-people/nicolai]]"` never resolved even though the page exists.
|
||||
const content = '---\nrelated: "[[90-people/nicolai]]"\ntype: concept\n---\nContent.';
|
||||
const allSlugs = new Set(['wiki/note', '90-people/nicolai']);
|
||||
const links = await extractLinksFromFile(content, 'wiki/note.md', allSlugs, { includeFrontmatter: true });
|
||||
const related = links.filter(l => l.link_type === 'related_to');
|
||||
expect(related).toHaveLength(1);
|
||||
expect(related[0].to_slug).toBe('90-people/nicolai');
|
||||
});
|
||||
|
||||
it('frontmatter extraction is default OFF (back-compat)', async () => {
|
||||
// Without includeFrontmatter, fs-source no longer auto-extracts frontmatter.
|
||||
// Matches db-source behavior. User opts in with --include-frontmatter flag.
|
||||
|
||||
@@ -9,7 +9,6 @@ import {
|
||||
parseTimelineEntries,
|
||||
isAutoLinkEnabled,
|
||||
FRONTMATTER_LINK_MAP,
|
||||
unwrapWikilink,
|
||||
type SlugResolver,
|
||||
} from '../src/core/link-extraction.ts';
|
||||
import type { BrainEngine } from '../src/core/engine.ts';
|
||||
@@ -1292,175 +1291,3 @@ describe('parseTimelineEntries — Format 3: inline [Source: ..., YYYY-MM-DD] ci
|
||||
expect(parseTimelineEntries('[Source: import batch, 2025-07-01]')).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
// ─── Frontmatter [[wikilink]] + slug-path resolution ──────────────────────
|
||||
// Mainstream Obsidian authors frontmatter links as `related: ["[[Page]]"]`,
|
||||
// and PARA-numbered vaults use digit-leading / nested slug paths like
|
||||
// `[[90-people/nicolai]]`. Both were silently dropped: brackets were treated
|
||||
// as part of the value and the step-1 slug regex (`^[a-z]…`) rejected
|
||||
// digit-leading / nested paths, while full-path fuzzy scored below threshold.
|
||||
// Fix: unwrapWikilink() before resolution + an exact getPage() for any
|
||||
// slug-shaped value (exact-match only → no false positives).
|
||||
|
||||
describe('unwrapWikilink', () => {
|
||||
test('wrapped title → bare title', () => {
|
||||
expect(unwrapWikilink('[[Monday Range]]')).toBe('Monday Range');
|
||||
});
|
||||
test('wrapped slug-path (digit-leading folder) → bare slug', () => {
|
||||
expect(unwrapWikilink('[[90-people/nicolai]]')).toBe('90-people/nicolai');
|
||||
});
|
||||
test('wrapped nested slug-path → bare slug', () => {
|
||||
expect(unwrapWikilink('[[01-trading/wiki/strategies/opening-range-breakout]]'))
|
||||
.toBe('01-trading/wiki/strategies/opening-range-breakout');
|
||||
});
|
||||
test('strips |alias', () => {
|
||||
expect(unwrapWikilink('[[90-people/nicolai|Nicolai]]')).toBe('90-people/nicolai');
|
||||
});
|
||||
test('strips #heading', () => {
|
||||
expect(unwrapWikilink('[[Page#Section]]')).toBe('Page');
|
||||
});
|
||||
test('strips ^block', () => {
|
||||
expect(unwrapWikilink('[[Page^abc123]]')).toBe('Page');
|
||||
});
|
||||
test('surrounding whitespace tolerated', () => {
|
||||
expect(unwrapWikilink(' [[Page]] ')).toBe('Page');
|
||||
});
|
||||
test('bare title passes through unchanged', () => {
|
||||
expect(unwrapWikilink('Monday Range')).toBe('Monday Range');
|
||||
});
|
||||
test('bare slug passes through unchanged', () => {
|
||||
expect(unwrapWikilink('90-people/nicolai')).toBe('90-people/nicolai');
|
||||
});
|
||||
test('partially-wrapped value is NOT unwrapped (anchored)', () => {
|
||||
// Not a wholly-wrapped value → left intact so existing behavior is exact.
|
||||
expect(unwrapWikilink('see [[Page]] for detail')).toBe('see [[Page]] for detail');
|
||||
});
|
||||
});
|
||||
|
||||
describe('makeResolver — slug-path exact getPage (step 1 broadened)', () => {
|
||||
function fakeEngine(
|
||||
slugs: string[],
|
||||
fuzzyMap: Map<string, { slug: string; similarity: number }> = new Map(),
|
||||
): BrainEngine {
|
||||
const lookup = new Set(slugs);
|
||||
return {
|
||||
async getPage(slug: string) { return lookup.has(slug) ? { slug } as any : null; },
|
||||
async findByTitleFuzzy(name: string) { return fuzzyMap.get(name) ?? null; },
|
||||
async searchKeyword() { return []; },
|
||||
} as unknown as BrainEngine;
|
||||
}
|
||||
|
||||
test('digit-leading folder slug resolves via exact getPage', async () => {
|
||||
const r = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
expect(await r.resolve('90-people/nicolai')).toBe('90-people/nicolai');
|
||||
});
|
||||
|
||||
test('nested (>2 segment) slug resolves via exact getPage', async () => {
|
||||
const r = makeResolver(fakeEngine(['01-trading/wiki/strategies/opening-range-breakout']));
|
||||
expect(await r.resolve('01-trading/wiki/strategies/opening-range-breakout'))
|
||||
.toBe('01-trading/wiki/strategies/opening-range-breakout');
|
||||
});
|
||||
|
||||
test('regression: single-segment lowercase slug still resolves', async () => {
|
||||
const r = makeResolver(fakeEngine(['people/pedro']));
|
||||
expect(await r.resolve('people/pedro')).toBe('people/pedro');
|
||||
});
|
||||
|
||||
test('exact-only: slug-shaped value with no matching page falls through (no false positive)', async () => {
|
||||
// `90-people/ghost` is slug-shaped but absent → step-1 getPage misses,
|
||||
// no fuzzy hit → null. Never invents an edge.
|
||||
const r = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
expect(await r.resolve('90-people/ghost')).toBeNull();
|
||||
});
|
||||
|
||||
test('non-slug value still routes to fuzzy', async () => {
|
||||
const r = makeResolver(fakeEngine(
|
||||
['01-trading/monday-range'],
|
||||
new Map([['Monday Range', { slug: '01-trading/monday-range', similarity: 1 }]]),
|
||||
));
|
||||
expect(await r.resolve('Monday Range')).toBe('01-trading/monday-range');
|
||||
});
|
||||
});
|
||||
|
||||
describe('extractFrontmatterLinks — [[wikilink]] related: values (end-to-end)', () => {
|
||||
function fakeEngine(
|
||||
slugs: string[],
|
||||
fuzzyMap: Map<string, { slug: string; similarity: number }> = new Map(),
|
||||
): BrainEngine {
|
||||
const lookup = new Set(slugs);
|
||||
return {
|
||||
async getPage(slug: string) { return lookup.has(slug) ? { slug } as any : null; },
|
||||
async findByTitleFuzzy(name: string) { return fuzzyMap.get(name) ?? null; },
|
||||
async searchKeyword() { return []; },
|
||||
} as unknown as BrainEngine;
|
||||
}
|
||||
|
||||
test('wrapped slug-path related: resolves (the core win)', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
const { candidates, unresolved } = await extractFrontmatterLinks(
|
||||
'wiki/originals/ideas/note', 'note' as never,
|
||||
{ related: '[[90-people/nicolai]]' }, resolver,
|
||||
);
|
||||
expect(unresolved).toHaveLength(0);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0]).toMatchObject({
|
||||
fromSlug: 'wiki/originals/ideas/note',
|
||||
targetSlug: '90-people/nicolai',
|
||||
linkType: 'related_to',
|
||||
linkSource: 'frontmatter',
|
||||
});
|
||||
});
|
||||
|
||||
test('wrapped nested slug-path related: resolves', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['01-trading/wiki/strategies/opening-range-breakout']));
|
||||
const { candidates } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: ['[[01-trading/wiki/strategies/opening-range-breakout]]'] }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0].targetSlug).toBe('01-trading/wiki/strategies/opening-range-breakout');
|
||||
});
|
||||
|
||||
test('wrapped value with |alias resolves to the target', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
const { candidates } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: '[[90-people/nicolai|Nicolai]]' }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0].targetSlug).toBe('90-people/nicolai');
|
||||
});
|
||||
|
||||
test('regression: bare slug related: still resolves', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
const { candidates } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: '90-people/nicolai' }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0].targetSlug).toBe('90-people/nicolai');
|
||||
});
|
||||
|
||||
test('regression: wrapped title resolves via fuzzy (brackets harmless)', async () => {
|
||||
const resolver = makeResolver(fakeEngine(
|
||||
['01-trading/monday-range'],
|
||||
new Map([['Monday Range', { slug: '01-trading/monday-range', similarity: 1 }]]),
|
||||
));
|
||||
const { candidates } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: '[[Monday Range]]' }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0].targetSlug).toBe('01-trading/monday-range');
|
||||
});
|
||||
|
||||
test('unknown wrapped slug → unresolved (no crash), original value preserved', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
const { candidates, unresolved } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: '[[99-archive/does-not-exist]]' }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(0);
|
||||
expect(unresolved).toHaveLength(1);
|
||||
expect(unresolved[0]).toEqual({ field: 'related', name: '[[99-archive/does-not-exist]]' });
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
/**
|
||||
* #2561 — sources.config.federated participates in UNQUALIFIED local CLI
|
||||
* search/query.
|
||||
*
|
||||
* Pre-fix: the local CLI always emitted a scalar `{sourceId}` scope (required
|
||||
* field, auto-filled 'default'), so a source registered with
|
||||
* `gbrain sources add --federated` was invisible to an unqualified
|
||||
* `gbrain search "X"` — contradicting docs/guides/multi-source-brains.md
|
||||
* ("Source participates in unqualified `gbrain search` results").
|
||||
*
|
||||
* Fix: the CLI context builder computes `ctx.localFederatedSourceIds`
|
||||
* (resolved source + every other federated source) whenever the source
|
||||
* resolved via a NON-explicit tier; `federatedSearchScope` widens the scalar
|
||||
* scope to that set for the `search` / `query` ops — trusted-local only
|
||||
* (`ctx.remote === false`), never for remote callers, never when a per-call
|
||||
* `source_id` or an explicit --source/env/dotfile was given.
|
||||
*/
|
||||
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
||||
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
import { localFederatedSourceIds } from '../src/core/source-resolver.ts';
|
||||
import {
|
||||
federatedSearchScope,
|
||||
operations,
|
||||
type OperationContext,
|
||||
} from '../src/core/operations.ts';
|
||||
|
||||
let engine: PGLiteEngine;
|
||||
const search = operations.find((o) => o.name === 'search')!;
|
||||
|
||||
function ctxOf(overrides: Partial<OperationContext> = {}): OperationContext {
|
||||
return {
|
||||
engine: engine as any,
|
||||
config: {} as any,
|
||||
logger: console as any,
|
||||
dryRun: false,
|
||||
remote: false,
|
||||
sourceId: 'default',
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
beforeAll(async () => {
|
||||
engine = new PGLiteEngine();
|
||||
await engine.connect({});
|
||||
await engine.initSchema();
|
||||
// Seeded 'default' source is federated=true. Add:
|
||||
// wiki — federated (must join unqualified search)
|
||||
// private — NOT federated (must stay invisible unless explicitly named)
|
||||
// oldnews — federated but archived (must stay excluded)
|
||||
await engine.executeRaw(
|
||||
`INSERT INTO sources (id, name, local_path, config) VALUES ('wiki', 'wiki', '/tmp/wiki', '{"federated": true}'::jsonb)`,
|
||||
);
|
||||
await engine.executeRaw(
|
||||
`INSERT INTO sources (id, name, local_path, config) VALUES ('private', 'private', '/tmp/private', '{}'::jsonb)`,
|
||||
);
|
||||
await engine.executeRaw(
|
||||
`INSERT INTO sources (id, name, local_path, config, archived) VALUES ('oldnews', 'oldnews', '/tmp/oldnews', '{"federated": true}'::jsonb, true)`,
|
||||
);
|
||||
const pages: Array<[slug: string, sourceId: string, where: string]> = [
|
||||
['notes/home', 'default', 'default'],
|
||||
['wiki/topic', 'wiki', 'wiki'],
|
||||
['private/topic', 'private', 'private'],
|
||||
['old/topic', 'oldnews', 'oldnews'],
|
||||
];
|
||||
for (const [slug, sourceId, where] of pages) {
|
||||
await engine.putPage(slug, {
|
||||
type: 'note', title: `Topic in ${where}`, compiled_truth: `the zebra telescope in ${where}`, frontmatter: {},
|
||||
}, { sourceId });
|
||||
await engine.upsertChunks(slug, [
|
||||
{ chunk_index: 0, chunk_text: `the zebra telescope in ${where}`, chunk_source: 'compiled_truth' },
|
||||
], { sourceId });
|
||||
}
|
||||
// Keyword-only search path: no embedding provider needed in tests.
|
||||
await engine.setConfig('search.mcp_keyword_only', 'true');
|
||||
}, 60_000);
|
||||
|
||||
afterAll(async () => {
|
||||
if (engine) await engine.disconnect();
|
||||
}, 60_000);
|
||||
|
||||
describe('localFederatedSourceIds — CLI-side scope computation', () => {
|
||||
test('non-explicit tier: resolved source first, then other federated, archived excluded', async () => {
|
||||
expect(await localFederatedSourceIds(engine, 'default', 'seed_default')).toEqual(['default', 'wiki']);
|
||||
});
|
||||
|
||||
test('non-federated resolved source still joins its own scope', async () => {
|
||||
expect(await localFederatedSourceIds(engine, 'private', 'brain_default')).toEqual(['private', 'default', 'wiki']);
|
||||
});
|
||||
|
||||
test('explicit tiers (--source / env / dotfile) never expand', async () => {
|
||||
expect(await localFederatedSourceIds(engine, 'default', 'flag')).toBeUndefined();
|
||||
expect(await localFederatedSourceIds(engine, 'default', 'env')).toBeUndefined();
|
||||
expect(await localFederatedSourceIds(engine, 'default', 'dotfile')).toBeUndefined();
|
||||
});
|
||||
|
||||
test('single federated source (the resolved one) keeps the scalar fast path', async () => {
|
||||
const solo = { executeRaw: async () => [{ id: 'default' }] } as any;
|
||||
expect(await localFederatedSourceIds(solo, 'default', 'seed_default')).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('federatedSearchScope — trust + explicitness matrix', () => {
|
||||
test('trusted local + unqualified widens to the federated set', () => {
|
||||
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
|
||||
expect(federatedSearchScope(ctx)).toEqual({ sourceIds: ['default', 'wiki'] });
|
||||
});
|
||||
|
||||
test('remote caller NEVER widens (fail-closed), even if the field is set', () => {
|
||||
const ctx = ctxOf({ remote: true, localFederatedSourceIds: ['default', 'wiki'] });
|
||||
expect(federatedSearchScope(ctx)).toEqual({ sourceId: 'default' });
|
||||
});
|
||||
|
||||
test('per-call source_id wins over the federated set', () => {
|
||||
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
|
||||
expect(federatedSearchScope(ctx, 'wiki')).toEqual({ sourceId: 'wiki' });
|
||||
});
|
||||
|
||||
test('per-call __all__ keeps the whole-brain semantics for trusted local', () => {
|
||||
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
|
||||
expect(federatedSearchScope(ctx, '__all__')).toEqual({});
|
||||
});
|
||||
|
||||
test('a federated OAuth grant wins over the local set', () => {
|
||||
const ctx = ctxOf({
|
||||
localFederatedSourceIds: ['default', 'wiki'],
|
||||
auth: { allowedSources: ['a', 'b'] } as OperationContext['auth'],
|
||||
});
|
||||
expect(federatedSearchScope(ctx)).toEqual({ sourceIds: ['a', 'b'] });
|
||||
});
|
||||
|
||||
test('no local federated set → unchanged scalar scope', () => {
|
||||
expect(federatedSearchScope(ctxOf())).toEqual({ sourceId: 'default' });
|
||||
});
|
||||
});
|
||||
|
||||
describe('search op — unqualified local search spans federated sources', () => {
|
||||
test('federated source results appear; non-federated + archived stay invisible', async () => {
|
||||
const ctx = ctxOf({
|
||||
localFederatedSourceIds: await localFederatedSourceIds(engine, 'default', 'seed_default'),
|
||||
});
|
||||
const results = (await search.handler(ctx, { query: 'zebra telescope' })) as Array<{ slug: string }>;
|
||||
const slugs = results.map((r) => r.slug);
|
||||
expect(slugs).toContain('notes/home');
|
||||
expect(slugs).toContain('wiki/topic'); // pre-#2561 this was missing
|
||||
expect(slugs).not.toContain('private/topic');
|
||||
expect(slugs).not.toContain('old/topic');
|
||||
});
|
||||
|
||||
test('explicit source resolution (no federated set on ctx) stays single-source', async () => {
|
||||
const results = (await search.handler(ctxOf(), { query: 'zebra telescope' })) as Array<{ slug: string }>;
|
||||
const slugs = results.map((r) => r.slug);
|
||||
expect(slugs).toEqual(['notes/home']);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user