Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 5d164956a1 fix(sources): federated-source pages visible to get_page/list_pages/resolve_slugs and no-grant MCP callers (#3242)
Pages ingested into a config.federated=true source were invisible to
normal reads: get_page/list_pages scoped to the scalar resolved source
('default'), while the fully UNSCOPED resolve_slugs leaked every
source's slugs — the reporter's exact observation matrix.

- federatedSearchScope now backs get_page, list_pages and resolve_slugs
  (not just search/query), so the unqualified read surface shares one
  visibility set: grant > federated set > scalar source. resolve_slugs
  gains the missing sourceScopeOpts-family scoping (leak sealed).
- The widening gate is now field-presence instead of ctx.remote:
  localFederatedSourceIds is populated only by server-side transports
  (never from caller params), so trust stays fail-closed while the
  stdio MCP transport (no GBRAIN_SOURCE) and the legacy HTTP token
  path (no operator-set permissions.source_id grant) can opt their
  unqualified callers into the operator-configured federated set.
  Tokens WITH a grant, per-call source_id, and OAuth allowedSources
  all still win and never widen.
- gbrain sync now attributes its ingest-log row to the synced source
  instead of the shared 'default' bucket (attribution sub-bug).

No engine SQL changes: getPage/listPages/resolveSlugs already accept
sourceIds[] in both engines (#1393/#876).

Fixes #3242

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-23 15:18:09 -07:00
17 changed files with 309 additions and 316 deletions
+20 -3
View File
@@ -249,6 +249,13 @@ async function resolveAIOptions(opts: ResolveAIOptionsArgs): Promise<ResolvedAIO
// --- Tier 1+2: explicit flags ---------------------------------------------
// #2301: an explicit embedding flag on THIS invocation overrides the
// persisted deferred-setup sentinel above. Without this, a stale
// `embedding_disabled: true` in config.json made every re-init defer
// embedding — including `gbrain init --embedding-model ...`, the exact
// recovery path the deferred-setup message tells users to take.
if (verbose || shorthand) delete out.noEmbedding;
if (verbose) {
out.embedding_model = verbose;
} else if (shorthand) {
@@ -435,7 +442,7 @@ function printNoEmbeddingProviderHint(typos: Array<{ userSet: string; suggested:
console.error(' gbrain init --pglite --embedding-model openai:text-embedding-3-large');
console.error('');
console.error('Or defer setup: gbrain init --pglite --no-embedding');
console.error(' (you can configure later with `gbrain config set embedding_model <id>`)');
console.error(' (you can configure later with `gbrain init --force --embedding-model <provider>:<model>`)');
// D13: surface near-miss env vars (e.g. OPENAPI_API_KEY → OPENAI_API_KEY).
if (typos.length > 0) {
console.error('');
@@ -833,7 +840,7 @@ async function initPGLite(opts: {
let resolvedModel: string | undefined;
if (opts.aiOpts?.noEmbedding) {
// D9 deferred-setup mode: skip preflight, no model/dim resolved.
console.log(` --no-embedding: deferred setup — configure with \`gbrain config set embedding_model <id>\` before import`);
console.log(` --no-embedding: deferred setup — run \`gbrain init --force --embedding-model <provider>:<model>\` before import`);
} else if (opts.aiOpts?.embedding_model) {
const { resolveSchemaEmbeddingDim } = await import('../core/embedding-dim-check.ts');
const pre = resolveSchemaEmbeddingDim({
@@ -972,6 +979,12 @@ async function initPGLite(opts: {
// unless explicitly overridden by --schema-pack on re-init.
...(opts.schemaPack ? { schema_pack: opts.schemaPack } : {}),
};
// #2301: a resolved embedding model supersedes any stale deferred-setup
// sentinel carried over via ...existingFile — otherwise the sentinel
// re-defers embedding on every future init/embed forever.
if (!opts.aiOpts?.noEmbedding && resolvedModel && resolvedDim) {
delete config.embedding_disabled;
}
// PR1: new installs publish their skill catalog over MCP by default
// (existing config wins on re-init, so a prior opt-out is preserved).
config.mcp = { publish_skills: true, ...(config.mcp ?? {}) };
@@ -1056,7 +1069,7 @@ async function initPostgres(opts: {
let resolvedDim: number | undefined;
let resolvedModel: string | undefined;
if (opts.aiOpts?.noEmbedding) {
console.log(` --no-embedding: deferred setup — configure with \`gbrain config set embedding_model <id>\` before import`);
console.log(` --no-embedding: deferred setup — run \`gbrain init --force --embedding-model <provider>:<model>\` before import`);
} else if (opts.aiOpts?.embedding_model) {
const { resolveSchemaEmbeddingDim } = await import('../core/embedding-dim-check.ts');
const pre = resolveSchemaEmbeddingDim({
@@ -1220,6 +1233,10 @@ async function initPostgres(opts: {
// v0.42 (T17): same schema_pack default as PGLite path.
...(opts.schemaPack ? { schema_pack: opts.schemaPack } : {}),
};
// #2301: same stale-sentinel drop as the PGLite path above.
if (!opts.aiOpts?.noEmbedding && resolvedModel && resolvedDim) {
delete config.embedding_disabled;
}
// PR1: new installs publish their skill catalog over MCP by default
// (existing config wins on re-init, so a prior opt-out is preserved).
config.mcp = { publish_skills: true, ...(config.mcp ?? {}) };
+3
View File
@@ -3384,6 +3384,9 @@ async function performSyncInner(engine: BrainEngine, opts: SyncOpts): Promise<Sy
// Log ingest
await engine.logIngest({
// #3242 (attribution sub-bug): credit the sync to the source it wrote
// to, not the shared 'default' bucket.
...(opts.sourceId ? { source_id: opts.sourceId } : {}),
source_type: 'git_sync',
source_ref: `${repoPath} @ ${headCommit.slice(0, 8)}`,
pages_updated: pagesAffected,
+5 -40
View File
@@ -486,44 +486,9 @@ const DEFAULT_PARALLELISM = 4;
* src/core/errors.ts (the v0.19.0 envelope every new agent-facing
* surface uses) rather than introducing a new BrainstormError class.
*/
/** File-config slice the orchestrator reads (see loadConfig in core/config.ts). */
export interface BrainstormRunConfig {
embedding_model?: string;
chat_model?: string;
emotional_weight?: { user_holder?: string };
}
/**
* Model used for the cost preview + hard cost ceiling. Mirrors what the
* gateway will actually run: explicit --model override, else the configured
* chat_model (gateway default), else the hardcoded gateway fallback. Before
* this resolved through config, a non-Sonnet chat_model got its preview
* priced against the wrong model. (Takeover of PR #1855 by @starm2010.)
*/
export function resolveBrainstormChatModel(
config: { chat_model?: string },
modelOverride?: string,
): string {
return modelOverride ?? config.chat_model ?? 'anthropic:claude-sonnet-4-6';
}
/**
* Judge-phase model precedence: --judge-model flag, else the
* `models.brainstorm.judge` config key, else undefined (falls back to
* `modelOverride` then the gateway default at the runJudge callsite).
*/
export async function resolveBrainstormJudgeModel(
engine: BrainEngine,
judgeModelFlag?: string,
): Promise<string | undefined> {
if (judgeModelFlag) return judgeModelFlag;
const configured = await engine.getConfig('models.brainstorm.judge');
return configured ?? undefined;
}
export async function runBrainstorm(
engine: BrainEngine,
config: BrainstormRunConfig,
config: { embedding_model?: string; emotional_weight?: { user_holder?: string } },
opts: BrainstormOptions
): Promise<BrainstormResult> {
// v0.39.3.0 (Phase 5, CV11+T4): outer try/catch around the orchestrator
@@ -545,7 +510,7 @@ export async function runBrainstorm(
async function runBrainstormImpl(
engine: BrainEngine,
config: BrainstormRunConfig,
config: { embedding_model?: string; emotional_weight?: { user_holder?: string } },
opts: BrainstormOptions,
): Promise<BrainstormResult> {
// v0.39.0.0 T10: install a gateway-layer BudgetTracker scope around the
@@ -565,7 +530,7 @@ async function runBrainstormImpl(
async function _runBrainstormInner(
engine: BrainEngine,
config: BrainstormRunConfig,
config: { embedding_model?: string; emotional_weight?: { user_holder?: string } },
opts: BrainstormOptions,
): Promise<BrainstormResult> {
const profile = opts.profile ?? BRAINSTORM_PROFILE;
@@ -574,7 +539,7 @@ async function _runBrainstormInner(
const embedFn = opts.embedQueryFn ?? embedQuery;
// ---- Phase 0: cost preview + TTY grace ----
const modelStr = resolveBrainstormChatModel(config, opts.modelOverride);
const modelStr = opts.modelOverride ?? 'anthropic:claude-sonnet-4-6';
const { aborted, estimate } = await previewCostAndWait({
profile,
model: modelStr,
@@ -883,7 +848,7 @@ async function _runBrainstormInner(
far_slug: i.far_slug,
}));
const judgeResult = await runJudge(profile.judge_config, judgeInput, {
modelOverride: (await resolveBrainstormJudgeModel(engine, opts.judgeModel)) ?? opts.modelOverride,
modelOverride: opts.judgeModel ?? opts.modelOverride,
chatFn: opts.chatFn,
activeBiasTags: activeBiasTags ?? undefined,
abortSignal: opts.abortSignal,
-1
View File
@@ -962,7 +962,6 @@ export const KNOWN_CONFIG_KEYS: readonly string[] = [
'models.subagent',
'models.expansion',
'models.chat',
'models.brainstorm.judge',
'models.eval.longmemeval',
'facts.extraction_model',
// #2113: output-token cap for the per-turn facts extractor (default 4000).
+10 -74
View File
@@ -39,11 +39,11 @@
import { randomUUID, createHash } from 'node:crypto';
import { BaseCyclePhase, type ScopedReadOpts, type BasePhaseOpts } from './base-phase.ts';
import { chat as gatewayChat, getChatModel, probeChatModel } from '../ai/gateway.ts';
import { normalizeModelId } from '../model-id.ts';
import { chat as gatewayChat, getChatModel } from '../ai/gateway.ts';
import { writeReceipt } from '../extract/receipt-writer.ts';
import { upsertExtractRollup } from '../extract/rollup-writer.ts';
import { GBrainError } from '../types.ts';
import type { Page, PageFilters } from '../types.ts';
import type { OperationContext } from '../operations.ts';
import type { BrainEngine } from '../engine.ts';
import type { PhaseStatus, CyclePhase } from '../cycle.ts';
@@ -160,48 +160,6 @@ export interface ProposeTakesResult {
warnings: string[];
}
/** Narrow projection of `pages` — the only columns this phase reads. */
interface ProposeTakesPageRow {
slug: string;
source_id: string;
compiled_truth: string | null;
}
/**
* Load proposal candidates with a narrow projection instead of
* `engine.listPages` (`SELECT p.*`). The phase only reads slug, source_id
* and compiled_truth — skipping timeline/frontmatter/title keeps large
* toasted columns out of the hot path. Scope precedence mirrors
* `sourceScopeOpts`: federated array (`sourceIds`) beats scalar
* (`sourceId`); ordering matches `PAGE_SORT_SQL.updated_desc` with an id
* tiebreak for determinism. (Takeover of PR #1979's projection by
* @shawnduggan.)
*/
async function listCandidatePages(
engine: BrainEngine,
scope: ScopedReadOpts,
limit: number,
): Promise<ProposeTakesPageRow[]> {
const where = ['deleted_at IS NULL'];
const params: unknown[] = [];
if (scope.sourceIds && scope.sourceIds.length > 0) {
params.push(scope.sourceIds);
where.push(`source_id = ANY($${params.length}::text[])`);
} else if (scope.sourceId) {
params.push(scope.sourceId);
where.push(`source_id = $${params.length}`);
}
params.push(limit);
return engine.executeRaw<ProposeTakesPageRow>(
`SELECT slug, source_id, compiled_truth
FROM pages
WHERE ${where.join(' AND ')}
ORDER BY updated_at DESC, id DESC
LIMIT $${params.length}`,
params,
);
}
/**
* Compute the content_hash key for the idempotency cache. SHA-256 of the
* page body suffices — page slug + prompt_version are separate columns in
@@ -372,34 +330,6 @@ class ProposeTakesPhase extends BaseCyclePhase {
const phaseStartMs = Date.now();
const proposalRunId = `propose-${new Date().toISOString().slice(0, 19).replace(/[-:T]/g, '')}-${randomUUID().slice(0, 8)}`;
const modelId = opts.model ?? getChatModel();
// With the default (gateway) extractor, skip cheaply when the resolved
// model's provider can't run — same probe semantics as patterns.ts /
// think/index.ts: unknown provider/model or Anthropic-without-key skips;
// other providers' auth surfaces lazily at chat() time. An injected
// extractor bypasses the gateway, so it is never gated. (Takeover of
// PR #1979's intent by @shawnduggan.)
if (!opts.extractor) {
const probe = probeChatModel(normalizeModelId(modelId));
if (!probe.ok) {
return {
summary: `propose_takes skipped: ${probe.detail}`,
details: {
reason: 'no_provider',
model: modelId,
pages_scanned: 0,
cache_hits: 0,
cache_misses: 0,
proposals_inserted: 0,
budget_exhausted: false,
warnings: [],
},
status: 'skipped',
};
}
}
const result: ProposeTakesResult = {
pages_scanned: 0,
cache_hits: 0,
@@ -410,12 +340,19 @@ class ProposeTakesPhase extends BaseCyclePhase {
};
// Load pages eligible for proposal. Source-scoped per BaseCyclePhase.
const pages = await listCandidatePages(engine, scope, pageLimit);
const pageFilters: PageFilters = {
...scope,
limit: pageLimit,
sort: 'updated_desc',
};
const pages: Page[] = await engine.listPages(pageFilters);
if (opts.reporter) {
opts.reporter.start('propose_takes.pages' as never, pages.length);
}
const modelId = opts.model ?? getChatModel();
for (const page of pages) {
// Phase deadline check. Break (not throw) so the phase returns a
// partial result with deadline_hit:true; work already banked stays.
@@ -572,5 +509,4 @@ export const __testing = {
contentHash,
hasCompleteFence,
extractExistingTakesForDedup,
listCandidatePages,
};
+2 -3
View File
@@ -71,9 +71,8 @@ export function assertEmbeddingEnabled(cfg: { embedding_disabled?: boolean } | n
throw new EmbeddingDisabledError(
'This brain was initialized with `--no-embedding` (deferred setup).\n' +
'Configure an embedding provider before running embed / import:\n' +
' gbrain config set embedding_model <provider>:<model>\n' +
' gbrain config set embedding_dimensions <N>\n' +
' gbrain init --force --embedding-model <provider>:<model> # re-init to size schema\n',
' gbrain init --force --embedding-model <provider>:<model> # re-init to size schema\n' +
'(`gbrain config set embedding_model` is refused — schema-sizing fields are set at init.)\n',
);
}
}
+41 -28
View File
@@ -433,20 +433,25 @@ export interface OperationContext {
*/
sourceId: string;
/**
* #2561 — federated read scope for UNQUALIFIED local CLI reads.
* #2561 / #3242 — federated read scope for UNQUALIFIED reads.
*
* Set ONLY by the local CLI's context builder (src/cli.ts makeContext), and
* only when the source resolved via a non-explicit tier (local_path /
* brain_default / sole_non_default / seed_default — NOT --source, NOT
* GBRAIN_SOURCE, NOT a .gbrain-source dotfile). Contains the resolved
* source first, then every other `config.federated = true` source, so an
* unqualified `gbrain search "X"` spans federated sources as
* docs/guides/multi-source-brains.md promises.
* Set ONLY by trusted server-side context builders — never from caller
* params — and only when the caller carries no explicit source scope:
* - local CLI (src/cli.ts makeContext) when the source resolved via a
* non-explicit tier (local_path / brain_default / sole_non_default /
* seed_default — NOT --source, NOT GBRAIN_SOURCE, NOT a dotfile);
* - stdio MCP (src/mcp/server.ts) when GBRAIN_SOURCE is unset;
* - HTTP MCP (src/mcp/http-transport.ts) for legacy bearer tokens with
* NO operator-set `permissions.source_id` grant (the historical
* 'default' floor). Tokens WITH an explicit grant never widen.
*
* Consumed exclusively by `federatedSearchScope` and ONLY when
* `ctx.remote === false` — a remote caller's scope stays governed by
* `ctx.auth.allowedSources` / scalar `ctx.sourceId` (source-isolation
* invariant, fail-closed).
* Contains the resolved source first, then every other
* `config.federated = true` source, so an unqualified read/search spans
* federated sources as docs/guides/multi-source-brains.md promises.
*
* Consumed exclusively by `federatedSearchScope`. Fail-closed remains:
* a grant (`ctx.auth.allowedSources`) or a per-call `source_id` always
* wins, and a context without this field never widens.
*/
localFederatedSourceIds?: string[];
}
@@ -565,25 +570,26 @@ export function resolveRequestedScope(
}
/**
* #2561 — source scope for the search-shaped read ops (`search`, `query`).
* #2561 / #3242 — source scope for the page-visibility read ops (`search`,
* `query`, `get_page`, `list_pages`, `resolve_slugs`).
*
* Delegates to `resolveRequestedScope` (the single trust+grant resolver), then
* widens an UNQUALIFIED trusted-local scalar scope to the CLI-computed
* federated set (`ctx.localFederatedSourceIds`, resolved source first). This is
* what makes `sources add --federated` mean something for local search: a
* federated source participates in unqualified `gbrain search "X"` results.
* widens an UNQUALIFIED scalar scope to the transport-computed federated set
* (`ctx.localFederatedSourceIds`, resolved source first). This is what makes
* `sources add --federated` mean something: a federated source participates in
* unqualified reads (#3242 — pages ingested into a `federated: true` source
* were invisible to get_page/search/list_pages while resolve_slugs leaked them).
*
* The expansion NEVER applies when:
* - the caller is not strictly trusted-local (`ctx.remote !== false`) —
* remote scope stays grant-governed (fail-closed source isolation);
* - a per-call `source_id` was passed (explicit wins, including `__all__`);
* - the resolver already produced a federated array (OAuth grant);
* - the CLI resolved the source from an explicit signal (--source / env /
* dotfile) — makeContext leaves `localFederatedSourceIds` unset then.
* - the resolver already produced a federated array (OAuth grant governs);
* - the transport didn't populate `localFederatedSourceIds` (see that
* field's doc: it is only set for callers with NO explicit source scope,
* and never from caller-controlled params — so trust stays fail-closed).
*
* Deliberately NOT inside `sourceScopeOpts`: code-intel ops collapse a
* multi-element scope to an error (`resolveCodeIntelScope`), and non-search
* reads (get_page, get_links, …) keep their long-standing scalar behavior.
* multi-element scope to an error (`resolveCodeIntelScope`), and the remaining
* scalar reads (get_links, get_chunks, …) keep their long-standing behavior.
*/
export function federatedSearchScope(
ctx: OperationContext,
@@ -591,7 +597,6 @@ export function federatedSearchScope(
): { sourceId?: string; sourceIds?: string[] } {
const scope = resolveRequestedScope(ctx, sourceIdParam);
if (
ctx.remote === false &&
sourceIdParam === undefined &&
scope.sourceId !== undefined &&
scope.sourceIds === undefined &&
@@ -748,7 +753,9 @@ const get_page: Operation = {
// with a federated `allowedSources` grant (and no single ctx.sourceId) got
// an UNSCOPED exact lookup — a cross-source read of any page by slug. getPage
// now honors sourceIds[] (both engines), so the same scope closes both paths.
const sourceOpts = sourceScopeOpts(ctx);
// #3242: federatedSearchScope (not bare sourceScopeOpts) so an unqualified
// read sees pages in `federated: true` sources, matching search/query.
const sourceOpts = federatedSearchScope(ctx);
const fuzzyScope = sourceOpts;
let page = await ctx.engine.getPage(slug, { includeDeleted, ...sourceOpts });
@@ -1503,7 +1510,9 @@ const list_pages: Operation = {
// enumerate src-B pages. Pre-fix, ctx.sourceId / ctx.auth?.allowedSources
// were ignored at this op handler and the engine returned every source's
// pages indiscriminately.
const scope = sourceScopeOpts(ctx);
// #3242: federatedSearchScope so unqualified listing spans federated
// sources (same visibility set as search / get_page). Grants still win.
const scope = federatedSearchScope(ctx);
const pages = await ctx.engine.listPages({
type: p.type as any,
tag: p.tag as string,
@@ -2737,7 +2746,11 @@ const resolve_slugs: Operation = {
partial: { type: 'string', required: true },
},
handler: async (ctx, p) => {
return ctx.engine.resolveSlugs(p.partial as string);
// #3242: was fully UNSCOPED — the one read that leaked every source's
// slugs to any caller (the reporter's "resolve_slugs sees them but
// get_page doesn't" matrix). Route through the same visibility set as
// get_page/search: grant > federated set > scalar source.
return ctx.engine.resolveSlugs(p.partial as string, federatedSearchScope(ctx));
},
scope: 'read',
};
+2 -2
View File
@@ -4835,11 +4835,11 @@ export class PGLiteEngine implements BrainEngine {
const { rows } = await this.db.query(
`SELECT t.id AS take_id, t.page_id, p.slug AS page_slug, t.row_num,
t.claim, t.kind, t.holder, t.weight,
word_similarity($1, t.claim)::real AS score
similarity(t.claim, $1)::real AS score
FROM takes t
JOIN pages p ON p.id = t.page_id
WHERE t.active
AND $1 <% t.claim
AND t.claim % $1
AND ($2::text[] IS NULL OR t.holder = ANY($2::text[]))
AND ($4::text[] IS NULL OR p.source_id = ANY($4::text[]))
AND ($5::text IS NULL OR p.source_id = $5::text)
+2 -2
View File
@@ -4962,11 +4962,11 @@ export class PostgresEngine implements BrainEngine {
const rows = await sql`
SELECT t.id AS take_id, t.page_id, p.slug AS page_slug, t.row_num,
t.claim, t.kind, t.holder, t.weight,
word_similarity(${query}, t.claim)::real AS score
similarity(t.claim, ${query})::real AS score
FROM takes t
JOIN pages p ON p.id = t.page_id
WHERE t.active
AND ${query} <% t.claim
AND t.claim % ${query}
AND (
${opts.takesHoldersAllowList ?? null}::text[] IS NULL
OR t.holder = ANY(${opts.takesHoldersAllowList ?? null}::text[])
+8
View File
@@ -54,6 +54,13 @@ export interface DispatchOpts {
* resolves it from the per-token allow-list (eE3).
*/
sourceId?: string;
/**
* #3242: federated read set for callers with NO explicit source scope
* (stdio without GBRAIN_SOURCE; legacy HTTP tokens without an operator-set
* `permissions.source_id` grant). Transport-computed, never derived from
* caller params. See OperationContext.localFederatedSourceIds.
*/
localFederatedSourceIds?: string[];
/**
* v0.31 (eD3): hook called by the dispatcher AFTER op.handler succeeds
* to compute `_meta.brain_hot_memory` for the response. Wrapped in its
@@ -216,6 +223,7 @@ export function buildOperationContext(
// CLI / HTTP / stdio transports SHOULD pass an explicit sourceId via opts;
// this fallback covers code paths that historically passed undefined.
sourceId: opts.sourceId ?? 'default',
...(opts.localFederatedSourceIds ? { localFederatedSourceIds: opts.localFederatedSourceIds } : {}),
auth: opts.auth,
};
}
+22
View File
@@ -84,6 +84,14 @@ interface AuthResult {
* Bounded to the stored grant never widened to "all".
*/
auth?: AuthInfo;
/**
* #3242: true when the token row carries an operator-set
* `permissions.source_id` (string OR array even a malformed one, which
* fails closed to 'default' without widening). false = the historical
* no-grant 'default' floor; ONLY that case gets the federated read set
* (config.federated sources) threaded as localFederatedSourceIds.
*/
hasSourceGrant?: boolean;
}
/* Legacy token source-scope parsing lives in core/legacy-token-scope.ts and is
@@ -229,6 +237,9 @@ export async function startHttpTransport(opts: HttpTransportOptions) {
// source unless the token carries an explicit grant (#1336 above).
sourceId,
auth,
// #3242: distinguish "operator granted a scope" from "historical
// no-grant floor" — only the latter widens to federated sources.
hasSourceGrant: perms?.source_id != null,
};
} catch {
return { ok: false };
@@ -379,10 +390,21 @@ export async function startHttpTransport(opts: HttpTransportOptions) {
// takes_search / query (when it returns takes) can server-side filter.
// v0.34.1 (#861): thread source-isolation scope. Legacy access_tokens
// path defaults to 'default' per AuthResult.sourceId above.
// #3242: a token with NO operator-set source grant reads across the
// federated set (config.federated sources), not just the scalar
// 'default' floor. Granted tokens (hasSourceGrant) never widen.
let localFederated: string[] | undefined;
if (auth.hasSourceGrant === false && auth.sourceId) {
try {
const { localFederatedSourceIds } = await import('../core/source-resolver.ts');
localFederated = await localFederatedSourceIds(engine, auth.sourceId, 'seed_default');
} catch { /* scalar scope stands */ }
}
const result = await dispatchToolCall(engine, toolName, args, {
remote: true,
takesHoldersAllowList: auth.takesHoldersAllowList,
sourceId: auth.sourceId,
...(localFederated ? { localFederatedSourceIds: localFederated } : {}),
// #1336: thread the token's federated_read grant so read ops scope
// to the operator-granted sources via sourceScopeOpts.
auth: auth.auth,
+15
View File
@@ -35,6 +35,20 @@ export async function startMcpServer(engine: BrainEngine) {
// shape and cast through `any` (the SDK accepts it via the ServerResult union).
server.setRequestHandler(CallToolRequestSchema, async (request: any): Promise<any> => {
const { name, arguments: params } = request.params;
// #3242: when the operator didn't pin a source via GBRAIN_SOURCE, stdio
// reads span every `config.federated = true` source (same visibility set
// as unqualified local CLI reads). GBRAIN_SOURCE set = explicit scope,
// no widening. Best-effort: a resolver failure keeps the scalar scope.
// ponytail: one tiny SELECT per tool call; cache it if it ever shows up.
let localFederated: string[] | undefined;
try {
const { localFederatedSourceIds } = await import('../core/source-resolver.ts');
localFederated = await localFederatedSourceIds(
engine,
process.env.GBRAIN_SOURCE || 'default',
process.env.GBRAIN_SOURCE ? 'env' : 'seed_default',
);
} catch { /* scalar scope stands */ }
// v0.28: stdio MCP has no per-token auth (local pipe). Default the
// takes-holder allow-list to ['world'] so agent-facing callers don't
// see private hunches via takes_list / takes_search / query. Operators
@@ -51,6 +65,7 @@ export async function startMcpServer(engine: BrainEngine) {
// Operators who want a different source on stdio MCP should set
// GBRAIN_SOURCE in the env or use --source via `gbrain call`.
sourceId: process.env.GBRAIN_SOURCE || 'default',
...(localFederated ? { localFederatedSourceIds: localFederated } : {}),
// v0.31 (eD3): _meta.brain_hot_memory injection so Claude Desktop /
// Code see the brain's relevant hot memory automatically alongside
// every tool-call response. Best-effort; absorbs errors.
-65
View File
@@ -1,65 +0,0 @@
/**
* Brainstorm model configurability (takeover of PR #1855 by @starm2010).
*
* - The cost preview + hard cost ceiling price the model that will actually
* run: --model override configured chat_model gateway fallback. Before
* this, the preview always priced anthropic:claude-sonnet-4-6 even when
* the configured chat_model was something else.
* - The judge phase honors the `models.brainstorm.judge` config key when no
* --judge-model flag is passed.
*/
import { describe, test, expect } from 'bun:test';
import {
resolveBrainstormChatModel,
resolveBrainstormJudgeModel,
} from '../../src/core/brainstorm/orchestrator.ts';
import type { BrainEngine } from '../../src/core/engine.ts';
function mockEngine(configValues: Record<string, string>): { engine: BrainEngine; reads: string[] } {
const reads: string[] = [];
const engine = {
async getConfig(key: string): Promise<string | null> {
reads.push(key);
return configValues[key] ?? null;
},
} as unknown as BrainEngine;
return { engine, reads };
}
describe('resolveBrainstormChatModel', () => {
test('--model override wins over config', () => {
expect(resolveBrainstormChatModel({ chat_model: 'openai:gpt-5' }, 'anthropic:claude-opus-4-6'))
.toBe('anthropic:claude-opus-4-6');
});
test('configured chat_model wins over the hardcoded fallback', () => {
expect(resolveBrainstormChatModel({ chat_model: 'openai:gpt-5' }))
.toBe('openai:gpt-5');
});
test('falls back to the gateway default model when nothing is configured', () => {
expect(resolveBrainstormChatModel({})).toBe('anthropic:claude-sonnet-4-6');
});
});
describe('resolveBrainstormJudgeModel', () => {
test('--judge-model flag wins without touching config', async () => {
const { engine, reads } = mockEngine({ 'models.brainstorm.judge': 'openai:gpt-5' });
const out = await resolveBrainstormJudgeModel(engine, 'anthropic:claude-opus-4-6');
expect(out).toBe('anthropic:claude-opus-4-6');
expect(reads).toHaveLength(0);
});
test('models.brainstorm.judge config key is honored when no flag is passed', async () => {
const { engine, reads } = mockEngine({ 'models.brainstorm.judge': 'openai:gpt-5' });
const out = await resolveBrainstormJudgeModel(engine);
expect(out).toBe('openai:gpt-5');
expect(reads).toEqual(['models.brainstorm.judge']);
});
test('returns undefined (defer to modelOverride / gateway default) when unset', async () => {
const { engine } = mockEngine({});
expect(await resolveBrainstormJudgeModel(engine)).toBeUndefined();
});
});
+111
View File
@@ -0,0 +1,111 @@
/**
* #2301 re-init with an explicit --embedding-model must recover a brain
* that was initialized with --no-embedding (deferred setup).
*
* Pre-fix: resolveAIOptions honored the persisted `embedding_disabled: true`
* sentinel BEFORE the explicit flag and never cleared noEmbedding, and the
* persistence merge carried the sentinel forward via ...existingFile. Result:
* every re-init (including the recovery command the deferred-setup error
* itself recommends) silently re-deferred embedding, forever.
*
* Hermetic: in-process runInit, GBRAIN_HOME pinned to a tmpdir (same pattern
* as test/e2e/fresh-install-pglite.test.ts).
*/
import { afterEach, beforeEach, describe, expect, test } from 'bun:test';
import { mkdtempSync, rmSync, readFileSync } from 'fs';
import { tmpdir } from 'os';
import { join } from 'path';
import { configureGateway, resetGateway } from '../../src/core/ai/gateway.ts';
describe('E2E: re-init with --embedding-model after --no-embedding init (#2301)', () => {
let tmpHome: string;
let origHome: string | undefined;
let origZeKey: string | undefined;
let origOpenaiKey: string | undefined;
let origVoyageKey: string | undefined;
beforeEach(() => {
tmpHome = mkdtempSync(join(tmpdir(), 'gbrain-e2e-reinit-'));
origHome = process.env.GBRAIN_HOME;
origZeKey = process.env.ZEROENTROPY_API_KEY;
origOpenaiKey = process.env.OPENAI_API_KEY;
origVoyageKey = process.env.VOYAGE_API_KEY;
delete process.env.OPENAI_API_KEY;
delete process.env.VOYAGE_API_KEY;
process.env.GBRAIN_HOME = tmpHome;
process.env.ZEROENTROPY_API_KEY = 'sk-test-ze';
resetGateway();
});
afterEach(() => {
rmSync(tmpHome, { recursive: true, force: true });
if (origHome === undefined) delete process.env.GBRAIN_HOME;
else process.env.GBRAIN_HOME = origHome;
if (origZeKey === undefined) delete process.env.ZEROENTROPY_API_KEY;
else process.env.ZEROENTROPY_API_KEY = origZeKey;
if (origOpenaiKey !== undefined) process.env.OPENAI_API_KEY = origOpenaiKey;
if (origVoyageKey !== undefined) process.env.VOYAGE_API_KEY = origVoyageKey;
// Restore legacy-preload gateway state (mirrors fresh-install-pglite.test.ts).
configureGateway({
embedding_model: 'openai:text-embedding-3-large',
embedding_dimensions: 1536,
env: { ...process.env },
});
});
async function runInitCapturing(args: string[]): Promise<string> {
const { runInit } = await import('../../src/commands/init.ts');
const origLog = console.log;
const origWarn = console.warn;
const stdoutBuf: string[] = [];
console.log = (...a: unknown[]) => {
stdoutBuf.push(a.map(x => (typeof x === 'string' ? x : JSON.stringify(x))).join(' '));
};
console.warn = () => {};
try {
await runInit(args);
} finally {
console.log = origLog;
console.warn = origWarn;
}
return stdoutBuf.join('\n');
}
const cfgPath = () => join(tmpHome, '.gbrain', 'config.json');
const readCfg = () => JSON.parse(readFileSync(cfgPath(), 'utf-8'));
test('explicit --embedding-model clears the persisted embedding_disabled sentinel', async () => {
// Step 1: deferred-setup init writes the sentinel.
const out1 = await runInitCapturing(['--pglite', '--non-interactive', '--no-embedding']);
expect(out1).toContain('deferred setup');
const cfg1 = readCfg();
expect(cfg1.embedding_disabled).toBe(true);
expect(cfg1.embedding_model).toBeUndefined();
// Step 2: re-init with an explicit embedding model — the recovery path.
// Pre-fix this printed the deferred-setup line again and re-persisted
// embedding_disabled: true.
const out2 = await runInitCapturing([
'--pglite', '--non-interactive', '--skip-embed-check',
'--embedding-model', 'zeroentropyai:zembed-1',
'--embedding-dimensions', '1280',
]);
expect(out2).not.toContain('deferred setup');
expect(out2).toContain('zeroentropyai:zembed-1');
const cfg2 = readCfg();
expect(cfg2.embedding_model).toBe('zeroentropyai:zembed-1');
expect(cfg2.embedding_dimensions).toBe(1280);
expect(cfg2.embedding_disabled).toBeUndefined();
}, 60000);
test('re-init WITHOUT flags still honors the deferred-setup sentinel (no regression)', async () => {
await runInitCapturing(['--pglite', '--non-interactive', '--no-embedding']);
const out = await runInitCapturing(['--pglite', '--non-interactive']);
expect(out).toContain('deferred setup');
const cfg = readCfg();
expect(cfg.embedding_disabled).toBe(true);
expect(cfg.embedding_model).toBeUndefined();
}, 60000);
});
+68 -5
View File
@@ -11,9 +11,17 @@
* Fix: the CLI context builder computes `ctx.localFederatedSourceIds`
* (resolved source + every other federated source) whenever the source
* resolved via a NON-explicit tier; `federatedSearchScope` widens the scalar
* scope to that set for the `search` / `query` ops trusted-local only
* (`ctx.remote === false`), never for remote callers, never when a per-call
* `source_id` or an explicit --source/env/dotfile was given.
* scope to that set never when a per-call `source_id`, a grant array, or an
* explicit --source/env/dotfile was given.
*
* #3242 extends the same visibility set to `get_page` / `list_pages` /
* `resolve_slugs` (pages ingested into a `federated: true` source were
* invisible to normal reads while the unscoped resolve_slugs leaked them),
* and to transports whose caller carries NO explicit source scope (stdio
* without GBRAIN_SOURCE; legacy HTTP tokens without a `permissions.source_id`
* grant) those transports now populate `localFederatedSourceIds` themselves,
* so the widening gate is field-presence (transport-decided, never
* param-controlled), not `ctx.remote`.
*/
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
@@ -105,11 +113,19 @@ describe('federatedSearchScope — trust + explicitness matrix', () => {
expect(federatedSearchScope(ctx)).toEqual({ sourceIds: ['default', 'wiki'] });
});
test('remote caller NEVER widens (fail-closed), even if the field is set', () => {
const ctx = ctxOf({ remote: true, localFederatedSourceIds: ['default', 'wiki'] });
test('remote caller WITHOUT the field never widens (fail-closed)', () => {
const ctx = ctxOf({ remote: true });
expect(federatedSearchScope(ctx)).toEqual({ sourceId: 'default' });
});
test('#3242: remote caller widens when its transport populated the field (no-grant floor)', () => {
// The field is set only by server-side transports (stdio without
// GBRAIN_SOURCE / legacy HTTP token without a source grant) — never from
// caller params — so presence of the field IS the trust decision.
const ctx = ctxOf({ remote: true, localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx)).toEqual({ sourceIds: ['default', 'wiki'] });
});
test('per-call source_id wins over the federated set', () => {
const ctx = ctxOf({ localFederatedSourceIds: ['default', 'wiki'] });
expect(federatedSearchScope(ctx, 'wiki')).toEqual({ sourceId: 'wiki' });
@@ -152,3 +168,50 @@ describe('search op — unqualified local search spans federated sources', () =>
expect(slugs).toEqual(['notes/home']);
});
});
// #3242 — pages in a federated source must be visible to the normal read ops,
// not just search/query; and resolve_slugs must be SCOPED (pre-fix it was the
// one read that leaked every source's slugs).
describe('#3242 — get_page / list_pages / resolve_slugs share the federated visibility set', () => {
const getPage = operations.find((o) => o.name === 'get_page')!;
const listPages = operations.find((o) => o.name === 'list_pages')!;
const resolveSlugsOp = operations.find((o) => o.name === 'resolve_slugs')!;
function federatedCtx(overrides: Partial<OperationContext> = {}): OperationContext {
return ctxOf({ localFederatedSourceIds: ['default', 'wiki'], ...overrides });
}
test('get_page: federated-source page readable on an unqualified ctx (pre-fix: page_not_found)', async () => {
const page = (await getPage.handler(federatedCtx(), { slug: 'wiki/topic' })) as { slug: string };
expect(page.slug).toBe('wiki/topic');
});
test('get_page: non-federated source stays invisible', async () => {
await expect(getPage.handler(federatedCtx(), { slug: 'private/topic' })).rejects.toThrow(/not found/i);
});
test('get_page: scalar ctx (explicit source, no field) keeps single-source behavior', async () => {
await expect(getPage.handler(ctxOf(), { slug: 'wiki/topic' })).rejects.toThrow(/not found/i);
});
test('list_pages: federated-source pages listed on an unqualified ctx (pre-fix: missing)', async () => {
const rows = (await listPages.handler(federatedCtx(), {})) as Array<{ slug: string }>;
const slugs = rows.map((r) => r.slug);
expect(slugs).toContain('notes/home');
expect(slugs).toContain('wiki/topic');
expect(slugs).not.toContain('private/topic');
expect(slugs).not.toContain('old/topic');
});
test('resolve_slugs: scoped to the visibility set (pre-fix: leaked every source)', async () => {
const federated = (await resolveSlugsOp.handler(federatedCtx(), { partial: 'topic' })) as string[];
expect(federated).toContain('wiki/topic');
expect(federated).not.toContain('private/topic');
// A remote scalar caller (no field, no grant) must no longer see foreign slugs.
const scalar = (await resolveSlugsOp.handler(ctxOf({ remote: true }), { partial: 'topic' })) as string[];
expect(scalar).not.toContain('wiki/topic');
expect(scalar).not.toContain('private/topic');
expect(scalar).not.toContain('old/topic');
});
});
-77
View File
@@ -15,7 +15,6 @@
*/
import { describe, test, expect } from 'bun:test';
import { withEnv, emptyHome } from './helpers/with-env.ts';
import {
runPhaseProposeTakes,
parseExtractorOutput,
@@ -53,14 +52,6 @@ function buildMockEngine(opts: {
},
async executeRaw<T>(sql: string, params?: unknown[]): Promise<T[]> {
captured.push({ sql, params: params ?? [] });
// Narrow candidate-page projection (replaces listPages in the phase).
if (sql.includes('SELECT slug, source_id, compiled_truth')) {
return opts.pages.map((p) => ({
slug: p.slug,
source_id: p.source_id,
compiled_truth: p.compiled_truth,
})) as T[];
}
// SELECT idempotency check
if (sql.includes('SELECT id FROM take_proposals')) {
const [sourceId, slug, ch, pv] = params ?? [];
@@ -485,72 +476,4 @@ New prose appended here.`;
resetGateway();
}
});
test('default extractor skips cleanly when the Anthropic chat model has no key', async () => {
// Empty GBRAIN_HOME so hasAnthropicKey's config-file fallback can't find
// the operator's real key.
await withEnv({ GBRAIN_HOME: emptyHome(), ANTHROPIC_API_KEY: undefined }, async () => {
configureGateway({ chat_model: 'anthropic:claude-sonnet-4-6', env: {} });
try {
const { engine, captured } = buildMockEngine({
pages: [buildPage({ slug: 'wiki/a', body: 'claim-ish prose' })],
});
const result = await runPhaseProposeTakes(buildCtx(engine));
expect(result.status).toBe('skipped');
expect((result.details as Record<string, unknown>).reason).toBe('no_provider');
// Skips BEFORE touching the engine — no page scan, no cache probes.
expect(captured).toHaveLength(0);
} finally {
resetGateway();
}
});
});
test('an injected extractor is never gated on provider availability', async () => {
await withEnv({ GBRAIN_HOME: emptyHome(), ANTHROPIC_API_KEY: undefined }, async () => {
configureGateway({ chat_model: 'anthropic:claude-sonnet-4-6', env: {} });
try {
const { engine } = buildMockEngine({
pages: [buildPage({ slug: 'wiki/b', body: 'still processed' })],
});
const extractor: ProposeTakesExtractor = async () => [];
const result = await runPhaseProposeTakes(buildCtx(engine), { extractor });
expect(result.status).toBe('ok');
expect((result.details as Record<string, unknown>).pages_scanned).toBe(1);
} finally {
resetGateway();
}
});
});
test('loads proposal candidates with a narrow page projection', async () => {
const pages = [buildPage({ slug: 'wiki/narrow', body: 'A narrow projection avoids unrelated page columns.' })];
const { engine, captured } = buildMockEngine({ pages });
const extractor: ProposeTakesExtractor = async () => [];
await runPhaseProposeTakes(buildCtx(engine), { extractor });
const pageSelect = captured.find(c => c.sql.includes('FROM pages'));
expect(pageSelect).toBeDefined();
expect(pageSelect!.sql).toContain('SELECT slug, source_id, compiled_truth');
expect(pageSelect!.sql).not.toContain('*');
// Scalar sourceId scope from ctx binds as a plain equality param.
expect(pageSelect!.params[0]).toBe('default');
});
test('narrow projection: federated sourceIds beat scalar sourceId', async () => {
const { engine, captured } = buildMockEngine({ pages: [] });
const extractor: ProposeTakesExtractor = async () => [];
const ctx = {
...buildCtx(engine),
auth: { allowedSources: ['team-a', 'team-b'] },
} as OperationContext;
await runPhaseProposeTakes(ctx, { extractor });
const pageSelect = captured.find(c => c.sql.includes('FROM pages'));
expect(pageSelect).toBeDefined();
expect(pageSelect!.sql).toContain('source_id = ANY(');
expect(pageSelect!.params[0]).toEqual(['team-a', 'team-b']);
});
});
-16
View File
@@ -100,22 +100,6 @@ describe('searchTakes', () => {
const worldHits = await engine.searchTakes('founder', { takesHoldersAllowList: ['world'] });
expect(worldHits.every(h => h.holder === 'world')).toBe(true);
});
// #3267: whole-string trigram % structurally can't match a short keyword
// against a long claim (similarity between the full strings stays under the
// 0.3 threshold). word_similarity (<%) matches the keyword against the
// best-matching word span instead.
test('single-word keyword matches a long claim containing it (#3267)', async () => {
await engine.addTakesBatch([
{
page_id: acmePageId, row_num: 50,
claim: 'Acme will consolidate the mid-market vertical SaaS landscape through disciplined acquisitions and a shared billing platform over the next five years',
kind: 'bet', holder: 'garry', weight: 0.6,
},
]);
const hits = await engine.searchTakes('consolidate');
expect(hits.some(h => h.claim.includes('consolidate the mid-market'))).toBe(true);
});
});
describe('updateTake', () => {