Compare commits

..
Author SHA1 Message Date
d9834a7a15 feat(ai): add reranker touchpoint to LiteLLM proxy recipe (takeover of #2455)
LiteLLM normalizes Cohere/Voyage/Jina rerank backends to the wire shape
gateway.rerank() already speaks, so a reranker touchpoint on the litellm
recipe makes any proxied rerank model reachable via
`search.reranker.model litellm:<model>` with no adapter.

Repairs from the original PR:
- path is the LEAF '/rerank' (not '/v1/rerank'): LiteLLM serves both
  /rerank and /v1/rerank, and the recipe's setup_hint allows
  LITELLM_BASE_URL with or without the /v1 suffix — pinning '/v1/rerank'
  doubled to /v1/v1/rerank (404) on /v1-suffixed bases.
- setup_hint appends the rerank guidance to master's current line instead
  of replacing it with a stale pre-/v1-suffix version.
- cost_per_1m_tokens_usd stays undefined (pricing-unknown), matching the
  recipe's embedding/chat touchpoints and budget-tracker's deliberate
  litellm exclusion from the free-provider sets (a proxy can front a paid
  provider; the touchpoint field isn't consumed by rerank pricing anyway).

Test drives gateway.rerank()'s real URL builder via the stubbed transport
for both base-URL forms; the /v1-suffixed case fails with the original
PR's path.

Co-authored-by: ozp <ozp@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:28:27 -07:00
5 changed files with 119 additions and 234 deletions
-64
View File
@@ -24,7 +24,6 @@ import type { GBrainConfig } from './core/config.ts';
import type { AIGatewayConfig } from './core/ai/types.ts';
import type { BrainEngine } from './core/engine.ts';
import { operations, OperationError } from './core/operations.ts';
import { resolveSourceIdEngineFree } from './core/source-resolver.ts';
import { formatVolunteeredPage } from './core/context/volunteer.ts';
import type { Operation, OperationContext } from './core/operations.ts';
import { shouldForceExitAfterMain, finishCliTeardown, flushThenExit, currentExitCode, setCliExitVerdict } from './core/cli-force-exit.ts';
@@ -383,15 +382,6 @@ async function main() {
if (op.localOnly) {
refuseThinClient(command, cfgPre!.remote_mcp!.mcp_url);
}
// #2098: the local path resolves --source / GBRAIN_SOURCE / .gbrain-source
// inside makeContext (ctx.sourceId), which this route never reaches — so
// scope must be mapped onto the op's source_id wire param before the call.
try {
applyThinClientSourceScope(op, params);
} catch (e: unknown) {
console.error(e instanceof Error ? e.message : String(e));
process.exit(1);
}
await runThinClientRouted(op, params, cfgPre!, cliOpts);
return;
}
@@ -812,60 +802,6 @@ export function parseOpArgs(op: Operation, args: string[]): Record<string, unkno
return params;
}
/**
* #2098: thin-client source scoping. Locally, --source / GBRAIN_SOURCE /
* .gbrain-source resolve to ctx.sourceId in makeContext; the thin-client
* route short-circuits before that, so `gbrain query --source X` against a
* remote brain silently searched unscoped. This runs the engine-free tiers
* (flag → env → dotfile; the DB-backed tiers can't run without an engine —
* the server's grant scoping covers the rest) and maps the result onto the
* op's `source_id` wire param.
*
* Ops that declare their OWN `source` param (facts add, etc.) are left
* untouched — their --source is an op param, not scope. An explicit --source
* on an op with no source_id wire param throws (loud beats silent drop);
* ambient env/dotfile scope with nowhere to send it is ignored, matching the
* pre-fix behavior for non-scopeable ops. Exported for tests.
*/
// Ops whose `source_id` wire param is NOT read-scope semantics: get_skill's
// source_id flips the lookup from host catalog to brain-resident-pack
// (getResidentSkillDetail). Ambient env/dotfile scope must never leak into
// these; an explicit --source-id still passes through untouched above.
const NON_SCOPE_SOURCE_ID_OPS = new Set(['get_skill']);
export function applyThinClientSourceScope(
op: Operation,
params: Record<string, unknown>,
cwd?: string,
): void {
if ('source' in op.params) return; // the op owns --source; not a scope flag
const explicit = typeof params.source === 'string' && params.source.length > 0
? (params.source as string)
: null;
delete params.source; // never a wire param on these ops — don't leak it
// Explicit per-call scope already on the wire wins over ambient tiers.
if (params.source_id !== undefined || params.all_sources === true) {
if (explicit) {
throw new Error('Pass either --source or --source-id/--all-sources, not both.');
}
return;
}
const resolved = resolveSourceIdEngineFree(explicit, cwd);
if (!resolved) return;
if (!('source_id' in op.params) || NON_SCOPE_SOURCE_ID_OPS.has(op.name)) {
if (explicit) {
const hint = NON_SCOPE_SOURCE_ID_OPS.has(op.name)
? `(its source_id parameter is not a scope filter; pass --source-id explicitly if you mean it)`
: `(the remote op has no source_id parameter; the server scopes it to your grant)`;
throw new Error(
`gbrain ${op.cliHints?.name || op.name} does not accept --source on a thin-client install ${hint}.`,
);
}
return; // ambient env/dotfile scope with nowhere to send it
}
params.source_id = resolved;
}
async function makeContext(engine: BrainEngine, params: Record<string, unknown>): Promise<OperationContext> {
// v0.31.8 (D11): resolve sourceId via the canonical 6-tier chain. Honors
// --source / GBRAIN_SOURCE / .gbrain-source / path-match / brain default /
+31 -1
View File
@@ -56,6 +56,36 @@ export const litellmProxy: Recipe = {
cost_per_1m_output_usd: undefined,
price_last_verified: '2026-06-14',
},
// LiteLLM normalizes Cohere / Voyage / Jina / etc. rerank backends to the
// same wire shape gbrain's gateway.rerank() already speaks (the
// ZeroEntropy/llama.cpp contract):
// { model, query, documents, top_n } → { results: [{ index, relevance_score }] }
// So any rerank model the user registers in their LiteLLM config is
// reachable via `gbrain config set search.reranker.model litellm:<model>`
// with no request/response adapter — same as embeddings ride the proxy.
reranker: {
models: [], // user-provided; whatever rerank models the proxy serves
// No canonical default — the proxy defines its own model ids. The user
// sets search.reranker.model explicitly (mirrors the embedding
// touchpoint's user_provided_models contract).
default_model: '',
// The proxied backend bills (Cohere/Voyage/…); pricing-unknown is the
// honest state — same stance as this recipe's embedding/chat
// touchpoints and budget-tracker's deliberate litellm exclusion from
// the free-provider sets.
cost_per_1m_tokens_usd: undefined,
price_last_verified: '2026-06-27',
max_payload_bytes: 5_000_000,
// LEAF path only (matches llama-server-reranker's convention). LiteLLM
// serves both `/rerank` and `/v1/rerank`, and LITELLM_BASE_URL may be
// set with or without the `/v1` suffix (the setup_hint allows both), so
// the leaf form yields a valid route either way:
// http://localhost:4000 + /rerank → /rerank ✓
// http://localhost:4000/v1 + /rerank → /v1/rerank ✓
// Pinning '/v1/rerank' here would double to /v1/v1/rerank → 404 on
// /v1-suffixed bases.
path: '/rerank',
},
},
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>.',
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>. For rerank: register a rerank model in LiteLLM and set search.reranker.model litellm:<model-name>.',
};
-27
View File
@@ -160,33 +160,6 @@ export async function resolveSourceId(
return 'default';
}
/**
* Engine-free tiers (1-3) of the resolution chain: explicit flag →
* GBRAIN_SOURCE env → .gbrain-source dotfile walk. Used by the thin-client
* CLI path (#2098), which has no local engine to run tiers 4-6 or
* assertSourceExists against — the remote server enforces existence + grant.
* Returns null when no engine-free tier fires.
*/
export function resolveSourceIdEngineFree(
explicit: string | null | undefined,
cwd: string = process.cwd(),
): string | null {
if (explicit) {
if (!SOURCE_ID_RE.test(explicit)) {
throw new Error(`Invalid --source value "${explicit}". Must match [a-z0-9-]{1,32}.`);
}
return explicit;
}
const env = process.env.GBRAIN_SOURCE;
if (env && env.length > 0) {
if (!SOURCE_ID_RE.test(env)) {
throw new Error(`Invalid GBRAIN_SOURCE value "${env}". Must match [a-z0-9-]{1,32}.`);
}
return env;
}
return readDotfileWalk(cwd);
}
/**
* Returns the id of the SINGLE registered non-default source with a
* local_path, when exactly one such row exists. Returns null when:
+88
View File
@@ -0,0 +1,88 @@
/**
* litellm-proxy reranker touchpoint smoke.
*
* Sibling of recipe-llama-server-reranker.test.ts. Pins the reranker
* touchpoint on the LiteLLM proxy recipe so:
* - the touchpoint exists with the LEAF '/rerank' path (LiteLLM serves both
* /rerank and /v1/rerank, so the leaf form is valid whether or not the
* user's LITELLM_BASE_URL carries the /v1 suffix the setup_hint allows)
* - a /v1-suffixed base URL does NOT produce /v1/v1/rerank (the original
* community PR pinned '/v1/rerank' which 404s on /v1-suffixed bases)
* - models: [] (user-provided; proxy defines the model ids)
* - pricing stays undefined (proxy can front a paid provider — same honest
* pricing-unknown stance as the embedding/chat touchpoints)
*
* The gateway.rerank() URL tests drive the real URL builder via the stubbed
* transport (same seam as test/ai/rerank.test.ts).
*/
import { describe, expect, test, afterEach } from 'bun:test';
import { getRecipe } from '../../src/core/ai/recipes/index.ts';
import {
configureGateway,
resetGateway,
rerank,
__setRerankTransportForTests,
} from '../../src/core/ai/gateway.ts';
afterEach(() => {
__setRerankTransportForTests(null);
resetGateway();
});
describe('recipe: litellm reranker touchpoint', () => {
test('declares reranker touchpoint with leaf /rerank path', () => {
const r = getRecipe('litellm')!;
const tp = r.touchpoints.reranker;
expect(tp).toBeDefined();
expect(tp!.path).toBe('/rerank');
expect(tp!.max_payload_bytes).toBe(5_000_000);
});
test('reranker touchpoint uses empty models[] for user-provided model ids', () => {
const r = getRecipe('litellm')!;
expect(r.touchpoints.reranker!.models).toEqual([]);
});
test('pricing stays undefined — proxy can front a paid provider', () => {
const r = getRecipe('litellm')!;
expect(r.touchpoints.reranker!.cost_per_1m_tokens_usd).toBeUndefined();
});
test('setup_hint keeps the /v1-suffix guidance AND mentions rerank', () => {
const r = getRecipe('litellm')!;
expect(r.setup_hint).toMatch(/\/v1 suffix/);
expect(r.setup_hint).toMatch(/search\.reranker\.model litellm:/);
});
});
describe('gateway.rerank() URL via litellm recipe', () => {
async function capturedRerankUrl(baseUrl?: string): Promise<string> {
configureGateway({
reranker_model: 'litellm:my-reranker',
env: {},
...(baseUrl ? { base_urls: { litellm: baseUrl } } : {}),
});
let capturedUrl = '';
__setRerankTransportForTests(async (url) => {
capturedUrl = url;
return new Response(
JSON.stringify({ results: [{ index: 0, relevance_score: 0.9 }] }),
{ status: 200, headers: { 'content-type': 'application/json' } },
);
});
await rerank({ query: 'q', documents: ['d'] });
return capturedUrl;
}
test('default base (no /v1 suffix) → /rerank', async () => {
const url = await capturedRerankUrl();
expect(url).toBe('http://localhost:4000/rerank');
});
test('/v1-suffixed base → /v1/rerank, NOT /v1/v1/rerank', async () => {
const url = await capturedRerankUrl('http://localhost:4000/v1');
expect(url).toBe('http://localhost:4000/v1/rerank');
expect(url).not.toContain('/v1/v1/');
});
});
-142
View File
@@ -1,142 +0,0 @@
/**
* #2098: thin-client routing dropped --source / GBRAIN_SOURCE / .gbrain-source.
*
* The local CLI path resolves source scope in makeContext (ctx.sourceId); the
* thin-client route short-circuits before that and sent params verbatim, so
* `gbrain query --source X` against a remote brain silently searched unscoped
* (the server op ignores the unknown `source` key).
*
* applyThinClientSourceScope runs the engine-free tiers (flag → env → dotfile)
* and maps the result onto the op's `source_id` wire param. These tests fail
* without the fix (params.source_id stays undefined / params.source leaks).
*/
import { describe, test, expect } from 'bun:test';
import { mkdtempSync, rmSync, writeFileSync } from 'fs';
import { join } from 'path';
import { tmpdir } from 'os';
import { applyThinClientSourceScope, parseOpArgs } from '../src/cli.ts';
import { operationsByName } from '../src/core/operations.ts';
import { withEnv } from './helpers/with-env.ts';
const queryOp = operationsByName.query;
describe('applyThinClientSourceScope (#2098)', () => {
test('--source maps onto the query op wire param source_id', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const params = parseOpArgs(queryOp, ['find things', '--source', 'wiki']);
expect(params.source).toBe('wiki'); // pre-fix state: wrong key
applyThinClientSourceScope(queryOp, params, '/');
expect(params.source_id).toBe('wiki');
expect('source' in params).toBe(false); // never leaks the unknown key
});
});
test('GBRAIN_SOURCE env tier fires when no flag is passed', async () => {
await withEnv({ GBRAIN_SOURCE: 'gstack' }, () => {
const params = parseOpArgs(queryOp, ['find things']);
applyThinClientSourceScope(queryOp, params, '/');
expect(params.source_id).toBe('gstack');
});
});
test('.gbrain-source dotfile tier fires when flag and env are absent', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const tmp = mkdtempSync(join(tmpdir(), 'gbrain-thin-scope-'));
try {
writeFileSync(join(tmp, '.gbrain-source'), 'essays\n');
const params = parseOpArgs(queryOp, ['find things']);
applyThinClientSourceScope(queryOp, params, tmp);
expect(params.source_id).toBe('essays');
} finally {
rmSync(tmp, { recursive: true, force: true });
}
});
});
test('explicit --source-id on the wire wins over ambient env scope', async () => {
await withEnv({ GBRAIN_SOURCE: 'gstack' }, () => {
const params = parseOpArgs(queryOp, ['find things', '--source-id', 'wiki']);
applyThinClientSourceScope(queryOp, params, '/');
expect(params.source_id).toBe('wiki');
});
});
test('--source together with --source-id is rejected loudly', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const params = parseOpArgs(queryOp, ['q', '--source', 'a', '--source-id', 'b']);
expect(() => applyThinClientSourceScope(queryOp, params, '/')).toThrow(/not both/);
});
});
test('invalid --source value is rejected loudly', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const params = parseOpArgs(queryOp, ['q', '--source', 'Bad_Value!']);
expect(() => applyThinClientSourceScope(queryOp, params, '/')).toThrow(/Invalid --source/);
});
});
test('--source on an op with no source_id wire param errors instead of silently dropping', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const op = operationsByName.add_tag;
expect('source_id' in op.params).toBe(false);
const params = { slug: 'x', tag: 'y', source: 'wiki' };
expect(() => applyThinClientSourceScope(op, params, '/')).toThrow(/--source/);
});
});
test('ambient env scope on an op with no source_id wire param is ignored (no throw)', async () => {
await withEnv({ GBRAIN_SOURCE: 'wiki' }, () => {
const op = operationsByName.add_tag;
const params: Record<string, unknown> = { slug: 'x', tag: 'y' };
applyThinClientSourceScope(op, params, '/');
expect(params.source_id).toBeUndefined();
});
});
test('ops that declare their OWN source param are left untouched', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const op = operationsByName.put_raw_data;
expect('source' in op.params).toBe(true);
const params: Record<string, unknown> = { slug: 'x', source: 'crustdata', data: {} };
applyThinClientSourceScope(op, params, '/');
expect(params.source).toBe('crustdata');
expect(params.source_id).toBeUndefined();
});
});
test('get_skill: ambient scope never leaks into its non-scope source_id param', async () => {
await withEnv({ GBRAIN_SOURCE: 'wiki' }, () => {
const op = operationsByName.get_skill;
expect('source_id' in op.params).toBe(true); // has the param, but it is a mode switch
const params: Record<string, unknown> = { name: 'ingest' };
applyThinClientSourceScope(op, params, '/');
expect(params.source_id).toBeUndefined(); // would flip host catalog → brain-pack lookup
});
});
test('get_skill: explicit --source errors instead of masquerading as --source-id', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const op = operationsByName.get_skill;
const params: Record<string, unknown> = { name: 'ingest', source: 'wiki' };
expect(() => applyThinClientSourceScope(op, params, '/')).toThrow(/--source-id/);
});
});
test('get_skill: explicit --source-id passes through untouched', async () => {
await withEnv({ GBRAIN_SOURCE: 'gstack' }, () => {
const op = operationsByName.get_skill;
const params: Record<string, unknown> = { name: 'ingest', source_id: 'wiki' };
applyThinClientSourceScope(op, params, '/');
expect(params.source_id).toBe('wiki');
});
});
test('no scope from any tier leaves params unchanged', async () => {
await withEnv({ GBRAIN_SOURCE: undefined }, () => {
const params = parseOpArgs(queryOp, ['find things']);
applyThinClientSourceScope(queryOp, params, '/');
expect(params.source_id).toBeUndefined();
});
});
});