Compare commits

..
Author SHA1 Message Date
d9834a7a15 feat(ai): add reranker touchpoint to LiteLLM proxy recipe (takeover of #2455)
LiteLLM normalizes Cohere/Voyage/Jina rerank backends to the wire shape
gateway.rerank() already speaks, so a reranker touchpoint on the litellm
recipe makes any proxied rerank model reachable via
`search.reranker.model litellm:<model>` with no adapter.

Repairs from the original PR:
- path is the LEAF '/rerank' (not '/v1/rerank'): LiteLLM serves both
  /rerank and /v1/rerank, and the recipe's setup_hint allows
  LITELLM_BASE_URL with or without the /v1 suffix — pinning '/v1/rerank'
  doubled to /v1/v1/rerank (404) on /v1-suffixed bases.
- setup_hint appends the rerank guidance to master's current line instead
  of replacing it with a stale pre-/v1-suffix version.
- cost_per_1m_tokens_usd stays undefined (pricing-unknown), matching the
  recipe's embedding/chat touchpoints and budget-tracker's deliberate
  litellm exclusion from the free-provider sets (a proxy can front a paid
  provider; the touchpoint field isn't consumed by rerank pricing anyway).

Test drives gateway.rerank()'s real URL builder via the stubbed transport
for both base-URL forms; the /v1-suffixed case fails with the original
PR's path.

Co-authored-by: ozp <ozp@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:28:27 -07:00
4 changed files with 122 additions and 220 deletions
+31 -1
View File
@@ -56,6 +56,36 @@ export const litellmProxy: Recipe = {
cost_per_1m_output_usd: undefined,
price_last_verified: '2026-06-14',
},
// LiteLLM normalizes Cohere / Voyage / Jina / etc. rerank backends to the
// same wire shape gbrain's gateway.rerank() already speaks (the
// ZeroEntropy/llama.cpp contract):
// { model, query, documents, top_n } → { results: [{ index, relevance_score }] }
// So any rerank model the user registers in their LiteLLM config is
// reachable via `gbrain config set search.reranker.model litellm:<model>`
// with no request/response adapter — same as embeddings ride the proxy.
reranker: {
models: [], // user-provided; whatever rerank models the proxy serves
// No canonical default — the proxy defines its own model ids. The user
// sets search.reranker.model explicitly (mirrors the embedding
// touchpoint's user_provided_models contract).
default_model: '',
// The proxied backend bills (Cohere/Voyage/…); pricing-unknown is the
// honest state — same stance as this recipe's embedding/chat
// touchpoints and budget-tracker's deliberate litellm exclusion from
// the free-provider sets.
cost_per_1m_tokens_usd: undefined,
price_last_verified: '2026-06-27',
max_payload_bytes: 5_000_000,
// LEAF path only (matches llama-server-reranker's convention). LiteLLM
// serves both `/rerank` and `/v1/rerank`, and LITELLM_BASE_URL may be
// set with or without the `/v1` suffix (the setup_hint allows both), so
// the leaf form yields a valid route either way:
// http://localhost:4000 + /rerank → /rerank ✓
// http://localhost:4000/v1 + /rerank → /v1/rerank ✓
// Pinning '/v1/rerank' here would double to /v1/v1/rerank → 404 on
// /v1-suffixed bases.
path: '/rerank',
},
},
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>.',
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>. For rerank: register a rerank model in LiteLLM and set search.reranker.model litellm:<model-name>.',
};
+2 -112
View File
@@ -1,100 +1,8 @@
import type { Recipe } from '../types.ts';
/**
* MiniMax transport shim (#1977). MiniMax's `/v1/embeddings` endpoint is NOT
* OpenAI-compatible at the wire level despite the recipe's
* `implementation: 'openai-compatible'`:
* - Request: requires `texts` (the AI SDK sends `input`) plus an optional
* `type: 'db' | 'query'` asymmetric-retrieval field, and rejects OpenAI's
* `encoding_format`.
* - Response: returns `{vectors: number[][], total_tokens}` where the AI
* SDK's Zod schema expects `{data: [{embedding, index}], usage}`.
*
* Chat (`/chat/completions`) IS OpenAI-compatible, and this same fetch is
* applied to every openai-compatible touchpoint by `applyOpenAICompatConfig`,
* so everything outside the embeddings path passes through untouched — and
* the response rewrite parses via `resp.clone()` only (never consume the
* body of a response we return as-is; the DeepSeek shim rule). Fail-open:
* any rewrite error returns the original request/response.
*
* @internal exported for tests.
*/
// Cast through `unknown` because Bun's `typeof fetch` carries a `preconnect`
// member the arrow function does not implement (matches deepseek.ts).
export const minimaxCompatFetch = (async (
input: RequestInfo | URL,
init?: RequestInit,
): Promise<Response> => {
const url =
typeof input === 'string' ? input : input instanceof URL ? input.href : input.url;
const isEmbeddings = url.includes('/embeddings');
// OUTBOUND (embeddings only): `input` → `texts`, default `type: 'db'`
// (the recipe's documented symmetric default — the AI SDK adapter strips
// the `type` threaded via providerOptions before it reaches the wire,
// same class as #1400), and drop `encoding_format` (not a MiniMax param).
if (isEmbeddings && init?.body && typeof init.body === 'string') {
try {
const parsed = JSON.parse(init.body);
if (
parsed && typeof parsed === 'object' &&
parsed.input !== undefined && parsed.texts === undefined
) {
parsed.texts = Array.isArray(parsed.input) ? parsed.input : [parsed.input];
delete parsed.input;
delete parsed.encoding_format;
if (parsed.type === undefined) parsed.type = 'db';
// Drop Content-Length so fetch recomputes from the new body.
const headers = new Headers(init.headers ?? {});
headers.delete('content-length');
init = { ...init, body: JSON.stringify(parsed), headers };
}
} catch {
// Body wasn't JSON — pass through untouched.
}
}
const res = await fetch(input as any, init as any);
// INBOUND (embeddings only): `{vectors: [[...]]}` → `{data: [{embedding}]}`.
// Anything else (chat completions, MiniMax base_resp errors, non-JSON)
// returns the ORIGINAL response with its body unread.
if (!isEmbeddings || !res.ok) return res;
const ctype = res.headers.get('content-type') ?? '';
if (!ctype.toLowerCase().includes('application/json')) return res;
try {
const json = await res.clone().json();
if (!json || typeof json !== 'object' || !Array.isArray(json.vectors)) return res;
const totalTokens = typeof json.total_tokens === 'number' ? json.total_tokens : 0;
const rewritten = {
object: 'list',
data: (json.vectors as number[][]).map((embedding, index) => ({
object: 'embedding',
embedding,
index,
})),
model: typeof json.model === 'string' ? json.model : 'embo-01',
usage: { prompt_tokens: totalTokens, total_tokens: totalTokens },
};
// Fresh header set: the body changed, so upstream content-length /
// content-encoding would now be wrong.
const headers = new Headers(res.headers);
headers.delete('content-length');
headers.delete('content-encoding');
return new Response(JSON.stringify(rewritten), {
status: res.status,
statusText: res.statusText,
headers,
});
} catch {
return res;
}
}) as unknown as typeof fetch;
/**
* MiniMax (海螺AI). `/embeddings` endpoint at api.minimaxi.com (wire shape
* normalized by `minimaxCompatFetch` above); OpenAI-compatible
* `/chat/completions`. The flagship embedding model is `embo-01` (1536 dims).
* MiniMax (海螺AI). OpenAI-compatible /embeddings endpoint at
* api.minimax.chat. The flagship embedding model is `embo-01` (1536 dims).
*
* MiniMax's API takes an extra `type: 'db' | 'query'` field for asymmetric
* retrieval. gbrain currently has no notion of "this is a document vs a
@@ -130,25 +38,7 @@ export const minimax: Recipe = {
// halving in the gateway catches token-limit errors at runtime.
max_batch_tokens: 4096,
},
chat: {
// Model list from MiniMax's /v1/models (#1977). Chat is genuinely
// OpenAI-compatible — no wire rewrite needed (minimaxCompatFetch
// passes non-embedding requests through untouched).
models: [
'MiniMax-M3',
'MiniMax-M2.7',
'MiniMax-M2.7-highspeed',
'MiniMax-M2.5',
'MiniMax-M2.5-highspeed',
'MiniMax-M2.1',
'MiniMax-M2.1-highspeed',
'MiniMax-M2',
],
supports_tools: false,
supports_subagent_loop: false,
},
},
setup_hint:
'Get an API key at https://www.minimaxi.com, then `export MINIMAX_API_KEY=...`',
compat: { fetch: minimaxCompatFetch },
};
+88
View File
@@ -0,0 +1,88 @@
/**
* litellm-proxy reranker touchpoint smoke.
*
* Sibling of recipe-llama-server-reranker.test.ts. Pins the reranker
* touchpoint on the LiteLLM proxy recipe so:
* - the touchpoint exists with the LEAF '/rerank' path (LiteLLM serves both
* /rerank and /v1/rerank, so the leaf form is valid whether or not the
* user's LITELLM_BASE_URL carries the /v1 suffix the setup_hint allows)
* - a /v1-suffixed base URL does NOT produce /v1/v1/rerank (the original
* community PR pinned '/v1/rerank' which 404s on /v1-suffixed bases)
* - models: [] (user-provided; proxy defines the model ids)
* - pricing stays undefined (proxy can front a paid provider — same honest
* pricing-unknown stance as the embedding/chat touchpoints)
*
* The gateway.rerank() URL tests drive the real URL builder via the stubbed
* transport (same seam as test/ai/rerank.test.ts).
*/
import { describe, expect, test, afterEach } from 'bun:test';
import { getRecipe } from '../../src/core/ai/recipes/index.ts';
import {
configureGateway,
resetGateway,
rerank,
__setRerankTransportForTests,
} from '../../src/core/ai/gateway.ts';
afterEach(() => {
__setRerankTransportForTests(null);
resetGateway();
});
describe('recipe: litellm reranker touchpoint', () => {
test('declares reranker touchpoint with leaf /rerank path', () => {
const r = getRecipe('litellm')!;
const tp = r.touchpoints.reranker;
expect(tp).toBeDefined();
expect(tp!.path).toBe('/rerank');
expect(tp!.max_payload_bytes).toBe(5_000_000);
});
test('reranker touchpoint uses empty models[] for user-provided model ids', () => {
const r = getRecipe('litellm')!;
expect(r.touchpoints.reranker!.models).toEqual([]);
});
test('pricing stays undefined — proxy can front a paid provider', () => {
const r = getRecipe('litellm')!;
expect(r.touchpoints.reranker!.cost_per_1m_tokens_usd).toBeUndefined();
});
test('setup_hint keeps the /v1-suffix guidance AND mentions rerank', () => {
const r = getRecipe('litellm')!;
expect(r.setup_hint).toMatch(/\/v1 suffix/);
expect(r.setup_hint).toMatch(/search\.reranker\.model litellm:/);
});
});
describe('gateway.rerank() URL via litellm recipe', () => {
async function capturedRerankUrl(baseUrl?: string): Promise<string> {
configureGateway({
reranker_model: 'litellm:my-reranker',
env: {},
...(baseUrl ? { base_urls: { litellm: baseUrl } } : {}),
});
let capturedUrl = '';
__setRerankTransportForTests(async (url) => {
capturedUrl = url;
return new Response(
JSON.stringify({ results: [{ index: 0, relevance_score: 0.9 }] }),
{ status: 200, headers: { 'content-type': 'application/json' } },
);
});
await rerank({ query: 'q', documents: ['d'] });
return capturedUrl;
}
test('default base (no /v1 suffix) → /rerank', async () => {
const url = await capturedRerankUrl();
expect(url).toBe('http://localhost:4000/rerank');
});
test('/v1-suffixed base → /v1/rerank, NOT /v1/v1/rerank', async () => {
const url = await capturedRerankUrl('http://localhost:4000/v1');
expect(url).toBe('http://localhost:4000/v1/rerank');
expect(url).not.toContain('/v1/v1/');
});
});
+1 -107
View File
@@ -6,14 +6,10 @@
* - default auth: MINIMAX_API_KEY → "Bearer <key>"; missing → AIConfigError
* - dimsProviderOptions threads `type: 'db'` for embo-01 (the asymmetric
* retrieval field default) — pins the v1 indexing-only behavior
* - #1977: chat touchpoint declared; minimaxCompatFetch rewrites the
* embedding wire shape both directions, passes chat through with the
* response body UNREAD (the consumed-body regression), fail-open.
*/
import { afterEach, describe, expect, test } from 'bun:test';
import { describe, expect, test } from 'bun:test';
import { getRecipe } from '../../src/core/ai/recipes/index.ts';
import { minimaxCompatFetch } from '../../src/core/ai/recipes/minimax.ts';
import { defaultResolveAuth } from '../../src/core/ai/gateway.ts';
import { dimsProviderOptions } from '../../src/core/ai/dims.ts';
import { AIConfigError } from '../../src/core/ai/errors.ts';
@@ -60,106 +56,4 @@ describe('recipe: minimax', () => {
expect(dimsProviderOptions('openai-compatible', 'voyage-3-lite', 512)).toBeUndefined();
expect(dimsProviderOptions('openai-compatible', 'nomic-embed-text', 768)).toBeUndefined();
});
test('chat touchpoint declared (#1977) so assertTouchpoint permits gbrain think', () => {
const r = getRecipe('minimax')!;
expect(r.touchpoints.chat).toBeDefined();
expect(r.touchpoints.chat!.models).toContain('MiniMax-M3');
expect(r.touchpoints.chat!.supports_tools).toBe(false);
expect(r.touchpoints.chat!.supports_subagent_loop).toBe(false);
});
test('recipe ships minimaxCompatFetch via compat.fetch (no env-templated base URL)', () => {
const r = getRecipe('minimax')!;
expect(r.compat?.fetch).toBe(minimaxCompatFetch);
// base_urls config override must keep working: no resolveOpenAICompatConfig.
expect(r.resolveOpenAICompatConfig).toBeUndefined();
});
});
describe('minimaxCompatFetch (#1977)', () => {
const realFetch = globalThis.fetch;
afterEach(() => { globalThis.fetch = realFetch; });
function stubFetch(body: unknown, init?: { status?: number; contentType?: string }) {
const calls: { url: string; init?: RequestInit }[] = [];
globalThis.fetch = (async (input: any, i?: RequestInit) => {
calls.push({ url: String(input), init: i });
return new Response(typeof body === 'string' ? body : JSON.stringify(body), {
status: init?.status ?? 200,
headers: { 'content-type': init?.contentType ?? 'application/json' },
});
}) as unknown as typeof fetch;
return calls;
}
const EMBED_URL = 'https://api.minimaxi.com/v1/embeddings';
const CHAT_URL = 'https://api.minimaxi.com/v1/chat/completions';
test('embedding request: input → texts, type:db injected, encoding_format dropped', async () => {
const calls = stubFetch({ vectors: [[0.1, 0.2]] });
await minimaxCompatFetch(EMBED_URL, {
method: 'POST',
headers: { 'content-type': 'application/json', 'content-length': '99' },
body: JSON.stringify({ model: 'embo-01', input: ['hello', 'world'], encoding_format: 'float' }),
});
const wire = JSON.parse(calls[0]!.init!.body as string);
expect(wire.texts).toEqual(['hello', 'world']);
expect(wire.input).toBeUndefined();
expect(wire.encoding_format).toBeUndefined();
expect(wire.type).toBe('db');
expect(new Headers(calls[0]!.init!.headers).get('content-length')).toBeNull();
});
test('embedding response: {vectors} rewritten to OpenAI {data:[{embedding}]}', async () => {
stubFetch({ vectors: [[0.1, 0.2], [0.3, 0.4]], total_tokens: 7 });
const res = await minimaxCompatFetch(EMBED_URL, {
method: 'POST',
body: JSON.stringify({ model: 'embo-01', input: ['a', 'b'] }),
});
const json = await res.json();
expect(json.data).toEqual([
{ object: 'embedding', embedding: [0.1, 0.2], index: 0 },
{ object: 'embedding', embedding: [0.3, 0.4], index: 1 },
]);
expect(json.usage).toEqual({ prompt_tokens: 7, total_tokens: 7 });
});
test('chat completion passes through with body UNREAD (consumed-body regression)', async () => {
stubFetch({ choices: [{ message: { role: 'assistant', content: 'hi' } }] });
const res = await minimaxCompatFetch(CHAT_URL, {
method: 'POST',
body: JSON.stringify({ model: 'MiniMax-M3', messages: [{ role: 'user', content: 'say hi' }] }),
});
expect(res.bodyUsed).toBe(false); // the broken PR #2882 wrapper consumed this
const json = await res.json(); // must NOT throw "Body already used"
expect(json.choices[0].message.content).toBe('hi');
});
test('chat request body is never rewritten (messages untouched, no type injected)', async () => {
const calls = stubFetch({ choices: [] });
const body = JSON.stringify({ model: 'MiniMax-M3', messages: [{ role: 'user', content: 'x' }] });
await minimaxCompatFetch(CHAT_URL, { method: 'POST', body });
expect(calls[0]!.init!.body).toBe(body);
});
test('embedding error response ({vectors:null, base_resp}) passes through re-readable', async () => {
stubFetch({ vectors: null, base_resp: { status_code: 2013, status_msg: 'invalid params' } });
const res = await minimaxCompatFetch(EMBED_URL, {
method: 'POST',
body: JSON.stringify({ model: 'embo-01', input: ['a'] }),
});
expect(res.bodyUsed).toBe(false);
const json = await res.json();
expect(json.base_resp.status_code).toBe(2013);
});
test('fail-open: non-JSON response body passes through untouched', async () => {
stubFetch('not json', { contentType: 'application/json' });
const res = await minimaxCompatFetch(EMBED_URL, {
method: 'POST',
body: JSON.stringify({ model: 'embo-01', input: ['a'] }),
});
expect(await res.text()).toBe('not json');
});
});