mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-17 18:32:41 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d9834a7a15 |
@@ -56,6 +56,36 @@ export const litellmProxy: Recipe = {
|
||||
cost_per_1m_output_usd: undefined,
|
||||
price_last_verified: '2026-06-14',
|
||||
},
|
||||
// LiteLLM normalizes Cohere / Voyage / Jina / etc. rerank backends to the
|
||||
// same wire shape gbrain's gateway.rerank() already speaks (the
|
||||
// ZeroEntropy/llama.cpp contract):
|
||||
// { model, query, documents, top_n } → { results: [{ index, relevance_score }] }
|
||||
// So any rerank model the user registers in their LiteLLM config is
|
||||
// reachable via `gbrain config set search.reranker.model litellm:<model>`
|
||||
// with no request/response adapter — same as embeddings ride the proxy.
|
||||
reranker: {
|
||||
models: [], // user-provided; whatever rerank models the proxy serves
|
||||
// No canonical default — the proxy defines its own model ids. The user
|
||||
// sets search.reranker.model explicitly (mirrors the embedding
|
||||
// touchpoint's user_provided_models contract).
|
||||
default_model: '',
|
||||
// The proxied backend bills (Cohere/Voyage/…); pricing-unknown is the
|
||||
// honest state — same stance as this recipe's embedding/chat
|
||||
// touchpoints and budget-tracker's deliberate litellm exclusion from
|
||||
// the free-provider sets.
|
||||
cost_per_1m_tokens_usd: undefined,
|
||||
price_last_verified: '2026-06-27',
|
||||
max_payload_bytes: 5_000_000,
|
||||
// LEAF path only (matches llama-server-reranker's convention). LiteLLM
|
||||
// serves both `/rerank` and `/v1/rerank`, and LITELLM_BASE_URL may be
|
||||
// set with or without the `/v1` suffix (the setup_hint allows both), so
|
||||
// the leaf form yields a valid route either way:
|
||||
// http://localhost:4000 + /rerank → /rerank ✓
|
||||
// http://localhost:4000/v1 + /rerank → /v1/rerank ✓
|
||||
// Pinning '/v1/rerank' here would double to /v1/v1/rerank → 404 on
|
||||
// /v1-suffixed bases.
|
||||
path: '/rerank',
|
||||
},
|
||||
},
|
||||
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>.',
|
||||
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>. For rerank: register a rerank model in LiteLLM and set search.reranker.model litellm:<model-name>.',
|
||||
};
|
||||
|
||||
@@ -36,7 +36,7 @@ import {
|
||||
} from './embedding-context.ts';
|
||||
import { loadSearchModeConfig, resolveSearchMode } from './search/mode.ts';
|
||||
import { normalizeAliasList } from './search/alias-normalize.ts';
|
||||
import { isUndefinedTableError, warnOncePerProcess, validateSlug } from './utils.ts';
|
||||
import { isUndefinedTableError, warnOncePerProcess } from './utils.ts';
|
||||
import { computeCorpusGeneration } from './contextual-retrieval-service.ts';
|
||||
import { runGuardrails } from './guardrails.ts';
|
||||
|
||||
@@ -295,12 +295,6 @@ export async function importFromContent(
|
||||
remote?: boolean;
|
||||
} = {},
|
||||
): Promise<ImportResult> {
|
||||
// Normalize BEFORE any tx write: putPage lowercases via validateSlug but
|
||||
// upsertChunks used to query by the caller's raw slug, so a mixed-case slug
|
||||
// created the page row then failed the chunk upsert with "Page not found",
|
||||
// rolling back the whole import (#430).
|
||||
slug = validateSlug(slug);
|
||||
|
||||
// v0.18.0+ multi-source: when caller is syncing under a non-default source,
|
||||
// every per-page tx call must carry `sourceId` so writes target the right
|
||||
// (source_id, slug) row. Pre-fix, putPage relied on the schema DEFAULT and
|
||||
|
||||
@@ -2235,9 +2235,6 @@ export class PGLiteEngine implements BrainEngine {
|
||||
}
|
||||
|
||||
private async _upsertChunksOnce(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string }): Promise<void> {
|
||||
// Normalize the same way putPage does — pages.slug is stored lowercased,
|
||||
// so a raw mixed-case slug here would miss the row it just wrote (#430).
|
||||
slug = validateSlug(slug);
|
||||
const sourceId = opts?.sourceId ?? 'default';
|
||||
|
||||
// Source-scope the page-id lookup so duplicate slugs in different sources
|
||||
|
||||
@@ -2385,9 +2385,6 @@ export class PostgresEngine implements BrainEngine {
|
||||
}
|
||||
|
||||
private async _upsertChunksOnce(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string }): Promise<void> {
|
||||
// Normalize the same way putPage does — pages.slug is stored lowercased,
|
||||
// so a raw mixed-case slug here would miss the row it just wrote (#430).
|
||||
slug = validateSlug(slug);
|
||||
const sql = this.sql;
|
||||
const sourceId = opts?.sourceId ?? 'default';
|
||||
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
/**
|
||||
* litellm-proxy reranker touchpoint smoke.
|
||||
*
|
||||
* Sibling of recipe-llama-server-reranker.test.ts. Pins the reranker
|
||||
* touchpoint on the LiteLLM proxy recipe so:
|
||||
* - the touchpoint exists with the LEAF '/rerank' path (LiteLLM serves both
|
||||
* /rerank and /v1/rerank, so the leaf form is valid whether or not the
|
||||
* user's LITELLM_BASE_URL carries the /v1 suffix the setup_hint allows)
|
||||
* - a /v1-suffixed base URL does NOT produce /v1/v1/rerank (the original
|
||||
* community PR pinned '/v1/rerank' which 404s on /v1-suffixed bases)
|
||||
* - models: [] (user-provided; proxy defines the model ids)
|
||||
* - pricing stays undefined (proxy can front a paid provider — same honest
|
||||
* pricing-unknown stance as the embedding/chat touchpoints)
|
||||
*
|
||||
* The gateway.rerank() URL tests drive the real URL builder via the stubbed
|
||||
* transport (same seam as test/ai/rerank.test.ts).
|
||||
*/
|
||||
|
||||
import { describe, expect, test, afterEach } from 'bun:test';
|
||||
import { getRecipe } from '../../src/core/ai/recipes/index.ts';
|
||||
import {
|
||||
configureGateway,
|
||||
resetGateway,
|
||||
rerank,
|
||||
__setRerankTransportForTests,
|
||||
} from '../../src/core/ai/gateway.ts';
|
||||
|
||||
afterEach(() => {
|
||||
__setRerankTransportForTests(null);
|
||||
resetGateway();
|
||||
});
|
||||
|
||||
describe('recipe: litellm reranker touchpoint', () => {
|
||||
test('declares reranker touchpoint with leaf /rerank path', () => {
|
||||
const r = getRecipe('litellm')!;
|
||||
const tp = r.touchpoints.reranker;
|
||||
expect(tp).toBeDefined();
|
||||
expect(tp!.path).toBe('/rerank');
|
||||
expect(tp!.max_payload_bytes).toBe(5_000_000);
|
||||
});
|
||||
|
||||
test('reranker touchpoint uses empty models[] for user-provided model ids', () => {
|
||||
const r = getRecipe('litellm')!;
|
||||
expect(r.touchpoints.reranker!.models).toEqual([]);
|
||||
});
|
||||
|
||||
test('pricing stays undefined — proxy can front a paid provider', () => {
|
||||
const r = getRecipe('litellm')!;
|
||||
expect(r.touchpoints.reranker!.cost_per_1m_tokens_usd).toBeUndefined();
|
||||
});
|
||||
|
||||
test('setup_hint keeps the /v1-suffix guidance AND mentions rerank', () => {
|
||||
const r = getRecipe('litellm')!;
|
||||
expect(r.setup_hint).toMatch(/\/v1 suffix/);
|
||||
expect(r.setup_hint).toMatch(/search\.reranker\.model litellm:/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('gateway.rerank() URL via litellm recipe', () => {
|
||||
async function capturedRerankUrl(baseUrl?: string): Promise<string> {
|
||||
configureGateway({
|
||||
reranker_model: 'litellm:my-reranker',
|
||||
env: {},
|
||||
...(baseUrl ? { base_urls: { litellm: baseUrl } } : {}),
|
||||
});
|
||||
let capturedUrl = '';
|
||||
__setRerankTransportForTests(async (url) => {
|
||||
capturedUrl = url;
|
||||
return new Response(
|
||||
JSON.stringify({ results: [{ index: 0, relevance_score: 0.9 }] }),
|
||||
{ status: 200, headers: { 'content-type': 'application/json' } },
|
||||
);
|
||||
});
|
||||
await rerank({ query: 'q', documents: ['d'] });
|
||||
return capturedUrl;
|
||||
}
|
||||
|
||||
test('default base (no /v1 suffix) → /rerank', async () => {
|
||||
const url = await capturedRerankUrl();
|
||||
expect(url).toBe('http://localhost:4000/rerank');
|
||||
});
|
||||
|
||||
test('/v1-suffixed base → /v1/rerank, NOT /v1/v1/rerank', async () => {
|
||||
const url = await capturedRerankUrl('http://localhost:4000/v1');
|
||||
expect(url).toBe('http://localhost:4000/v1/rerank');
|
||||
expect(url).not.toContain('/v1/v1/');
|
||||
});
|
||||
});
|
||||
@@ -6,7 +6,6 @@
|
||||
|
||||
import { describe, test, expect, beforeAll, afterAll, beforeEach } from 'bun:test';
|
||||
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
import { importFromContent } from '../src/core/import-file.ts';
|
||||
import type { BrainEngine } from '../src/core/engine.ts';
|
||||
import type { PageInput, ChunkInput } from '../src/core/types.ts';
|
||||
|
||||
@@ -182,24 +181,6 @@ describe('PGLiteEngine: Pages', () => {
|
||||
const page = await engine.putPage('Test/UPPER', testPage);
|
||||
expect(page.slug).toBe('test/upper');
|
||||
});
|
||||
|
||||
test('importFromContent normalizes mixed-case slugs before all tx writes (#430)', async () => {
|
||||
const result = await importFromContent(
|
||||
engine,
|
||||
'TestNamespace/Page-Name',
|
||||
'---\ntype: note\ntitle: Mixed Case\n---\n\nbody text',
|
||||
{ noEmbed: true },
|
||||
);
|
||||
expect(result.status).toBe('imported');
|
||||
expect(result.slug).toBe('testnamespace/page-name');
|
||||
|
||||
const page = await engine.getPage('testnamespace/page-name');
|
||||
expect(page).not.toBeNull();
|
||||
expect(page!.title).toBe('Mixed Case');
|
||||
|
||||
const chunks = await engine.getChunks('testnamespace/page-name');
|
||||
expect(chunks.length).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
@@ -383,17 +364,6 @@ describe('PGLiteEngine: Chunks', () => {
|
||||
expect(chunks[1].chunk_text).toBe('Chunk one');
|
||||
});
|
||||
|
||||
test('upsertChunks normalizes mixed-case slugs like putPage (#430)', async () => {
|
||||
await engine.putPage('Test/ChunkCase', testPage);
|
||||
await engine.upsertChunks('Test/ChunkCase', [
|
||||
{ chunk_index: 0, chunk_text: 'Mixed-case chunk', chunk_source: 'compiled_truth' },
|
||||
]);
|
||||
|
||||
const chunks = await engine.getChunks('test/chunkcase');
|
||||
expect(chunks.length).toBe(1);
|
||||
expect(chunks[0].chunk_text).toBe('Mixed-case chunk');
|
||||
});
|
||||
|
||||
test('upsertChunks removes orphan chunks', async () => {
|
||||
await engine.putPage('test/orphan', testPage);
|
||||
await engine.upsertChunks('test/orphan', [
|
||||
|
||||
Reference in New Issue
Block a user