Compare commits

..
Author SHA1 Message Date
d9834a7a15 feat(ai): add reranker touchpoint to LiteLLM proxy recipe (takeover of #2455)
LiteLLM normalizes Cohere/Voyage/Jina rerank backends to the wire shape
gateway.rerank() already speaks, so a reranker touchpoint on the litellm
recipe makes any proxied rerank model reachable via
`search.reranker.model litellm:<model>` with no adapter.

Repairs from the original PR:
- path is the LEAF '/rerank' (not '/v1/rerank'): LiteLLM serves both
  /rerank and /v1/rerank, and the recipe's setup_hint allows
  LITELLM_BASE_URL with or without the /v1 suffix — pinning '/v1/rerank'
  doubled to /v1/v1/rerank (404) on /v1-suffixed bases.
- setup_hint appends the rerank guidance to master's current line instead
  of replacing it with a stale pre-/v1-suffix version.
- cost_per_1m_tokens_usd stays undefined (pricing-unknown), matching the
  recipe's embedding/chat touchpoints and budget-tracker's deliberate
  litellm exclusion from the free-provider sets (a proxy can front a paid
  provider; the touchpoint field isn't consumed by rerank pricing anyway).

Test drives gateway.rerank()'s real URL builder via the stubbed transport
for both base-URL forms; the /v1-suffixed case fails with the original
PR's path.

Co-authored-by: ozp <ozp@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:28:27 -07:00
5 changed files with 123 additions and 163 deletions
+1 -4
View File
@@ -935,10 +935,7 @@ export function formatResult(opName: string, result: unknown): string {
lines.push(`Link coverage (entities): ${(h.link_coverage * 100).toFixed(1)}%`);
}
if (h.timeline_coverage !== undefined) {
lines.push(`Timeline coverage (entity pages): ${(h.timeline_coverage * 100).toFixed(1)}%`);
}
if (h.timeline_coverage_score !== undefined) {
lines.push(`Timeline density (all pages): ${h.timeline_coverage_score}/15 (whole-brain brain-score component)`);
lines.push(`Timeline coverage (entities): ${(h.timeline_coverage * 100).toFixed(1)}%`);
}
if (Array.isArray(h.most_connected) && h.most_connected.length > 0) {
lines.push('Most connected entities:');
+3 -3
View File
@@ -5868,12 +5868,12 @@ export async function buildChecks(
message: `Only code/test fixture entity pages found (${entityCount}); graph_coverage not applicable`,
});
} else if (linkCoverage >= 0.5 && timelineCoverage >= 0.5) {
checks.push({ name: 'graph_coverage', status: 'ok', message: `Entity link coverage ${linkPct}%, entity timeline coverage ${timelinePct}%` });
checks.push({ name: 'graph_coverage', status: 'ok', message: `Entity link coverage ${linkPct}%, timeline ${timelinePct}%` });
} else {
checks.push({
name: 'graph_coverage',
status: 'warn',
message: `Entity link coverage ${linkPct}%, entity timeline coverage ${timelinePct}% (${eligibleEntityCount} entity pages). Run: gbrain extract all`,
message: `Entity link coverage ${linkPct}%, timeline ${timelinePct}% (${eligibleEntityCount} entity pages). Run: gbrain extract all`,
});
}
@@ -5885,7 +5885,7 @@ export async function buildChecks(
const parts = [
`embed ${health.embed_coverage_score}/35`,
`links ${health.link_density_score}/25`,
`timeline density (all pages) ${health.timeline_coverage_score}/15`,
`timeline ${health.timeline_coverage_score}/15`,
`orphans ${health.no_orphans_score}/15`,
`dead-links ${health.no_dead_links_score}/10`,
];
+31 -1
View File
@@ -56,6 +56,36 @@ export const litellmProxy: Recipe = {
cost_per_1m_output_usd: undefined,
price_last_verified: '2026-06-14',
},
// LiteLLM normalizes Cohere / Voyage / Jina / etc. rerank backends to the
// same wire shape gbrain's gateway.rerank() already speaks (the
// ZeroEntropy/llama.cpp contract):
// { model, query, documents, top_n } → { results: [{ index, relevance_score }] }
// So any rerank model the user registers in their LiteLLM config is
// reachable via `gbrain config set search.reranker.model litellm:<model>`
// with no request/response adapter — same as embeddings ride the proxy.
reranker: {
models: [], // user-provided; whatever rerank models the proxy serves
// No canonical default — the proxy defines its own model ids. The user
// sets search.reranker.model explicitly (mirrors the embedding
// touchpoint's user_provided_models contract).
default_model: '',
// The proxied backend bills (Cohere/Voyage/…); pricing-unknown is the
// honest state — same stance as this recipe's embedding/chat
// touchpoints and budget-tracker's deliberate litellm exclusion from
// the free-provider sets.
cost_per_1m_tokens_usd: undefined,
price_last_verified: '2026-06-27',
max_payload_bytes: 5_000_000,
// LEAF path only (matches llama-server-reranker's convention). LiteLLM
// serves both `/rerank` and `/v1/rerank`, and LITELLM_BASE_URL may be
// set with or without the `/v1` suffix (the setup_hint allows both), so
// the leaf form yields a valid route either way:
// http://localhost:4000 + /rerank → /rerank ✓
// http://localhost:4000/v1 + /rerank → /v1/rerank ✓
// Pinning '/v1/rerank' here would double to /v1/v1/rerank → 404 on
// /v1-suffixed bases.
path: '/rerank',
},
},
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>.',
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>. For rerank: register a rerank model in LiteLLM and set search.reranker.model litellm:<model-name>.',
};
+88
View File
@@ -0,0 +1,88 @@
/**
* litellm-proxy reranker touchpoint smoke.
*
* Sibling of recipe-llama-server-reranker.test.ts. Pins the reranker
* touchpoint on the LiteLLM proxy recipe so:
* - the touchpoint exists with the LEAF '/rerank' path (LiteLLM serves both
* /rerank and /v1/rerank, so the leaf form is valid whether or not the
* user's LITELLM_BASE_URL carries the /v1 suffix the setup_hint allows)
* - a /v1-suffixed base URL does NOT produce /v1/v1/rerank (the original
* community PR pinned '/v1/rerank' which 404s on /v1-suffixed bases)
* - models: [] (user-provided; proxy defines the model ids)
* - pricing stays undefined (proxy can front a paid provider — same honest
* pricing-unknown stance as the embedding/chat touchpoints)
*
* The gateway.rerank() URL tests drive the real URL builder via the stubbed
* transport (same seam as test/ai/rerank.test.ts).
*/
import { describe, expect, test, afterEach } from 'bun:test';
import { getRecipe } from '../../src/core/ai/recipes/index.ts';
import {
configureGateway,
resetGateway,
rerank,
__setRerankTransportForTests,
} from '../../src/core/ai/gateway.ts';
afterEach(() => {
__setRerankTransportForTests(null);
resetGateway();
});
describe('recipe: litellm reranker touchpoint', () => {
test('declares reranker touchpoint with leaf /rerank path', () => {
const r = getRecipe('litellm')!;
const tp = r.touchpoints.reranker;
expect(tp).toBeDefined();
expect(tp!.path).toBe('/rerank');
expect(tp!.max_payload_bytes).toBe(5_000_000);
});
test('reranker touchpoint uses empty models[] for user-provided model ids', () => {
const r = getRecipe('litellm')!;
expect(r.touchpoints.reranker!.models).toEqual([]);
});
test('pricing stays undefined — proxy can front a paid provider', () => {
const r = getRecipe('litellm')!;
expect(r.touchpoints.reranker!.cost_per_1m_tokens_usd).toBeUndefined();
});
test('setup_hint keeps the /v1-suffix guidance AND mentions rerank', () => {
const r = getRecipe('litellm')!;
expect(r.setup_hint).toMatch(/\/v1 suffix/);
expect(r.setup_hint).toMatch(/search\.reranker\.model litellm:/);
});
});
describe('gateway.rerank() URL via litellm recipe', () => {
async function capturedRerankUrl(baseUrl?: string): Promise<string> {
configureGateway({
reranker_model: 'litellm:my-reranker',
env: {},
...(baseUrl ? { base_urls: { litellm: baseUrl } } : {}),
});
let capturedUrl = '';
__setRerankTransportForTests(async (url) => {
capturedUrl = url;
return new Response(
JSON.stringify({ results: [{ index: 0, relevance_score: 0.9 }] }),
{ status: 200, headers: { 'content-type': 'application/json' } },
);
});
await rerank({ query: 'q', documents: ['d'] });
return capturedUrl;
}
test('default base (no /v1 suffix) → /rerank', async () => {
const url = await capturedRerankUrl();
expect(url).toBe('http://localhost:4000/rerank');
});
test('/v1-suffixed base → /v1/rerank, NOT /v1/v1/rerank', async () => {
const url = await capturedRerankUrl('http://localhost:4000/v1');
expect(url).toBe('http://localhost:4000/v1/rerank');
expect(url).not.toContain('/v1/v1/');
});
});
@@ -1,155 +0,0 @@
/**
* Issue #2298 — timeline metric presentation contract.
*
* Authoritative upstream semantics (src/core/types.ts):
* - Metric A `timeline_coverage` (entity-scoped, fraction 01):
* eligible entity pages WITH a timeline entry / eligible entity pages
* -> surfaced by `graph_coverage` check AND `get_health` CLI entity line.
* - Metric B `timeline_coverage_score` (whole-brain, 015 brain-score component):
* all pages WITH a timeline entry / all pages
* -> surfaced by `brain_score` component breakdown AND (separately) CLI.
*
* The two have DIFFERENT numerators/denominators. This PR labels each
* explicitly and keeps BOTH the entity CLI line and the whole-brain line.
*
* Tests (no private EriadorMu data, no production/home DB, no network):
* - numeric denominator assertions (Metric A = 50%, Metric B = 4/15)
* - doctor rendered-message assertions (exact labels, no ambiguous old label)
* - CLI rendered-output assertions (exact lines, guard matrix)
* - red/green: same assertions FAIL on origin/master, PASS on this branch
*
* Scoring formula UNCHANGED. Canonical PGLite fixture via resetPgliteState.
*/
import { describe, expect, test, beforeAll, afterAll, beforeEach } from 'bun:test';
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
import { sqlQueryForEngine } from '../src/core/sql-query.ts';
import { resetPgliteState } from './helpers/reset-pglite.ts';
import { buildChecks } from '../src/commands/doctor.ts';
import { formatResult } from '../src/cli.ts';
let engine: PGLiteEngine;
async function seedFourPages(eng: PGLiteEngine): Promise<void> {
const sql = sqlQueryForEngine(eng);
// 2 eligible entity pages, 2 technical/non-entity pages.
// Only ONE entity page has a timeline entry; only ONE total page does.
await sql`
INSERT INTO pages (slug, source_id, type, title, compiled_truth, frontmatter, content_hash, created_at, updated_at)
VALUES
('acme-example', 'default', 'company', 'Acme', '', '{}', 'h1', now(), now()),
('alice-example', 'default', 'person', 'Alice', '', '{}', 'h2', now(), now()),
('technical-a', 'default', 'note', 'Tech A', '', '{}', 'h3', now(), now()),
('technical-b', 'default', 'note', 'Tech B', '', '{}', 'h4', now(), now())
`;
const companyId = (await sql`SELECT id FROM pages WHERE slug='acme-example'`)[0].id as number;
await sql`INSERT INTO timeline_entries (page_id, date, source, summary, detail)
VALUES (${companyId}, CURRENT_DATE, 'test', 'milestone', '{}')`;
}
beforeAll(async () => {
engine = new PGLiteEngine();
await engine.connect({});
await engine.initSchema();
});
afterAll(async () => {
await engine.disconnect();
});
beforeEach(async () => {
await resetPgliteState(engine);
});
describe('issue #2298 — numeric denominator semantics', () => {
test('entity timeline coverage = 1/2 = 50% (2 eligible entities, 1 with timeline)', async () => {
await seedFourPages(engine);
const health = await engine.getHealth();
expect(health.timeline_coverage).toBeDefined();
expect(Math.round((health.timeline_coverage ?? 0) * 100)).toBe(50);
});
test('whole-brain timeline density = 1/4 -> score 4/15 (4 total pages, 1 with timeline)', async () => {
await seedFourPages(engine);
const health = await engine.getHealth();
expect(health.timeline_coverage_score).toBeDefined();
expect(health.timeline_coverage_score).toBe(4);
});
test('the two metrics use independent denominators', async () => {
await seedFourPages(engine);
const health = await engine.getHealth();
expect(Math.round((health.timeline_coverage ?? 0) * 100)).toBe(50);
expect(health.timeline_coverage_score ?? 0).toBe(4);
// 50% (entity, /2) != 26.7% (whole-brain, /4). Provably distinct.
expect(Math.round(((health.timeline_coverage_score ?? 0) / 15) * 100)).not.toBe(50);
});
});
describe('issue #2298 — doctor rendered-message contract', () => {
test('graph_coverage renders entity-scoped label with 50%', async () => {
await seedFourPages(engine);
const checks = await buildChecks(engine, [], null);
const graph = checks.find((c) => c.name === 'graph_coverage');
expect(graph, 'graph_coverage check must be present').toBeDefined();
expect(graph!.message).toContain('entity timeline coverage 50%');
// ambiguous old label must NOT be present
expect(graph!.message).not.toMatch(/timeline 50%/);
expect(graph!.message).not.toMatch(/timeline \(entity, brain score\)/);
});
test('brain_score renders whole-brain density label 4/15', async () => {
await seedFourPages(engine);
const checks = await buildChecks(engine, [], null);
const brain = checks.find((c) => c.name === 'brain_score');
expect(brain, 'brain_score check must be present').toBeDefined();
expect(brain!.message).toContain('timeline density (all pages) 4/15');
// wrong labels must NOT be present
expect(brain!.message).not.toMatch(/timeline 4\/15/);
expect(brain!.message).not.toMatch(/timeline \(entity, brain score\)/);
// brain-score component must NOT carry the word "entity" (it is whole-brain)
const timelinePart = brain!.message.split('timeline density (all pages) 4/15')[0] + 'timeline density (all pages) 4/15';
expect(timelinePart).not.toMatch(/entity/);
});
});
describe('issue #2298 — CLI get_health rendered-output contract', () => {
function fakeHealth(overrides: Record<string, unknown>): any {
return {
embed_coverage: 1, missing_embeddings: 0, stale_pages: 0, orphan_pages: 0,
link_coverage: 1, timeline_coverage: 0.5, timeline_coverage_score: 4,
most_connected: [], ...overrides,
};
}
test('both entity and whole-brain lines render, no undefined/15', () => {
const out = formatResult('get_health', fakeHealth({}));
expect(out).toContain('Timeline coverage (entity pages): 50.0%');
expect(out).toContain('Timeline density (all pages): 4/15');
expect(out).not.toContain('undefined/15');
expect(out).not.toContain('Timeline coverage (entities)');
expect(out).not.toMatch(/timeline \(entity, brain score\)/);
expect(out).not.toMatch(/bare "timeline 4\/15"/);
});
test('guard matrix: entity present, whole-brain absent -> only entity line', () => {
const out = formatResult('get_health', fakeHealth({ timeline_coverage_score: undefined }));
expect(out).toContain('Timeline coverage (entity pages): 50.0%');
expect(out).not.toContain('Timeline density (all pages)');
expect(out).not.toContain('undefined/15');
});
test('guard matrix: whole-brain present, entity absent -> only whole-brain line', () => {
const out = formatResult('get_health', fakeHealth({ timeline_coverage: undefined }));
expect(out).toContain('Timeline density (all pages): 4/15');
expect(out).not.toContain('Timeline coverage (entity pages)');
expect(out).not.toContain('undefined/15');
});
test('guard matrix: both absent -> neither timeline line, never undefined/15', () => {
const out = formatResult('get_health', fakeHealth({ timeline_coverage: undefined, timeline_coverage_score: undefined }));
expect(out).not.toContain('Timeline coverage (entity pages)');
expect(out).not.toContain('Timeline density (all pages)');
expect(out).not.toContain('undefined/15');
});
});