mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-16 09:52:22 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fffa779172 |
@@ -1551,6 +1551,24 @@ export async function checkRerankerHealth(engine: BrainEngine): Promise<Check> {
|
||||
};
|
||||
}
|
||||
|
||||
// Historical #2059 rows were logged as `unknown` before missing reranker
|
||||
// auth was classified at the gateway. Surface repeated unknowns instead of
|
||||
// reporting "ok" while every rerank fails open.
|
||||
const unknownFails = failures.filter((f) => f.reason === 'unknown');
|
||||
if (unknownFails.length >= 3) {
|
||||
const setupHint = unknownFails.some((f) => {
|
||||
const summary = String(f.error_summary ?? '');
|
||||
return summary.includes('ZEROENTROPY_API_KEY') || summary.toLowerCase().includes('api key');
|
||||
})
|
||||
? ' Fix: verify ZEROENTROPY_API_KEY and run `gbrain models doctor`.'
|
||||
: '';
|
||||
return {
|
||||
name: 'reranker_health',
|
||||
status: 'warn',
|
||||
message: `${unknownFails.length} unknown reranker failure(s) in last 7 days.${setupHint}`,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
name: 'reranker_health',
|
||||
status: 'ok',
|
||||
|
||||
@@ -3656,7 +3656,15 @@ export async function rerank(input: RerankInput): Promise<RerankResult[]> {
|
||||
// whose request/response shape differs from ZE/llama.cpp (e.g. Voyage with
|
||||
// `top_k` / `data[]`) needs separate adapter hooks in a follow-up plan.
|
||||
const url = `${compat.baseURL.replace(/\/$/, '')}${tp.path ?? '/models/rerank'}`;
|
||||
const auth = applyResolveAuth(recipe, cfg, 'reranker');
|
||||
let auth: { apiKey?: string; headers?: Record<string, string> };
|
||||
try {
|
||||
auth = applyResolveAuth(recipe, cfg, 'reranker');
|
||||
} catch (err) {
|
||||
if (err instanceof AIConfigError) {
|
||||
throw new RerankError(err.message, 'auth');
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
// applyResolveAuth returns { apiKey } for Bearer-style auth (SDK's native
|
||||
// path) or { headers } for custom-header providers (Azure). v0.37.6.0:
|
||||
// recipes can ALSO declare default_headers (attribution etc.) which flow
|
||||
|
||||
@@ -56,36 +56,6 @@ export const litellmProxy: Recipe = {
|
||||
cost_per_1m_output_usd: undefined,
|
||||
price_last_verified: '2026-06-14',
|
||||
},
|
||||
// LiteLLM normalizes Cohere / Voyage / Jina / etc. rerank backends to the
|
||||
// same wire shape gbrain's gateway.rerank() already speaks (the
|
||||
// ZeroEntropy/llama.cpp contract):
|
||||
// { model, query, documents, top_n } → { results: [{ index, relevance_score }] }
|
||||
// So any rerank model the user registers in their LiteLLM config is
|
||||
// reachable via `gbrain config set search.reranker.model litellm:<model>`
|
||||
// with no request/response adapter — same as embeddings ride the proxy.
|
||||
reranker: {
|
||||
models: [], // user-provided; whatever rerank models the proxy serves
|
||||
// No canonical default — the proxy defines its own model ids. The user
|
||||
// sets search.reranker.model explicitly (mirrors the embedding
|
||||
// touchpoint's user_provided_models contract).
|
||||
default_model: '',
|
||||
// The proxied backend bills (Cohere/Voyage/…); pricing-unknown is the
|
||||
// honest state — same stance as this recipe's embedding/chat
|
||||
// touchpoints and budget-tracker's deliberate litellm exclusion from
|
||||
// the free-provider sets.
|
||||
cost_per_1m_tokens_usd: undefined,
|
||||
price_last_verified: '2026-06-27',
|
||||
max_payload_bytes: 5_000_000,
|
||||
// LEAF path only (matches llama-server-reranker's convention). LiteLLM
|
||||
// serves both `/rerank` and `/v1/rerank`, and LITELLM_BASE_URL may be
|
||||
// set with or without the `/v1` suffix (the setup_hint allows both), so
|
||||
// the leaf form yields a valid route either way:
|
||||
// http://localhost:4000 + /rerank → /rerank ✓
|
||||
// http://localhost:4000/v1 + /rerank → /v1/rerank ✓
|
||||
// Pinning '/v1/rerank' here would double to /v1/v1/rerank → 404 on
|
||||
// /v1-suffixed bases.
|
||||
path: '/rerank',
|
||||
},
|
||||
},
|
||||
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>. For rerank: register a rerank model in LiteLLM and set search.reranker.model litellm:<model-name>.',
|
||||
setup_hint: 'Run LiteLLM (https://docs.litellm.ai) in front of any provider; set LITELLM_BASE_URL (include the /v1 suffix if your proxy serves the OpenAI route there, e.g. http://localhost:4000/v1) + pass --embedding-model litellm:<model> and --embedding-dimensions <N>.',
|
||||
};
|
||||
|
||||
@@ -1,88 +0,0 @@
|
||||
/**
|
||||
* litellm-proxy reranker touchpoint smoke.
|
||||
*
|
||||
* Sibling of recipe-llama-server-reranker.test.ts. Pins the reranker
|
||||
* touchpoint on the LiteLLM proxy recipe so:
|
||||
* - the touchpoint exists with the LEAF '/rerank' path (LiteLLM serves both
|
||||
* /rerank and /v1/rerank, so the leaf form is valid whether or not the
|
||||
* user's LITELLM_BASE_URL carries the /v1 suffix the setup_hint allows)
|
||||
* - a /v1-suffixed base URL does NOT produce /v1/v1/rerank (the original
|
||||
* community PR pinned '/v1/rerank' which 404s on /v1-suffixed bases)
|
||||
* - models: [] (user-provided; proxy defines the model ids)
|
||||
* - pricing stays undefined (proxy can front a paid provider — same honest
|
||||
* pricing-unknown stance as the embedding/chat touchpoints)
|
||||
*
|
||||
* The gateway.rerank() URL tests drive the real URL builder via the stubbed
|
||||
* transport (same seam as test/ai/rerank.test.ts).
|
||||
*/
|
||||
|
||||
import { describe, expect, test, afterEach } from 'bun:test';
|
||||
import { getRecipe } from '../../src/core/ai/recipes/index.ts';
|
||||
import {
|
||||
configureGateway,
|
||||
resetGateway,
|
||||
rerank,
|
||||
__setRerankTransportForTests,
|
||||
} from '../../src/core/ai/gateway.ts';
|
||||
|
||||
afterEach(() => {
|
||||
__setRerankTransportForTests(null);
|
||||
resetGateway();
|
||||
});
|
||||
|
||||
describe('recipe: litellm reranker touchpoint', () => {
|
||||
test('declares reranker touchpoint with leaf /rerank path', () => {
|
||||
const r = getRecipe('litellm')!;
|
||||
const tp = r.touchpoints.reranker;
|
||||
expect(tp).toBeDefined();
|
||||
expect(tp!.path).toBe('/rerank');
|
||||
expect(tp!.max_payload_bytes).toBe(5_000_000);
|
||||
});
|
||||
|
||||
test('reranker touchpoint uses empty models[] for user-provided model ids', () => {
|
||||
const r = getRecipe('litellm')!;
|
||||
expect(r.touchpoints.reranker!.models).toEqual([]);
|
||||
});
|
||||
|
||||
test('pricing stays undefined — proxy can front a paid provider', () => {
|
||||
const r = getRecipe('litellm')!;
|
||||
expect(r.touchpoints.reranker!.cost_per_1m_tokens_usd).toBeUndefined();
|
||||
});
|
||||
|
||||
test('setup_hint keeps the /v1-suffix guidance AND mentions rerank', () => {
|
||||
const r = getRecipe('litellm')!;
|
||||
expect(r.setup_hint).toMatch(/\/v1 suffix/);
|
||||
expect(r.setup_hint).toMatch(/search\.reranker\.model litellm:/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('gateway.rerank() URL via litellm recipe', () => {
|
||||
async function capturedRerankUrl(baseUrl?: string): Promise<string> {
|
||||
configureGateway({
|
||||
reranker_model: 'litellm:my-reranker',
|
||||
env: {},
|
||||
...(baseUrl ? { base_urls: { litellm: baseUrl } } : {}),
|
||||
});
|
||||
let capturedUrl = '';
|
||||
__setRerankTransportForTests(async (url) => {
|
||||
capturedUrl = url;
|
||||
return new Response(
|
||||
JSON.stringify({ results: [{ index: 0, relevance_score: 0.9 }] }),
|
||||
{ status: 200, headers: { 'content-type': 'application/json' } },
|
||||
);
|
||||
});
|
||||
await rerank({ query: 'q', documents: ['d'] });
|
||||
return capturedUrl;
|
||||
}
|
||||
|
||||
test('default base (no /v1 suffix) → /rerank', async () => {
|
||||
const url = await capturedRerankUrl();
|
||||
expect(url).toBe('http://localhost:4000/rerank');
|
||||
});
|
||||
|
||||
test('/v1-suffixed base → /v1/rerank, NOT /v1/v1/rerank', async () => {
|
||||
const url = await capturedRerankUrl('http://localhost:4000/v1');
|
||||
expect(url).toBe('http://localhost:4000/v1/rerank');
|
||||
expect(url).not.toContain('/v1/v1/');
|
||||
});
|
||||
});
|
||||
@@ -154,6 +154,28 @@ describe('gateway.rerank() — happy path', () => {
|
||||
describe('gateway.rerank() — error classification', () => {
|
||||
beforeEach(() => configureZE());
|
||||
|
||||
test('missing required reranker API key → RerankError(auth) before HTTP call', async () => {
|
||||
configureGateway({
|
||||
reranker_model: 'zeroentropyai:zerank-2',
|
||||
env: {},
|
||||
});
|
||||
let called = false;
|
||||
__setRerankTransportForTests(async () => {
|
||||
called = true;
|
||||
return mockResp({ results: [{ index: 0, relevance_score: 0.5 }] });
|
||||
});
|
||||
|
||||
try {
|
||||
await rerank({ query: 'q', documents: ['d'] });
|
||||
throw new Error('should have thrown');
|
||||
} catch (err) {
|
||||
expect(err).toBeInstanceOf(RerankError);
|
||||
expect((err as RerankError).reason).toBe('auth');
|
||||
expect((err as Error).message).toContain('ZEROENTROPY_API_KEY');
|
||||
expect(called).toBe(false);
|
||||
}
|
||||
});
|
||||
|
||||
test('401 → auth', async () => {
|
||||
__setRerankTransportForTests(async () => new Response('Unauthorized', { status: 401 }));
|
||||
try {
|
||||
|
||||
@@ -2,6 +2,11 @@ import { describe, test, expect, beforeAll, afterAll, beforeEach } from 'bun:tes
|
||||
import { mkdirSync, rmSync, writeFileSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
import { tmpdir } from 'os';
|
||||
import * as fs from 'node:fs';
|
||||
import * as os from 'node:os';
|
||||
import * as path from 'node:path';
|
||||
import { withEnv } from './helpers/with-env.ts';
|
||||
import { logRerankFailure } from '../src/core/rerank-audit.ts';
|
||||
|
||||
describe('doctor command', () => {
|
||||
test('doctor module exports runDoctor', async () => {
|
||||
@@ -47,6 +52,34 @@ describe('doctor command', () => {
|
||||
expect(check.issues![0].action).toContain('trigger');
|
||||
});
|
||||
|
||||
test('reranker_health warns on repeated unknown rerank failures', async () => {
|
||||
const { checkRerankerHealth } = await import('../src/commands/doctor.ts');
|
||||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gbrain-rerank-doctor-'));
|
||||
try {
|
||||
await withEnv({ GBRAIN_AUDIT_DIR: tmpDir }, async () => {
|
||||
for (let i = 0; i < 3; i++) {
|
||||
logRerankFailure({
|
||||
model: 'zeroentropyai:zerank-2',
|
||||
reason: 'unknown',
|
||||
query_hash: `unknown${i}`,
|
||||
doc_count: 30,
|
||||
error_summary: 'ZeroEntropy reranker requires ZEROENTROPY_API_KEY.',
|
||||
});
|
||||
}
|
||||
const check = await checkRerankerHealth({
|
||||
async getConfig(key: string): Promise<string | null> {
|
||||
return key === 'search.reranker.enabled' ? 'true' : null;
|
||||
},
|
||||
} as any);
|
||||
expect(check.status).toBe('warn');
|
||||
expect(check.message).toContain('unknown');
|
||||
expect(check.message).toContain('ZEROENTROPY_API_KEY');
|
||||
});
|
||||
} finally {
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test('runDoctor accepts null engine for filesystem-only mode', async () => {
|
||||
const { runDoctor } = await import('../src/commands/doctor.ts');
|
||||
// runDoctor should accept null engine — it runs filesystem checks only.
|
||||
|
||||
@@ -12,9 +12,14 @@
|
||||
*/
|
||||
|
||||
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as os from 'node:os';
|
||||
import * as path from 'node:path';
|
||||
import { applyReranker, type RerankerOpts } from '../../src/core/search/rerank.ts';
|
||||
import { RerankError, type RerankResult } from '../../src/core/ai/gateway.ts';
|
||||
import { readRecentRerankFailures } from '../../src/core/rerank-audit.ts';
|
||||
import type { SearchResult } from '../../src/core/types.ts';
|
||||
import { withEnv } from '../helpers/with-env.ts';
|
||||
|
||||
function makeResult(slug: string, score: number, chunk: string): SearchResult {
|
||||
return {
|
||||
@@ -160,6 +165,36 @@ describe('applyReranker — fail-open on every RerankError reason', () => {
|
||||
expect(out).toEqual(results);
|
||||
});
|
||||
|
||||
test('missing gateway reranker API key fail-opens and audits auth', async () => {
|
||||
const { configureGateway } = await import('../../src/core/ai/gateway.ts');
|
||||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gbrain-rerank-search-'));
|
||||
try {
|
||||
await withEnv({ GBRAIN_AUDIT_DIR: tmpDir }, async () => {
|
||||
configureGateway({
|
||||
reranker_model: 'zeroentropyai:zerank-2',
|
||||
env: {},
|
||||
});
|
||||
|
||||
const results = [makeResult('a', 1.0, 'doc a')];
|
||||
const out = await applyReranker('q', results, {
|
||||
enabled: true,
|
||||
topNIn: 1,
|
||||
topNOut: null,
|
||||
model: 'zeroentropyai:zerank-2',
|
||||
});
|
||||
|
||||
expect(out).toEqual(results);
|
||||
const failures = readRecentRerankFailures(1);
|
||||
expect(failures).toHaveLength(1);
|
||||
expect(failures[0]!.reason).toBe('auth');
|
||||
expect(failures[0]!.error_summary).toContain('ZEROENTROPY_API_KEY');
|
||||
});
|
||||
} finally {
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||||
configureGateway({ env: { ZEROENTROPY_API_KEY: 'test-key' } });
|
||||
}
|
||||
});
|
||||
|
||||
test('fail-open on non-RerankError throw too', async () => {
|
||||
const results = [makeResult('a', 1.0, 'a')];
|
||||
const opts: RerankerOpts = {
|
||||
|
||||
Reference in New Issue
Block a user