Compare commits

..
Author SHA1 Message Date
583975dee8 fix(sync): guard putPage 0-row RETURNING + reclassify sync failure copy (#2189)
Root cause of #2189's opaque per-file crash: putPage's INSERT … ON CONFLICT
DO UPDATE … RETURNING can yield 0 rows when brain-local DB state (e.g. a
BEFORE trigger) suppresses the write, and rowToPage(rows[0]) then died with
"undefined is not an object (evaluating 'row.deleted_at')". Both engines now
throw a descriptive error naming the slug + source_id so the failure is
diagnosable per-file instead of an anonymous TypeError.

Also lands the salvageable parts of community PR #2586 (takeover): the
"failed to parse / fix the frontmatter" copy misclassified runtime import
errors as YAML problems — reworded to "failed to import" — plus its
first-code-sync PGLite smoke test.

New regression test reproduces the exact 0-row RETURNING state via a
suppressing trigger: fails pre-fix with the reported crash signature,
passes with the guard.

Co-authored-by: javieraldape <javieraldape@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:41:32 -07:00
7 changed files with 127 additions and 224 deletions
+5 -5
View File
@@ -197,7 +197,7 @@ export interface SyncResult {
/** Pages re-embedded during this sync's auto-embed step. 0 if --no-embed or skipped. */
embedded: number;
pagesAffected: string[];
failedFiles?: number; // count of parse failures (Bug 9)
failedFiles?: number; // count of per-file import/sync failures (Bug 9)
/**
* v0.41.13.0 partial-sync fields (only set when status === 'partial').
*
@@ -3183,7 +3183,7 @@ async function performSyncInner(engine: BrainEngine, opts: SyncOpts): Promise<Sy
await clearOpCheckpoint(engine, ckpt.target);
};
// issue #1939 adversarial finding #1: a file that failed to parse (open ledger
// issue #1939 adversarial finding #1: a file that failed to import (open ledger
// row) and is then deleted/renamed-away never re-enters failedFiles and never
// imports, so its row would never clear and would age doctor to a permanent
// FAIL. Treat removed paths as resolved so the ledger self-heals.
@@ -3215,9 +3215,9 @@ async function performSyncInner(engine: BrainEngine, opts: SyncOpts): Promise<Sy
} else {
const fileFailCount = failedFiles.filter(f => isSkippablePath(f.path)).length;
serr(
`\nSync blocked: ${fileFailCount} file(s) failed to parse:\n` +
`\nSync blocked: ${fileFailCount} file(s) failed to import:\n` +
`${codeBreakdown}\n\n` +
`Fix the frontmatter and re-run, or use 'gbrain sync --skip-failed' to ` +
`Fix the listed file errors and re-run, or use 'gbrain sync --skip-failed' to ` +
`acknowledge and move on. A file that keeps failing auto-skips after ` +
`${resolveAutoSkipThreshold()} consecutive syncs.`,
);
@@ -5355,7 +5355,7 @@ function printSyncResult(result: SyncResult, sink: NodeJS.WriteStream = process.
case 'dry_run':
break; // already printed in performSync
case 'blocked_by_failures':
write(`Sync BLOCKED at ${result.toCommit.slice(0, 8)}: ${result.failedFiles ?? 0} file(s) failed to parse.`);
write(`Sync BLOCKED at ${result.toCommit.slice(0, 8)}: ${result.failedFiles ?? 0} file(s) failed to import.`);
write(` See ~/.gbrain/sync-failures.jsonl for details, or run 'gbrain doctor'.`);
write(` Fix the files then re-run 'gbrain sync', or 'gbrain sync --skip-failed' to move on.`);
break;
+2 -112
View File
@@ -1,100 +1,8 @@
import type { Recipe } from '../types.ts';
/**
* MiniMax transport shim (#1977). MiniMax's `/v1/embeddings` endpoint is NOT
* OpenAI-compatible at the wire level despite the recipe's
* `implementation: 'openai-compatible'`:
* - Request: requires `texts` (the AI SDK sends `input`) plus an optional
* `type: 'db' | 'query'` asymmetric-retrieval field, and rejects OpenAI's
* `encoding_format`.
* - Response: returns `{vectors: number[][], total_tokens}` where the AI
* SDK's Zod schema expects `{data: [{embedding, index}], usage}`.
*
* Chat (`/chat/completions`) IS OpenAI-compatible, and this same fetch is
* applied to every openai-compatible touchpoint by `applyOpenAICompatConfig`,
* so everything outside the embeddings path passes through untouched — and
* the response rewrite parses via `resp.clone()` only (never consume the
* body of a response we return as-is; the DeepSeek shim rule). Fail-open:
* any rewrite error returns the original request/response.
*
* @internal exported for tests.
*/
// Cast through `unknown` because Bun's `typeof fetch` carries a `preconnect`
// member the arrow function does not implement (matches deepseek.ts).
export const minimaxCompatFetch = (async (
input: RequestInfo | URL,
init?: RequestInit,
): Promise<Response> => {
const url =
typeof input === 'string' ? input : input instanceof URL ? input.href : input.url;
const isEmbeddings = url.includes('/embeddings');
// OUTBOUND (embeddings only): `input` → `texts`, default `type: 'db'`
// (the recipe's documented symmetric default — the AI SDK adapter strips
// the `type` threaded via providerOptions before it reaches the wire,
// same class as #1400), and drop `encoding_format` (not a MiniMax param).
if (isEmbeddings && init?.body && typeof init.body === 'string') {
try {
const parsed = JSON.parse(init.body);
if (
parsed && typeof parsed === 'object' &&
parsed.input !== undefined && parsed.texts === undefined
) {
parsed.texts = Array.isArray(parsed.input) ? parsed.input : [parsed.input];
delete parsed.input;
delete parsed.encoding_format;
if (parsed.type === undefined) parsed.type = 'db';
// Drop Content-Length so fetch recomputes from the new body.
const headers = new Headers(init.headers ?? {});
headers.delete('content-length');
init = { ...init, body: JSON.stringify(parsed), headers };
}
} catch {
// Body wasn't JSON — pass through untouched.
}
}
const res = await fetch(input as any, init as any);
// INBOUND (embeddings only): `{vectors: [[...]]}` → `{data: [{embedding}]}`.
// Anything else (chat completions, MiniMax base_resp errors, non-JSON)
// returns the ORIGINAL response with its body unread.
if (!isEmbeddings || !res.ok) return res;
const ctype = res.headers.get('content-type') ?? '';
if (!ctype.toLowerCase().includes('application/json')) return res;
try {
const json = await res.clone().json();
if (!json || typeof json !== 'object' || !Array.isArray(json.vectors)) return res;
const totalTokens = typeof json.total_tokens === 'number' ? json.total_tokens : 0;
const rewritten = {
object: 'list',
data: (json.vectors as number[][]).map((embedding, index) => ({
object: 'embedding',
embedding,
index,
})),
model: typeof json.model === 'string' ? json.model : 'embo-01',
usage: { prompt_tokens: totalTokens, total_tokens: totalTokens },
};
// Fresh header set: the body changed, so upstream content-length /
// content-encoding would now be wrong.
const headers = new Headers(res.headers);
headers.delete('content-length');
headers.delete('content-encoding');
return new Response(JSON.stringify(rewritten), {
status: res.status,
statusText: res.statusText,
headers,
});
} catch {
return res;
}
}) as unknown as typeof fetch;
/**
* MiniMax (海螺AI). `/embeddings` endpoint at api.minimaxi.com (wire shape
* normalized by `minimaxCompatFetch` above); OpenAI-compatible
* `/chat/completions`. The flagship embedding model is `embo-01` (1536 dims).
* MiniMax (海螺AI). OpenAI-compatible /embeddings endpoint at
* api.minimax.chat. The flagship embedding model is `embo-01` (1536 dims).
*
* MiniMax's API takes an extra `type: 'db' | 'query'` field for asymmetric
* retrieval. gbrain currently has no notion of "this is a document vs a
@@ -130,25 +38,7 @@ export const minimax: Recipe = {
// halving in the gateway catches token-limit errors at runtime.
max_batch_tokens: 4096,
},
chat: {
// Model list from MiniMax's /v1/models (#1977). Chat is genuinely
// OpenAI-compatible — no wire rewrite needed (minimaxCompatFetch
// passes non-embedding requests through untouched).
models: [
'MiniMax-M3',
'MiniMax-M2.7',
'MiniMax-M2.7-highspeed',
'MiniMax-M2.5',
'MiniMax-M2.5-highspeed',
'MiniMax-M2.1',
'MiniMax-M2.1-highspeed',
'MiniMax-M2',
],
supports_tools: false,
supports_subagent_loop: false,
},
},
setup_hint:
'Get an API key at https://www.minimaxi.com, then `export MINIMAX_API_KEY=...`',
compat: { fetch: minimaxCompatFetch },
};
+10
View File
@@ -1058,6 +1058,16 @@ export class PGLiteEngine implements BrainEngine {
RETURNING id, source_id, slug, type, title, compiled_truth, timeline, frontmatter, content_hash, created_at, updated_at, effective_date, effective_date_source, import_filename, source_kind, source_uri, ingested_via, ingested_at`,
[sourceId, slug, page.type, pageKind, page.title, page.compiled_truth, page.timeline || '', JSON.stringify(frontmatter), hash, effectiveDate, effectiveDateSource, importFilename, chunkerVersion, sourcePath, sourceKind, sourceUri, ingestedVia, ingestedAt]
);
// #2189: an INSERT … ON CONFLICT DO UPDATE … RETURNING that yields 0 rows
// (e.g. a BEFORE trigger suppressing the write) previously crashed in
// rowToPage with an opaque "undefined is not an object (row.deleted_at)".
// Throw a diagnosable error naming the row instead. Mirrors postgres-engine.ts.
if (!rows[0]) {
throw new Error(
`putPage: INSERT … RETURNING produced no row for slug='${slug}' source_id='${sourceId}'. ` +
`A trigger or rule on the pages table may be suppressing the write.`
);
}
return rowToPage(rows[0] as Record<string, unknown>);
}
+10
View File
@@ -1119,6 +1119,16 @@ export class PostgresEngine implements BrainEngine {
ingested_at = COALESCE(EXCLUDED.ingested_at, pages.ingested_at)
RETURNING id, source_id, slug, type, title, compiled_truth, timeline, frontmatter, content_hash, created_at, updated_at, effective_date, effective_date_source, import_filename, source_kind, source_uri, ingested_via, ingested_at
`;
// #2189: an INSERT … ON CONFLICT DO UPDATE … RETURNING that yields 0 rows
// (e.g. a BEFORE trigger suppressing the write) previously crashed in
// rowToPage with an opaque "undefined is not an object (row.deleted_at)".
// Throw a diagnosable error naming the row instead. Mirrors pglite-engine.ts.
if (!rows[0]) {
throw new Error(
`putPage: INSERT … RETURNING produced no row for slug='${slug}' source_id='${sourceId}'. ` +
`A trigger or rule on the pages table may be suppressing the write.`
);
}
return rowToPage(rows[0]);
}
+1 -107
View File
@@ -6,14 +6,10 @@
* - default auth: MINIMAX_API_KEY → "Bearer <key>"; missing → AIConfigError
* - dimsProviderOptions threads `type: 'db'` for embo-01 (the asymmetric
* retrieval field default) — pins the v1 indexing-only behavior
* - #1977: chat touchpoint declared; minimaxCompatFetch rewrites the
* embedding wire shape both directions, passes chat through with the
* response body UNREAD (the consumed-body regression), fail-open.
*/
import { afterEach, describe, expect, test } from 'bun:test';
import { describe, expect, test } from 'bun:test';
import { getRecipe } from '../../src/core/ai/recipes/index.ts';
import { minimaxCompatFetch } from '../../src/core/ai/recipes/minimax.ts';
import { defaultResolveAuth } from '../../src/core/ai/gateway.ts';
import { dimsProviderOptions } from '../../src/core/ai/dims.ts';
import { AIConfigError } from '../../src/core/ai/errors.ts';
@@ -60,106 +56,4 @@ describe('recipe: minimax', () => {
expect(dimsProviderOptions('openai-compatible', 'voyage-3-lite', 512)).toBeUndefined();
expect(dimsProviderOptions('openai-compatible', 'nomic-embed-text', 768)).toBeUndefined();
});
test('chat touchpoint declared (#1977) so assertTouchpoint permits gbrain think', () => {
const r = getRecipe('minimax')!;
expect(r.touchpoints.chat).toBeDefined();
expect(r.touchpoints.chat!.models).toContain('MiniMax-M3');
expect(r.touchpoints.chat!.supports_tools).toBe(false);
expect(r.touchpoints.chat!.supports_subagent_loop).toBe(false);
});
test('recipe ships minimaxCompatFetch via compat.fetch (no env-templated base URL)', () => {
const r = getRecipe('minimax')!;
expect(r.compat?.fetch).toBe(minimaxCompatFetch);
// base_urls config override must keep working: no resolveOpenAICompatConfig.
expect(r.resolveOpenAICompatConfig).toBeUndefined();
});
});
describe('minimaxCompatFetch (#1977)', () => {
const realFetch = globalThis.fetch;
afterEach(() => { globalThis.fetch = realFetch; });
function stubFetch(body: unknown, init?: { status?: number; contentType?: string }) {
const calls: { url: string; init?: RequestInit }[] = [];
globalThis.fetch = (async (input: any, i?: RequestInit) => {
calls.push({ url: String(input), init: i });
return new Response(typeof body === 'string' ? body : JSON.stringify(body), {
status: init?.status ?? 200,
headers: { 'content-type': init?.contentType ?? 'application/json' },
});
}) as unknown as typeof fetch;
return calls;
}
const EMBED_URL = 'https://api.minimaxi.com/v1/embeddings';
const CHAT_URL = 'https://api.minimaxi.com/v1/chat/completions';
test('embedding request: input → texts, type:db injected, encoding_format dropped', async () => {
const calls = stubFetch({ vectors: [[0.1, 0.2]] });
await minimaxCompatFetch(EMBED_URL, {
method: 'POST',
headers: { 'content-type': 'application/json', 'content-length': '99' },
body: JSON.stringify({ model: 'embo-01', input: ['hello', 'world'], encoding_format: 'float' }),
});
const wire = JSON.parse(calls[0]!.init!.body as string);
expect(wire.texts).toEqual(['hello', 'world']);
expect(wire.input).toBeUndefined();
expect(wire.encoding_format).toBeUndefined();
expect(wire.type).toBe('db');
expect(new Headers(calls[0]!.init!.headers).get('content-length')).toBeNull();
});
test('embedding response: {vectors} rewritten to OpenAI {data:[{embedding}]}', async () => {
stubFetch({ vectors: [[0.1, 0.2], [0.3, 0.4]], total_tokens: 7 });
const res = await minimaxCompatFetch(EMBED_URL, {
method: 'POST',
body: JSON.stringify({ model: 'embo-01', input: ['a', 'b'] }),
});
const json = await res.json();
expect(json.data).toEqual([
{ object: 'embedding', embedding: [0.1, 0.2], index: 0 },
{ object: 'embedding', embedding: [0.3, 0.4], index: 1 },
]);
expect(json.usage).toEqual({ prompt_tokens: 7, total_tokens: 7 });
});
test('chat completion passes through with body UNREAD (consumed-body regression)', async () => {
stubFetch({ choices: [{ message: { role: 'assistant', content: 'hi' } }] });
const res = await minimaxCompatFetch(CHAT_URL, {
method: 'POST',
body: JSON.stringify({ model: 'MiniMax-M3', messages: [{ role: 'user', content: 'say hi' }] }),
});
expect(res.bodyUsed).toBe(false); // the broken PR #2882 wrapper consumed this
const json = await res.json(); // must NOT throw "Body already used"
expect(json.choices[0].message.content).toBe('hi');
});
test('chat request body is never rewritten (messages untouched, no type injected)', async () => {
const calls = stubFetch({ choices: [] });
const body = JSON.stringify({ model: 'MiniMax-M3', messages: [{ role: 'user', content: 'x' }] });
await minimaxCompatFetch(CHAT_URL, { method: 'POST', body });
expect(calls[0]!.init!.body).toBe(body);
});
test('embedding error response ({vectors:null, base_resp}) passes through re-readable', async () => {
stubFetch({ vectors: null, base_resp: { status_code: 2013, status_msg: 'invalid params' } });
const res = await minimaxCompatFetch(EMBED_URL, {
method: 'POST',
body: JSON.stringify({ model: 'embo-01', input: ['a'] }),
});
expect(res.bodyUsed).toBe(false);
const json = await res.json();
expect(json.base_resp.status_code).toBe(2013);
});
test('fail-open: non-JSON response body passes through untouched', async () => {
stubFetch('not json', { contentType: 'application/json' });
const res = await minimaxCompatFetch(EMBED_URL, {
method: 'POST',
body: JSON.stringify({ model: 'embo-01', input: ['a'] }),
});
expect(await res.text()).toBe('not json');
});
});
+74
View File
@@ -0,0 +1,74 @@
// #2189 regression guard: putPage's INSERT … ON CONFLICT DO UPDATE … RETURNING
// can yield 0 rows when brain-local DB state (e.g. a BEFORE INSERT trigger)
// suppresses the write. Pre-fix, rowToPage(rows[0]) crashed with the opaque
// "undefined is not an object (evaluating 'row.deleted_at')" that failed
// ~all files of a code sync. Post-fix, putPage throws a descriptive error
// naming the slug + source_id so the failure is diagnosable per-file.
//
// Same guard lands in postgres-engine.ts (engine-parity invariant); this test
// exercises the PGLite side, where the issue was reported.
import { describe, expect, test, beforeAll, afterAll } from 'bun:test';
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
let engine: PGLiteEngine;
beforeAll(async () => {
engine = new PGLiteEngine();
await engine.connect({});
await engine.initSchema();
// Simulate the reporter's state-dependent failure: a trigger that
// suppresses inserts for one slug, making RETURNING produce no row.
await engine.executeRaw(`
CREATE OR REPLACE FUNCTION suppress_pages_insert() RETURNS trigger AS $$
BEGIN
IF NEW.slug = 'suppressed-page' THEN RETURN NULL; END IF;
RETURN NEW;
END;
$$ LANGUAGE plpgsql;
`);
await engine.executeRaw(`
CREATE TRIGGER suppress_pages_insert_trg
BEFORE INSERT ON pages
FOR EACH ROW EXECUTE FUNCTION suppress_pages_insert();
`);
});
afterAll(async () => {
await engine.executeRaw('DROP TRIGGER IF EXISTS suppress_pages_insert_trg ON pages');
await engine.executeRaw('DROP FUNCTION IF EXISTS suppress_pages_insert');
await engine.disconnect();
});
describe('putPage RETURNING guard (#2189)', () => {
test('0-row RETURNING throws a descriptive error, not row.deleted_at TypeError', async () => {
let err: Error | undefined;
try {
await engine.putPage('suppressed-page', {
type: 'code',
title: 'Suppressed',
compiled_truth: 'x',
timeline: '',
});
} catch (e) {
err = e as Error;
}
expect(err).toBeDefined();
expect(err!.message).toContain('putPage');
expect(err!.message).toContain("slug='suppressed-page'");
expect(err!.message).toContain("source_id='default'");
// The pre-fix crash signature must be gone.
expect(err!.message).not.toContain('deleted_at');
});
test('unsuppressed slugs still upsert normally with the trigger installed', async () => {
const page = await engine.putPage('normal-page', {
type: 'concept',
title: 'Normal',
compiled_truth: 'y',
timeline: '',
});
expect(page.slug).toBe('normal-page');
expect(page.source_id).toBe('default');
});
});
+25
View File
@@ -375,6 +375,31 @@ describe('performSync dry-run never writes', () => {
expect(messages.some(m => m.includes('git pull failed'))).toBe(false);
});
test('first PGLite code sync imports code files without runtime failures', async () => {
const { performSync } = await import('../src/commands/sync.ts');
mkdirSync(join(repoPath, 'src'), { recursive: true });
writeFileSync(
join(repoPath, 'src/example.ts'),
'export function add(left: number, right: number) { return left + right; }\n',
);
execSync('git add -A && git commit -m "add code file"', { cwd: repoPath, stdio: 'pipe' });
const result = await performSync(engine, {
repoPath,
noPull: true,
noEmbed: true,
noExtract: true,
strategy: 'code',
});
expect(result.status).toBe('first_sync');
expect(result.added).toBe(1);
expect(result.failedFiles ?? 0).toBe(0);
const page = await engine.getPage('src-example-ts');
expect(page?.type).toBe('code');
expect(page?.frontmatter).toMatchObject({ file: 'src/example.ts', language: 'typescript' });
});
test('incremental dry-run does NOT write to DB or advance the bookmark', async () => {
const { performSync } = await import('../src/commands/sync.ts');
// First do a real sync to seed the bookmark.