Compare commits

..
Author SHA1 Message Date
8900e478d1 fix(import): normalize mixed-case slugs before chunk upsert (#430)
putPage lowercases slugs via validateSlug, but upsertChunks queried
pages by the caller's raw slug — so a mixed-case slug through
importFromContent created the page row, then failed the chunk upsert
with 'Page not found' and rolled back the whole import.

Normalize via validateSlug at importFromContent entry and inside
_upsertChunksOnce on BOTH engines (postgres + pglite parity).

Takeover of #855, rebased onto current master shapes (batchRetry
wrapper / _upsertChunksOnce, rewritten importFromContent opts block).

Co-authored-by: Kage18 <Kage18@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:44:24 -07:00
6 changed files with 49 additions and 101 deletions
+7 -1
View File
@@ -36,7 +36,7 @@ import {
} from './embedding-context.ts';
import { loadSearchModeConfig, resolveSearchMode } from './search/mode.ts';
import { normalizeAliasList } from './search/alias-normalize.ts';
import { isUndefinedTableError, warnOncePerProcess } from './utils.ts';
import { isUndefinedTableError, warnOncePerProcess, validateSlug } from './utils.ts';
import { computeCorpusGeneration } from './contextual-retrieval-service.ts';
import { runGuardrails } from './guardrails.ts';
@@ -295,6 +295,12 @@ export async function importFromContent(
remote?: boolean;
} = {},
): Promise<ImportResult> {
// Normalize BEFORE any tx write: putPage lowercases via validateSlug but
// upsertChunks used to query by the caller's raw slug, so a mixed-case slug
// created the page row then failed the chunk upsert with "Page not found",
// rolling back the whole import (#430).
slug = validateSlug(slug);
// v0.18.0+ multi-source: when caller is syncing under a non-default source,
// every per-page tx call must carry `sourceId` so writes target the right
// (source_id, slug) row. Pre-fix, putPage relied on the schema DEFAULT and
+3
View File
@@ -2235,6 +2235,9 @@ export class PGLiteEngine implements BrainEngine {
}
private async _upsertChunksOnce(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string }): Promise<void> {
// Normalize the same way putPage does — pages.slug is stored lowercased,
// so a raw mixed-case slug here would miss the row it just wrote (#430).
slug = validateSlug(slug);
const sourceId = opts?.sourceId ?? 'default';
// Source-scope the page-id lookup so duplicate slugs in different sources
+3
View File
@@ -2385,6 +2385,9 @@ export class PostgresEngine implements BrainEngine {
}
private async _upsertChunksOnce(slug: string, chunks: ChunkInput[], opts?: { sourceId?: string }): Promise<void> {
// Normalize the same way putPage does — pages.slug is stored lowercased,
// so a raw mixed-case slug here would miss the row it just wrote (#430).
slug = validateSlug(slug);
const sql = this.sql;
const sourceId = opts?.sourceId ?? 'default';
+6 -17
View File
@@ -18,7 +18,6 @@
import type { BrainEngine } from '../engine.ts';
import { loadActivePackBestEffort } from './best-effort.ts';
import type { OperationContext } from '../operations.ts';
import { isUndefinedTableError } from '../utils.ts';
export interface StatsOpts {
/** Single source scope. Omit + omit sourceIds for whole-brain aggregate. */
@@ -165,17 +164,9 @@ async function fetchCountRows(engine: BrainEngine, opts: StatsOpts): Promise<Raw
`;
try {
return await engine.executeRaw<RawCountRow>(sql, params);
} catch (err) {
// ONLY swallow the genuine "pages table doesn't exist yet" case
// (empty / pre-init brain). #2466: the old bare `catch {}` masked
// EVERY error — so any engine-level failure (connection, version
// skew, a query incompatibility) was silently converted to 0 rows,
// printing "Total pages: 0" on a populated brain and cascading into
// false "100% coverage" + a starved `schema suggest`. Surface
// everything that is not a missing-table error so the real failure
// is visible instead of hidden behind a fake zero.
if (isUndefinedTableError(err)) return [];
throw err;
} catch {
// Empty / pre-init brain: pages table may not exist yet.
return [];
}
}
@@ -213,11 +204,9 @@ async function detectDeadPrefixes(
if (cnt === 0) {
hints.push({ type: t.name, prefix });
}
} catch (err) {
// #2466: only skip on the genuine "no pages table yet" case;
// rethrow any other engine error so it isn't silently masked.
if (isUndefinedTableError(err)) continue;
throw err;
} catch {
// Skip on engine error (no pages table yet, etc.).
continue;
}
}
}
+30
View File
@@ -6,6 +6,7 @@
import { describe, test, expect, beforeAll, afterAll, beforeEach } from 'bun:test';
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
import { importFromContent } from '../src/core/import-file.ts';
import type { BrainEngine } from '../src/core/engine.ts';
import type { PageInput, ChunkInput } from '../src/core/types.ts';
@@ -181,6 +182,24 @@ describe('PGLiteEngine: Pages', () => {
const page = await engine.putPage('Test/UPPER', testPage);
expect(page.slug).toBe('test/upper');
});
test('importFromContent normalizes mixed-case slugs before all tx writes (#430)', async () => {
const result = await importFromContent(
engine,
'TestNamespace/Page-Name',
'---\ntype: note\ntitle: Mixed Case\n---\n\nbody text',
{ noEmbed: true },
);
expect(result.status).toBe('imported');
expect(result.slug).toBe('testnamespace/page-name');
const page = await engine.getPage('testnamespace/page-name');
expect(page).not.toBeNull();
expect(page!.title).toBe('Mixed Case');
const chunks = await engine.getChunks('testnamespace/page-name');
expect(chunks.length).toBeGreaterThan(0);
});
});
// ─────────────────────────────────────────────────────────────────
@@ -364,6 +383,17 @@ describe('PGLiteEngine: Chunks', () => {
expect(chunks[1].chunk_text).toBe('Chunk one');
});
test('upsertChunks normalizes mixed-case slugs like putPage (#430)', async () => {
await engine.putPage('Test/ChunkCase', testPage);
await engine.upsertChunks('Test/ChunkCase', [
{ chunk_index: 0, chunk_text: 'Mixed-case chunk', chunk_source: 'compiled_truth' },
]);
const chunks = await engine.getChunks('test/chunkcase');
expect(chunks.length).toBe(1);
expect(chunks[0].chunk_text).toBe('Mixed-case chunk');
});
test('upsertChunks removes orphan chunks', async () => {
await engine.putPage('test/orphan', testPage);
await engine.upsertChunks('test/orphan', [
-83
View File
@@ -222,89 +222,6 @@ describe('runStatsCore — JSON envelope shape', () => {
});
});
describe('runStatsCore — #2466 catch-narrowing (real count + error surfacing)', () => {
// #2466: `gbrain schema stats` reported "Total pages: 0" on a populated
// PGLite brain. The bug was a bare `catch {}` in fetchCountRows (and a
// sibling in detectDeadPrefixes) that converted ANY engine error into 0
// rows. The COUNT query itself is valid on PGLite (proven below), so the
// regression pins two things: (a) a populated brain reports the real,
// non-zero count through the full runStatsCore path; (b) a non-missing-
// table engine error is rethrown, not masked into a fake zero.
it('reports the real non-zero count on a populated PGLite brain (no false 0)', async () => {
await withEnv({ GBRAIN_SCHEMA_PACK: undefined }, async () => {
// Seed a realistic mix: typed, untyped, multiple types — like the
// 169-page brain in the bug report (scaled down).
for (let i = 0; i < 12; i++) {
const type = i % 3 === 0 ? '' : (i % 3 === 1 ? 'person' : 'company');
await seedPage(`notes/p${i}`, { type, sourcePath: `notes/p${i}.md` });
}
const result = await runStatsCore(ctxOf());
// The core regression: NOT zero.
expect(result.aggregate.total_pages).toBe(12);
expect(result.aggregate.typed_pages).toBe(8);
expect(result.aggregate.untyped_pages).toBe(4);
// And coverage is the honest ratio, not the vacuous 1.0 a 0/0 prints.
expect(result.aggregate.coverage).not.toBe(1.0);
});
});
it('fetchCountRows rethrows a non-missing-table engine error instead of masking it as 0 pages', async () => {
await withEnv({ GBRAIN_SCHEMA_PACK: undefined }, async () => {
// No pack → detectDeadPrefixes is skipped, isolating the throw to the
// fetchCountRows catch we narrowed. The count query (the GROUP BY one)
// throws a column-level error (SQLSTATE 42703) — the exact class the
// old bare `catch {}` swallowed into 0 rows; everything else succeeds.
__setPackLocatorForTests(() => null);
const boom = Object.assign(new Error('column "type" does not exist'), { code: '42703' });
const stubEngine = {
executeRaw: async (sql: string) => {
if (/GROUP BY source_id/.test(sql)) throw boom; // the fetchCountRows query
return [];
},
} as unknown as PGLiteEngine;
const ctx = { ...ctxOf(), engine: stubEngine } as unknown as OperationContext;
await expect(runStatsCore(ctx)).rejects.toThrow('column "type" does not exist');
});
});
it('fetchCountRows still degrades to empty (no throw) on a genuine missing pages table', async () => {
await withEnv({ GBRAIN_SCHEMA_PACK: undefined }, async () => {
// Pre-init brain shape: the count query hits a missing pages table
// (SQLSTATE 42P01). This is the ONLY case the narrowed catch swallows.
__setPackLocatorForTests(() => null);
const missing = Object.assign(new Error('relation "pages" does not exist'), { code: '42P01' });
const stubEngine = {
executeRaw: async (sql: string) => {
if (/GROUP BY source_id/.test(sql)) throw missing;
return [];
},
} as unknown as PGLiteEngine;
const ctx = { ...ctxOf(), engine: stubEngine } as unknown as OperationContext;
const result = await runStatsCore(ctx);
expect(result.aggregate.total_pages).toBe(0);
expect(result.per_source).toEqual([]);
});
});
it('detectDeadPrefixes rethrows a non-missing-table error (sibling catch)', async () => {
await withEnv({ GBRAIN_HOME: tmpDir, GBRAIN_SCHEMA_PACK: 'tiny' }, async () => {
seedTinyPack('tiny', [{ name: 'person', prefix: 'people/' }]);
// fetchCountRows (the GROUP BY query) succeeds → []; the per-prefix
// dead-prefix LIKE query then throws a non-missing-table error, which
// must surface through the narrowed sibling catch.
const stubEngine = {
executeRaw: async (sql: string) => {
if (/GROUP BY source_id/.test(sql)) return []; // count query: empty brain, fine
throw Object.assign(new Error('division by zero'), { code: '22012' }); // the LIKE query
},
} as unknown as PGLiteEngine;
const ctx = { ...ctxOf(), engine: stubEngine } as unknown as OperationContext;
await expect(runStatsCore(ctx)).rejects.toThrow('division by zero');
});
});
});
describe('runStatsCore — type/untyped split', () => {
it('treats empty-string type as untyped (not its own bucket)', async () => {
await withEnv({ GBRAIN_SCHEMA_PACK: undefined }, async () => {