Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 a42352081c test(extract): pin the version-arm stamp floor with an explicit regression test
The floor in extractStaleFromDB was only covered incidentally — the #1768
test's updated_at (2026-06-02) fell below the bumped VERSION_TS, but its
comment still says the date was chosen to sit ABOVE the watermark. A future
date 'repair' there would silently drop floor coverage. This test pins it
directly: pre-watermark page clears after --stale and stays cleared.
Verified fail-without-fix against master's extract.ts.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 11:59:09 -07:00
Garry TanandClaude Fable 5 070f4c5678 fix(extract): floor --stale stamp at LINK_EXTRACTOR_VERSION_TS so version bumps don't create permanently-stale pages
The version-ts bump in this PR exposed a latent conflict: extractStaleFromDB
stamps links_extracted_at = the page's read updated_at (#1768 µs fix), but a
page last edited BEFORE the new LINK_EXTRACTOR_VERSION_TS then lands below the
version watermark and the 'links_extracted_at < versionTs' arm re-flags it
stale on every run — an infinite re-extraction loop. Stamp
max(updated_at, versionTs); versionTs is always a past release date, so the
D4 concurrent-edit race guard still holds.

Fixes the test/extract-stale.test.ts CI failure on this branch.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 10:56:38 -07:00
Garry TanandClaude Fable 5 4b2df89935 fix(extract): resolve path-style wikilinks outside DIR_PATTERN + add ops/ to whitelist (#2576 bug 2)
- Add ops to DIR_PATTERN so [[ops/services/...]] wikilinks and bare
  ops/... slug refs are recognized as qualified entity references.
- queryBasenameIndex (the shared basename matcher behind
  resolveBasenameMatches, the FS resolver, and the doctor check) now
  falls back to the path tail when a path-style ref misses — before,
  normalizeBasename stripped slashes into a garbage key that could
  never hit the tail-keyed index.
- Bump LINK_EXTRACTOR_VERSION_TS so extract --stale re-sweeps
  previously-stamped pages with the new extraction logic.

Bugs 1+3 of #2576 are covered by PR #2717 (--stale nullResolver).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:33:45 -07:00
6 changed files with 64 additions and 105 deletions
+9 -1
View File
@@ -1743,7 +1743,15 @@ async function extractStaleFromDB(
// `page.updated_at.toISOString()` — the JS Date is ms-truncated, so the
// µs-precision DB updated_at stayed strictly greater and the page never
// cleared on Postgres. Stamping the exact value makes them equal.
processedRefs.push({ slug: page.slug, source_id: page.source_id, extractedAt: page.updated_at_iso });
//
// Version-arm floor: a page last edited BEFORE LINK_EXTRACTOR_VERSION_TS
// would otherwise be stamped below the version watermark and stay
// permanently stale (`links_extracted_at < versionTs` re-fires every run).
// Stamp max(updated_at, versionTs) — versionTs is always a past release
// date, so a concurrent edit's now() still exceeds the stamp and D4 holds.
// Tie at ms precision picks updated_at_iso (its µs ≥ versionTs's .000000).
const stampTs = new Date(page.updated_at_iso) >= new Date(versionTs) ? page.updated_at_iso : versionTs;
processedRefs.push({ slug: page.slug, source_id: page.source_id, extractedAt: stampTs });
}
// Flush NON-swallowing (CDX-4): a throw here propagates out of the sweep so
+13 -4
View File
@@ -28,7 +28,7 @@ import { ensureWellFormed } from './text-safe.ts';
* OR updated_at > links_extracted_at`. It is an ISO-8601 string (NOT a number) —
* the column is TIMESTAMPTZ and the predicate binds it as `::timestamptz`.
*/
export const LINK_EXTRACTOR_VERSION_TS = '2026-05-31T00:00:00Z';
export const LINK_EXTRACTOR_VERSION_TS = '2026-07-21T00:00:00Z';
// ─── Entity references ──────────────────────────────────────────
@@ -80,10 +80,10 @@ export type LinkResolutionType = 'qualified' | 'unqualified';
* Directory prefix whitelist. These are the top-level slug dirs the extractor
* recognizes as entity references. Upstream canonical + our extensions:
* - Gbrain canonical: people, companies, meetings, concepts, deal, civic, project, source, media, yc, projects
* - Our domain extensions: tech, finance, personal, openclaw (domain-organized wikis)
* - Our domain extensions: tech, finance, personal, openclaw, ops (domain-organized wikis)
* - Our entity prefix: entities (we kept some legacy entities/projects/ pages)
*/
const DIR_PATTERN = '(?:people|companies|meetings|concepts|deal|civic|project|projects|source|media|yc|tech|finance|personal|openclaw|entities)';
const DIR_PATTERN = '(?:people|companies|meetings|concepts|deal|civic|project|projects|source|media|yc|tech|finance|personal|openclaw|entities|ops)';
/**
* Match `[Name](path)` markdown links pointing to entity directories.
@@ -865,7 +865,16 @@ export function queryBasenameIndex(idx: Map<string, string[]>, name: string): st
if (!name || typeof name !== 'string') return [];
const trimmed = name.trim();
if (!trimmed) return [];
const hit = idx.get(trimmed) ?? idx.get(trimmed.toLowerCase()) ?? idx.get(normalizeBasename(trimmed));
let hit = idx.get(trimmed) ?? idx.get(trimmed.toLowerCase()) ?? idx.get(normalizeBasename(trimmed));
// Issue #2576 bug 2: path-style refs (`runbooks/2026-05-01-x`) from dirs
// outside DIR_PATTERN reach here, but normalizeBasename strips slashes
// into a garbage key (`runbooks2026-05-01-x`) that can never hit the
// tail-keyed index. Fall back to the path tail so qualified refs resolve
// by basename like everything else.
if (!hit && trimmed.includes('/')) {
const tail = trimmed.slice(trimmed.lastIndexOf('/') + 1).trim();
if (tail) hit = idx.get(tail) ?? idx.get(tail.toLowerCase()) ?? idx.get(normalizeBasename(tail));
}
return hit ? [...hit].sort(basenameSort) : [];
}
+6 -17
View File
@@ -18,7 +18,6 @@
import type { BrainEngine } from '../engine.ts';
import { loadActivePackBestEffort } from './best-effort.ts';
import type { OperationContext } from '../operations.ts';
import { isUndefinedTableError } from '../utils.ts';
export interface StatsOpts {
/** Single source scope. Omit + omit sourceIds for whole-brain aggregate. */
@@ -165,17 +164,9 @@ async function fetchCountRows(engine: BrainEngine, opts: StatsOpts): Promise<Raw
`;
try {
return await engine.executeRaw<RawCountRow>(sql, params);
} catch (err) {
// ONLY swallow the genuine "pages table doesn't exist yet" case
// (empty / pre-init brain). #2466: the old bare `catch {}` masked
// EVERY error — so any engine-level failure (connection, version
// skew, a query incompatibility) was silently converted to 0 rows,
// printing "Total pages: 0" on a populated brain and cascading into
// false "100% coverage" + a starved `schema suggest`. Surface
// everything that is not a missing-table error so the real failure
// is visible instead of hidden behind a fake zero.
if (isUndefinedTableError(err)) return [];
throw err;
} catch {
// Empty / pre-init brain: pages table may not exist yet.
return [];
}
}
@@ -213,11 +204,9 @@ async function detectDeadPrefixes(
if (cnt === 0) {
hints.push({ type: t.name, prefix });
}
} catch (err) {
// #2466: only skip on the genuine "no pages table yet" case;
// rethrow any other engine error so it isn't silently masked.
if (isUndefinedTableError(err)) continue;
throw err;
} catch {
// Skip on engine error (no pages table yet, etc.).
continue;
}
}
}
+18
View File
@@ -209,6 +209,24 @@ describe('gbrain extract --stale', () => {
expect(usRows[0]?.eq).toBe(true);
});
test('version-arm floor: page edited BEFORE LINK_EXTRACTOR_VERSION_TS clears after --stale (issue #2576 bug 3)', async () => {
// A page whose updated_at predates the version watermark used to be
// stamped at its updated_at (< versionTs), so the version arm re-fired
// every run — permanently stale. The sweep now floors the stamp at
// versionTs. (The #1768 test above also covers this since the v0.42.x
// VERSION_TS bump moved its date below the watermark, but this pins the
// behavior explicitly so a date "repair" there can't drop coverage.)
await engine.putPage('people/alice', personPage('Alice'));
await engine.executeRaw(`UPDATE pages SET updated_at = '2000-01-01T00:00:00Z' WHERE slug = 'people/alice'`);
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(1);
await runExtract(engine, ['--stale']);
// Pre-floor this stayed 1 forever (stamp < versionTs → version arm re-fires).
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(0);
await runExtract(engine, ['--stale']);
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(0);
});
test('CDX-4 (D2): a link-flush throw aborts the sweep and leaves pages UNSTAMPED', async () => {
await engine.putPage('people/alice', personPage('Alice'));
await engine.putPage('companies/acme', companyPage('Acme', '[Alice](people/alice) founded [Acme](companies/acme).'));
+18
View File
@@ -140,6 +140,15 @@ describe('extractEntityRefs', () => {
expect(wikiRefs[0].needsResolution).toBe(true);
});
test('recognizes ops/ qualified wikilinks (issue #2576 bug 2)', () => {
// `ops` was missing from DIR_PATTERN, so [[ops/...]] fell through to
// the generic 2c pass (needsResolution) instead of being a real ref.
const refs = extractEntityRefs('Deployed via [[ops/services/pointer-agent]].');
expect(refs.length).toBe(1);
expect(refs[0].slug).toBe('ops/services/pointer-agent');
expect(refs[0].needsResolution).toBeUndefined();
});
test('skips qualified-syntax tokens (those belong to 2a)', () => {
// [[wiki:topics/ai]] looks like 2a's qualified shape — even though
// it wouldn't satisfy DIR_PATTERN, 2c must not claim it either
@@ -1069,6 +1078,15 @@ describe('makeResolver — fallback chain', () => {
]);
});
test('resolveBasenameMatches: path-style ref falls back to the tail (issue #2576 bug 2)', async () => {
// normalizeBasename strips slashes, so `runbooks/2026-05-01-pointer-agent`
// used to normalize to a garbage key that never hit the tail-keyed index.
const engine = makeFakeEngineWithSlugs(['ops/changes/2026-05-01-pointer-agent']);
const r = makeResolver(engine);
expect(await r.resolveBasenameMatches!('runbooks/2026-05-01-pointer-agent'))
.toEqual(['ops/changes/2026-05-01-pointer-agent']);
});
test('resolveBasenameMatches: case-insensitive fallback', async () => {
const engine = makeFakeEngineWithSlugs(['companies/fast-weigh']);
const r = makeResolver(engine);
-83
View File
@@ -222,89 +222,6 @@ describe('runStatsCore — JSON envelope shape', () => {
});
});
describe('runStatsCore — #2466 catch-narrowing (real count + error surfacing)', () => {
// #2466: `gbrain schema stats` reported "Total pages: 0" on a populated
// PGLite brain. The bug was a bare `catch {}` in fetchCountRows (and a
// sibling in detectDeadPrefixes) that converted ANY engine error into 0
// rows. The COUNT query itself is valid on PGLite (proven below), so the
// regression pins two things: (a) a populated brain reports the real,
// non-zero count through the full runStatsCore path; (b) a non-missing-
// table engine error is rethrown, not masked into a fake zero.
it('reports the real non-zero count on a populated PGLite brain (no false 0)', async () => {
await withEnv({ GBRAIN_SCHEMA_PACK: undefined }, async () => {
// Seed a realistic mix: typed, untyped, multiple types — like the
// 169-page brain in the bug report (scaled down).
for (let i = 0; i < 12; i++) {
const type = i % 3 === 0 ? '' : (i % 3 === 1 ? 'person' : 'company');
await seedPage(`notes/p${i}`, { type, sourcePath: `notes/p${i}.md` });
}
const result = await runStatsCore(ctxOf());
// The core regression: NOT zero.
expect(result.aggregate.total_pages).toBe(12);
expect(result.aggregate.typed_pages).toBe(8);
expect(result.aggregate.untyped_pages).toBe(4);
// And coverage is the honest ratio, not the vacuous 1.0 a 0/0 prints.
expect(result.aggregate.coverage).not.toBe(1.0);
});
});
it('fetchCountRows rethrows a non-missing-table engine error instead of masking it as 0 pages', async () => {
await withEnv({ GBRAIN_SCHEMA_PACK: undefined }, async () => {
// No pack → detectDeadPrefixes is skipped, isolating the throw to the
// fetchCountRows catch we narrowed. The count query (the GROUP BY one)
// throws a column-level error (SQLSTATE 42703) — the exact class the
// old bare `catch {}` swallowed into 0 rows; everything else succeeds.
__setPackLocatorForTests(() => null);
const boom = Object.assign(new Error('column "type" does not exist'), { code: '42703' });
const stubEngine = {
executeRaw: async (sql: string) => {
if (/GROUP BY source_id/.test(sql)) throw boom; // the fetchCountRows query
return [];
},
} as unknown as PGLiteEngine;
const ctx = { ...ctxOf(), engine: stubEngine } as unknown as OperationContext;
await expect(runStatsCore(ctx)).rejects.toThrow('column "type" does not exist');
});
});
it('fetchCountRows still degrades to empty (no throw) on a genuine missing pages table', async () => {
await withEnv({ GBRAIN_SCHEMA_PACK: undefined }, async () => {
// Pre-init brain shape: the count query hits a missing pages table
// (SQLSTATE 42P01). This is the ONLY case the narrowed catch swallows.
__setPackLocatorForTests(() => null);
const missing = Object.assign(new Error('relation "pages" does not exist'), { code: '42P01' });
const stubEngine = {
executeRaw: async (sql: string) => {
if (/GROUP BY source_id/.test(sql)) throw missing;
return [];
},
} as unknown as PGLiteEngine;
const ctx = { ...ctxOf(), engine: stubEngine } as unknown as OperationContext;
const result = await runStatsCore(ctx);
expect(result.aggregate.total_pages).toBe(0);
expect(result.per_source).toEqual([]);
});
});
it('detectDeadPrefixes rethrows a non-missing-table error (sibling catch)', async () => {
await withEnv({ GBRAIN_HOME: tmpDir, GBRAIN_SCHEMA_PACK: 'tiny' }, async () => {
seedTinyPack('tiny', [{ name: 'person', prefix: 'people/' }]);
// fetchCountRows (the GROUP BY query) succeeds → []; the per-prefix
// dead-prefix LIKE query then throws a non-missing-table error, which
// must surface through the narrowed sibling catch.
const stubEngine = {
executeRaw: async (sql: string) => {
if (/GROUP BY source_id/.test(sql)) return []; // count query: empty brain, fine
throw Object.assign(new Error('division by zero'), { code: '22012' }); // the LIKE query
},
} as unknown as PGLiteEngine;
const ctx = { ...ctxOf(), engine: stubEngine } as unknown as OperationContext;
await expect(runStatsCore(ctx)).rejects.toThrow('division by zero');
});
});
});
describe('runStatsCore — type/untyped split', () => {
it('treats empty-string type as untyped (not its own bucket)', async () => {
await withEnv({ GBRAIN_SCHEMA_PACK: undefined }, async () => {