Compare commits

..
Author SHA1 Message Date
a11ec9c468 fix(dream): stamp incremental extraction watermark (#2636)
The Dream cycle disables sync's inline extraction and routes changed
slugs through extractForSlugs, which flushed link/timeline batches but
never stamped links_extracted_at — so incrementally extracted pages
stayed permanently visible to `extract --stale` / doctor.

Collect processedRefs per successfully processed page and stamp them
via stampExtracted (best-effort) after both batch flushes, non-dry-run
mode 'all' only. Source-id threading from the original PR #2637 already
landed on master via #1503/#1747, so this rebase carries only the
missing watermark stamp plus regression tests.

Takeover of #2637.

Co-authored-by: JavanC <JavanC@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:34:22 -07:00
5 changed files with 62 additions and 58 deletions
+13 -9
View File
@@ -1025,6 +1025,10 @@ async function extractForSlugs(
let linksCreated = 0;
let timelineCreated = 0;
let pagesProcessed = 0;
// #2636: successfully processed pages get their extraction watermark
// stamped after the final flush (mode 'all' only — a partial-mode run
// hasn't done the full extraction the watermark asserts).
const processedRefs: Array<{ slug: string; source_id: string }> = [];
// Issue #972: read the basename flag once per extract run.
const globalBasename = await isGlobalBasenameEnabled(engine);
@@ -1113,6 +1117,7 @@ async function extractForSlugs(
}
pagesProcessed++;
if (!dryRun) processedRefs.push({ slug, source_id: sourceId ?? 'default' });
} catch { /* skip unreadable */ }
progress.tick(1);
},
@@ -1120,6 +1125,13 @@ async function extractForSlugs(
await flushLinks();
await flushTimeline();
// #2636: the Dream cycle disables sync's inline extraction and routes
// changed slugs through this incremental path — without a stamp here,
// those pages never get links_extracted_at and stay permanently visible
// to `extract --stale` / doctor. Stamp only after BOTH batches flushed.
if (!dryRun && mode === 'all') {
await stampExtracted(engine, processedRefs);
}
progress.finish();
if (!jsonMode) {
@@ -1743,15 +1755,7 @@ async function extractStaleFromDB(
// `page.updated_at.toISOString()` — the JS Date is ms-truncated, so the
// µs-precision DB updated_at stayed strictly greater and the page never
// cleared on Postgres. Stamping the exact value makes them equal.
//
// Version-arm floor: a page last edited BEFORE LINK_EXTRACTOR_VERSION_TS
// would otherwise be stamped below the version watermark and stay
// permanently stale (`links_extracted_at < versionTs` re-fires every run).
// Stamp max(updated_at, versionTs) — versionTs is always a past release
// date, so a concurrent edit's now() still exceeds the stamp and D4 holds.
// Tie at ms precision picks updated_at_iso (its µs ≥ versionTs's .000000).
const stampTs = new Date(page.updated_at_iso) >= new Date(versionTs) ? page.updated_at_iso : versionTs;
processedRefs.push({ slug: page.slug, source_id: page.source_id, extractedAt: stampTs });
processedRefs.push({ slug: page.slug, source_id: page.source_id, extractedAt: page.updated_at_iso });
}
// Flush NON-swallowing (CDX-4): a throw here propagates out of the sweep so
+4 -13
View File
@@ -28,7 +28,7 @@ import { ensureWellFormed } from './text-safe.ts';
* OR updated_at > links_extracted_at`. It is an ISO-8601 string (NOT a number) —
* the column is TIMESTAMPTZ and the predicate binds it as `::timestamptz`.
*/
export const LINK_EXTRACTOR_VERSION_TS = '2026-07-21T00:00:00Z';
export const LINK_EXTRACTOR_VERSION_TS = '2026-05-31T00:00:00Z';
// ─── Entity references ──────────────────────────────────────────
@@ -80,10 +80,10 @@ export type LinkResolutionType = 'qualified' | 'unqualified';
* Directory prefix whitelist. These are the top-level slug dirs the extractor
* recognizes as entity references. Upstream canonical + our extensions:
* - Gbrain canonical: people, companies, meetings, concepts, deal, civic, project, source, media, yc, projects
* - Our domain extensions: tech, finance, personal, openclaw, ops (domain-organized wikis)
* - Our domain extensions: tech, finance, personal, openclaw (domain-organized wikis)
* - Our entity prefix: entities (we kept some legacy entities/projects/ pages)
*/
const DIR_PATTERN = '(?:people|companies|meetings|concepts|deal|civic|project|projects|source|media|yc|tech|finance|personal|openclaw|entities|ops)';
const DIR_PATTERN = '(?:people|companies|meetings|concepts|deal|civic|project|projects|source|media|yc|tech|finance|personal|openclaw|entities)';
/**
* Match `[Name](path)` markdown links pointing to entity directories.
@@ -865,16 +865,7 @@ export function queryBasenameIndex(idx: Map<string, string[]>, name: string): st
if (!name || typeof name !== 'string') return [];
const trimmed = name.trim();
if (!trimmed) return [];
let hit = idx.get(trimmed) ?? idx.get(trimmed.toLowerCase()) ?? idx.get(normalizeBasename(trimmed));
// Issue #2576 bug 2: path-style refs (`runbooks/2026-05-01-x`) from dirs
// outside DIR_PATTERN reach here, but normalizeBasename strips slashes
// into a garbage key (`runbooks2026-05-01-x`) that can never hit the
// tail-keyed index. Fall back to the path tail so qualified refs resolve
// by basename like everything else.
if (!hit && trimmed.includes('/')) {
const tail = trimmed.slice(trimmed.lastIndexOf('/') + 1).trim();
if (tail) hit = idx.get(tail) ?? idx.get(tail.toLowerCase()) ?? idx.get(normalizeBasename(tail));
}
const hit = idx.get(trimmed) ?? idx.get(trimmed.toLowerCase()) ?? idx.get(normalizeBasename(trimmed));
return hit ? [...hit].sort(basenameSort) : [];
}
+45
View File
@@ -60,6 +60,51 @@ async function seedPage(slug: string, body: string): Promise<void> {
}
describe('runExtractCore — incremental cycle path (#417)', () => {
test('Dream incremental all-mode stamps the source-scoped extraction watermark (#2636)', async () => {
await engine.executeRaw(
`INSERT INTO sources (id, name, local_path) VALUES ($1, $2, $3)`,
['repo-a', 'repo-a', tempDir],
);
await engine.putPage('people/alice-example', {
type: 'person',
title: 'alice-example',
compiled_truth: '# alice',
timeline: '',
frontmatter: {},
content_hash: 'h',
}, { sourceId: 'repo-a' });
writeFileSync(join(tempDir, 'people/alice-example.md'), '# alice');
await runExtractCore(engine as unknown as BrainEngine, {
mode: 'all',
dir: tempDir,
slugs: ['people/alice-example'],
sourceId: 'repo-a',
});
const rows = await engine.executeRaw<{ links_extracted_at: string | null }>(
`SELECT links_extracted_at FROM pages WHERE slug = $1 AND source_id = $2`,
['people/alice-example', 'repo-a'],
);
expect(rows[0]?.links_extracted_at).not.toBeNull();
expect(await engine.countStalePagesForExtraction({ sourceId: 'repo-a' })).toBe(0);
});
test('Dream incremental dry-run does NOT stamp the watermark', async () => {
await seedPage('people/alice-example', '# alice');
await runExtractCore(engine as unknown as BrainEngine, {
mode: 'all',
dir: tempDir,
slugs: ['people/alice-example'],
dryRun: true,
});
const rows = await engine.executeRaw<{ links_extracted_at: string | null }>(
`SELECT links_extracted_at FROM pages WHERE slug = $1`,
['people/alice-example'],
);
expect(rows[0]?.links_extracted_at ?? null).toBeNull();
});
test('1. slugs: [] returns immediately with zero counts (early-return path)', async () => {
await seedPage('people/alice-example', '# alice');
const result = await runExtractCore(engine as unknown as BrainEngine, {
-18
View File
@@ -209,24 +209,6 @@ describe('gbrain extract --stale', () => {
expect(usRows[0]?.eq).toBe(true);
});
test('version-arm floor: page edited BEFORE LINK_EXTRACTOR_VERSION_TS clears after --stale (issue #2576 bug 3)', async () => {
// A page whose updated_at predates the version watermark used to be
// stamped at its updated_at (< versionTs), so the version arm re-fired
// every run — permanently stale. The sweep now floors the stamp at
// versionTs. (The #1768 test above also covers this since the v0.42.x
// VERSION_TS bump moved its date below the watermark, but this pins the
// behavior explicitly so a date "repair" there can't drop coverage.)
await engine.putPage('people/alice', personPage('Alice'));
await engine.executeRaw(`UPDATE pages SET updated_at = '2000-01-01T00:00:00Z' WHERE slug = 'people/alice'`);
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(1);
await runExtract(engine, ['--stale']);
// Pre-floor this stayed 1 forever (stamp < versionTs → version arm re-fires).
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(0);
await runExtract(engine, ['--stale']);
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(0);
});
test('CDX-4 (D2): a link-flush throw aborts the sweep and leaves pages UNSTAMPED', async () => {
await engine.putPage('people/alice', personPage('Alice'));
await engine.putPage('companies/acme', companyPage('Acme', '[Alice](people/alice) founded [Acme](companies/acme).'));
-18
View File
@@ -140,15 +140,6 @@ describe('extractEntityRefs', () => {
expect(wikiRefs[0].needsResolution).toBe(true);
});
test('recognizes ops/ qualified wikilinks (issue #2576 bug 2)', () => {
// `ops` was missing from DIR_PATTERN, so [[ops/...]] fell through to
// the generic 2c pass (needsResolution) instead of being a real ref.
const refs = extractEntityRefs('Deployed via [[ops/services/pointer-agent]].');
expect(refs.length).toBe(1);
expect(refs[0].slug).toBe('ops/services/pointer-agent');
expect(refs[0].needsResolution).toBeUndefined();
});
test('skips qualified-syntax tokens (those belong to 2a)', () => {
// [[wiki:topics/ai]] looks like 2a's qualified shape — even though
// it wouldn't satisfy DIR_PATTERN, 2c must not claim it either
@@ -1078,15 +1069,6 @@ describe('makeResolver — fallback chain', () => {
]);
});
test('resolveBasenameMatches: path-style ref falls back to the tail (issue #2576 bug 2)', async () => {
// normalizeBasename strips slashes, so `runbooks/2026-05-01-pointer-agent`
// used to normalize to a garbage key that never hit the tail-keyed index.
const engine = makeFakeEngineWithSlugs(['ops/changes/2026-05-01-pointer-agent']);
const r = makeResolver(engine);
expect(await r.resolveBasenameMatches!('runbooks/2026-05-01-pointer-agent'))
.toEqual(['ops/changes/2026-05-01-pointer-agent']);
});
test('resolveBasenameMatches: case-insensitive fallback', async () => {
const engine = makeFakeEngineWithSlugs(['companies/fast-weigh']);
const r = makeResolver(engine);