mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-16 18:02:30 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b05ebf6e4a |
+4
-1
@@ -935,7 +935,10 @@ export function formatResult(opName: string, result: unknown): string {
|
||||
lines.push(`Link coverage (entities): ${(h.link_coverage * 100).toFixed(1)}%`);
|
||||
}
|
||||
if (h.timeline_coverage !== undefined) {
|
||||
lines.push(`Timeline coverage (entities): ${(h.timeline_coverage * 100).toFixed(1)}%`);
|
||||
lines.push(`Timeline coverage (entity pages): ${(h.timeline_coverage * 100).toFixed(1)}%`);
|
||||
}
|
||||
if (h.timeline_coverage_score !== undefined) {
|
||||
lines.push(`Timeline density (all pages): ${h.timeline_coverage_score}/15 (whole-brain brain-score component)`);
|
||||
}
|
||||
if (Array.isArray(h.most_connected) && h.most_connected.length > 0) {
|
||||
lines.push('Most connected entities:');
|
||||
|
||||
@@ -5868,12 +5868,12 @@ export async function buildChecks(
|
||||
message: `Only code/test fixture entity pages found (${entityCount}); graph_coverage not applicable`,
|
||||
});
|
||||
} else if (linkCoverage >= 0.5 && timelineCoverage >= 0.5) {
|
||||
checks.push({ name: 'graph_coverage', status: 'ok', message: `Entity link coverage ${linkPct}%, timeline ${timelinePct}%` });
|
||||
checks.push({ name: 'graph_coverage', status: 'ok', message: `Entity link coverage ${linkPct}%, entity timeline coverage ${timelinePct}%` });
|
||||
} else {
|
||||
checks.push({
|
||||
name: 'graph_coverage',
|
||||
status: 'warn',
|
||||
message: `Entity link coverage ${linkPct}%, timeline ${timelinePct}% (${eligibleEntityCount} entity pages). Run: gbrain extract all`,
|
||||
message: `Entity link coverage ${linkPct}%, entity timeline coverage ${timelinePct}% (${eligibleEntityCount} entity pages). Run: gbrain extract all`,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -5885,7 +5885,7 @@ export async function buildChecks(
|
||||
const parts = [
|
||||
`embed ${health.embed_coverage_score}/35`,
|
||||
`links ${health.link_density_score}/25`,
|
||||
`timeline ${health.timeline_coverage_score}/15`,
|
||||
`timeline density (all pages) ${health.timeline_coverage_score}/15`,
|
||||
`orphans ${health.no_orphans_score}/15`,
|
||||
`dead-links ${health.no_dead_links_score}/10`,
|
||||
];
|
||||
|
||||
@@ -1743,15 +1743,7 @@ async function extractStaleFromDB(
|
||||
// `page.updated_at.toISOString()` — the JS Date is ms-truncated, so the
|
||||
// µs-precision DB updated_at stayed strictly greater and the page never
|
||||
// cleared on Postgres. Stamping the exact value makes them equal.
|
||||
//
|
||||
// Version-arm floor: a page last edited BEFORE LINK_EXTRACTOR_VERSION_TS
|
||||
// would otherwise be stamped below the version watermark and stay
|
||||
// permanently stale (`links_extracted_at < versionTs` re-fires every run).
|
||||
// Stamp max(updated_at, versionTs) — versionTs is always a past release
|
||||
// date, so a concurrent edit's now() still exceeds the stamp and D4 holds.
|
||||
// Tie at ms precision picks updated_at_iso (its µs ≥ versionTs's .000000).
|
||||
const stampTs = new Date(page.updated_at_iso) >= new Date(versionTs) ? page.updated_at_iso : versionTs;
|
||||
processedRefs.push({ slug: page.slug, source_id: page.source_id, extractedAt: stampTs });
|
||||
processedRefs.push({ slug: page.slug, source_id: page.source_id, extractedAt: page.updated_at_iso });
|
||||
}
|
||||
|
||||
// Flush NON-swallowing (CDX-4): a throw here propagates out of the sweep so
|
||||
|
||||
@@ -28,7 +28,7 @@ import { ensureWellFormed } from './text-safe.ts';
|
||||
* OR updated_at > links_extracted_at`. It is an ISO-8601 string (NOT a number) —
|
||||
* the column is TIMESTAMPTZ and the predicate binds it as `::timestamptz`.
|
||||
*/
|
||||
export const LINK_EXTRACTOR_VERSION_TS = '2026-07-21T00:00:00Z';
|
||||
export const LINK_EXTRACTOR_VERSION_TS = '2026-05-31T00:00:00Z';
|
||||
|
||||
// ─── Entity references ──────────────────────────────────────────
|
||||
|
||||
@@ -80,10 +80,10 @@ export type LinkResolutionType = 'qualified' | 'unqualified';
|
||||
* Directory prefix whitelist. These are the top-level slug dirs the extractor
|
||||
* recognizes as entity references. Upstream canonical + our extensions:
|
||||
* - Gbrain canonical: people, companies, meetings, concepts, deal, civic, project, source, media, yc, projects
|
||||
* - Our domain extensions: tech, finance, personal, openclaw, ops (domain-organized wikis)
|
||||
* - Our domain extensions: tech, finance, personal, openclaw (domain-organized wikis)
|
||||
* - Our entity prefix: entities (we kept some legacy entities/projects/ pages)
|
||||
*/
|
||||
const DIR_PATTERN = '(?:people|companies|meetings|concepts|deal|civic|project|projects|source|media|yc|tech|finance|personal|openclaw|entities|ops)';
|
||||
const DIR_PATTERN = '(?:people|companies|meetings|concepts|deal|civic|project|projects|source|media|yc|tech|finance|personal|openclaw|entities)';
|
||||
|
||||
/**
|
||||
* Match `[Name](path)` markdown links pointing to entity directories.
|
||||
@@ -865,16 +865,7 @@ export function queryBasenameIndex(idx: Map<string, string[]>, name: string): st
|
||||
if (!name || typeof name !== 'string') return [];
|
||||
const trimmed = name.trim();
|
||||
if (!trimmed) return [];
|
||||
let hit = idx.get(trimmed) ?? idx.get(trimmed.toLowerCase()) ?? idx.get(normalizeBasename(trimmed));
|
||||
// Issue #2576 bug 2: path-style refs (`runbooks/2026-05-01-x`) from dirs
|
||||
// outside DIR_PATTERN reach here, but normalizeBasename strips slashes
|
||||
// into a garbage key (`runbooks2026-05-01-x`) that can never hit the
|
||||
// tail-keyed index. Fall back to the path tail so qualified refs resolve
|
||||
// by basename like everything else.
|
||||
if (!hit && trimmed.includes('/')) {
|
||||
const tail = trimmed.slice(trimmed.lastIndexOf('/') + 1).trim();
|
||||
if (tail) hit = idx.get(tail) ?? idx.get(tail.toLowerCase()) ?? idx.get(normalizeBasename(tail));
|
||||
}
|
||||
const hit = idx.get(trimmed) ?? idx.get(trimmed.toLowerCase()) ?? idx.get(normalizeBasename(trimmed));
|
||||
return hit ? [...hit].sort(basenameSort) : [];
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,155 @@
|
||||
/**
|
||||
* Issue #2298 — timeline metric presentation contract.
|
||||
*
|
||||
* Authoritative upstream semantics (src/core/types.ts):
|
||||
* - Metric A `timeline_coverage` (entity-scoped, fraction 0–1):
|
||||
* eligible entity pages WITH a timeline entry / eligible entity pages
|
||||
* -> surfaced by `graph_coverage` check AND `get_health` CLI entity line.
|
||||
* - Metric B `timeline_coverage_score` (whole-brain, 0–15 brain-score component):
|
||||
* all pages WITH a timeline entry / all pages
|
||||
* -> surfaced by `brain_score` component breakdown AND (separately) CLI.
|
||||
*
|
||||
* The two have DIFFERENT numerators/denominators. This PR labels each
|
||||
* explicitly and keeps BOTH the entity CLI line and the whole-brain line.
|
||||
*
|
||||
* Tests (no private EriadorMu data, no production/home DB, no network):
|
||||
* - numeric denominator assertions (Metric A = 50%, Metric B = 4/15)
|
||||
* - doctor rendered-message assertions (exact labels, no ambiguous old label)
|
||||
* - CLI rendered-output assertions (exact lines, guard matrix)
|
||||
* - red/green: same assertions FAIL on origin/master, PASS on this branch
|
||||
*
|
||||
* Scoring formula UNCHANGED. Canonical PGLite fixture via resetPgliteState.
|
||||
*/
|
||||
|
||||
import { describe, expect, test, beforeAll, afterAll, beforeEach } from 'bun:test';
|
||||
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
import { sqlQueryForEngine } from '../src/core/sql-query.ts';
|
||||
import { resetPgliteState } from './helpers/reset-pglite.ts';
|
||||
import { buildChecks } from '../src/commands/doctor.ts';
|
||||
import { formatResult } from '../src/cli.ts';
|
||||
|
||||
let engine: PGLiteEngine;
|
||||
|
||||
async function seedFourPages(eng: PGLiteEngine): Promise<void> {
|
||||
const sql = sqlQueryForEngine(eng);
|
||||
// 2 eligible entity pages, 2 technical/non-entity pages.
|
||||
// Only ONE entity page has a timeline entry; only ONE total page does.
|
||||
await sql`
|
||||
INSERT INTO pages (slug, source_id, type, title, compiled_truth, frontmatter, content_hash, created_at, updated_at)
|
||||
VALUES
|
||||
('acme-example', 'default', 'company', 'Acme', '', '{}', 'h1', now(), now()),
|
||||
('alice-example', 'default', 'person', 'Alice', '', '{}', 'h2', now(), now()),
|
||||
('technical-a', 'default', 'note', 'Tech A', '', '{}', 'h3', now(), now()),
|
||||
('technical-b', 'default', 'note', 'Tech B', '', '{}', 'h4', now(), now())
|
||||
`;
|
||||
const companyId = (await sql`SELECT id FROM pages WHERE slug='acme-example'`)[0].id as number;
|
||||
await sql`INSERT INTO timeline_entries (page_id, date, source, summary, detail)
|
||||
VALUES (${companyId}, CURRENT_DATE, 'test', 'milestone', '{}')`;
|
||||
}
|
||||
|
||||
beforeAll(async () => {
|
||||
engine = new PGLiteEngine();
|
||||
await engine.connect({});
|
||||
await engine.initSchema();
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
await engine.disconnect();
|
||||
});
|
||||
|
||||
beforeEach(async () => {
|
||||
await resetPgliteState(engine);
|
||||
});
|
||||
|
||||
describe('issue #2298 — numeric denominator semantics', () => {
|
||||
test('entity timeline coverage = 1/2 = 50% (2 eligible entities, 1 with timeline)', async () => {
|
||||
await seedFourPages(engine);
|
||||
const health = await engine.getHealth();
|
||||
expect(health.timeline_coverage).toBeDefined();
|
||||
expect(Math.round((health.timeline_coverage ?? 0) * 100)).toBe(50);
|
||||
});
|
||||
|
||||
test('whole-brain timeline density = 1/4 -> score 4/15 (4 total pages, 1 with timeline)', async () => {
|
||||
await seedFourPages(engine);
|
||||
const health = await engine.getHealth();
|
||||
expect(health.timeline_coverage_score).toBeDefined();
|
||||
expect(health.timeline_coverage_score).toBe(4);
|
||||
});
|
||||
|
||||
test('the two metrics use independent denominators', async () => {
|
||||
await seedFourPages(engine);
|
||||
const health = await engine.getHealth();
|
||||
expect(Math.round((health.timeline_coverage ?? 0) * 100)).toBe(50);
|
||||
expect(health.timeline_coverage_score ?? 0).toBe(4);
|
||||
// 50% (entity, /2) != 26.7% (whole-brain, /4). Provably distinct.
|
||||
expect(Math.round(((health.timeline_coverage_score ?? 0) / 15) * 100)).not.toBe(50);
|
||||
});
|
||||
});
|
||||
|
||||
describe('issue #2298 — doctor rendered-message contract', () => {
|
||||
test('graph_coverage renders entity-scoped label with 50%', async () => {
|
||||
await seedFourPages(engine);
|
||||
const checks = await buildChecks(engine, [], null);
|
||||
const graph = checks.find((c) => c.name === 'graph_coverage');
|
||||
expect(graph, 'graph_coverage check must be present').toBeDefined();
|
||||
expect(graph!.message).toContain('entity timeline coverage 50%');
|
||||
// ambiguous old label must NOT be present
|
||||
expect(graph!.message).not.toMatch(/timeline 50%/);
|
||||
expect(graph!.message).not.toMatch(/timeline \(entity, brain score\)/);
|
||||
});
|
||||
|
||||
test('brain_score renders whole-brain density label 4/15', async () => {
|
||||
await seedFourPages(engine);
|
||||
const checks = await buildChecks(engine, [], null);
|
||||
const brain = checks.find((c) => c.name === 'brain_score');
|
||||
expect(brain, 'brain_score check must be present').toBeDefined();
|
||||
expect(brain!.message).toContain('timeline density (all pages) 4/15');
|
||||
// wrong labels must NOT be present
|
||||
expect(brain!.message).not.toMatch(/timeline 4\/15/);
|
||||
expect(brain!.message).not.toMatch(/timeline \(entity, brain score\)/);
|
||||
// brain-score component must NOT carry the word "entity" (it is whole-brain)
|
||||
const timelinePart = brain!.message.split('timeline density (all pages) 4/15')[0] + 'timeline density (all pages) 4/15';
|
||||
expect(timelinePart).not.toMatch(/entity/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('issue #2298 — CLI get_health rendered-output contract', () => {
|
||||
function fakeHealth(overrides: Record<string, unknown>): any {
|
||||
return {
|
||||
embed_coverage: 1, missing_embeddings: 0, stale_pages: 0, orphan_pages: 0,
|
||||
link_coverage: 1, timeline_coverage: 0.5, timeline_coverage_score: 4,
|
||||
most_connected: [], ...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
test('both entity and whole-brain lines render, no undefined/15', () => {
|
||||
const out = formatResult('get_health', fakeHealth({}));
|
||||
expect(out).toContain('Timeline coverage (entity pages): 50.0%');
|
||||
expect(out).toContain('Timeline density (all pages): 4/15');
|
||||
expect(out).not.toContain('undefined/15');
|
||||
expect(out).not.toContain('Timeline coverage (entities)');
|
||||
expect(out).not.toMatch(/timeline \(entity, brain score\)/);
|
||||
expect(out).not.toMatch(/bare "timeline 4\/15"/);
|
||||
});
|
||||
|
||||
test('guard matrix: entity present, whole-brain absent -> only entity line', () => {
|
||||
const out = formatResult('get_health', fakeHealth({ timeline_coverage_score: undefined }));
|
||||
expect(out).toContain('Timeline coverage (entity pages): 50.0%');
|
||||
expect(out).not.toContain('Timeline density (all pages)');
|
||||
expect(out).not.toContain('undefined/15');
|
||||
});
|
||||
|
||||
test('guard matrix: whole-brain present, entity absent -> only whole-brain line', () => {
|
||||
const out = formatResult('get_health', fakeHealth({ timeline_coverage: undefined }));
|
||||
expect(out).toContain('Timeline density (all pages): 4/15');
|
||||
expect(out).not.toContain('Timeline coverage (entity pages)');
|
||||
expect(out).not.toContain('undefined/15');
|
||||
});
|
||||
|
||||
test('guard matrix: both absent -> neither timeline line, never undefined/15', () => {
|
||||
const out = formatResult('get_health', fakeHealth({ timeline_coverage: undefined, timeline_coverage_score: undefined }));
|
||||
expect(out).not.toContain('Timeline coverage (entity pages)');
|
||||
expect(out).not.toContain('Timeline density (all pages)');
|
||||
expect(out).not.toContain('undefined/15');
|
||||
});
|
||||
});
|
||||
@@ -209,24 +209,6 @@ describe('gbrain extract --stale', () => {
|
||||
expect(usRows[0]?.eq).toBe(true);
|
||||
});
|
||||
|
||||
test('version-arm floor: page edited BEFORE LINK_EXTRACTOR_VERSION_TS clears after --stale (issue #2576 bug 3)', async () => {
|
||||
// A page whose updated_at predates the version watermark used to be
|
||||
// stamped at its updated_at (< versionTs), so the version arm re-fired
|
||||
// every run — permanently stale. The sweep now floors the stamp at
|
||||
// versionTs. (The #1768 test above also covers this since the v0.42.x
|
||||
// VERSION_TS bump moved its date below the watermark, but this pins the
|
||||
// behavior explicitly so a date "repair" there can't drop coverage.)
|
||||
await engine.putPage('people/alice', personPage('Alice'));
|
||||
await engine.executeRaw(`UPDATE pages SET updated_at = '2000-01-01T00:00:00Z' WHERE slug = 'people/alice'`);
|
||||
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(1);
|
||||
|
||||
await runExtract(engine, ['--stale']);
|
||||
// Pre-floor this stayed 1 forever (stamp < versionTs → version arm re-fires).
|
||||
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(0);
|
||||
await runExtract(engine, ['--stale']);
|
||||
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(0);
|
||||
});
|
||||
|
||||
test('CDX-4 (D2): a link-flush throw aborts the sweep and leaves pages UNSTAMPED', async () => {
|
||||
await engine.putPage('people/alice', personPage('Alice'));
|
||||
await engine.putPage('companies/acme', companyPage('Acme', '[Alice](people/alice) founded [Acme](companies/acme).'));
|
||||
|
||||
@@ -140,15 +140,6 @@ describe('extractEntityRefs', () => {
|
||||
expect(wikiRefs[0].needsResolution).toBe(true);
|
||||
});
|
||||
|
||||
test('recognizes ops/ qualified wikilinks (issue #2576 bug 2)', () => {
|
||||
// `ops` was missing from DIR_PATTERN, so [[ops/...]] fell through to
|
||||
// the generic 2c pass (needsResolution) instead of being a real ref.
|
||||
const refs = extractEntityRefs('Deployed via [[ops/services/pointer-agent]].');
|
||||
expect(refs.length).toBe(1);
|
||||
expect(refs[0].slug).toBe('ops/services/pointer-agent');
|
||||
expect(refs[0].needsResolution).toBeUndefined();
|
||||
});
|
||||
|
||||
test('skips qualified-syntax tokens (those belong to 2a)', () => {
|
||||
// [[wiki:topics/ai]] looks like 2a's qualified shape — even though
|
||||
// it wouldn't satisfy DIR_PATTERN, 2c must not claim it either
|
||||
@@ -1078,15 +1069,6 @@ describe('makeResolver — fallback chain', () => {
|
||||
]);
|
||||
});
|
||||
|
||||
test('resolveBasenameMatches: path-style ref falls back to the tail (issue #2576 bug 2)', async () => {
|
||||
// normalizeBasename strips slashes, so `runbooks/2026-05-01-pointer-agent`
|
||||
// used to normalize to a garbage key that never hit the tail-keyed index.
|
||||
const engine = makeFakeEngineWithSlugs(['ops/changes/2026-05-01-pointer-agent']);
|
||||
const r = makeResolver(engine);
|
||||
expect(await r.resolveBasenameMatches!('runbooks/2026-05-01-pointer-agent'))
|
||||
.toEqual(['ops/changes/2026-05-01-pointer-agent']);
|
||||
});
|
||||
|
||||
test('resolveBasenameMatches: case-insensitive fallback', async () => {
|
||||
const engine = makeFakeEngineWithSlugs(['companies/fast-weigh']);
|
||||
const r = makeResolver(engine);
|
||||
|
||||
Reference in New Issue
Block a user