Compare commits

..
Author SHA1 Message Date
b05ebf6e4a fix(doctor): scope timeline labels to disambiguate entity coverage vs brain-score component (#2298)
doctor and the get_health CLI surface printed two different timeline
metrics under one ambiguous 'timeline' label: the entity-scoped
timeline_coverage fraction (eligible entity pages with a timeline entry)
and the whole-brain timeline_coverage_score brain-score component (all
pages with a timeline entry, 0-15). Different numerators AND
denominators, indistinguishable in output.

Label-only fix, scoring unchanged:
- graph_coverage check: 'entity timeline coverage N%'
- brain_score breakdown: 'timeline density (all pages) N/15'
- get_health CLI: 'Timeline coverage (entity pages)' plus a new
  'Timeline density (all pages): N/15' line when the score is present

Adds test/doctor-timeline-metric-labels-2298.test.ts pinning the
denominator semantics, the rendered doctor messages, and the CLI
guard matrix (fails on master, passes here).

Takeover of #2761.

Co-authored-by: TurgutKural <TurgutKural@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:18:54 -07:00
5 changed files with 162 additions and 310 deletions
+4 -1
View File
@@ -935,7 +935,10 @@ export function formatResult(opName: string, result: unknown): string {
lines.push(`Link coverage (entities): ${(h.link_coverage * 100).toFixed(1)}%`);
}
if (h.timeline_coverage !== undefined) {
lines.push(`Timeline coverage (entities): ${(h.timeline_coverage * 100).toFixed(1)}%`);
lines.push(`Timeline coverage (entity pages): ${(h.timeline_coverage * 100).toFixed(1)}%`);
}
if (h.timeline_coverage_score !== undefined) {
lines.push(`Timeline density (all pages): ${h.timeline_coverage_score}/15 (whole-brain brain-score component)`);
}
if (Array.isArray(h.most_connected) && h.most_connected.length > 0) {
lines.push('Most connected entities:');
-109
View File
@@ -151,100 +151,6 @@ export function shouldSpawnAutopilotWorker(args: string[]): boolean {
return !args.includes('--no-worker');
}
/**
* #1525 — positional subcommand translation.
*
* Pre-fix, `gbrain autopilot status` silently fell through to "start daemon"
* because `runAutopilot()` only branched on flag forms (`--status`, etc.).
* `status` was treated as a stray positional and ignored.
*
* This translator maps known positional subcommands to their flag form so
* `autopilot status` is equivalent to `autopilot --status`, then rejects
* any unrecognized positional with a fail-loud error before any side
* effect (lockfile, daemon spawn, sync dispatch) runs.
*
* Scope decisions:
* - Known aliases: `status` → `--status`, `install` → `--install`,
* `uninstall` → `--uninstall`, `start` → (drop; default daemon launch).
* - `stop` is intentionally NOT aliased here. Stopping a running daemon
* is a new behavior (read PID from lock, SIGTERM, drain) that deserves
* its own design and PR. Users typing `gbrain autopilot stop` today get
* the unknown-positional error with the canonical alternatives.
* - At most one positional allowed; multiple positionals fail loud.
*/
// Every flag that consumes the NEXT argv token. Missing one here makes the
// translator misread the flag's value as a positional subcommand and exit 2
// (e.g. `--install --target linux-cron`). Keep in sync with parseArg call sites.
const AUTOPILOT_VALUE_FLAGS = new Set(['--repo', '--interval', '--target']);
const AUTOPILOT_POSITIONAL_ALIASES: Record<string, string | null> = {
status: '--status',
install: '--install',
uninstall: '--uninstall',
start: null, // drop the positional; default behavior is daemon launch
};
export type PositionalTranslation =
| { ok: true; args: string[] }
| {
ok: false;
reason: 'unknown_subcommand' | 'multiple_subcommands';
message: string;
};
export function translatePositionalSubcommands(args: string[]): PositionalTranslation {
const out: string[] = [];
let positionalSeen = false;
let i = 0;
while (i < args.length) {
const a = args[i];
if (AUTOPILOT_VALUE_FLAGS.has(a)) {
// Pass through the flag and its value untouched. If the value is
// missing at end-of-argv, fall through so the existing parseArg
// path can report the broken usage.
out.push(a);
if (i + 1 < args.length) {
out.push(args[i + 1]);
i += 2;
} else {
i += 1;
}
continue;
}
if (a.startsWith('-')) {
out.push(a);
i += 1;
continue;
}
// Positional subcommand.
if (positionalSeen) {
const known = Object.keys(AUTOPILOT_POSITIONAL_ALIASES).join(', ');
return {
ok: false,
reason: 'multiple_subcommands',
message: `Multiple subcommands given. Use only one of: ${known}.`,
};
}
positionalSeen = true;
if (a in AUTOPILOT_POSITIONAL_ALIASES) {
const alias = AUTOPILOT_POSITIONAL_ALIASES[a];
if (alias) out.push(alias);
i += 1;
continue;
}
const known = Object.keys(AUTOPILOT_POSITIONAL_ALIASES).join(', ');
return {
ok: false,
reason: 'unknown_subcommand',
message:
`Unknown subcommand: \`${a}\`.\n` +
`Allowed subcommands: ${known}.\n` +
`Or use the flag form: --status, --install, --uninstall.\n` +
`Run \`gbrain autopilot --help\` for full usage.`,
};
}
return { ok: true, args: out };
}
export function isPidAlive(pid: number): boolean {
if (!Number.isFinite(pid) || pid <= 0) return false;
try {
@@ -457,11 +363,6 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
' gbrain autopilot --install [--repo <path>]\n' +
' gbrain autopilot --uninstall\n' +
' gbrain autopilot --status [--json]\n\n' +
'Subcommand aliases:\n' +
' gbrain autopilot status → --status\n' +
' gbrain autopilot install → --install\n' +
' gbrain autopilot uninstall → --uninstall\n' +
' gbrain autopilot start → (default daemon launch)\n\n' +
'Self-maintaining brain daemon. Runs the full maintenance cycle\n' +
'(lint + backlinks + sync + extract + embed + orphans) on an interval.\n\n' +
'For a one-shot cron-triggered cycle, see `gbrain dream`.',
@@ -469,16 +370,6 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
return;
}
// #1525: translate positional subcommands to their flag form BEFORE any
// side effect (lockfile, daemon spawn, sync dispatch). Unknown positionals
// fail loud here rather than silently starting the daemon.
const translated = translatePositionalSubcommands(args);
if (!translated.ok) {
console.error(translated.message);
process.exit(2);
}
args = translated.args;
if (args.includes('--install')) {
await installDaemon(engine, args);
return;
+3 -3
View File
@@ -5868,12 +5868,12 @@ export async function buildChecks(
message: `Only code/test fixture entity pages found (${entityCount}); graph_coverage not applicable`,
});
} else if (linkCoverage >= 0.5 && timelineCoverage >= 0.5) {
checks.push({ name: 'graph_coverage', status: 'ok', message: `Entity link coverage ${linkPct}%, timeline ${timelinePct}%` });
checks.push({ name: 'graph_coverage', status: 'ok', message: `Entity link coverage ${linkPct}%, entity timeline coverage ${timelinePct}%` });
} else {
checks.push({
name: 'graph_coverage',
status: 'warn',
message: `Entity link coverage ${linkPct}%, timeline ${timelinePct}% (${eligibleEntityCount} entity pages). Run: gbrain extract all`,
message: `Entity link coverage ${linkPct}%, entity timeline coverage ${timelinePct}% (${eligibleEntityCount} entity pages). Run: gbrain extract all`,
});
}
@@ -5885,7 +5885,7 @@ export async function buildChecks(
const parts = [
`embed ${health.embed_coverage_score}/35`,
`links ${health.link_density_score}/25`,
`timeline ${health.timeline_coverage_score}/15`,
`timeline density (all pages) ${health.timeline_coverage_score}/15`,
`orphans ${health.no_orphans_score}/15`,
`dead-links ${health.no_dead_links_score}/10`,
];
@@ -1,197 +0,0 @@
/**
* Tests for translatePositionalSubcommands() — the v0.41.x #1525 fix that
* prevents `gbrain autopilot status` from silently starting the daemon.
*
* IRON RULE regression guard: the exact ticket repro (`gbrain autopilot
* status`) MUST translate to `--status`, not fall through to the default
* daemon launch. Verified by the "ticket-exact repro" case below.
*/
import { describe, test, expect } from 'bun:test';
import { translatePositionalSubcommands } from '../src/commands/autopilot.ts';
describe('translatePositionalSubcommands — known aliases', () => {
test('IRON RULE — `autopilot status` translates to `--status` (ticket #1525 repro)', () => {
const r = translatePositionalSubcommands(['status']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--status']);
});
test('`install` translates to `--install`', () => {
const r = translatePositionalSubcommands(['install']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--install']);
});
test('`uninstall` translates to `--uninstall`', () => {
const r = translatePositionalSubcommands(['uninstall']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--uninstall']);
});
test('`start` drops the positional (default daemon launch)', () => {
const r = translatePositionalSubcommands(['start']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual([]);
});
test('`start --json` drops only the positional, keeps the flag', () => {
const r = translatePositionalSubcommands(['start', '--json']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--json']);
});
});
describe('translatePositionalSubcommands — flag/positional interleaving', () => {
test('`status --json` preserves the trailing flag', () => {
const r = translatePositionalSubcommands(['status', '--json']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--status', '--json']);
});
test('`--json status` preserves the leading flag', () => {
const r = translatePositionalSubcommands(['--json', 'status']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--json', '--status']);
});
test('`--repo /foo status` does not mis-classify the path as positional', () => {
const r = translatePositionalSubcommands(['--repo', '/foo', 'status']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--repo', '/foo', '--status']);
});
test('`--interval 300 install` does not mis-classify the number as positional', () => {
const r = translatePositionalSubcommands(['--interval', '300', 'install']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--interval', '300', '--install']);
});
test('`--install --target linux-cron` does not mis-classify the target as positional', () => {
// --target is installDaemon's value flag; its value must never be read
// as a positional subcommand (regression guard for the review fix).
const r = translatePositionalSubcommands(['--install', '--target', 'linux-cron']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--install', '--target', 'linux-cron']);
});
test('`install --target macos` keeps the alias translation and the target value', () => {
const r = translatePositionalSubcommands(['install', '--target', 'macos']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--install', '--target', 'macos']);
});
test('value-flag at end of argv with missing value passes through (so parseArg can report it)', () => {
const r = translatePositionalSubcommands(['--repo']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--repo']);
});
test('value-flag whose value looks like an alias is NOT translated', () => {
// `--repo status` means "use repo path 'status'", not "show status".
// Translator must not destructure the value of --repo.
const r = translatePositionalSubcommands(['--repo', 'status']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--repo', 'status']);
});
});
describe('translatePositionalSubcommands — pass-through cases', () => {
test('empty args returns empty args', () => {
const r = translatePositionalSubcommands([]);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual([]);
});
test('flag-only invocation passes through unchanged', () => {
const r = translatePositionalSubcommands(['--status', '--json']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['--status', '--json']);
});
test('short flag `-h` passes through unchanged', () => {
const r = translatePositionalSubcommands(['-h']);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(['-h']);
});
test('all known bare flags pass through unchanged', () => {
const flags = ['--help', '--install', '--uninstall', '--status', '--json', '--inline', '--no-worker'];
const r = translatePositionalSubcommands(flags);
expect(r.ok).toBe(true);
if (r.ok) expect(r.args).toEqual(flags);
});
});
describe('translatePositionalSubcommands — rejection of unknown positionals', () => {
test('unknown positional `foo` fails with reason=unknown_subcommand + structured message', () => {
const r = translatePositionalSubcommands(['foo']);
expect(r.ok).toBe(false);
if (!r.ok) {
expect(r.reason).toBe('unknown_subcommand');
expect(r.message).toContain('Unknown subcommand: `foo`');
expect(r.message).toContain('status');
expect(r.message).toContain('install');
expect(r.message).toContain('uninstall');
expect(r.message).toContain('--help');
}
});
test('unknown positional `stop` fails with reason=unknown_subcommand (NOT silently aliased)', () => {
// Stop is mentioned in the ticket but deliberately NOT aliased in this
// PR — stopping a running daemon is a new behavior, not just an alias.
// Until that feature lands separately, `stop` must fail loud rather
// than starting the daemon (the bug we're fixing).
const r = translatePositionalSubcommands(['stop']);
expect(r.ok).toBe(false);
if (!r.ok) {
expect(r.reason).toBe('unknown_subcommand');
expect(r.message).toContain('Unknown subcommand: `stop`');
}
});
test('unknown positional `status-detail` (close-but-not-matching) fails', () => {
const r = translatePositionalSubcommands(['status-detail']);
expect(r.ok).toBe(false);
if (!r.ok) {
expect(r.reason).toBe('unknown_subcommand');
expect(r.message).toContain('Unknown subcommand: `status-detail`');
}
});
test('multiple positionals fail with reason=multiple_subcommands (`start install`)', () => {
const r = translatePositionalSubcommands(['start', 'install']);
expect(r.ok).toBe(false);
if (!r.ok) {
expect(r.reason).toBe('multiple_subcommands');
expect(r.message).toContain('Multiple subcommands');
}
});
test('multiple positionals fail even when both are known aliases (`status install`)', () => {
const r = translatePositionalSubcommands(['status', 'install']);
expect(r.ok).toBe(false);
if (!r.ok) {
expect(r.reason).toBe('multiple_subcommands');
expect(r.message).toContain('Multiple subcommands');
}
});
test('known-then-unknown rejects with multiple_subcommands (first-positional-wins)', () => {
// First positional is known, second is not. Rejection comes from the
// multiple-positional rule, which fires before the unknown check; the
// intent is "only one subcommand allowed."
const r = translatePositionalSubcommands(['status', 'garbage']);
expect(r.ok).toBe(false);
if (!r.ok) expect(r.reason).toBe('multiple_subcommands');
});
test('unknown-then-known rejects on the unknown (unknown fires before second-positional check)', () => {
const r = translatePositionalSubcommands(['garbage', 'status']);
expect(r.ok).toBe(false);
if (!r.ok) {
expect(r.reason).toBe('unknown_subcommand');
expect(r.message).toContain('garbage');
}
});
});
@@ -0,0 +1,155 @@
/**
* Issue #2298 — timeline metric presentation contract.
*
* Authoritative upstream semantics (src/core/types.ts):
* - Metric A `timeline_coverage` (entity-scoped, fraction 01):
* eligible entity pages WITH a timeline entry / eligible entity pages
* -> surfaced by `graph_coverage` check AND `get_health` CLI entity line.
* - Metric B `timeline_coverage_score` (whole-brain, 015 brain-score component):
* all pages WITH a timeline entry / all pages
* -> surfaced by `brain_score` component breakdown AND (separately) CLI.
*
* The two have DIFFERENT numerators/denominators. This PR labels each
* explicitly and keeps BOTH the entity CLI line and the whole-brain line.
*
* Tests (no private EriadorMu data, no production/home DB, no network):
* - numeric denominator assertions (Metric A = 50%, Metric B = 4/15)
* - doctor rendered-message assertions (exact labels, no ambiguous old label)
* - CLI rendered-output assertions (exact lines, guard matrix)
* - red/green: same assertions FAIL on origin/master, PASS on this branch
*
* Scoring formula UNCHANGED. Canonical PGLite fixture via resetPgliteState.
*/
import { describe, expect, test, beforeAll, afterAll, beforeEach } from 'bun:test';
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
import { sqlQueryForEngine } from '../src/core/sql-query.ts';
import { resetPgliteState } from './helpers/reset-pglite.ts';
import { buildChecks } from '../src/commands/doctor.ts';
import { formatResult } from '../src/cli.ts';
let engine: PGLiteEngine;
async function seedFourPages(eng: PGLiteEngine): Promise<void> {
const sql = sqlQueryForEngine(eng);
// 2 eligible entity pages, 2 technical/non-entity pages.
// Only ONE entity page has a timeline entry; only ONE total page does.
await sql`
INSERT INTO pages (slug, source_id, type, title, compiled_truth, frontmatter, content_hash, created_at, updated_at)
VALUES
('acme-example', 'default', 'company', 'Acme', '', '{}', 'h1', now(), now()),
('alice-example', 'default', 'person', 'Alice', '', '{}', 'h2', now(), now()),
('technical-a', 'default', 'note', 'Tech A', '', '{}', 'h3', now(), now()),
('technical-b', 'default', 'note', 'Tech B', '', '{}', 'h4', now(), now())
`;
const companyId = (await sql`SELECT id FROM pages WHERE slug='acme-example'`)[0].id as number;
await sql`INSERT INTO timeline_entries (page_id, date, source, summary, detail)
VALUES (${companyId}, CURRENT_DATE, 'test', 'milestone', '{}')`;
}
beforeAll(async () => {
engine = new PGLiteEngine();
await engine.connect({});
await engine.initSchema();
});
afterAll(async () => {
await engine.disconnect();
});
beforeEach(async () => {
await resetPgliteState(engine);
});
describe('issue #2298 — numeric denominator semantics', () => {
test('entity timeline coverage = 1/2 = 50% (2 eligible entities, 1 with timeline)', async () => {
await seedFourPages(engine);
const health = await engine.getHealth();
expect(health.timeline_coverage).toBeDefined();
expect(Math.round((health.timeline_coverage ?? 0) * 100)).toBe(50);
});
test('whole-brain timeline density = 1/4 -> score 4/15 (4 total pages, 1 with timeline)', async () => {
await seedFourPages(engine);
const health = await engine.getHealth();
expect(health.timeline_coverage_score).toBeDefined();
expect(health.timeline_coverage_score).toBe(4);
});
test('the two metrics use independent denominators', async () => {
await seedFourPages(engine);
const health = await engine.getHealth();
expect(Math.round((health.timeline_coverage ?? 0) * 100)).toBe(50);
expect(health.timeline_coverage_score ?? 0).toBe(4);
// 50% (entity, /2) != 26.7% (whole-brain, /4). Provably distinct.
expect(Math.round(((health.timeline_coverage_score ?? 0) / 15) * 100)).not.toBe(50);
});
});
describe('issue #2298 — doctor rendered-message contract', () => {
test('graph_coverage renders entity-scoped label with 50%', async () => {
await seedFourPages(engine);
const checks = await buildChecks(engine, [], null);
const graph = checks.find((c) => c.name === 'graph_coverage');
expect(graph, 'graph_coverage check must be present').toBeDefined();
expect(graph!.message).toContain('entity timeline coverage 50%');
// ambiguous old label must NOT be present
expect(graph!.message).not.toMatch(/timeline 50%/);
expect(graph!.message).not.toMatch(/timeline \(entity, brain score\)/);
});
test('brain_score renders whole-brain density label 4/15', async () => {
await seedFourPages(engine);
const checks = await buildChecks(engine, [], null);
const brain = checks.find((c) => c.name === 'brain_score');
expect(brain, 'brain_score check must be present').toBeDefined();
expect(brain!.message).toContain('timeline density (all pages) 4/15');
// wrong labels must NOT be present
expect(brain!.message).not.toMatch(/timeline 4\/15/);
expect(brain!.message).not.toMatch(/timeline \(entity, brain score\)/);
// brain-score component must NOT carry the word "entity" (it is whole-brain)
const timelinePart = brain!.message.split('timeline density (all pages) 4/15')[0] + 'timeline density (all pages) 4/15';
expect(timelinePart).not.toMatch(/entity/);
});
});
describe('issue #2298 — CLI get_health rendered-output contract', () => {
function fakeHealth(overrides: Record<string, unknown>): any {
return {
embed_coverage: 1, missing_embeddings: 0, stale_pages: 0, orphan_pages: 0,
link_coverage: 1, timeline_coverage: 0.5, timeline_coverage_score: 4,
most_connected: [], ...overrides,
};
}
test('both entity and whole-brain lines render, no undefined/15', () => {
const out = formatResult('get_health', fakeHealth({}));
expect(out).toContain('Timeline coverage (entity pages): 50.0%');
expect(out).toContain('Timeline density (all pages): 4/15');
expect(out).not.toContain('undefined/15');
expect(out).not.toContain('Timeline coverage (entities)');
expect(out).not.toMatch(/timeline \(entity, brain score\)/);
expect(out).not.toMatch(/bare "timeline 4\/15"/);
});
test('guard matrix: entity present, whole-brain absent -> only entity line', () => {
const out = formatResult('get_health', fakeHealth({ timeline_coverage_score: undefined }));
expect(out).toContain('Timeline coverage (entity pages): 50.0%');
expect(out).not.toContain('Timeline density (all pages)');
expect(out).not.toContain('undefined/15');
});
test('guard matrix: whole-brain present, entity absent -> only whole-brain line', () => {
const out = formatResult('get_health', fakeHealth({ timeline_coverage: undefined }));
expect(out).toContain('Timeline density (all pages): 4/15');
expect(out).not.toContain('Timeline coverage (entity pages)');
expect(out).not.toContain('undefined/15');
});
test('guard matrix: both absent -> neither timeline line, never undefined/15', () => {
const out = formatResult('get_health', fakeHealth({ timeline_coverage: undefined, timeline_coverage_score: undefined }));
expect(out).not.toContain('Timeline coverage (entity pages)');
expect(out).not.toContain('Timeline density (all pages)');
expect(out).not.toContain('undefined/15');
});
});