mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-17 10:22:34 +00:00
Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3fa01a4538 | ||
|
|
ded4aeaeae |
@@ -186,6 +186,18 @@ export function isSourceStale(src: SourceRow, now = Date.now(), floorMin = FULL_
|
||||
return ageMin >= floorMin;
|
||||
}
|
||||
|
||||
/**
|
||||
* #2060: count sources past the per-source cycle freshness floor. Consumed
|
||||
* by autopilot's dispatch decision — a stale source forces the fanout path
|
||||
* even when the doctor plan is small (score 70–94, plan ≤ 3, est < 300s),
|
||||
* so targeted mode can't leave cycle_freshness stale indefinitely.
|
||||
* dispatchPerSource's own throttles (skipped_fresh / fanoutMax / failure
|
||||
* cooldown) bound the resulting work.
|
||||
*/
|
||||
export function countStaleSources(sources: SourceRow[], now = Date.now(), floorMin = FULL_CYCLE_FLOOR_MIN): number {
|
||||
return sources.filter((s) => isSourceStale(s, now, floorMin)).length;
|
||||
}
|
||||
|
||||
/**
|
||||
* Most recent SUCCESSFUL cycle for a source. Prefers `last_source_cycle_at`
|
||||
* (per-source phases, written by the split cycle) and falls back to the legacy
|
||||
|
||||
@@ -901,13 +901,27 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
|
||||
const FULL_CYCLE_FLOOR_MIN = 60;
|
||||
const minutesSinceLastFull = (Date.now() - lastFullCycleAt) / 60000;
|
||||
|
||||
// #2060: stale per-source cycle freshness is a dispatch input. Without
|
||||
// it, a brain sitting at score 70–94 with a small targeted plan (≤3
|
||||
// steps, <300s) stays in targeted mode indefinitely and no per-source
|
||||
// cycle is ever dispatched — cycle_freshness never advances. A stale
|
||||
// source forces the fanout path; dispatchPerSource's throttles
|
||||
// (skipped_fresh / fanoutMax / failure cooldown) bound the work.
|
||||
// Fail-open to 0: a read failure must not block dispatch.
|
||||
let staleCycleSources = 0;
|
||||
try {
|
||||
const { countStaleSources } = await import('./autopilot-fanout.ts');
|
||||
staleCycleSources = countStaleSources(await engine.listAllSources({ localPathOnly: true }));
|
||||
} catch { /* fail-open: freshness is a dispatch hint, not a gate */ }
|
||||
|
||||
const shouldFullCycle =
|
||||
(score >= 95 && plan.length === 0 && minutesSinceLastFull >= FULL_CYCLE_FLOOR_MIN) ||
|
||||
plan.length > 3 ||
|
||||
estTotal >= 300 ||
|
||||
score < 70;
|
||||
score < 70 ||
|
||||
staleCycleSources > 0;
|
||||
|
||||
const shouldSleep = score >= 95 && plan.length === 0 && minutesSinceLastFull < FULL_CYCLE_FLOOR_MIN;
|
||||
const shouldSleep = score >= 95 && plan.length === 0 && minutesSinceLastFull < FULL_CYCLE_FLOOR_MIN && staleCycleSources === 0;
|
||||
|
||||
if (shouldSleep) {
|
||||
if (jsonMode) {
|
||||
|
||||
+3
-22
@@ -433,10 +433,7 @@ export async function extractLinksFromFile(
|
||||
async resolve(name: string, dirHint?: string | string[]): Promise<string | null> {
|
||||
if (!name) return null;
|
||||
const trimmed = name.trim();
|
||||
// Same broadened slug-shape as makeResolver step 1: accepts
|
||||
// digit-leading folders (`90-people/nicolai`) and nested paths.
|
||||
// Exact Set membership guards it — no false positives.
|
||||
if (/\//.test(trimmed) && /^[a-z0-9][a-z0-9/_-]*$/.test(trimmed) && allSlugs.has(trimmed)) {
|
||||
if (/^[a-z][a-z0-9-]*\/[a-z0-9][a-z0-9-]*$/.test(trimmed) && allSlugs.has(trimmed)) {
|
||||
return trimmed;
|
||||
}
|
||||
const hints = Array.isArray(dirHint) ? dirHint : (dirHint ? [dirHint] : []);
|
||||
@@ -585,17 +582,6 @@ export interface ExtractOpts {
|
||||
* before (single-'default'-source brains unaffected).
|
||||
*/
|
||||
sourceId?: string;
|
||||
/**
|
||||
* v0.42 — also extract frontmatter links on the incremental (slugs) path.
|
||||
* `extractForSlugs` extracts BODY links only by default; set this true to also
|
||||
* parse each changed page's frontmatter so `sources:`/`related:` edges stay fresh
|
||||
* when YAML is edited externally and synced in. Applied PER changed page, so the
|
||||
* incremental walk stays bounded (no switch to a full DB scan). Only honored on
|
||||
* the incremental path (`slugs` defined); the full-walk path already covers
|
||||
* frontmatter via its own dispatch. Gated upstream by the config key
|
||||
* `autopilot.incremental_extract_include_frontmatter` (default off).
|
||||
*/
|
||||
includeFrontmatter?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -634,7 +620,7 @@ export async function runExtractCore(engine: BrainEngine, opts: ExtractOpts): Pr
|
||||
// Nothing changed — skip entirely.
|
||||
return result;
|
||||
}
|
||||
const r = await extractForSlugs(engine, opts.dir, opts.slugs, opts.mode, dryRun, jsonMode, workers, opts.signal, opts.sourceId, opts.includeFrontmatter);
|
||||
const r = await extractForSlugs(engine, opts.dir, opts.slugs, opts.mode, dryRun, jsonMode, workers, opts.signal, opts.sourceId);
|
||||
result.links_created = r.links_created;
|
||||
result.timeline_entries_created = r.timeline_created;
|
||||
result.pages_processed = r.pages;
|
||||
@@ -1025,11 +1011,6 @@ async function extractForSlugs(
|
||||
signal?: AbortSignal,
|
||||
// #1747/#1503: stamp resolved brain source id on batch rows (see ExtractOpts.sourceId).
|
||||
sourceId?: string,
|
||||
// v0.42: when true, also extract frontmatter links per changed page so
|
||||
// externally-edited YAML (`sources:`/`related:`) stays fresh on the cycle.
|
||||
// Default false preserves the body-only incremental behavior. Gated upstream
|
||||
// by `autopilot.incremental_extract_include_frontmatter`.
|
||||
includeFrontmatter: boolean = false,
|
||||
): Promise<{ links_created: number; timeline_created: number; pages: number }> {
|
||||
// Build the full slug set for link resolution (fast: just readdir, no file reads)
|
||||
const allFiles = walkMarkdownFiles(brainDir);
|
||||
@@ -1104,7 +1085,7 @@ async function extractForSlugs(
|
||||
const content = readFileSync(fullPath, 'utf-8');
|
||||
|
||||
if (doLinks) {
|
||||
const links = await extractLinksFromFile(content, relPath, allSlugs, { globalBasename, includeFrontmatter });
|
||||
const links = await extractLinksFromFile(content, relPath, allSlugs, { globalBasename });
|
||||
for (const link of links) {
|
||||
if (dryRun) {
|
||||
if (!jsonMode) console.log(` ${link.from_slug} → ${link.to_slug} (${link.link_type})`);
|
||||
|
||||
@@ -121,18 +121,6 @@ export interface GBrainConfig {
|
||||
/** Daily spend cap (USD); bounds drains/day = floor(cap / ~$0.30). Default 2.0. */
|
||||
max_usd_per_day?: number;
|
||||
};
|
||||
/**
|
||||
* v0.42 — keep frontmatter links fresh on the incremental cycle. The cycle's
|
||||
* extract phase re-extracts only the slugs a sync changed, but `extractForSlugs`
|
||||
* extracts BODY links only — frontmatter (`sources:`/`related:` etc.) link edges
|
||||
* silently drift stale when a page's YAML is edited externally and synced in.
|
||||
* Set true to also extract frontmatter links per changed page each cycle, keeping
|
||||
* externally-edited YAML edges fresh without a full rescan. Default false
|
||||
* (preserves current behavior). Read via the file/env/DB plane in the cycle's
|
||||
* extract dispatch. Disable/enable with
|
||||
* `gbrain config set autopilot.incremental_extract_include_frontmatter <bool>`.
|
||||
*/
|
||||
incremental_extract_include_frontmatter?: boolean;
|
||||
};
|
||||
eval?: {
|
||||
/** false disables capture entirely. Defaults to true. */
|
||||
|
||||
+19
-26
@@ -854,7 +854,10 @@ interface SyncPhaseResult extends PhaseResult {
|
||||
/**
|
||||
* Resolve the source id for a brain directory by looking up the sources
|
||||
* table. Returns undefined when no registered source matches (falls back
|
||||
* to pre-v0.18 global config.sync.* keys).
|
||||
* to pre-v0.18 global config.sync.* keys) OR when MORE than one source
|
||||
* claims the path — an ambiguous match must not scope phases or stamp
|
||||
* last_full_cycle_at for an arbitrarily-picked source (the "freshness
|
||||
* stamp that lies" this resolution exists to prevent).
|
||||
*/
|
||||
async function resolveSourceForDir(
|
||||
engine: BrainEngine,
|
||||
@@ -865,10 +868,10 @@ async function resolveSourceForDir(
|
||||
if (brainDir === null) return undefined;
|
||||
try {
|
||||
const rows = await engine.executeRaw<{ id: string }>(
|
||||
`SELECT id FROM sources WHERE local_path = $1 LIMIT 1`,
|
||||
`SELECT id FROM sources WHERE local_path = $1 LIMIT 2`,
|
||||
[brainDir],
|
||||
);
|
||||
return rows[0]?.id;
|
||||
return rows.length === 1 ? rows[0]!.id : undefined;
|
||||
} catch {
|
||||
// sources table might not exist on very old brains — fall through.
|
||||
return undefined;
|
||||
@@ -996,21 +999,6 @@ async function runPhaseExtract(
|
||||
): Promise<PhaseResult> {
|
||||
try {
|
||||
const { runExtractCore } = await import('../commands/extract.ts');
|
||||
const { loadConfig } = await import('./config.ts');
|
||||
// Default off: the incremental cycle extracts body links only unless the
|
||||
// operator opts in to keeping externally-edited frontmatter links fresh too.
|
||||
// Both planes, file wins (env > file > DB precedence, per loadConfigWithEngine):
|
||||
// `gbrain config set autopilot.incremental_extract_include_frontmatter true`
|
||||
// writes the DB plane (engine.setConfig), so a file-plane-only read here
|
||||
// would make the documented enable command a silent no-op (#2120 class).
|
||||
const fileVal = loadConfig()?.autopilot?.incremental_extract_include_frontmatter;
|
||||
let includeFrontmatter = fileVal === true;
|
||||
if (fileVal === undefined) {
|
||||
try {
|
||||
includeFrontmatter =
|
||||
(await engine.getConfig('autopilot.incremental_extract_include_frontmatter')) === 'true';
|
||||
} catch { /* config table unreadable → default off */ }
|
||||
}
|
||||
// Extract is read-mostly against the filesystem + write to links table.
|
||||
// Honor dryRun by skipping with a 'skipped' entry: extract doesn't have
|
||||
// a clean dry-run mode today and runCycle should be honest about it.
|
||||
@@ -1031,7 +1019,6 @@ async function runPhaseExtract(
|
||||
slugs: changedSlugs, // undefined = full walk (first run / manual)
|
||||
signal,
|
||||
sourceId,
|
||||
includeFrontmatter, // honored on the incremental (slugs) path only
|
||||
});
|
||||
const linksCreated = result?.links_created ?? 0;
|
||||
const timelineCreated = result?.timeline_entries_created ?? 0;
|
||||
@@ -2381,17 +2368,23 @@ export async function runCycle(
|
||||
}
|
||||
|
||||
// v0.38 (codex r1 P0-5): persist per-source cycle completion timestamp
|
||||
// when the cycle ran successfully against an explicit source. Read by
|
||||
// autopilot's per-source freshness gate next tick. Skipped when:
|
||||
// - opts.sourceId is unset (legacy callers — autopilot still here)
|
||||
// - engine is null (no-DB path)
|
||||
// when the cycle ran successfully against a resolvable source. Read by
|
||||
// autopilot's per-source freshness gate next tick.
|
||||
//
|
||||
// #1993: keyed off `cycleSourceId` (opts.sourceId ?? the source resolved
|
||||
// from brainDir) — the SAME id the cycle locked + scoped its phases to —
|
||||
// NOT raw opts.sourceId. The autopilot's inline cycle sets brainDir but
|
||||
// passes no explicit sourceId, so keying off opts.sourceId alone never
|
||||
// advanced last_full_cycle_at and cycle_freshness stayed stale even while
|
||||
// the autopilot cycled every interval. Skipped when:
|
||||
// - no source resolves (engine null, or no checkout AND no opts.sourceId)
|
||||
// - status is 'failed' or 'skipped' (don't mark a non-run as fresh)
|
||||
// - dryRun (writes are out of scope)
|
||||
//
|
||||
// Best-effort: a write failure does NOT change the CycleReport status.
|
||||
// The cost of writing the wrong timestamp post-failure is higher than
|
||||
// the cost of missing a successful write (next cycle will redo work).
|
||||
if (opts.sourceId && engine && !dryRun && !aborted && (status === 'ok' || status === 'clean' || status === 'partial')) {
|
||||
if (cycleSourceId && engine && !dryRun && !aborted && (status === 'ok' || status === 'clean' || status === 'partial')) {
|
||||
try {
|
||||
const nowIso = new Date().toISOString();
|
||||
// #2194 fix #3 (the cycle split): `last_source_cycle_at` is the NEW gate
|
||||
@@ -2401,13 +2394,13 @@ export async function runCycle(
|
||||
// phases (those gate on autopilot.last_global_at), so writing it on a
|
||||
// source-only cycle does not re-introduce the freshness poisoning codex
|
||||
// flagged in the rejected skip-based design.
|
||||
await engine.updateSourceConfig(opts.sourceId, {
|
||||
await engine.updateSourceConfig(cycleSourceId, {
|
||||
last_source_cycle_at: nowIso,
|
||||
last_full_cycle_at: nowIso,
|
||||
});
|
||||
} catch (e) {
|
||||
// Best-effort; cycle already succeeded by the time we get here.
|
||||
console.warn(`[cycle] failed to write last_source_cycle_at for source ${opts.sourceId}: ${e instanceof Error ? e.message : String(e)}`);
|
||||
console.warn(`[cycle] failed to write last_source_cycle_at for source ${cycleSourceId}: ${e instanceof Error ? e.message : String(e)}`);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -942,17 +942,8 @@ export function makeResolver(
|
||||
|
||||
const hints = Array.isArray(dirHint) ? dirHint : (dirHint ? [dirHint] : []);
|
||||
|
||||
// Step 1: already a slug? Try an exact page lookup for any slug-shaped
|
||||
// value (contains '/', slug charset). Broadened beyond the original
|
||||
// single-segment lowercase-leading form (`^[a-z][a-z0-9-]*\/[a-z0-9]...`)
|
||||
// to also accept digit-leading folders (`90-people/nicolai`,
|
||||
// `01-trading/...`) and nested paths (`a/b/c`) — common in PARA-numbered
|
||||
// vaults. This is an EXACT getPage match only — no fuzzy — so it never
|
||||
// produces a false positive; a non-existent slug just falls through to
|
||||
// the steps below. Fixes frontmatter `related: [[dir/slug]]` values
|
||||
// (unwrapped by unwrapWikilink) that name a real page the strict regex
|
||||
// could not reach and whose full-path fuzzy score is below threshold.
|
||||
if (/\//.test(trimmed) && /^[a-z0-9][a-z0-9/_-]*$/.test(trimmed)) {
|
||||
// Step 1: already a slug? (dir/name shape, lowercase, hyphenated)
|
||||
if (/^[a-z][a-z0-9-]*\/[a-z0-9][a-z0-9-]*$/.test(trimmed)) {
|
||||
const page = await engine.getPage(trimmed);
|
||||
if (page) {
|
||||
cache.set(cacheKey, trimmed);
|
||||
@@ -1012,25 +1003,6 @@ export function makeResolver(
|
||||
|
||||
// ─── Frontmatter extractor ──────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Unwrap an Obsidian `[[wikilink]]` frontmatter value to its bare link
|
||||
* target so the resolver (which expects bare titles / dir slugs) can match
|
||||
* it. Mainstream Obsidian authors frontmatter links as `related: ["[[Page]]"]`;
|
||||
* without this, the resolver treats the brackets as part of the value and a
|
||||
* `[[90-people/nicolai]]` is normalized into `90peoplenicolai`, so it never
|
||||
* resolves. Strips a trailing `|alias`, `#heading`, or `^block` suffix — the
|
||||
* link target only. The regex is anchored to a wholly-wrapped value
|
||||
* (`^\s*\[\[…\]\]\s*$`), so bare titles and any value not fully wrapped pass
|
||||
* through unchanged and existing behavior is preserved exactly.
|
||||
*/
|
||||
export function unwrapWikilink(value: string): string {
|
||||
const match = /^\s*\[\[(.+?)\]\]\s*$/.exec(value);
|
||||
if (!match) return value;
|
||||
// Take the link target: drop |alias, then #heading / ^block suffixes.
|
||||
const target = match[1].split('|')[0].split('#')[0].split('^')[0];
|
||||
return target.trim();
|
||||
}
|
||||
|
||||
export interface UnresolvedFrontmatterRef {
|
||||
/** The frontmatter field name. */
|
||||
field: string;
|
||||
@@ -1088,12 +1060,7 @@ export async function extractFrontmatterLinks(
|
||||
}
|
||||
if (!name) continue; // skip numbers, nulls, malformed objects
|
||||
|
||||
// Accept Obsidian `[[wikilink]]` values in frontmatter link fields by
|
||||
// unwrapping to the bare target before resolution. Bare titles pass
|
||||
// through unchanged; the original `name` is preserved for the
|
||||
// unresolved report and edge context.
|
||||
const linkTarget = unwrapWikilink(name);
|
||||
const resolved = await resolver.resolve(linkTarget, mapping.dirHint);
|
||||
const resolved = await resolver.resolve(name, mapping.dirHint);
|
||||
if (!resolved) {
|
||||
unresolved.push({ field, name });
|
||||
continue;
|
||||
|
||||
@@ -54,6 +54,20 @@ describe('autopilot.ts ↔ dispatchPerSource wiring', () => {
|
||||
expect(AUTOPILOT_SRC).toMatch(/lastFullCycleAt\s*=\s*Date\.now\(\)/);
|
||||
});
|
||||
|
||||
test('stale per-source cycle freshness is a shouldFullCycle input (#2060)', () => {
|
||||
// Targeted mode (score 70–94, plan ≤3, est <300s) must not be able to
|
||||
// starve per-source cycle dispatch: a stale source (per countStaleSources
|
||||
// over listAllSources) forces the fanout path, and the sleep gate must
|
||||
// not fire while stale sources exist. Without these terms, cycle
|
||||
// freshness never advances for a brain that always lands in targeted mode.
|
||||
expect(AUTOPILOT_SRC).toMatch(/countStaleSources/);
|
||||
const fullCycleDeclIdx = AUTOPILOT_SRC.indexOf('const shouldFullCycle');
|
||||
expect(fullCycleDeclIdx).toBeGreaterThan(-1);
|
||||
const decl = AUTOPILOT_SRC.slice(fullCycleDeclIdx, fullCycleDeclIdx + 700);
|
||||
expect(decl).toMatch(/staleCycleSources\s*>\s*0/);
|
||||
expect(decl).toMatch(/const shouldSleep[^;]*staleCycleSources\s*===\s*0/);
|
||||
});
|
||||
|
||||
test('does NOT regress to the single-job dispatch on the full-cycle path', () => {
|
||||
// Pre-PR: the shouldFullCycle branch did:
|
||||
// const job = await queue.add('autopilot-cycle', { repoPath }, {
|
||||
|
||||
@@ -14,6 +14,7 @@ import { describe, test, expect } from 'bun:test';
|
||||
import {
|
||||
readLastFullCycleAt,
|
||||
isSourceStale,
|
||||
countStaleSources,
|
||||
selectSourcesForDispatch,
|
||||
resolveFanoutMax,
|
||||
dispatchPerSource,
|
||||
@@ -74,6 +75,23 @@ describe('isSourceStale', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('countStaleSources (#2060 dispatch-decision input)', () => {
|
||||
const NOW = Date.parse('2026-05-22T12:00:00.000Z');
|
||||
test('counts never-cycled + past-floor sources, ignores fresh', () => {
|
||||
const sources = [
|
||||
src('never-cycled'), // stale (null)
|
||||
src('old', new Date(NOW - 2 * 60 * 60_000).toISOString()), // stale (2h)
|
||||
src('fresh', new Date(NOW - 30 * 60_000).toISOString()), // fresh (30min)
|
||||
];
|
||||
expect(countStaleSources(sources, NOW)).toBe(2);
|
||||
});
|
||||
test('returns 0 for all-fresh and for empty list', () => {
|
||||
const fresh = src('a', new Date(NOW - 10 * 60_000).toISOString());
|
||||
expect(countStaleSources([fresh], NOW)).toBe(0);
|
||||
expect(countStaleSources([], NOW)).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('selectSourcesForDispatch', () => {
|
||||
const NOW = Date.parse('2026-05-22T12:00:00.000Z');
|
||||
const fresh = (id: string, agoMin: number) =>
|
||||
|
||||
@@ -3,8 +3,10 @@
|
||||
* cycles. Closes codex round-1 P0-5 (write site for last_full_cycle_at
|
||||
* was unspecified pre-PR).
|
||||
*
|
||||
* Conditions for write:
|
||||
* - opts.sourceId is set (legacy callers without sourceId skip the write)
|
||||
* Conditions for write (keyed off `cycleSourceId` = opts.sourceId ?? the
|
||||
* source resolved from brainDir, so the autopilot's inline cycle — brainDir
|
||||
* set, no explicit sourceId — also advances the timestamp, #1993):
|
||||
* - a source resolves (explicit sourceId, or brainDir matches a source)
|
||||
* - engine is non-null (no-DB path skips)
|
||||
* - status is 'ok' | 'clean' | 'partial' (failed/skipped don't mark fresh)
|
||||
* - dryRun is false
|
||||
@@ -90,17 +92,45 @@ describe('runCycle last_full_cycle_at exit hook', () => {
|
||||
});
|
||||
});
|
||||
|
||||
test('legacy caller (no sourceId) does NOT write any source timestamp', async () => {
|
||||
test('no explicit sourceId but brainDir resolves a source → writes the resolved source timestamp', async () => {
|
||||
await withEnv({ GBRAIN_HOME: gbrainHome }, async () => {
|
||||
await seedSource('default-like');
|
||||
// No sourceId passed; should remain untouched.
|
||||
// The autopilot's inline cycle sets brainDir but passes no sourceId.
|
||||
// runCycle resolves the source from brainDir (local_path match) into
|
||||
// cycleSourceId and stamps last_full_cycle_at for it — otherwise
|
||||
// cycle_freshness reports the brain stale even while the autopilot
|
||||
// cycles every interval (#1993).
|
||||
await seedSource('resolved-from-dir'); // local_path = brainDir
|
||||
expect(await readLastFullCycleAt('resolved-from-dir')).toBeNull();
|
||||
|
||||
const t0 = Date.now();
|
||||
const report = await runCycle(engine, {
|
||||
brainDir,
|
||||
phases: ['lint'],
|
||||
});
|
||||
expect(['ok', 'clean']).toContain(report.status);
|
||||
|
||||
const after = await readLastFullCycleAt('resolved-from-dir');
|
||||
expect(after).not.toBeNull();
|
||||
expect(new Date(after!).getTime()).toBeGreaterThanOrEqual(t0);
|
||||
});
|
||||
});
|
||||
|
||||
test('no sourceId and brainDir matches no source → does not write', async () => {
|
||||
await withEnv({ GBRAIN_HOME: gbrainHome }, async () => {
|
||||
// A source exists but its local_path does NOT match brainDir, so
|
||||
// resolveSourceForDir returns undefined, cycleSourceId is undefined,
|
||||
// and no per-source timestamp is written.
|
||||
await engine.executeRaw(
|
||||
`INSERT INTO sources (id, name, local_path, config, archived, created_at)
|
||||
VALUES ('unmatched', 'unmatched', '/no/such/repo', '{}'::jsonb, false, NOW())
|
||||
ON CONFLICT (id) DO UPDATE SET local_path = EXCLUDED.local_path`,
|
||||
[],
|
||||
);
|
||||
await runCycle(engine, {
|
||||
brainDir,
|
||||
phases: ['lint'],
|
||||
});
|
||||
// No per-source write happens; default source's config stays empty.
|
||||
const after = await readLastFullCycleAt('default-like');
|
||||
expect(after).toBeNull();
|
||||
expect(await readLastFullCycleAt('unmatched')).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -191,37 +191,3 @@ describe('runExtractCore — incremental cycle path (#417)', () => {
|
||||
expect(result.links_created).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
describe('runExtractCore — incremental frontmatter gate (includeFrontmatter)', () => {
|
||||
// alice has a `source:` frontmatter edge but NO body links. The incremental
|
||||
// path extracts body links only by default, so the frontmatter edge is the
|
||||
// sole signal that distinguishes the gate off vs on.
|
||||
const aliceFm = '---\nsource: companies/acme-example\n---\n# alice';
|
||||
|
||||
test('9. default (flag omitted) does NOT extract frontmatter links on the incremental path', async () => {
|
||||
await seedPage('companies/acme-example', '# acme');
|
||||
await seedPage('people/alice-example', aliceFm);
|
||||
const result = await runExtractCore(engine as unknown as BrainEngine, {
|
||||
mode: 'all',
|
||||
dir: tempDir,
|
||||
slugs: ['people/alice-example'],
|
||||
});
|
||||
// alice's only potential edge is her frontmatter `source:`; with the gate off
|
||||
// it must not be extracted (preserves the body-only incremental behavior).
|
||||
expect(result.pages_processed).toBe(1);
|
||||
expect(result.links_created).toBe(0);
|
||||
});
|
||||
|
||||
test('10. includeFrontmatter: true extracts the frontmatter link on the incremental path', async () => {
|
||||
await seedPage('companies/acme-example', '# acme');
|
||||
await seedPage('people/alice-example', aliceFm);
|
||||
const result = await runExtractCore(engine as unknown as BrainEngine, {
|
||||
mode: 'all',
|
||||
dir: tempDir,
|
||||
slugs: ['people/alice-example'],
|
||||
includeFrontmatter: true,
|
||||
});
|
||||
// Same page, gate on → the `source:` frontmatter edge is now extracted.
|
||||
expect(result.pages_processed).toBe(1);
|
||||
expect(result.links_created).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -76,18 +76,6 @@ describe('extractLinksFromFile', () => {
|
||||
}
|
||||
});
|
||||
|
||||
it('resolves wrapped [[wikilink]] digit-leading slug-path in frontmatter (fs resolver, broadened step 1)', async () => {
|
||||
// Same bug class as makeResolver step 1 (#1983): the fs resolver's strict
|
||||
// `^[a-z]…` slug regex rejected digit-leading / nested paths, so a PARA-vault
|
||||
// `related: "[[90-people/nicolai]]"` never resolved even though the page exists.
|
||||
const content = '---\nrelated: "[[90-people/nicolai]]"\ntype: concept\n---\nContent.';
|
||||
const allSlugs = new Set(['wiki/note', '90-people/nicolai']);
|
||||
const links = await extractLinksFromFile(content, 'wiki/note.md', allSlugs, { includeFrontmatter: true });
|
||||
const related = links.filter(l => l.link_type === 'related_to');
|
||||
expect(related).toHaveLength(1);
|
||||
expect(related[0].to_slug).toBe('90-people/nicolai');
|
||||
});
|
||||
|
||||
it('frontmatter extraction is default OFF (back-compat)', async () => {
|
||||
// Without includeFrontmatter, fs-source no longer auto-extracts frontmatter.
|
||||
// Matches db-source behavior. User opts in with --include-frontmatter flag.
|
||||
|
||||
@@ -9,7 +9,6 @@ import {
|
||||
parseTimelineEntries,
|
||||
isAutoLinkEnabled,
|
||||
FRONTMATTER_LINK_MAP,
|
||||
unwrapWikilink,
|
||||
type SlugResolver,
|
||||
} from '../src/core/link-extraction.ts';
|
||||
import type { BrainEngine } from '../src/core/engine.ts';
|
||||
@@ -1292,175 +1291,3 @@ describe('parseTimelineEntries — Format 3: inline [Source: ..., YYYY-MM-DD] ci
|
||||
expect(parseTimelineEntries('[Source: import batch, 2025-07-01]')).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
// ─── Frontmatter [[wikilink]] + slug-path resolution ──────────────────────
|
||||
// Mainstream Obsidian authors frontmatter links as `related: ["[[Page]]"]`,
|
||||
// and PARA-numbered vaults use digit-leading / nested slug paths like
|
||||
// `[[90-people/nicolai]]`. Both were silently dropped: brackets were treated
|
||||
// as part of the value and the step-1 slug regex (`^[a-z]…`) rejected
|
||||
// digit-leading / nested paths, while full-path fuzzy scored below threshold.
|
||||
// Fix: unwrapWikilink() before resolution + an exact getPage() for any
|
||||
// slug-shaped value (exact-match only → no false positives).
|
||||
|
||||
describe('unwrapWikilink', () => {
|
||||
test('wrapped title → bare title', () => {
|
||||
expect(unwrapWikilink('[[Monday Range]]')).toBe('Monday Range');
|
||||
});
|
||||
test('wrapped slug-path (digit-leading folder) → bare slug', () => {
|
||||
expect(unwrapWikilink('[[90-people/nicolai]]')).toBe('90-people/nicolai');
|
||||
});
|
||||
test('wrapped nested slug-path → bare slug', () => {
|
||||
expect(unwrapWikilink('[[01-trading/wiki/strategies/opening-range-breakout]]'))
|
||||
.toBe('01-trading/wiki/strategies/opening-range-breakout');
|
||||
});
|
||||
test('strips |alias', () => {
|
||||
expect(unwrapWikilink('[[90-people/nicolai|Nicolai]]')).toBe('90-people/nicolai');
|
||||
});
|
||||
test('strips #heading', () => {
|
||||
expect(unwrapWikilink('[[Page#Section]]')).toBe('Page');
|
||||
});
|
||||
test('strips ^block', () => {
|
||||
expect(unwrapWikilink('[[Page^abc123]]')).toBe('Page');
|
||||
});
|
||||
test('surrounding whitespace tolerated', () => {
|
||||
expect(unwrapWikilink(' [[Page]] ')).toBe('Page');
|
||||
});
|
||||
test('bare title passes through unchanged', () => {
|
||||
expect(unwrapWikilink('Monday Range')).toBe('Monday Range');
|
||||
});
|
||||
test('bare slug passes through unchanged', () => {
|
||||
expect(unwrapWikilink('90-people/nicolai')).toBe('90-people/nicolai');
|
||||
});
|
||||
test('partially-wrapped value is NOT unwrapped (anchored)', () => {
|
||||
// Not a wholly-wrapped value → left intact so existing behavior is exact.
|
||||
expect(unwrapWikilink('see [[Page]] for detail')).toBe('see [[Page]] for detail');
|
||||
});
|
||||
});
|
||||
|
||||
describe('makeResolver — slug-path exact getPage (step 1 broadened)', () => {
|
||||
function fakeEngine(
|
||||
slugs: string[],
|
||||
fuzzyMap: Map<string, { slug: string; similarity: number }> = new Map(),
|
||||
): BrainEngine {
|
||||
const lookup = new Set(slugs);
|
||||
return {
|
||||
async getPage(slug: string) { return lookup.has(slug) ? { slug } as any : null; },
|
||||
async findByTitleFuzzy(name: string) { return fuzzyMap.get(name) ?? null; },
|
||||
async searchKeyword() { return []; },
|
||||
} as unknown as BrainEngine;
|
||||
}
|
||||
|
||||
test('digit-leading folder slug resolves via exact getPage', async () => {
|
||||
const r = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
expect(await r.resolve('90-people/nicolai')).toBe('90-people/nicolai');
|
||||
});
|
||||
|
||||
test('nested (>2 segment) slug resolves via exact getPage', async () => {
|
||||
const r = makeResolver(fakeEngine(['01-trading/wiki/strategies/opening-range-breakout']));
|
||||
expect(await r.resolve('01-trading/wiki/strategies/opening-range-breakout'))
|
||||
.toBe('01-trading/wiki/strategies/opening-range-breakout');
|
||||
});
|
||||
|
||||
test('regression: single-segment lowercase slug still resolves', async () => {
|
||||
const r = makeResolver(fakeEngine(['people/pedro']));
|
||||
expect(await r.resolve('people/pedro')).toBe('people/pedro');
|
||||
});
|
||||
|
||||
test('exact-only: slug-shaped value with no matching page falls through (no false positive)', async () => {
|
||||
// `90-people/ghost` is slug-shaped but absent → step-1 getPage misses,
|
||||
// no fuzzy hit → null. Never invents an edge.
|
||||
const r = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
expect(await r.resolve('90-people/ghost')).toBeNull();
|
||||
});
|
||||
|
||||
test('non-slug value still routes to fuzzy', async () => {
|
||||
const r = makeResolver(fakeEngine(
|
||||
['01-trading/monday-range'],
|
||||
new Map([['Monday Range', { slug: '01-trading/monday-range', similarity: 1 }]]),
|
||||
));
|
||||
expect(await r.resolve('Monday Range')).toBe('01-trading/monday-range');
|
||||
});
|
||||
});
|
||||
|
||||
describe('extractFrontmatterLinks — [[wikilink]] related: values (end-to-end)', () => {
|
||||
function fakeEngine(
|
||||
slugs: string[],
|
||||
fuzzyMap: Map<string, { slug: string; similarity: number }> = new Map(),
|
||||
): BrainEngine {
|
||||
const lookup = new Set(slugs);
|
||||
return {
|
||||
async getPage(slug: string) { return lookup.has(slug) ? { slug } as any : null; },
|
||||
async findByTitleFuzzy(name: string) { return fuzzyMap.get(name) ?? null; },
|
||||
async searchKeyword() { return []; },
|
||||
} as unknown as BrainEngine;
|
||||
}
|
||||
|
||||
test('wrapped slug-path related: resolves (the core win)', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
const { candidates, unresolved } = await extractFrontmatterLinks(
|
||||
'wiki/originals/ideas/note', 'note' as never,
|
||||
{ related: '[[90-people/nicolai]]' }, resolver,
|
||||
);
|
||||
expect(unresolved).toHaveLength(0);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0]).toMatchObject({
|
||||
fromSlug: 'wiki/originals/ideas/note',
|
||||
targetSlug: '90-people/nicolai',
|
||||
linkType: 'related_to',
|
||||
linkSource: 'frontmatter',
|
||||
});
|
||||
});
|
||||
|
||||
test('wrapped nested slug-path related: resolves', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['01-trading/wiki/strategies/opening-range-breakout']));
|
||||
const { candidates } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: ['[[01-trading/wiki/strategies/opening-range-breakout]]'] }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0].targetSlug).toBe('01-trading/wiki/strategies/opening-range-breakout');
|
||||
});
|
||||
|
||||
test('wrapped value with |alias resolves to the target', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
const { candidates } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: '[[90-people/nicolai|Nicolai]]' }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0].targetSlug).toBe('90-people/nicolai');
|
||||
});
|
||||
|
||||
test('regression: bare slug related: still resolves', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
const { candidates } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: '90-people/nicolai' }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0].targetSlug).toBe('90-people/nicolai');
|
||||
});
|
||||
|
||||
test('regression: wrapped title resolves via fuzzy (brackets harmless)', async () => {
|
||||
const resolver = makeResolver(fakeEngine(
|
||||
['01-trading/monday-range'],
|
||||
new Map([['Monday Range', { slug: '01-trading/monday-range', similarity: 1 }]]),
|
||||
));
|
||||
const { candidates } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: '[[Monday Range]]' }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(1);
|
||||
expect(candidates[0].targetSlug).toBe('01-trading/monday-range');
|
||||
});
|
||||
|
||||
test('unknown wrapped slug → unresolved (no crash), original value preserved', async () => {
|
||||
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
|
||||
const { candidates, unresolved } = await extractFrontmatterLinks(
|
||||
'wiki/note', 'note' as never,
|
||||
{ related: '[[99-archive/does-not-exist]]' }, resolver,
|
||||
);
|
||||
expect(candidates).toHaveLength(0);
|
||||
expect(unresolved).toHaveLength(1);
|
||||
expect(unresolved[0]).toEqual({ field: 'related', name: '[[99-archive/does-not-exist]]' });
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user