Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 bcdb435d73 fix(dream): deterministic last-writer attribution for colliding slugs
collectChildPutPageSlugs paired each slug to a jobId first-seen over an
unordered result set. When two children collide on a final slug, the pages
row holds the LAST put_page write, so the provenance stamp
(transcript_id/transcript_source/date) could attribute an arbitrary other
transcript. ORDER BY id + last-writer-wins aligns the stamp with the
surviving content; pinned by a collision test.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 12:14:19 -07:00
Garry TanandClaude Fable 5 093d693502 test(progress): assert net-zero live-reporter delta, not absolute zero
The 'only one process-level signal handler' test asserted
__liveReporterCountForTest() === 0, which encodes 'no other test file in
this bun process left a live reporter' — a shard-composition property,
not this test's invariant. PR #3096's new test files reshuffled the LPT
shard packing so a shardmate's live reporter now lands before
progress.test.ts in CI shard 5, failing the test deterministically
(both run attempts). Snapshot the count before the 50 lifecycles and
assert the delta is zero, mirroring the installedBefore baseline the
test already uses for the signal handler.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 11:22:03 -07:00
9268552d70 feat(dream): orchestrator-owned transcript metadata + transcriptSource discovery (#2285)
Takeover of #2286, rebased onto master so #1586's cycle-source scoping is
preserved (refs keep the cycleSourceId threading; the original branch's
rewritten collectChildPutPageSlugs hardcoded source_id 'default').

- DiscoveredTranscript carries transcriptSource, derived from the
  <corpus>/<source>/<date>/<id>.md layout; null for ad-hoc inputs.
- Transcript date inference now prefers the '| First message |' row in the
  transcript's '## Metadata' table (stable across mtime-restamping
  re-syncs), with the filename-regex date as fallback — implementing the
  cascade #2286 promised but didn't ship.
- The synthesize orchestrator stamps deterministic frontmatter
  (transcript_id, transcript_hash, transcript_source, chunk, date) through
  the existing #2569 stampDreamProvenance jsonb funnel — no second putPage
  write; reverseWriteRefs re-reads the row so DB and disk stay lockstep.
  Subagents retain authority over type/title/tags/body.

Co-authored-by: brettdavies <brettdavies@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:25:51 -07:00
15 changed files with 376 additions and 333 deletions
+3 -22
View File
@@ -433,10 +433,7 @@ export async function extractLinksFromFile(
async resolve(name: string, dirHint?: string | string[]): Promise<string | null> {
if (!name) return null;
const trimmed = name.trim();
// Same broadened slug-shape as makeResolver step 1: accepts
// digit-leading folders (`90-people/nicolai`) and nested paths.
// Exact Set membership guards it — no false positives.
if (/\//.test(trimmed) && /^[a-z0-9][a-z0-9/_-]*$/.test(trimmed) && allSlugs.has(trimmed)) {
if (/^[a-z][a-z0-9-]*\/[a-z0-9][a-z0-9-]*$/.test(trimmed) && allSlugs.has(trimmed)) {
return trimmed;
}
const hints = Array.isArray(dirHint) ? dirHint : (dirHint ? [dirHint] : []);
@@ -585,17 +582,6 @@ export interface ExtractOpts {
* before (single-'default'-source brains unaffected).
*/
sourceId?: string;
/**
* v0.42 — also extract frontmatter links on the incremental (slugs) path.
* `extractForSlugs` extracts BODY links only by default; set this true to also
* parse each changed page's frontmatter so `sources:`/`related:` edges stay fresh
* when YAML is edited externally and synced in. Applied PER changed page, so the
* incremental walk stays bounded (no switch to a full DB scan). Only honored on
* the incremental path (`slugs` defined); the full-walk path already covers
* frontmatter via its own dispatch. Gated upstream by the config key
* `autopilot.incremental_extract_include_frontmatter` (default off).
*/
includeFrontmatter?: boolean;
}
/**
@@ -634,7 +620,7 @@ export async function runExtractCore(engine: BrainEngine, opts: ExtractOpts): Pr
// Nothing changed — skip entirely.
return result;
}
const r = await extractForSlugs(engine, opts.dir, opts.slugs, opts.mode, dryRun, jsonMode, workers, opts.signal, opts.sourceId, opts.includeFrontmatter);
const r = await extractForSlugs(engine, opts.dir, opts.slugs, opts.mode, dryRun, jsonMode, workers, opts.signal, opts.sourceId);
result.links_created = r.links_created;
result.timeline_entries_created = r.timeline_created;
result.pages_processed = r.pages;
@@ -1025,11 +1011,6 @@ async function extractForSlugs(
signal?: AbortSignal,
// #1747/#1503: stamp resolved brain source id on batch rows (see ExtractOpts.sourceId).
sourceId?: string,
// v0.42: when true, also extract frontmatter links per changed page so
// externally-edited YAML (`sources:`/`related:`) stays fresh on the cycle.
// Default false preserves the body-only incremental behavior. Gated upstream
// by `autopilot.incremental_extract_include_frontmatter`.
includeFrontmatter: boolean = false,
): Promise<{ links_created: number; timeline_created: number; pages: number }> {
// Build the full slug set for link resolution (fast: just readdir, no file reads)
const allFiles = walkMarkdownFiles(brainDir);
@@ -1104,7 +1085,7 @@ async function extractForSlugs(
const content = readFileSync(fullPath, 'utf-8');
if (doLinks) {
const links = await extractLinksFromFile(content, relPath, allSlugs, { globalBasename, includeFrontmatter });
const links = await extractLinksFromFile(content, relPath, allSlugs, { globalBasename });
for (const link of links) {
if (dryRun) {
if (!jsonMode) console.log(` ${link.from_slug}${link.to_slug} (${link.link_type})`);
-12
View File
@@ -121,18 +121,6 @@ export interface GBrainConfig {
/** Daily spend cap (USD); bounds drains/day = floor(cap / ~$0.30). Default 2.0. */
max_usd_per_day?: number;
};
/**
* v0.42 — keep frontmatter links fresh on the incremental cycle. The cycle's
* extract phase re-extracts only the slugs a sync changed, but `extractForSlugs`
* extracts BODY links only — frontmatter (`sources:`/`related:` etc.) link edges
* silently drift stale when a page's YAML is edited externally and synced in.
* Set true to also extract frontmatter links per changed page each cycle, keeping
* externally-edited YAML edges fresh without a full rescan. Default false
* (preserves current behavior). Read via the file/env/DB plane in the cycle's
* extract dispatch. Disable/enable with
* `gbrain config set autopilot.incremental_extract_include_frontmatter <bool>`.
*/
incremental_extract_include_frontmatter?: boolean;
};
eval?: {
/** false disables capture entirely. Defaults to true. */
-16
View File
@@ -996,21 +996,6 @@ async function runPhaseExtract(
): Promise<PhaseResult> {
try {
const { runExtractCore } = await import('../commands/extract.ts');
const { loadConfig } = await import('./config.ts');
// Default off: the incremental cycle extracts body links only unless the
// operator opts in to keeping externally-edited frontmatter links fresh too.
// Both planes, file wins (env > file > DB precedence, per loadConfigWithEngine):
// `gbrain config set autopilot.incremental_extract_include_frontmatter true`
// writes the DB plane (engine.setConfig), so a file-plane-only read here
// would make the documented enable command a silent no-op (#2120 class).
const fileVal = loadConfig()?.autopilot?.incremental_extract_include_frontmatter;
let includeFrontmatter = fileVal === true;
if (fileVal === undefined) {
try {
includeFrontmatter =
(await engine.getConfig('autopilot.incremental_extract_include_frontmatter')) === 'true';
} catch { /* config table unreadable → default off */ }
}
// Extract is read-mostly against the filesystem + write to links table.
// Honor dryRun by skipping with a 'skipped' entry: extract doesn't have
// a clean dry-run mode today and runCycle should be honest about it.
@@ -1031,7 +1016,6 @@ async function runPhaseExtract(
slugs: changedSlugs, // undefined = full walk (first run / manual)
signal,
sourceId,
includeFrontmatter, // honored on the incremental (slugs) path only
});
const linksCreated = result?.links_created ?? 0;
const timelineCreated = result?.timeline_entries_created ?? 0;
+101 -20
View File
@@ -428,8 +428,13 @@ export async function runPhaseSynthesize(
const queue = new MinionQueue(engine);
const childIds: number[] = [];
/** Map child job_id → chunk metadata for D6 orchestrator-side slug rewrite. */
const chunkInfo = new Map<number, { idx: number; hash6: string }>();
/**
* Map child job_id → transcript metadata. Drives D6 orchestrator-side
* slug rewrite for chunked transcripts AND the deterministic frontmatter
* stampDreamProvenance merges into each written page. Populated for
* every child (single-chunk children carry chunkTotal=1).
*/
const childMeta = new Map<number, ChildMeta>();
/** Skip reasons for the cycle report (D5 cap hits, D8 legacy-key skips). */
const skipReports: Array<{ filePath: string; reason: string }> = [];
@@ -513,9 +518,14 @@ export async function runPhaseSynthesize(
{ allowProtectedSubmit: true },
);
childIds.push(child.id);
if (isChunked) {
chunkInfo.set(child.id, { idx: i, hash6 });
}
childMeta.set(child.id, {
idx: i,
hash6,
chunkTotal: chunks.length,
transcriptSource: t.transcriptSource,
transcriptId: stripContentVersionSuffix(t.basename),
inferredDate: t.inferredDate,
});
}
}
@@ -544,14 +554,14 @@ export async function runPhaseSynthesize(
// Collect slugs from put_page tool executions across the children
// (codex finding #2: deterministic provenance, NOT pages.updated_at).
// D6 orchestrator slug rewrite: chunkInfo drives post-hoc rewrite of
// D6 orchestrator slug rewrite: childMeta drives post-hoc rewrite of
// bare-hash slugs to `<hash6>-c<idx>` so chunked siblings can't collide
// even if Sonnet drops the chunk suffix.
// v0.32.8: refs carry source_id so reverseWriteRefs picks the correct
// (source, slug) row. #1586: refs are stamped with the cycle's resolved
// source (children write there via SubagentHandlerData.source_id).
const cycleSourceId = opts.sourceId ?? 'default';
const writtenRefs = await collectChildPutPageSlugs(engine, childIds, chunkInfo, cycleSourceId);
const writtenRefs = await collectChildPutPageSlugs(engine, childIds, childMeta, cycleSourceId);
const summaryDate = opts.date ?? today();
@@ -559,7 +569,12 @@ export async function runPhaseSynthesize(
// of every child-written page BEFORE reverse-rendering, so generated pages
// are queryable (`frontmatter->>'dream_generated'`) and a later put_page
// write-through (which re-renders from the DB row) can't erase the stamp.
await stampDreamProvenance(engine, writtenRefs, summaryDate);
// #2285: the stamp also carries the orchestrator-owned deterministic
// frontmatter (transcript_id, transcript_source, transcript_hash, date,
// chunk) derived from childMeta — subagent drift on those fields can't
// leak, and reverseWriteRefs below re-reads the row so the same fields
// land in the on-disk markdown.
await stampDreamProvenance(engine, writtenRefs, summaryDate, childMeta);
// Dual-write: reverse-render each DB row → markdown file.
const reverseWriteCount = await reverseWriteRefs(engine, opts.brainDir, writtenRefs, cycleSourceId);
@@ -1095,15 +1110,17 @@ function sanitizeForSlug(s: string): string {
* fake"): we no longer need detection because the rewrite enforces
* uniqueness at slug-write time.
*
* `chunkInfo` maps child job_id → { chunk_index, hash6 }. Single-chunk
* children are absent from the map and pass through unchanged.
* `childMeta` maps child job_id → per-child transcript metadata. Chunked
* children (chunkTotal > 1) get the slug rewrite; single-chunk children
* pass through unchanged. Each returned ref carries the job_id that wrote
* it so stampDreamProvenance can pair the slug back to its childMeta entry.
*/
async function collectChildPutPageSlugs(
engine: BrainEngine,
childIds: number[],
chunkInfo: Map<number, { idx: number; hash6: string }>,
childMeta: Map<number, ChildMeta>,
sourceId = 'default',
): Promise<Array<{ slug: string; source_id: string }>> {
): Promise<Array<{ slug: string; source_id: string; jobId: number }>> {
if (childIds.length === 0) return [];
// Raw fetch — NO SELECT DISTINCT. Preserves per-child slug duplicates so
// the orchestrator sees what each child wrote. COALESCE handles both
@@ -1122,16 +1139,73 @@ async function collectChildPutPageSlugs(
FROM subagent_tool_executions
WHERE job_id = ANY($1::int[])
AND tool_name = 'brain_put_page'
AND status = 'complete'`,
AND status = 'complete'
ORDER BY id`,
[childIds],
);
const rewritten = new Set<string>();
const rewritten = new Map<string, number>();
for (const r of rows) {
if (typeof r.slug !== 'string' || r.slug.length === 0) continue;
const ci = chunkInfo.get(r.job_id);
rewritten.add(ci ? rewriteChunkedSlug(r.slug, ci.hash6, ci.idx) : r.slug);
const meta = childMeta.get(r.job_id);
const finalSlug = meta && meta.chunkTotal > 1
? rewriteChunkedSlug(r.slug, meta.hash6, meta.idx)
: r.slug;
// Last writer wins, in execution-row order (ORDER BY id): if two children
// collide on a final slug, the pages row holds the LAST put_page write, so
// the stamp must attribute that child's transcript — not an arbitrary one.
rewritten.set(finalSlug, r.job_id);
}
return Array.from(rewritten).sort().map(slug => ({ slug, source_id: sourceId }));
return [...rewritten.entries()]
.sort(([a], [b]) => a.localeCompare(b))
.map(([slug, jobId]) => ({ slug, source_id: sourceId, jobId }));
}
/**
* Per-child orchestrator state. Drives D6 chunked-slug rewrite (idx + hash6)
* AND the deterministic frontmatter stampDreamProvenance merges into each
* written page. Populated for every child, not just chunked ones.
*/
interface ChildMeta {
idx: number;
hash6: string;
chunkTotal: number;
transcriptSource: string | null;
transcriptId: string;
inferredDate: string | null;
}
/**
* Strip the content-version suffix that claude-code-archive appends when a
* conversation is edited (`<uuid>--<contentHash>.md`). The session UUID is
* the stable transcript identifier; the suffix changes with content. Used to
* populate `transcript_id` so edits of the same session collapse to one id.
*/
function stripContentVersionSuffix(basename: string): string {
return basename.replace(/--[a-f0-9]+$/i, '');
}
/**
* Deterministic frontmatter for one synthesized page (#2285). Every field
* here is owned by the orchestrator — the subagent's value for any of these
* is overwritten. The subagent retains authority over type / title / tags /
* body. `date` feeds the effective-date precedence chain
* (src/core/effective-date.ts) so re-imports keep the conversation date even
* when sync tools re-stamp file mtimes.
*/
function buildDeterministicFrontmatter(
meta: ChildMeta,
cycleDate: string,
): Record<string, unknown> {
const overrides: Record<string, unknown> = {
dream_generated: true,
dream_cycle_date: cycleDate,
transcript_id: meta.transcriptId,
transcript_hash: meta.hash6,
};
if (meta.transcriptSource) overrides.transcript_source = meta.transcriptSource;
if (meta.chunkTotal > 1) overrides.chunk = `${meta.idx + 1}/${meta.chunkTotal}`;
if (meta.inferredDate) overrides.date = meta.inferredDate;
return overrides;
}
/**
@@ -1177,12 +1251,19 @@ async function hasLegacySingleChunkCompletion(
*/
async function stampDreamProvenance(
engine: BrainEngine,
refs: Array<{ slug: string; source_id: string }>,
refs: Array<{ slug: string; source_id: string; jobId?: number }>,
cycleDate: string,
childMeta?: Map<number, ChildMeta>,
): Promise<void> {
if (refs.length === 0) return;
const { executeRawJsonb } = await import('../sql-query.ts');
for (const { slug, source_id } of refs) {
for (const { slug, source_id, jobId } of refs) {
// #2285: when the ref pairs back to a child, the stamp also carries the
// orchestrator-owned deterministic frontmatter for that transcript.
const meta = jobId !== undefined ? childMeta?.get(jobId) : undefined;
const stamp = meta
? buildDeterministicFrontmatter(meta, cycleDate)
: { dream_generated: true, dream_cycle_date: cycleDate };
try {
await executeRawJsonb(
engine,
@@ -1190,7 +1271,7 @@ async function stampDreamProvenance(
SET frontmatter = COALESCE(frontmatter, '{}'::jsonb) || $3::jsonb
WHERE slug = $1 AND source_id = $2`,
[slug, source_id],
[{ dream_generated: true, dream_cycle_date: cycleDate }],
[stamp],
);
} catch (e) {
const msg = e instanceof Error ? e.message : String(e);
+58 -5
View File
@@ -10,7 +10,7 @@
*/
import { readFileSync, readdirSync, statSync } from 'node:fs';
import { join, basename } from 'node:path';
import { join, basename, dirname } from 'node:path';
import { createHash } from 'node:crypto';
import { pruneDir } from '../sync.ts';
@@ -23,8 +23,22 @@ export interface DiscoveredTranscript {
content: string;
/** Filename basename without extension; used as a topic-slug seed. */
basename: string;
/** Inferred date if the basename matches `YYYY-MM-DD...` (or null). */
/**
* Inferred conversation date (YYYY-MM-DD) or null. Precedence: the
* `| First message | <ISO> |` row in the transcript's `## Metadata`
* table (stable across mtime-restamping re-syncs) wins; a leading
* `YYYY-MM-DD` in the basename is the fallback.
*/
inferredDate: string | null;
/**
* Transcript source archive name, derived from the path's grandparent
* directory (the immediate parent of the date directory). For the
* canonical layout `<corpus>/<source>/<date>/<id>.md` this yields the
* source-name segment — e.g. `claude-code` for the claude-code-archive
* output, `meetings` for meeting recordings. Null when the file does
* not live under a `<source>/<date>/` pair (ad-hoc inputs).
*/
transcriptSource: string | null;
}
export interface DiscoverOpts {
@@ -161,6 +175,36 @@ function matchesAnyExclude(text: string, patterns: RegExp[]): boolean {
return false;
}
/**
* Content-based conversation date: the `| First message | <ISO timestamp> |`
* row claude-code-archive writes into the transcript's `## Metadata` table.
* Stable across rsync/Dropbox/Syncthing/B2 re-syncs that re-stamp mtime,
* unlike anything derived from file metadata. Returns YYYY-MM-DD or null.
*/
const FIRST_MESSAGE_RE = /^\|\s*First message\s*\|\s*(\d{4}-\d{2}-\d{2})/im;
export function inferContentDate(content: string): string | null {
const m = FIRST_MESSAGE_RE.exec(content);
return m ? m[1] : null;
}
/**
* Derive the archive source name from a transcript path. Returns the basename
* of the directory two levels above the file when the immediate parent is a
* date directory and the grandparent looks like a source-name slug (lowercase
* alphanumeric segments separated by hyphens); otherwise null. This pins the
* canonical claude-code-archive layout `<corpus>/<source>/<date>/<id>.md`
* without claiming a source for ad-hoc inputs that don't match.
*/
export function deriveTranscriptSource(filePath: string): string | null {
const parentName = basename(dirname(filePath));
if (!/^\d{4}-\d{2}-\d{2}/.test(parentName)) return null;
const grandparentName = basename(dirname(dirname(filePath)));
if (!grandparentName) return null;
if (!/^[a-z0-9]+(-[a-z0-9]+)*$/.test(grandparentName)) return null;
return grandparentName;
}
function listTextFiles(dir: string): string[] {
// Recursive walk with descent-time pruning (closes codex C12/C13 spec gap).
// Accepts BOTH .txt and .md per transcript-discovery's domain rules — does
@@ -225,8 +269,11 @@ export function discoverTranscripts(opts: DiscoverOpts): DiscoveredTranscript[]
const ext = filePath.endsWith('.md') ? '.md' : '.txt';
const baseName = basename(filePath, ext);
const dateMatch = DATE_RE.exec(baseName);
const inferredDate = dateMatch ? dateMatch[1] : null;
if (!isInDateRange(inferredDate, opts)) continue;
const filenameDate = dateMatch ? dateMatch[1] : null;
// Fast path: date-named files outside the window skip before the read.
// ponytail: a date-named file whose content date differs is filtered on
// its filename date — acceptable; archive layouts use UUID basenames.
if (filenameDate && !isInDateRange(filenameDate, opts)) continue;
let content: string;
try {
@@ -241,12 +288,17 @@ export function discoverTranscripts(opts: DiscoverOpts): DiscoveredTranscript[]
}
if (matchesAnyExclude(content, excludeRes)) continue;
// Content-metadata date wins (survives mtime restamps); filename next.
const inferredDate = inferContentDate(content) ?? filenameDate;
if (!isInDateRange(inferredDate, opts)) continue;
results.push({
filePath,
contentHash: hashContent(content),
content,
basename: baseName,
inferredDate,
transcriptSource: deriveTranscriptSource(filePath),
});
}
}
@@ -290,6 +342,7 @@ export function readSingleTranscript(
contentHash: hashContent(content),
content,
basename: baseName,
inferredDate: dateMatch ? dateMatch[1] : null,
inferredDate: inferContentDate(content) ?? (dateMatch ? dateMatch[1] : null),
transcriptSource: deriveTranscriptSource(filePath),
};
}
+3 -36
View File
@@ -942,17 +942,8 @@ export function makeResolver(
const hints = Array.isArray(dirHint) ? dirHint : (dirHint ? [dirHint] : []);
// Step 1: already a slug? Try an exact page lookup for any slug-shaped
// value (contains '/', slug charset). Broadened beyond the original
// single-segment lowercase-leading form (`^[a-z][a-z0-9-]*\/[a-z0-9]...`)
// to also accept digit-leading folders (`90-people/nicolai`,
// `01-trading/...`) and nested paths (`a/b/c`) — common in PARA-numbered
// vaults. This is an EXACT getPage match only — no fuzzy — so it never
// produces a false positive; a non-existent slug just falls through to
// the steps below. Fixes frontmatter `related: [[dir/slug]]` values
// (unwrapped by unwrapWikilink) that name a real page the strict regex
// could not reach and whose full-path fuzzy score is below threshold.
if (/\//.test(trimmed) && /^[a-z0-9][a-z0-9/_-]*$/.test(trimmed)) {
// Step 1: already a slug? (dir/name shape, lowercase, hyphenated)
if (/^[a-z][a-z0-9-]*\/[a-z0-9][a-z0-9-]*$/.test(trimmed)) {
const page = await engine.getPage(trimmed);
if (page) {
cache.set(cacheKey, trimmed);
@@ -1012,25 +1003,6 @@ export function makeResolver(
// ─── Frontmatter extractor ──────────────────────────────────────
/**
* Unwrap an Obsidian `[[wikilink]]` frontmatter value to its bare link
* target so the resolver (which expects bare titles / dir slugs) can match
* it. Mainstream Obsidian authors frontmatter links as `related: ["[[Page]]"]`;
* without this, the resolver treats the brackets as part of the value and a
* `[[90-people/nicolai]]` is normalized into `90peoplenicolai`, so it never
* resolves. Strips a trailing `|alias`, `#heading`, or `^block` suffix — the
* link target only. The regex is anchored to a wholly-wrapped value
* (`^\s*\[\[…\]\]\s*$`), so bare titles and any value not fully wrapped pass
* through unchanged and existing behavior is preserved exactly.
*/
export function unwrapWikilink(value: string): string {
const match = /^\s*\[\[(.+?)\]\]\s*$/.exec(value);
if (!match) return value;
// Take the link target: drop |alias, then #heading / ^block suffixes.
const target = match[1].split('|')[0].split('#')[0].split('^')[0];
return target.trim();
}
export interface UnresolvedFrontmatterRef {
/** The frontmatter field name. */
field: string;
@@ -1088,12 +1060,7 @@ export async function extractFrontmatterLinks(
}
if (!name) continue; // skip numbers, nulls, malformed objects
// Accept Obsidian `[[wikilink]]` values in frontmatter link fields by
// unwrapping to the bare target before resolution. Bare titles pass
// through unchanged; the original `name` is preserved for the
// unresolved report and edge context.
const linkTarget = unwrapWikilink(name);
const resolved = await resolver.resolve(linkTarget, mapping.dirHint);
const resolved = await resolver.resolve(name, mapping.dirHint);
if (!resolved) {
unresolved.push({ field, name });
continue;
+1
View File
@@ -26,6 +26,7 @@ const transcript: DiscoveredTranscript = {
content: 'User: hello world',
contentHash: 'abcdef0123456789',
inferredDate: '2026-07-17',
transcriptSource: null,
} as DiscoveredTranscript;
describe('#2415: buildSynthesisPrompt output root', () => {
@@ -152,3 +152,94 @@ describe('#2569: stampDreamProvenance persists the marker into DB frontmatter',
await stampDreamProvenance(engine as any, refs, '2026-07-17'); // idempotent
});
});
describe('#2285: orchestrator-owned deterministic transcript frontmatter', () => {
const meta = {
idx: 1,
hash6: 'abc123',
chunkTotal: 3,
transcriptSource: 'claude-code',
transcriptId: 'session-uuid',
inferredDate: '2026-05-15',
};
test('collectChildPutPageSlugs pairs each ref back to the writing job', async () => {
const refs = await collectChildPutPageSlugs(
engine as any, [1001], new Map([[1001, { ...meta, chunkTotal: 1 }]]), 'mybrain',
);
expect(refs.length).toBeGreaterThan(0);
for (const r of refs) {
expect(r.jobId).toBe(1001);
expect(r.source_id).toBe('mybrain'); // #1586: cycle source, never hardcoded 'default'
}
});
test('slug collision across children attributes the LAST writer (matches surviving putPage)', async () => {
const db = (engine as any).db;
// Jobs 1001 then 1002 write the same slug; the pages row would hold
// 1002's content (last put_page wins), so the ref must carry jobId 1002.
await db.query(
`INSERT INTO subagent_tool_executions (job_id, message_idx, tool_use_id, tool_name, status, input)
VALUES (1001, 9, 'tool_dup_a', 'brain_put_page', 'complete', $1::jsonb)`,
[JSON.stringify({ slug: 'wiki/agents/test/collision', body: 'first' })],
);
await db.query(
`INSERT INTO subagent_tool_executions (job_id, message_idx, tool_use_id, tool_name, status, input)
VALUES (1002, 9, 'tool_dup_b', 'brain_put_page', 'complete', $1::jsonb)`,
[JSON.stringify({ slug: 'wiki/agents/test/collision', body: 'second' })],
);
const refs = await collectChildPutPageSlugs(engine as any, [1001, 1002], new Map());
const hit = refs.find((r: { slug: string }) => r.slug === 'wiki/agents/test/collision');
expect(hit?.jobId).toBe(1002);
});
test('stampDreamProvenance merges the transcript metadata into DB frontmatter', async () => {
const slug = 'wiki/originals/ideas/2026-07-17-transcript-meta-abc123';
await engine.putPage(slug, {
type: 'note',
title: 'Meta stamp',
compiled_truth: 'body',
timeline: '',
frontmatter: { keep_me: 'yes', transcript_id: 'subagent-drift' },
});
await stampDreamProvenance(
engine as any,
[{ slug, source_id: 'default', jobId: 42 }],
'2026-07-17',
new Map([[42, meta]]),
);
const rows = await engine.executeRaw<{ fm: Record<string, unknown> }>(
`SELECT frontmatter AS fm FROM pages WHERE slug = $1`, [slug],
);
const fm = rows[0].fm as Record<string, unknown>;
expect(fm.dream_generated).toBe(true);
expect(fm.dream_cycle_date).toBe('2026-07-17');
expect(fm.transcript_id).toBe('session-uuid'); // orchestrator wins over subagent drift
expect(fm.transcript_hash).toBe('abc123');
expect(fm.transcript_source).toBe('claude-code');
expect(fm.chunk).toBe('2/3');
expect(fm.date).toBe('2026-05-15');
expect(fm.keep_me).toBe('yes'); // subagent-owned keys survive
});
test('single-chunk children with no inferredDate stamp only the applicable fields', async () => {
const slug = 'wiki/originals/ideas/2026-07-17-minimal-meta-abc123';
await engine.putPage(slug, {
type: 'note', title: 'Minimal', compiled_truth: 'b', timeline: '', frontmatter: {},
});
await stampDreamProvenance(
engine as any,
[{ slug, source_id: 'default', jobId: 43 }],
'2026-07-17',
new Map([[43, { ...meta, chunkTotal: 1, transcriptSource: null, inferredDate: null }]]),
);
const rows = await engine.executeRaw<{ fm: Record<string, unknown> }>(
`SELECT frontmatter AS fm FROM pages WHERE slug = $1`, [slug],
);
const fm = rows[0].fm as Record<string, unknown>;
expect(fm.transcript_id).toBe('session-uuid');
expect(fm.chunk).toBeUndefined();
expect(fm.transcript_source).toBeUndefined();
expect(fm.date).toBeUndefined();
});
});
+2
View File
@@ -302,6 +302,7 @@ describe('judgeSignificance', () => {
content: 'A short conversation about something interesting.',
basename: 'x',
inferredDate: null,
transcriptSource: null,
};
}
@@ -415,6 +416,7 @@ describe('judgeSignificance — UTF-16 safety (v0.41.13)', () => {
content,
basename: 'long',
inferredDate: null,
transcriptSource: null,
};
}
@@ -45,6 +45,7 @@ const FIXTURE_TRANSCRIPT: DiscoveredTranscript = {
content: 'Synthetic transcript content for gateway-adapter parity tests.',
contentHash: 'sha-fixture-1',
inferredDate: '2026-05-24',
transcriptSource: null,
};
describe('makeJudgeClient — construction-time provider probe', () => {
@@ -0,0 +1,109 @@
/**
* #2285 — transcript metadata discovery.
*
* Pins the two discovery-side additions:
* 1. `transcriptSource` — derived from the `<source>/<date>/<file>` path
* layout; null for ad-hoc inputs that don't match.
* 2. Content-based date inference — the `| First message | <ISO> |` row in
* the transcript's `## Metadata` table wins over the filename-regex
* date (stable across mtime-restamping re-syncs); filename is the
* fallback.
*
* Pure filesystem; no engine, no LLM.
*/
import { describe, test, expect, beforeEach, afterEach } from 'bun:test';
import { mkdtempSync, rmSync, writeFileSync, mkdirSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join, dirname } from 'node:path';
import {
discoverTranscripts,
readSingleTranscript,
deriveTranscriptSource,
inferContentDate,
} from '../../src/core/cycle/transcript-discovery.ts';
let tmpDir: string;
beforeEach(() => {
tmpDir = mkdtempSync(join(tmpdir(), 'gbrain-transcript-meta-'));
});
afterEach(() => {
rmSync(tmpDir, { recursive: true, force: true });
});
function write(relPath: string, body: string): string {
const full = join(tmpDir, relPath);
mkdirSync(dirname(full), { recursive: true });
writeFileSync(full, body);
return full;
}
const FILLER = 'User: hello world. '.repeat(200);
const METADATA_BLOCK =
'## Metadata\n\n| Key | Value |\n| --- | --- |\n| First message | 2026-05-15T03:51:11.584Z |\n\n';
describe('deriveTranscriptSource', () => {
test('extracts the source slug from <source>/<date>/<file> layout', () => {
expect(deriveTranscriptSource('/corpus/claude-code/2026-06-12/abc.md')).toBe('claude-code');
expect(deriveTranscriptSource('/corpus/voice-notes/2026-06-12/xyz.md')).toBe('voice-notes');
});
test('null when the parent dir is not a date dir or grandparent is not a slug', () => {
expect(deriveTranscriptSource('/corpus/flat-file.md')).toBeNull();
expect(deriveTranscriptSource('/corpus/claude-code/not-a-date/abc.md')).toBeNull();
expect(deriveTranscriptSource('/corpus/Not A Slug/2026-06-12/abc.md')).toBeNull();
});
});
describe('inferContentDate', () => {
test('parses the | First message | row', () => {
expect(inferContentDate(METADATA_BLOCK)).toBe('2026-05-15');
});
test('null when absent', () => {
expect(inferContentDate(FILLER)).toBeNull();
});
});
describe('discoverTranscripts — transcriptSource + date cascade', () => {
test('populates transcriptSource per file; null for flat files', () => {
write('claude-code/2026-06-12/aaaa.md', FILLER);
write('2026-06-12-flat.md', FILLER);
const out = discoverTranscripts({ corpusDir: tmpDir, minChars: 100 });
const byBase = new Map(out.map(t => [t.basename, t.transcriptSource]));
expect(byBase.get('aaaa')).toBe('claude-code');
expect(byBase.get('2026-06-12-flat')).toBeNull();
});
test('content First-message date wins over the filename date', () => {
write('2026-01-01-named.md', METADATA_BLOCK + FILLER);
const out = discoverTranscripts({ corpusDir: tmpDir, minChars: 100 });
expect(out).toHaveLength(1);
expect(out[0].inferredDate).toBe('2026-05-15');
});
test('filename date remains the fallback when content has no metadata row', () => {
write('2026-01-01-named.md', FILLER);
const out = discoverTranscripts({ corpusDir: tmpDir, minChars: 100 });
expect(out[0].inferredDate).toBe('2026-01-01');
});
test('date filter matches on the content date for UUID-named transcripts', () => {
write('claude-code/2026-05-15/uuid-basename.md', METADATA_BLOCK + FILLER);
const hit = discoverTranscripts({ corpusDir: tmpDir, minChars: 100, date: '2026-05-15' });
expect(hit).toHaveLength(1);
const miss = discoverTranscripts({ corpusDir: tmpDir, minChars: 100, date: '2026-05-16' });
expect(miss).toHaveLength(0);
});
});
describe('readSingleTranscript — same metadata surface', () => {
test('carries transcriptSource and prefers the content date', () => {
const p = write('claude-code/2026-05-15/2026-01-01-single.md', METADATA_BLOCK + FILLER);
const t = readSingleTranscript(p, { minChars: 100 });
expect(t).not.toBeNull();
expect(t!.transcriptSource).toBe('claude-code');
expect(t!.inferredDate).toBe('2026-05-15');
});
});
-34
View File
@@ -191,37 +191,3 @@ describe('runExtractCore — incremental cycle path (#417)', () => {
expect(result.links_created).toBeGreaterThan(0);
});
});
describe('runExtractCore — incremental frontmatter gate (includeFrontmatter)', () => {
// alice has a `source:` frontmatter edge but NO body links. The incremental
// path extracts body links only by default, so the frontmatter edge is the
// sole signal that distinguishes the gate off vs on.
const aliceFm = '---\nsource: companies/acme-example\n---\n# alice';
test('9. default (flag omitted) does NOT extract frontmatter links on the incremental path', async () => {
await seedPage('companies/acme-example', '# acme');
await seedPage('people/alice-example', aliceFm);
const result = await runExtractCore(engine as unknown as BrainEngine, {
mode: 'all',
dir: tempDir,
slugs: ['people/alice-example'],
});
// alice's only potential edge is her frontmatter `source:`; with the gate off
// it must not be extracted (preserves the body-only incremental behavior).
expect(result.pages_processed).toBe(1);
expect(result.links_created).toBe(0);
});
test('10. includeFrontmatter: true extracts the frontmatter link on the incremental path', async () => {
await seedPage('companies/acme-example', '# acme');
await seedPage('people/alice-example', aliceFm);
const result = await runExtractCore(engine as unknown as BrainEngine, {
mode: 'all',
dir: tempDir,
slugs: ['people/alice-example'],
includeFrontmatter: true,
});
// Same page, gate on → the `source:` frontmatter edge is now extracted.
expect(result.pages_processed).toBe(1);
expect(result.links_created).toBeGreaterThan(0);
});
});
-12
View File
@@ -76,18 +76,6 @@ describe('extractLinksFromFile', () => {
}
});
it('resolves wrapped [[wikilink]] digit-leading slug-path in frontmatter (fs resolver, broadened step 1)', async () => {
// Same bug class as makeResolver step 1 (#1983): the fs resolver's strict
// `^[a-z]…` slug regex rejected digit-leading / nested paths, so a PARA-vault
// `related: "[[90-people/nicolai]]"` never resolved even though the page exists.
const content = '---\nrelated: "[[90-people/nicolai]]"\ntype: concept\n---\nContent.';
const allSlugs = new Set(['wiki/note', '90-people/nicolai']);
const links = await extractLinksFromFile(content, 'wiki/note.md', allSlugs, { includeFrontmatter: true });
const related = links.filter(l => l.link_type === 'related_to');
expect(related).toHaveLength(1);
expect(related[0].to_slug).toBe('90-people/nicolai');
});
it('frontmatter extraction is default OFF (back-compat)', async () => {
// Without includeFrontmatter, fs-source no longer auto-extracts frontmatter.
// Matches db-source behavior. User opts in with --include-frontmatter flag.
-173
View File
@@ -9,7 +9,6 @@ import {
parseTimelineEntries,
isAutoLinkEnabled,
FRONTMATTER_LINK_MAP,
unwrapWikilink,
type SlugResolver,
} from '../src/core/link-extraction.ts';
import type { BrainEngine } from '../src/core/engine.ts';
@@ -1292,175 +1291,3 @@ describe('parseTimelineEntries — Format 3: inline [Source: ..., YYYY-MM-DD] ci
expect(parseTimelineEntries('[Source: import batch, 2025-07-01]')).toHaveLength(0);
});
});
// ─── Frontmatter [[wikilink]] + slug-path resolution ──────────────────────
// Mainstream Obsidian authors frontmatter links as `related: ["[[Page]]"]`,
// and PARA-numbered vaults use digit-leading / nested slug paths like
// `[[90-people/nicolai]]`. Both were silently dropped: brackets were treated
// as part of the value and the step-1 slug regex (`^[a-z]…`) rejected
// digit-leading / nested paths, while full-path fuzzy scored below threshold.
// Fix: unwrapWikilink() before resolution + an exact getPage() for any
// slug-shaped value (exact-match only → no false positives).
describe('unwrapWikilink', () => {
test('wrapped title → bare title', () => {
expect(unwrapWikilink('[[Monday Range]]')).toBe('Monday Range');
});
test('wrapped slug-path (digit-leading folder) → bare slug', () => {
expect(unwrapWikilink('[[90-people/nicolai]]')).toBe('90-people/nicolai');
});
test('wrapped nested slug-path → bare slug', () => {
expect(unwrapWikilink('[[01-trading/wiki/strategies/opening-range-breakout]]'))
.toBe('01-trading/wiki/strategies/opening-range-breakout');
});
test('strips |alias', () => {
expect(unwrapWikilink('[[90-people/nicolai|Nicolai]]')).toBe('90-people/nicolai');
});
test('strips #heading', () => {
expect(unwrapWikilink('[[Page#Section]]')).toBe('Page');
});
test('strips ^block', () => {
expect(unwrapWikilink('[[Page^abc123]]')).toBe('Page');
});
test('surrounding whitespace tolerated', () => {
expect(unwrapWikilink(' [[Page]] ')).toBe('Page');
});
test('bare title passes through unchanged', () => {
expect(unwrapWikilink('Monday Range')).toBe('Monday Range');
});
test('bare slug passes through unchanged', () => {
expect(unwrapWikilink('90-people/nicolai')).toBe('90-people/nicolai');
});
test('partially-wrapped value is NOT unwrapped (anchored)', () => {
// Not a wholly-wrapped value → left intact so existing behavior is exact.
expect(unwrapWikilink('see [[Page]] for detail')).toBe('see [[Page]] for detail');
});
});
describe('makeResolver — slug-path exact getPage (step 1 broadened)', () => {
function fakeEngine(
slugs: string[],
fuzzyMap: Map<string, { slug: string; similarity: number }> = new Map(),
): BrainEngine {
const lookup = new Set(slugs);
return {
async getPage(slug: string) { return lookup.has(slug) ? { slug } as any : null; },
async findByTitleFuzzy(name: string) { return fuzzyMap.get(name) ?? null; },
async searchKeyword() { return []; },
} as unknown as BrainEngine;
}
test('digit-leading folder slug resolves via exact getPage', async () => {
const r = makeResolver(fakeEngine(['90-people/nicolai']));
expect(await r.resolve('90-people/nicolai')).toBe('90-people/nicolai');
});
test('nested (>2 segment) slug resolves via exact getPage', async () => {
const r = makeResolver(fakeEngine(['01-trading/wiki/strategies/opening-range-breakout']));
expect(await r.resolve('01-trading/wiki/strategies/opening-range-breakout'))
.toBe('01-trading/wiki/strategies/opening-range-breakout');
});
test('regression: single-segment lowercase slug still resolves', async () => {
const r = makeResolver(fakeEngine(['people/pedro']));
expect(await r.resolve('people/pedro')).toBe('people/pedro');
});
test('exact-only: slug-shaped value with no matching page falls through (no false positive)', async () => {
// `90-people/ghost` is slug-shaped but absent → step-1 getPage misses,
// no fuzzy hit → null. Never invents an edge.
const r = makeResolver(fakeEngine(['90-people/nicolai']));
expect(await r.resolve('90-people/ghost')).toBeNull();
});
test('non-slug value still routes to fuzzy', async () => {
const r = makeResolver(fakeEngine(
['01-trading/monday-range'],
new Map([['Monday Range', { slug: '01-trading/monday-range', similarity: 1 }]]),
));
expect(await r.resolve('Monday Range')).toBe('01-trading/monday-range');
});
});
describe('extractFrontmatterLinks — [[wikilink]] related: values (end-to-end)', () => {
function fakeEngine(
slugs: string[],
fuzzyMap: Map<string, { slug: string; similarity: number }> = new Map(),
): BrainEngine {
const lookup = new Set(slugs);
return {
async getPage(slug: string) { return lookup.has(slug) ? { slug } as any : null; },
async findByTitleFuzzy(name: string) { return fuzzyMap.get(name) ?? null; },
async searchKeyword() { return []; },
} as unknown as BrainEngine;
}
test('wrapped slug-path related: resolves (the core win)', async () => {
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
const { candidates, unresolved } = await extractFrontmatterLinks(
'wiki/originals/ideas/note', 'note' as never,
{ related: '[[90-people/nicolai]]' }, resolver,
);
expect(unresolved).toHaveLength(0);
expect(candidates).toHaveLength(1);
expect(candidates[0]).toMatchObject({
fromSlug: 'wiki/originals/ideas/note',
targetSlug: '90-people/nicolai',
linkType: 'related_to',
linkSource: 'frontmatter',
});
});
test('wrapped nested slug-path related: resolves', async () => {
const resolver = makeResolver(fakeEngine(['01-trading/wiki/strategies/opening-range-breakout']));
const { candidates } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: ['[[01-trading/wiki/strategies/opening-range-breakout]]'] }, resolver,
);
expect(candidates).toHaveLength(1);
expect(candidates[0].targetSlug).toBe('01-trading/wiki/strategies/opening-range-breakout');
});
test('wrapped value with |alias resolves to the target', async () => {
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
const { candidates } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: '[[90-people/nicolai|Nicolai]]' }, resolver,
);
expect(candidates).toHaveLength(1);
expect(candidates[0].targetSlug).toBe('90-people/nicolai');
});
test('regression: bare slug related: still resolves', async () => {
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
const { candidates } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: '90-people/nicolai' }, resolver,
);
expect(candidates).toHaveLength(1);
expect(candidates[0].targetSlug).toBe('90-people/nicolai');
});
test('regression: wrapped title resolves via fuzzy (brackets harmless)', async () => {
const resolver = makeResolver(fakeEngine(
['01-trading/monday-range'],
new Map([['Monday Range', { slug: '01-trading/monday-range', similarity: 1 }]]),
));
const { candidates } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: '[[Monday Range]]' }, resolver,
);
expect(candidates).toHaveLength(1);
expect(candidates[0].targetSlug).toBe('01-trading/monday-range');
});
test('unknown wrapped slug → unresolved (no crash), original value preserved', async () => {
const resolver = makeResolver(fakeEngine(['90-people/nicolai']));
const { candidates, unresolved } = await extractFrontmatterLinks(
'wiki/note', 'note' as never,
{ related: '[[99-archive/does-not-exist]]' }, resolver,
);
expect(candidates).toHaveLength(0);
expect(unresolved).toHaveLength(1);
expect(unresolved[0]).toEqual({ field: 'related', name: '[[99-archive/does-not-exist]]' });
});
});
+7 -3
View File
@@ -216,17 +216,21 @@ describe('progress reporter', () => {
});
test('only one process-level signal handler installed across many reporters', () => {
// Baseline: one handler already installed by prior tests in this file.
// Baseline: one handler already installed by prior tests in this file, and
// possibly live reporters from OTHER test files sharing this bun process
// (shard composition is not this test's invariant — assert the delta, not
// an absolute zero, or shard reshuffles make this fail spuriously).
const installedBefore = __signalHandlerInstalledForTest();
const liveBefore = __liveReporterCountForTest();
const { stream } = sink(false);
for (let i = 0; i < 50; i++) {
const p = createProgress({ mode: 'json', stream, minIntervalMs: 0, minItems: 1 });
p.start(`phase_${i}`, 1);
p.finish();
}
// After 50 reporter lifecycles, still exactly one handler and zero leaked live entries.
// After 50 reporter lifecycles, still exactly one handler and zero NET leaked live entries.
expect(__signalHandlerInstalledForTest()).toBe(installedBefore || true);
expect(__liveReporterCountForTest()).toBe(0);
expect(__liveReporterCountForTest()).toBe(liveBefore);
});
test('startHeartbeat() fires heartbeats and stop() clears', async () => {