mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-14 08:53:22 +00:00
Compare commits
29
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e22b8fb555 | ||
|
|
1e73e93344 | ||
|
|
52f9581966 | ||
|
|
2a9feb859f | ||
|
|
5d9dc4393e | ||
|
|
15b9316dbf | ||
|
|
08746b06d2 | ||
|
|
f739de5521 | ||
|
|
02d585c0a4 | ||
|
|
e573fa6988 | ||
|
|
8468ba25a9 | ||
|
|
d3b52edeba | ||
|
|
ff6320e552 | ||
|
|
36c750bbec | ||
|
|
7f2c81f929 | ||
|
|
1353366b5f | ||
|
|
93ae40dd3a | ||
|
|
8fcd2737bf | ||
|
|
b23f24f91b | ||
|
|
10d96545a4 | ||
|
|
6966623e0f | ||
|
|
6b2f3bc321 | ||
|
|
be8fffad71 | ||
|
|
e734937254 | ||
|
|
891c28b582 | ||
|
|
c78c3d0135 | ||
|
|
e2961c04bd | ||
|
|
172b55ba9d | ||
|
|
f718c595b3 |
@@ -21,11 +21,22 @@ jobs:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
test:
|
||||
# ubuntu-latest is free 2-core/7GB. Larger runners (16-cores, etc.) require
|
||||
# a provisioned runner pool in repo settings. Falling back to default keeps
|
||||
# the matrix shard speedup (~5-6x via parallelism) at zero cost.
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
shard: [1, 2, 3, 4]
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
|
||||
with:
|
||||
bun-version: latest
|
||||
- run: bun install
|
||||
- run: bun run test
|
||||
- name: Pre-test gates (shard 1 only — they're not test files)
|
||||
if: matrix.shard == 1
|
||||
run: scripts/check-jsonb-pattern.sh && scripts/check-progress-to-stdout.sh && scripts/check-wasm-embedded.sh && bun run typecheck
|
||||
- name: Run test shard ${{ matrix.shard }}/4
|
||||
run: scripts/test-shard.sh ${{ matrix.shard }} 4
|
||||
|
||||
@@ -17,3 +17,4 @@ eval/data/world-v1/world.html
|
||||
|
||||
# BrainBench amara-life-v1 Opus cache (regenerate via eval:generate-amara-life)
|
||||
eval/data/amara-life-v1/_cache/
|
||||
.claude/
|
||||
|
||||
+1156
File diff suppressed because it is too large
Load Diff
@@ -25,21 +25,30 @@ strict behavior when unset.
|
||||
- `src/core/operations.ts` — Contract-first operation definitions (the foundation). Also exports upload validators: `validateUploadPath`, `validatePageSlug`, `validateFilename`. `OperationContext.remote` flags untrusted callers.
|
||||
- `src/core/engine.ts` — Pluggable engine interface (BrainEngine). `clampSearchLimit(limit, default, cap)` takes an explicit cap so per-operation caps can be tighter than `MAX_SEARCH_LIMIT`. Exports `LinkBatchInput` / `TimelineBatchInput` for the v0.12.1 bulk-insert API (`addLinksBatch` / `addTimelineEntriesBatch`). As of v0.13.1, `BrainEngine` has a `readonly kind: 'postgres' | 'pglite'` discriminator so migrations (`src/core/migrate.ts`) and other consumers can branch on engine without `instanceof` + dynamic imports.
|
||||
- `src/core/engine-factory.ts` — Engine factory with dynamic imports (`'pglite'` | `'postgres'`)
|
||||
- `src/core/pglite-engine.ts` — PGLite (embedded Postgres 17.5 via WASM) implementation, all 40 BrainEngine methods. `addLinksBatch` / `addTimelineEntriesBatch` use multi-row `unnest()` with manual `$N` placeholders. As of v0.13.1, `connect()` wraps `PGlite.create()` in a try/catch that emits an actionable error naming the macOS 26.3 WASM bug (#223) and pointing at `gbrain doctor`; the lock is released on failure so the next process can retry cleanly.
|
||||
- `src/core/pglite-engine.ts` — PGLite (embedded Postgres 17.5 via WASM) implementation, all 40 BrainEngine methods. `addLinksBatch` / `addTimelineEntriesBatch` use multi-row `unnest()` with manual `$N` placeholders. As of v0.13.1, `connect()` wraps `PGlite.create()` in a try/catch that emits an actionable error naming the macOS 26.3 WASM bug (#223) and pointing at `gbrain doctor`; the lock is released on failure so the next process can retry cleanly. v0.22.0: `searchKeyword` and `searchKeywordChunks` multiply `ts_rank` by the source-factor CASE expression at the chunk-grain level; `searchVector` becomes a two-stage CTE — inner CTE keeps `ORDER BY cc.embedding <=> vec` so HNSW stays usable, outer SELECT re-ranks by `raw_score * source_factor`. Inner LIMIT scales with offset to preserve pagination contract. As of v0.22.6.1, `initSchema()` calls `applyForwardReferenceBootstrap()` BEFORE replaying SCHEMA_SQL — probes for the specific forward-referenced state the embedded schema blob needs (`pages.source_id`, `links.link_source`, `links.origin_page_id`, `content_chunks.symbol_name`, `content_chunks.language`, `sources` FK target table) and adds only what's missing. Closes the upgrade-wedge bug class that bit users 10+ times across 6 schema versions over 2 years (#239/#243/#266/#357/#366/#374/#375/#378/#395/#396). No-op on fresh installs and modern brains.
|
||||
- `src/core/pglite-schema.ts` — PGLite-specific DDL (pgvector, pg_trgm, triggers)
|
||||
- `src/core/postgres-engine.ts` — Postgres + pgvector implementation (Supabase / self-hosted). `addLinksBatch` / `addTimelineEntriesBatch` use `INSERT ... SELECT FROM unnest($1::text[], ...) JOIN pages ON CONFLICT DO NOTHING RETURNING 1` — 4-5 array params regardless of batch size, sidesteps the 65535-parameter cap. As of v0.12.3, `searchKeyword` / `searchVector` scope `statement_timeout` via `sql.begin` + `SET LOCAL` so the GUC dies with the transaction instead of leaking across the pooled postgres.js connection (contributed by @garagon). `getEmbeddingsByChunkIds` uses `tryParseEmbedding` so one corrupt row skips+warns instead of killing the query.
|
||||
- `src/core/postgres-engine.ts` — Postgres + pgvector implementation (Supabase / self-hosted). `addLinksBatch` / `addTimelineEntriesBatch` use `INSERT ... SELECT FROM unnest($1::text[], ...) JOIN pages ON CONFLICT DO NOTHING RETURNING 1` — 4-5 array params regardless of batch size, sidesteps the 65535-parameter cap. As of v0.12.3, `searchKeyword` / `searchVector` scope `statement_timeout` via `sql.begin` + `SET LOCAL` so the GUC dies with the transaction instead of leaking across the pooled postgres.js connection (contributed by @garagon). `getEmbeddingsByChunkIds` uses `tryParseEmbedding` so one corrupt row skips+warns instead of killing the query. v0.22.0: `searchKeyword`, `searchKeywordChunks`, and `searchVector` apply source-aware ranking by inlining the source-factor CASE and `NOT (col LIKE …)` hard-exclude clause from `src/core/search/sql-ranking.ts`. `searchVector` switches to a two-stage CTE (HNSW-safe inner ORDER BY, source-boost re-rank in the outer SELECT) and carries `p.source_id` through inner→outer for v0.18 multi-source callers. v0.22.1 (#406): `_savedConfig` retains the connect config; `reconnect()` tears down + recreates the pool from saved config (called by supervisor watchdog after 3 consecutive health-check failures). `executeRaw` is a single-statement passthrough — no per-call retry (D3 dropped that as unsound for non-idempotent statements; recovery is supervisor-driven). v0.22.1 (#363, contributed by @orendi84): `connect()` applies `resolveSessionTimeouts()` from `db.ts` as connection-time startup parameters (`statement_timeout`, `idle_in_transaction_session_timeout`) so orphan pgbouncer backends can't hold locks for hours. v0.22.1 (#409, contributed by @atrevino47): `countStaleChunks()` + `listStaleChunks()` server-side-filter on `embedding IS NULL` for `embed --stale`, eliminating ~76 MB/call client-side pull on a fully-embedded brain; `upsertChunks()` resets both `embedding` AND `embedded_at` to NULL when chunk_text changes without a new embedding (consistency). As of v0.22.6.1, `initSchema()` calls `applyForwardReferenceBootstrap()` BEFORE replaying SCHEMA_SQL on the same forward-reference probe set as the PGLite engine, so old Postgres brains pinned at v0.13/v0.18/v0.19 walk forward cleanly instead of wedging on `column "..." does not exist`.
|
||||
- `src/core/utils.ts` — Shared SQL utilities extracted from postgres-engine.ts. Exports `parseEmbedding(value)` (throws on unknown input, used by migration + ingest paths where data integrity matters) and as of v0.12.3 `tryParseEmbedding(value)` (returns `null` + warns once per process, used by search/rescore paths where availability matters more than strictness).
|
||||
- `src/core/db.ts` — Connection management, schema initialization
|
||||
- `src/core/db.ts` — Connection management, schema initialization. v0.22.1 (#363, contributed by @orendi84): `resolveSessionTimeouts()` returns `statement_timeout` + `idle_in_transaction_session_timeout` (defaults: 5min each, env-overridable via `GBRAIN_STATEMENT_TIMEOUT` / `GBRAIN_IDLE_TX_TIMEOUT` / `GBRAIN_CLIENT_CHECK_INTERVAL`). Both `connect()` (module singleton) and `PostgresEngine.connect()` (worker pool) consume the result via postgres.js's `connection` option, sending GUCs as startup parameters that survive PgBouncer transaction mode (unlike the prior `setSessionDefaults` post-pool SET, kept as a back-compat no-op shim).
|
||||
- `src/commands/migrate-engine.ts` — Bidirectional engine migration (`gbrain migrate --to supabase/pglite`)
|
||||
- `src/core/import-file.ts` — importFromFile + importFromContent (chunk + embed + tags)
|
||||
- `src/core/sync.ts` — Pure sync functions (manifest parsing, filtering, slug conversion)
|
||||
- `src/core/sync.ts` — Pure sync functions (manifest parsing, filtering, slug conversion). v0.22.12 (#500, foundation by @wintermute via #501): `classifyErrorCode(errorMsg)` regex-based classifier with 12 codes (`SLUG_MISMATCH`, `YAML_PARSE`, `YAML_DUPLICATE_KEY`, `MISSING_OPEN`, `MISSING_CLOSE`, `NESTED_QUOTES`, `EMPTY_FRONTMATTER`, `NULL_BYTES`, `INVALID_UTF8`, `STATEMENT_TIMEOUT`, `FILE_TOO_LARGE`, `SYMLINK_NOT_ALLOWED`) plus `UNKNOWN` fallback. `summarizeFailuresByCode(failures)` returns sorted `[{code, count}]`. `code?` optional field on `SyncFailure`; backfilled at ack time on pre-v0.22.12 entries. `acknowledgeSyncFailures()` returns `AcknowledgeResult { count, summary }`. Three regexes (`MISSING_OPEN`, `MISSING_CLOSE`, `EMPTY_FRONTMATTER`) broadened to match actual `markdown.ts:159-244` validator message strings, not just the literal code-name prefix. `FILE_TOO_LARGE` covers all three production size sites in `import-file.ts:199, 352, 401`; `SYMLINK_NOT_ALLOWED` covers the rejection at `:347`. Closes the silent-skip pattern that motivated #500.
|
||||
- `src/core/storage.ts` — Pluggable storage interface (S3, Supabase Storage, local)
|
||||
- `src/core/storage-config.ts` (v0.22.11) — Storage tiering: `loadStorageConfig` reads `gbrain.yml`, normalizes deprecated keys (`git_tracked` / `supabase_only`) to canonical (`db_tracked` / `db_only`) with once-per-process deprecation warning, and runs `normalizeAndValidateStorageConfig` (auto-fixes missing trailing `/`, throws `StorageConfigError` on tier overlap). Path-segment matcher: `media/x/` does NOT match `media/xerox/foo`. Replaces gray-matter (broken on delimiter-less YAML) with a dedicated parser for the `gbrain.yml` shape.
|
||||
- `src/core/disk-walk.ts` (v0.22.11) — `walkBrainRepo(repoPath)` returns `Map<slug, {size, mtimeMs}>` from one recursive `readdirSync`. Skips dot-dirs, `node_modules`, non-`.md` files. Used by `gbrain storage status` to replace per-page `existsSync + statSync` (~400K syscalls on 200K-page brains → tens).
|
||||
- `src/commands/storage.ts` (v0.22.11) — `gbrain storage status [--repo P] [--json]`. Split into pure data (`getStorageStatus`) + JSON formatter + human formatter (ASCII-only per D10) matching the `orphans.ts` pattern. `PageCountsByTier` and `DiskUsageByTier` are distinct nominal types so swaps fail at compile time.
|
||||
- `gbrain.yml` (brain repo root, v0.22.11) — Optional storage tiering config. Top-level `storage:` section with `db_tracked:` and `db_only:` array-valued keys. `gbrain sync` auto-manages `.gitignore` for `db_only` paths on successful sync (skips on dry-run, blocked-by-failures, submodule context, or `GBRAIN_NO_GITIGNORE=1`). `gbrain export --restore-only [--repo P] [--type T] [--slug-prefix S]` repopulates missing `db_only` files from the database.
|
||||
- `src/core/supabase-admin.ts` — Supabase admin API (project discovery, pgvector check)
|
||||
- `src/core/file-resolver.ts` — File resolution with fallback chain (local -> .redirect.yaml -> .redirect -> .supabase)
|
||||
- `src/core/chunkers/` — 3-tier chunking (recursive, semantic, LLM-guided)
|
||||
- `src/core/search/` — Hybrid search: vector + keyword + RRF + multi-query expansion + dedup
|
||||
- `src/core/chunkers/` — 3-tier chunking (recursive, semantic, LLM-guided). v0.19.0 adds `code.ts` — tree-sitter-based semantic chunker for 29 languages with embedded-asset WASMs (`src/assets/wasm/`), `@dqbd/tiktoken` cl100k_base tokenizer, small-sibling merging. `CHUNKER_VERSION` constant folded into `importCodeFile`'s `content_hash` so chunker shape changes force clean re-chunks across releases.
|
||||
- `src/core/errors.ts` (v0.19.0) — `StructuredAgentError` + `buildError` + `serializeError`. Every new v0.19.0 agent-facing surface (code-def, code-refs, usage errors) uses this envelope; matches v0.17.0 `CycleReport.PhaseResult.error` shape.
|
||||
- `src/assets/wasm/` (v0.19.0) — 36 tree-sitter grammar WASMs + tree-sitter runtime. Committed to the repo so `bun --compile` embeds them deterministically via `import path from ... with { type: 'file' }`. The CI guard `scripts/check-wasm-embedded.sh` fails the build if the compiled binary ever silently falls through to recursive chunks.
|
||||
- `src/commands/code-def.ts` + `src/commands/code-refs.ts` (v0.19.0) — symbol definition + references lookup. Query `content_chunks.symbol_name` or chunk_text ILIKE with `page_kind='code'` filter. Auto-JSON when stdout is not a TTY (gh-CLI convention). Bypass the standard `searchKeyword` `DISTINCT ON (slug)` collapse so multiple call-sites from the same file surface.
|
||||
- `src/core/search/` — Hybrid search: vector + keyword + RRF + multi-query expansion + dedup. As of v0.22.0, `searchKeyword` / `searchKeywordChunks` / `searchVector` apply source-aware ranking at the SQL layer (curated content like `originals/`, `concepts/`, `writing/` outranks bulk content like `wintermute/chat/`, `daily/`, `media/x/`). `searchVector` uses a two-stage CTE so source-boost re-ranking doesn't kill the HNSW index. Hard-exclude prefixes (`test/`, `archive/`, `attachments/`, `.raw/` by default) filter at retrieval, not post-rank. Both gates honor `detail !== 'high'` so temporal queries surface chat pages normally.
|
||||
- `src/core/search/intent.ts` — Query intent classifier (entity/temporal/event/general → auto-selects detail level)
|
||||
- `src/core/search/eval.ts` — Retrieval eval harness: P@k, R@k, MRR, nDCG@k metrics + runEval() orchestrator
|
||||
- `src/core/search/source-boost.ts` (v0.22.0) — Source-type boost map keyed by slug prefix. `DEFAULT_SOURCE_BOOSTS` (originals/ 1.5, concepts/ 1.3, writing/ 1.4, people/companies/deals/ 1.2, daily/ 0.8, media/x/ 0.7, wintermute/chat/ 0.5) and `DEFAULT_HARD_EXCLUDES` (test/, archive/, attachments/, .raw/). `parseSourceBoostEnv` / `parseHardExcludesEnv` parse comma-separated `prefix:factor` pairs from `GBRAIN_SOURCE_BOOST` / `GBRAIN_SEARCH_EXCLUDE` env vars. `resolveBoostMap` and `resolveHardExcludes` merge defaults + env + caller `SearchOpts.exclude_slug_prefixes`/`include_slug_prefixes`.
|
||||
- `src/core/search/sql-ranking.ts` (v0.22.0) — Pure SQL string builders. `buildSourceFactorCase(slugColumn, boostMap, detail)` emits a CASE expression with longest-prefix-match wins (returns literal `'1.0'` when `detail === 'high'` for temporal-bypass parity with COMPILED_TRUTH_BOOST). `buildHardExcludeClause(slugColumn, prefixes)` emits `NOT (col LIKE 'p1%' OR col LIKE 'p2%')` — OR-chain wrapped in NOT, NOT `NOT LIKE ALL/ANY` (those quantifiers don't express set-exclusion). LIKE meta-character escape covers all three of `%`, `_`, AND `\` (backslash matters because it's Postgres LIKE's default escape char). Single-quote doubling on SQL string literals so injection-style inputs are inert text.
|
||||
- `src/commands/eval.ts` — `gbrain eval` command: single-run table + A/B config comparison
|
||||
- `src/core/embedding.ts` — OpenAI text-embedding-3-large, batch, retry, backoff
|
||||
- `src/core/check-resolvable.ts` — Resolver validation: reachability, MECE overlap, DRY checks, structured fix objects. v0.14.1: `CROSS_CUTTING_PATTERNS.conventions` is an array (notability gate accepts both `conventions/quality.md` and `_brain-filing-rules.md`). New `extractDelegationTargets()` parses `> **Convention:**`, `> **Filing rule:**`, and inline backtick references. DRY suppression is proximity-based via `DRY_PROXIMITY_LINES = 40`.
|
||||
@@ -58,12 +67,14 @@ strict behavior when unset.
|
||||
- `src/core/transcription.ts` — Audio transcription: Groq Whisper (default), OpenAI fallback, ffmpeg segmentation for >25MB
|
||||
- `src/core/enrichment-service.ts` — Global enrichment service: entity slug generation, tier auto-escalation, batch throttling
|
||||
- `src/core/data-research.ts` — Recipe validation, field extraction (MRR/ARR regex), dedup, tracker parsing, HTML stripping
|
||||
- `src/commands/extract.ts` — `gbrain extract links|timeline|all [--source fs|db]`: batch link/timeline extraction. fs walks markdown files, db walks pages from the engine (mutation-immune snapshot iteration; use this for live brains with no local checkout). As of v0.12.1 there is no in-memory dedup pre-load — candidates are buffered 100 at a time and flushed via `addLinksBatch` / `addTimelineEntriesBatch`; `ON CONFLICT DO NOTHING` enforces uniqueness at the DB layer, and the `created` counter returns real rows inserted (truthful on re-runs).
|
||||
- `src/commands/embed.ts` — `gbrain embed [--stale|--all] [--slugs ...]`. v0.22.1 (#409, contributed by @atrevino47): `--stale` path now starts with `engine.countStaleChunks()` (single SELECT count(*) WHERE embedding IS NULL, ~50 bytes wire). On a fully-embedded brain that's a 1-line short-circuit — no further reads. When stale chunks exist, `engine.listStaleChunks()` returns just the chunks needing embeddings (slug + chunk_index + chunk_text + metadata, no `vector(1536)` payload). Caller groups by slug, embeds via OpenAI, re-upserts via `upsertChunks`. Replaces the prior page-walk that pulled every chunk's embedding column over the wire and discarded most.
|
||||
- `src/commands/extract.ts` — `gbrain extract links|timeline|all [--source fs|db]`: batch link/timeline extraction. fs walks markdown files, db walks pages from the engine (mutation-immune snapshot iteration; use this for live brains with no local checkout). As of v0.12.1 there is no in-memory dedup pre-load — candidates are buffered 100 at a time and flushed via `addLinksBatch` / `addTimelineEntriesBatch`; `ON CONFLICT DO NOTHING` enforces uniqueness at the DB layer, and the `created` counter returns real rows inserted (truthful on re-runs). v0.22.1 (#417): `ExtractOpts.slugs?: string[]` enables incremental extract — when set, `extractForSlugs()` reads ONLY those slugs' files (single combined links+timeline pass) instead of the full directory walk. CLI `gbrain extract` keeps full-walk behavior; the cycle path threads sync's `pagesAffected` through. `walkMarkdownFiles(brainDir)` still runs at line 455 to build `allSlugs` for link resolution — see `TODOS.md` for replacing it with `engine.getAllSlugs()`.
|
||||
- `src/commands/graph-query.ts` — `gbrain graph-query <slug> [--type T] [--depth N] [--direction in|out|both]`: typed-edge relationship traversal (renders indented tree)
|
||||
- `src/core/link-extraction.ts` — shared library for the v0.12.0 graph layer. extractEntityRefs (canonical, replaces backlinks.ts duplicate) matches both `[Name](people/slug)` markdown links and Obsidian `[[people/slug|Name]]` wikilinks as of v0.12.3. extractPageLinks, inferLinkType heuristics (attended/works_at/invested_in/founded/advises/source/mentions), parseTimelineEntries, isAutoLinkEnabled config helper. `DIR_PATTERN` covers `people`, `companies`, `deals`, `topics`, `concepts`, `projects`, `entities`, `tech`, `finance`, `personal`, `openclaw`. Used by extract.ts, operations.ts auto-link post-hook, and backlinks.ts.
|
||||
- `src/core/minions/` — Minions job queue: BullMQ-inspired, Postgres-native (queue, worker, backoff, types, protected-names, quiet-hours, stagger, handlers/shell).
|
||||
- `src/core/minions/queue.ts` — MinionQueue class (submit, claim, complete, fail, stall detection, parent-child, depth/child-cap, per-job timeouts, cascade-kill, attachments, idempotency keys, child_done inbox, removeOnComplete/Fail). `add()` takes a 4th `trusted` arg (separate from `opts` to prevent spread leakage); protected names in `PROTECTED_JOB_NAMES` require `{allowProtectedSubmit: true}` and the check runs trim-normalized (whitespace-bypass safe). v0.14.1 #219: `add()` plumbs `max_stalled` through with a `[1, 100]` clamp; omitted values let the schema DEFAULT (5) kick in. v0.19.0: `handleWallClockTimeouts(lockDurationMs)` is Layer 3 kill shot for jobs where `FOR UPDATE SKIP LOCKED` stall detection and the timeout sweep both fail to evict (wedged worker holding a row lock via a pending transaction). v0.19.1: `maxWaiting` coalesce path now uses `pg_advisory_xact_lock` keyed on `(name, queue)` to serialize concurrent submits for the same key, and filters on `queue` in addition to `name` so cross-queue same-name jobs don't suppress each other.
|
||||
- `src/core/minions/worker.ts` — MinionWorker class (handler registry, lock renewal, graceful shutdown, timeout safety net). v0.14.0 abort-path fix: aborted jobs now call `failJob` with reason (`timeout`/`cancel`/`lock-lost`/`shutdown`) instead of returning silently. `shutdownAbort` (instance field) fires on process SIGTERM/SIGINT and propagates to `ctx.shutdownSignal` — shell handler listens to it; non-shell handlers don't.
|
||||
- `src/core/minions/worker.ts` — MinionWorker class (handler registry, lock renewal, graceful shutdown, timeout safety net). v0.14.0 abort-path fix: aborted jobs now call `failJob` with reason (`timeout`/`cancel`/`lock-lost`/`shutdown`) instead of returning silently. `shutdownAbort` (instance field) fires on process SIGTERM/SIGINT and propagates to `ctx.shutdownSignal` — shell handler listens to it; non-shell handlers don't. v0.22.1 (#403): per-job timeout fires `abort.abort(new Error('timeout'))` then a 30-second grace-then-evict safety net force-evicts the job from `inFlight` and marks it dead in DB if the handler ignores the abort signal — frees the slot even when a handler wedges (the 98-waiting-0-active prod incident driver).
|
||||
- `src/core/minions/supervisor.ts` — MinionSupervisor process manager. Spawns `gbrain jobs work` as a child, restarts on crash with exponential backoff, periodic health check. v0.22.1 (#406): `consecutiveHealthFailures` counter; on 3 consecutive failures emits `health_warn` with `reason: 'db_connection_degraded'` and calls `engine.reconnect()` to swap in a fresh pool, then resets the counter. Worker exit classifier emits `likely_cause` field on `worker_exited` events: `oom_or_external_kill` (SIGKILL), `graceful_shutdown` (SIGTERM), `runtime_error` (code 1), `clean_exit` (code 0), `unknown`.
|
||||
- `src/core/minions/types.ts` — `MinionJobInput` + `MinionJobStatus` + handler context types. `MinionJobInput.max_stalled` (new in v0.14.1) is optional; omitted values let the schema DEFAULT (5) kick in, provided values are clamped to `[1, 100]`.
|
||||
- `src/core/minions/protected-names.ts` — side-effect-free constant module exporting `PROTECTED_JOB_NAMES` + `isProtectedJobName()`. Kept pure so queue core can import without loading handler modules.
|
||||
- `src/core/minions/handlers/shell.ts` — `shell` job handler. Spawns `/bin/sh -c cmd` (absolute path, PATH-override-safe) or `argv[0] argv[1..]` (no shell). Env allowlist: `PATH, HOME, USER, LANG, TZ, NODE_ENV` + caller `env:` overrides. UTF-8-safe stdout/stderr tail via `string_decoder.StringDecoder`. Abort (either `ctx.signal` or `ctx.shutdownSignal`) fires SIGTERM → 5s grace → SIGKILL on child. Requires `GBRAIN_ALLOW_SHELL_JOBS=1` on worker (gated by `registerBuiltinHandlers`).
|
||||
@@ -81,20 +92,27 @@ strict behavior when unset.
|
||||
- `src/core/minions/attachments.ts` — Attachment validation (path traversal, null byte, oversize, base64, duplicate detection)
|
||||
- `src/commands/agent.ts` (v0.16) — `gbrain agent run <prompt> [flags]` CLI. Submits `subagent` (or N children + 1 aggregator) under `{allowProtectedSubmit: true}`. Single-entry `--fanout-manifest` short-circuits. Children get `on_child_fail: 'continue'` + `max_stalled: 3`. `--follow` is the default on TTY; streams logs + polls `waitForCompletion` in parallel. Ctrl-C detaches, does not cancel.
|
||||
- `src/commands/agent-logs.ts` (v0.16) — `gbrain agent logs <job> [--follow] [--since]`. Merges JSONL heartbeat audit + `subagent_messages` into a chronological timeline. `parseSince` accepts ISO-8601 or relative (`5m`, `1h`, `2d`). Transcript tail renders only for terminal jobs.
|
||||
- `src/commands/jobs.ts` — `gbrain jobs` CLI subcommands + `gbrain jobs work` daemon. v0.13.1 surfaces the full `MinionJobInput` retry/backoff/timeout/idempotency surface as first-class CLI flags on `jobs submit`: `--max-stalled`, `--backoff-type fixed|exponential`, `--backoff-delay`, `--backoff-jitter`, `--timeout-ms`, `--idempotency-key`. `jobs smoke --sigkill-rescue` is the opt-in regression guard for #219. v0.16 wires `registerBuiltinHandlers` to always register `subagent` + `subagent_aggregator` (no env flag — `ANTHROPIC_API_KEY` is the natural cost gate, trust is via `PROTECTED_JOB_NAMES`) and loads `GBRAIN_PLUGIN_PATH` plugins at worker startup with a loud startup-line per plugin. `shell` handler still gated by `GBRAIN_ALLOW_SHELL_JOBS=1` (RCE surface, separate concern).
|
||||
- `src/commands/jobs.ts` — `gbrain jobs` CLI subcommands + `gbrain jobs work` daemon. v0.13.1 surfaces the full `MinionJobInput` retry/backoff/timeout/idempotency surface as first-class CLI flags on `jobs submit`: `--max-stalled`, `--backoff-type fixed|exponential`, `--backoff-delay`, `--backoff-jitter`, `--timeout-ms`, `--idempotency-key`. `jobs smoke --sigkill-rescue` is the opt-in regression guard for #219. v0.16 wires `registerBuiltinHandlers` to always register `subagent` + `subagent_aggregator` (no env flag — `ANTHROPIC_API_KEY` is the natural cost gate, trust is via `PROTECTED_JOB_NAMES`) and loads `GBRAIN_PLUGIN_PATH` plugins at worker startup with a loud startup-line per plugin. `shell` handler still gated by `GBRAIN_ALLOW_SHELL_JOBS=1` (RCE surface, separate concern). v0.22.10 (#521): the `autopilot-cycle` handler now forwards `job.data.phases` to `runCycle` (was previously discarded — caller-supplied phase selection silently became a full cycle). Phases are validated against `ALL_PHASES` from `src/core/cycle.ts`; invalid names are filtered out and an empty/missing array falls back to the default 6-phase cycle. v0.22.13 (PR #490 CODEX-1+CODEX-4): `sync` handler now resolves `sourceId` at entry by looking up `sources.local_path` (mirrors `cycle.ts:480`'s autopilot fix from PR #475) so multi-source brains read the per-source `last_commit` anchor instead of the global config key. Concurrency routed through the shared `autoConcurrency()` policy in `src/core/sync-concurrency.ts` instead of the prior hardcoded `4`; PGLite stays serial. `noEmbed` default is `true` (embed is a separate job — submit `gbrain embed --stale` after sync, or rely on the autopilot cycle's embed phase).
|
||||
- `src/commands/features.ts` — `gbrain features --json --auto-fix`: usage scan + feature adoption salesman
|
||||
- `src/commands/autopilot.ts` — `gbrain autopilot --install`: self-maintaining brain daemon (sync+extract+embed)
|
||||
- `src/mcp/server.ts` — MCP stdio server (generated from operations)
|
||||
- `src/commands/auth.ts` — Standalone token management (create/list/revoke/test)
|
||||
- `src/mcp/server.ts` — MCP stdio server (generated from operations). v0.22.7: tool-call handler delegates to `dispatchToolCall` from `src/mcp/dispatch.ts` so stdio + HTTP transports share one validation, context-build, and error-format path.
|
||||
- `src/mcp/dispatch.ts` (v0.22.7) — Shared tool-call dispatch consumed by both stdio (`server.ts`) and HTTP (`http-transport.ts`). Exports `dispatchToolCall(engine, name, params, opts)`, `buildOperationContext(engine, params, opts)`, and `validateParams(op, params)`. Single source of truth for `(ctx, params)` handler arg order and the 5-field `OperationContext` shape (engine + config + logger + dryRun + remote). Defaults to `remote: true` (untrusted); local CLI callers pass `remote: false`. Closed F1 (reversed handler args) + F2 (incomplete OperationContext) + F3 (no param validation) drift bugs in the original v0.22.5 HTTP transport.
|
||||
- `src/mcp/rate-limit.ts` (v0.22.7) — Bounded-LRU token-bucket limiter for `gbrain serve --http`. `buildDefaultLimiters()` returns the two-bucket pipeline used by http-transport: pre-auth IP (default 30/60s, fires BEFORE the DB lookup so brute-force load against `access_tokens` is actually capped) + post-auth token-id (default 60/60s). Tracks `lastTouchedMs` separately from `lastRefillMs` so an exhausted key can't be reset by hammering past the TTL. LRU cap (default 10K keys) bounds memory under attacker-controlled key growth; TTL prune at 2× window evicts abandoned buckets.
|
||||
- `src/mcp/http-transport.ts` (v0.22.7, rewrite) — `gbrain serve --http` HTTP transport. Postgres-only — fails fast at startup on PGLite (the `access_tokens` table only exists on Postgres). Bearer auth against SHA-256 hashes in `access_tokens`. CORS default-deny via `GBRAIN_HTTP_CORS_ORIGIN` allowlist. Body cap stream-counted (1 MiB default via `GBRAIN_HTTP_MAX_BODY_BYTES`) so chunked transfers without Content-Length still hit the cap. `last_used_at` SQL-level debounce (one UPDATE per token per 60s). Per-request audit row in `mcp_request_log` with token_name + operation + status + latency. Optional `GBRAIN_HTTP_TRUST_PROXY=1` honors `X-Forwarded-For` — only safe when bound to a private interface AND the proxy strips client-supplied XFF (otherwise enables IP spoofing past the pre-auth rate limit). `/health` does `SELECT 1` against Postgres and returns 503 + `status:unhealthy` when the DB is unreachable so orchestration doesn't see green pods while clients get misleading 401s. Replaces the standalone OAuth wrapper that was vulnerable to unauthenticated client registration.
|
||||
- `src/commands/auth.ts` — Token management for the HTTP transport. `gbrain auth create/list/revoke/test`. As of v0.22.7 wired into the main CLI (`src/cli.ts`); also runs standalone via `bun run src/commands/auth.ts ...` for environments without a compiled binary. Tokens stored as SHA-256 hashes in `access_tokens` (Postgres-only).
|
||||
- `src/commands/upgrade.ts` — Self-update CLI. `runPostUpgrade()` enumerates migrations from the TS registry (src/commands/migrations/index.ts) and tail-calls `runApplyMigrations(['--yes', '--non-interactive'])` so the mechanical side of every outstanding migration runs unconditionally.
|
||||
- `src/commands/migrations/` — TS migration registry (compiled into the binary; no filesystem walk of `skills/migrations/*.md` needed at runtime). `index.ts` lists migrations in semver order. `v0_11_0.ts` = Minions adoption orchestrator (8 phases). `v0_12_0.ts` = Knowledge Graph auto-wire orchestrator (5 phases: schema → config check → backfill links → backfill timeline → verify). `phaseASchema` has a 600s timeout (bumped from 60s in v0.12.1 for duplicate-heavy brains). `v0_12_2.ts` = JSONB double-encode repair orchestrator (4 phases: schema → repair-jsonb → verify → record). `v0_14_0.ts` = shell-jobs + autopilot cooperative (2 phases: schema ALTER minion_jobs.max_stalled SET DEFAULT 3 — superseded by v0.14.3's schema-level DEFAULT 5 + UPDATE backfill; pending-host-work ping for skills/migrations/v0.14.0.md). All orchestrators are idempotent and resumable from `partial` status. As of v0.14.2 (Bug 3), the RUNNER owns all ledger writes — orchestrators return `OrchestratorResult` and `apply-migrations.ts` persists a canonical `{version, status, phases}` shape after return. Orchestrators no longer call `appendCompletedMigration` directly. `statusForVersion` prefers `complete` over `partial` (never regresses). 3 consecutive partials → wedged → `--force-retry <version>` writes a `'retry'` reset marker. v0.14.3 (fix wave) ships schema-only migrations v14 (`pages_updated_at_index`) + v15 (`minion_jobs_max_stalled_default_5` with UPDATE backfill) via the `MIGRATIONS` array in `src/core/migrate.ts` — no orchestrator phases needed.
|
||||
- `src/commands/repair-jsonb.ts` — `gbrain repair-jsonb [--dry-run] [--json]`: rewrites `jsonb_typeof='string'` rows in place across 5 affected columns (pages.frontmatter, raw_data.data, ingest_log.pages_updated, files.metadata, page_versions.frontmatter). Fixes v0.12.0 double-encode bug on Postgres; PGLite no-ops. Idempotent.
|
||||
- `src/commands/orphans.ts` — `gbrain orphans [--json] [--count] [--include-pseudo]`: surfaces pages with zero inbound wikilinks, grouped by domain. Auto-generated/raw/pseudo pages filtered by default. Also exposed as `find_orphans` MCP operation. Shipped in v0.12.3 (contributed by @knee5).
|
||||
- `src/commands/doctor.ts` — `gbrain doctor [--json] [--fast] [--fix] [--dry-run] [--index-audit]`: health checks. v0.12.3 added `jsonb_integrity` + `markdown_body_completeness` reliability checks. v0.14.1: `--fix` delegates inlined cross-cutting rules to `> **Convention:** see [path](path).` callouts (pipes DRY violations into `src/core/dry-fix.ts`); `--fix --dry-run` previews without writing. v0.14.2: `schema_version` check fails loudly when `version=0` (migrations never ran — the #218 `bun install -g` signature) and routes users to `gbrain apply-migrations --yes`; new opt-in `--index-audit` flag (Postgres-only) reports zero-scan indexes from `pg_stat_user_indexes` (informational only, no auto-drop). v0.15.2: every DB check is wrapped in a progress phase; `markdown_body_completeness` runs under a 1s heartbeat timer so 10+ min scans are observable on 50K-page brains. v0.19.1 added `queue_health` (Postgres-only) with two subchecks: stalled-forever active jobs (started_at > 1h) and waiting-depth-per-name > threshold (default 10, override via `GBRAIN_QUEUE_WAITING_THRESHOLD`). Worker-heartbeat subcheck intentionally deferred to follow-up B7 because it needs a `minion_workers` table to produce ground-truth signal. Fix hints point at `gbrain repair-jsonb`, `gbrain sync --force`, `gbrain apply-migrations`, and `gbrain jobs get/cancel <id>`.
|
||||
- `src/core/migrate.ts` — schema-migration runner. Owns the `MIGRATIONS` array (source of truth for schema DDL). v0.14.2 extended the `Migration` interface with `sqlFor?: { postgres?, pglite? }` (engine-specific SQL overrides `sql`) and `transaction?: boolean` (set to false for `CREATE INDEX CONCURRENTLY`, which Postgres refuses inside a transaction; ignored on PGLite since it has no concurrent writers). Migration v14 (fix wave) uses a handler branching on `engine.kind` to run CONCURRENTLY on Postgres (with a pre-drop of any invalid remnant via `pg_index.indisvalid`) and plain `CREATE INDEX` on PGLite. v15 bumps `minion_jobs.max_stalled` default 1→5 and backfills existing non-terminal rows.
|
||||
- `src/commands/integrity.ts` — `gbrain integrity check|auto|review|extract`: bare-tweet detection, dead-link detection, three-bucket repair (auto-repair / review-queue / skip). `scanIntegrity()` is the shared library function called from `gbrain doctor` (sampled at limit=500) and `cmdCheck` (full scan). v0.22.8: batch-load fast path on Postgres uses `SELECT DISTINCT ON (slug)` in a single SQL query to fix the PgBouncer round-trip timeout (60s → ~6s) while preserving `engine.getAllSlugs()`'s `Set<string>` semantics on multi-source brains. Gated by `engine.kind === 'postgres'` at the call site so PGLite never enters batch; fallback `catch` logs at `GBRAIN_DEBUG=1` so real Postgres errors are diagnosable.
|
||||
- `src/commands/doctor.ts` — `gbrain doctor [--json] [--fast] [--fix] [--dry-run] [--index-audit]`: health checks. v0.12.3 added `jsonb_integrity` + `markdown_body_completeness` reliability checks. v0.14.1: `--fix` delegates inlined cross-cutting rules to `> **Convention:** see [path](path).` callouts (pipes DRY violations into `src/core/dry-fix.ts`); `--fix --dry-run` previews without writing. v0.14.2: `schema_version` check fails loudly when `version=0` (migrations never ran — the #218 `bun install -g` signature) and routes users to `gbrain apply-migrations --yes`; new opt-in `--index-audit` flag (Postgres-only) reports zero-scan indexes from `pg_stat_user_indexes` (informational only, no auto-drop). v0.15.2: every DB check is wrapped in a progress phase; `markdown_body_completeness` runs under a 1s heartbeat timer so 10+ min scans are observable on 50K-page brains. v0.19.1 added `queue_health` (Postgres-only) with two subchecks: stalled-forever active jobs (started_at > 1h) and waiting-depth-per-name > threshold (default 10, override via `GBRAIN_QUEUE_WAITING_THRESHOLD`). Worker-heartbeat subcheck intentionally deferred to follow-up B7 because it needs a `minion_workers` table to produce ground-truth signal. Fix hints point at `gbrain repair-jsonb`, `gbrain sync --force`, `gbrain apply-migrations`, and `gbrain jobs get/cancel <id>`. v0.22.12 (#500): `sync_failures` check shows `[CODE=N, ...]` breakdown for both unacked entries (warn) and acked-historical entries (ok), surfacing systemic failure modes (`SLUG_MISMATCH=2685`) instead of a bare count.
|
||||
- `src/core/migrate.ts` — schema-migration runner. Owns the `MIGRATIONS` array (source of truth for schema DDL). v0.14.2 extended the `Migration` interface with `sqlFor?: { postgres?, pglite? }` (engine-specific SQL overrides `sql`) and `transaction?: boolean` (set to false for `CREATE INDEX CONCURRENTLY`, which Postgres refuses inside a transaction; ignored on PGLite since it has no concurrent writers). Migration v14 (fix wave) uses a handler branching on `engine.kind` to run CONCURRENTLY on Postgres (with a pre-drop of any invalid remnant via `pg_index.indisvalid`) and plain `CREATE INDEX` on PGLite. v15 bumps `minion_jobs.max_stalled` default 1→5 and backfills existing non-terminal rows. v0.22.6.1: migration v24 (`rls_backfill_missing_tables`) uses `sqlFor: { pglite: '' }` to no-op on PGLite — PGLite has no RLS engine and is single-tenant by definition, and the v24 ALTERs target subagent tables that don't exist in pglite-schema.ts. Closes #395 (contributed by @jdcastro2).
|
||||
- `src/core/progress.ts` — Shared bulk-action progress reporter. Writes to stderr. Modes: `auto` (TTY: `\r`-rewriting; non-TTY: plain lines), `human`, `json` (JSONL), `quiet`. Rate-gated by `minIntervalMs` and `minItems`. `startHeartbeat(reporter, note)` helper for single long queries. `child()` composes phase paths. Singleton SIGINT/SIGTERM coordinator emits `abort` events for every live phase. EPIPE defense on both sync throws and stream `'error'` events. Zero dependencies. Introduced in v0.15.2.
|
||||
- `src/core/cli-options.ts` — Global CLI flag parser. `parseGlobalFlags(argv)` returns `{cliOpts, rest}` with `--quiet` / `--progress-json` / `--progress-interval=<ms>` stripped. `getCliOptions()` / `setCliOptions()` expose a module-level singleton so commands reach the resolved flags without parameter threading. `cliOptsToProgressOptions()` maps to reporter options. `childGlobalFlags()` returns the flag suffix to append to `execSync('gbrain ...')` calls in migration orchestrators. `OperationContext.cliOpts` extends shared-op dispatch for MCP callers.
|
||||
- `src/core/cycle.ts` — v0.17 brain maintenance cycle primitive. `runCycle(engine: BrainEngine | null, opts: CycleOpts): Promise<CycleReport>` composes 6 phases in semantically-driven order (lint → backlinks → sync → extract → embed → orphans). Three callers: `gbrain dream` CLI, `gbrain autopilot` daemon's inline path, and the Minions `autopilot-cycle` handler (`src/commands/jobs.ts`). One source of truth for what the brain does overnight. Coordination via `gbrain_cycle_locks` DB table (TTL-based; works through PgBouncer transaction pooling, unlike session-scoped `pg_try_advisory_lock`) + `~/.gbrain/cycle.lock` file lock with PID-liveness for PGLite / engine=null mode. `CycleReport.schema_version: "1"` is the stable agent-consumable shape. `PhaseResult.error: { class, code, message, hint?, docs_url? }` is Stripe-API-tier structured failure info. `yieldBetweenPhases` hook awaited between every phase — Minions handler uses this to renew its job lock and prevent v0.14 stall-death regression. Engine nullable: filesystem phases (lint, backlinks) run without DB; DB phases skip with `status: "skipped", reason: "no_database"`. Lock-skip: read-only phase selections (`--phase orphans`) bypass the cycle lock.
|
||||
- `src/core/db-lock.ts` (v0.22.13) — generic `tryAcquireDbLock(engine, lockId, ttlMinutes)` over the existing `gbrain_cycle_locks` table. Parameterized lock id so different scopes can nest cleanly: `gbrain-cycle` for the broad cycle (held by `cycle.ts`) and `gbrain-sync` (`SYNC_LOCK_ID` constant) for `performSync`'s narrower writer window. Same UPSERT-with-TTL semantics as the prior cycle-only helper, just generalized. Survives PgBouncer transaction pooling (unlike session-scoped `pg_try_advisory_lock`); crashed holders auto-release once their TTL expires.
|
||||
- `src/core/sync-concurrency.ts` (v0.22.13) — single source of truth for the parallel-sync policy. Exports `autoConcurrency(engine, fileCount, override?)` (PGLite always serial; explicit override clamped to >=1; auto path returns `DEFAULT_PARALLEL_WORKERS=4` when `fileCount > AUTO_CONCURRENCY_FILE_THRESHOLD=100`), `shouldRunParallel(workers, fileCount, explicit)` (Q1: explicit `--workers` bypasses the >50-file floor), and `parseWorkers(s)` (rejects `'0'`, `'-3'`, `'foo'`, `'1.5'`, trailing chars — replaces the prior parseInt-with-no-validation in both `sync.ts` and `import.ts`). Used by `performSync`, `performFullSync`, `runImport`, and the Minion `sync` handler so the three sites can no longer drift.
|
||||
- `src/commands/sync.ts` — `gbrain sync` CLI + the `performSync` / `performFullSync` library entrypoints (consumed by the autopilot cycle and the Minion sync handler). v0.22.13 (PR #490): `performSync` wraps its body in a `gbrain-sync` writer lock so two concurrent syncs (manual + autopilot, two terminals, two Conductor workspaces) cannot both write `last_commit` and let the last writer win. Head-drift gate after the import phase re-checks `git rev-parse HEAD`; if HEAD moved (someone ran `git checkout` / `git pull` mid-sync), the bookmark refuses to advance. Vanished files now record a failedFiles entry instead of silent-skip — the silent-skip-then-advance pathology that survived prior hardening passes is dead. Worker engines wrap in try/finally so disconnect always fires (panic-path leak fix). Both PGLite-detection sites use `engine.kind === 'pglite'`. CLI accepts `--workers N` (alias `--concurrency N`), validated via `parseWorkers`. Explicit `--workers` bypasses the auto-path file-count floor; auto path defers to `autoConcurrency()`. Banner moved to stderr.
|
||||
- `src/core/cycle.ts` — v0.17 brain maintenance cycle primitive. `runCycle(engine: BrainEngine | null, opts: CycleOpts): Promise<CycleReport>` composes 6 phases in semantically-driven order (lint → backlinks → sync → extract → embed → orphans). Three callers: `gbrain dream` CLI, `gbrain autopilot` daemon's inline path, and the Minions `autopilot-cycle` handler (`src/commands/jobs.ts`). One source of truth for what the brain does overnight. Coordination via `gbrain_cycle_locks` DB table (TTL-based; works through PgBouncer transaction pooling, unlike session-scoped `pg_try_advisory_lock`) + `~/.gbrain/cycle.lock` file lock with PID-liveness for PGLite / engine=null mode. `CycleReport.schema_version: "1"` is the stable agent-consumable shape. `PhaseResult.error: { class, code, message, hint?, docs_url? }` is Stripe-API-tier structured failure info. `yieldBetweenPhases` hook awaited between every phase — Minions handler uses this to renew its job lock and prevent v0.14 stall-death regression. Engine nullable: filesystem phases (lint, backlinks) run without DB; DB phases skip with `status: "skipped", reason: "no_database"`. Lock-skip: read-only phase selections (`--phase orphans`) bypass the cycle lock. v0.22.1 (#403): `CycleOpts.signal?: AbortSignal` propagates the worker's abort signal; `checkAborted()` fires between every phase and throws if the signal is aborted (cooperative — can't interrupt a phase mid-execution). v0.22.1 (#417): `runPhaseSync` returns `pagesAffected` via `SyncPhaseResult`; `runCycle` captures it and threads to `runPhaseExtract` as the 4th arg, enabling incremental extract on the cycle path. v0.22.1 (Codex F2): `runPhaseSync` takes `willRunExtractPhase: boolean` and sets `noExtract: phases.includes('extract')` so `gbrain dream --phase sync` doesn't silently lose extraction. v0.22.5 (#475): new `resolveSourceForDir(engine, brainDir)` helper queries `SELECT id FROM sources WHERE local_path = $1 LIMIT 1`; `runPhaseSync` threads result as `sourceId` to `performSync()` so sync reads the per-source `sources.last_commit` anchor instead of the drift-prone global `config.sync.last_commit` key. Bare try/catch lets pre-v0.18 brains fall through to the global key. Closes the prod hang where every autopilot cycle ran a 30-min full reimport because the global anchor commit had been GC'd from git history.
|
||||
- `src/commands/dream.ts` — v0.17 `gbrain dream` CLI. ~80-line thin alias over `runCycle`. brainDir resolution requires explicit `--dir` OR `sync.repo_path` config (no more walk-up-cwd-for-.git footgun). Flags: `--dry-run`, `--json`, `--phase <name>`, `--pull`, `--dir <path>`. Exit code 1 on status=failed (partial/warn not fatal — don't page on warnings).
|
||||
- `scripts/check-progress-to-stdout.sh` — CI guard against regressing to `\r`-on-stdout progress. Wired into `bun run test` via `scripts/check-progress-to-stdout.sh && bun test` in package.json.
|
||||
- `docs/progress-events.md` — Canonical JSON event schema reference. Stable from v0.15.2, additive only.
|
||||
@@ -205,6 +223,10 @@ Key commands added in v0.14.3 (fix wave):
|
||||
- `gbrain jobs submit` gains `--max-stalled`, `--backoff-type`, `--backoff-delay`, `--backoff-jitter`, `--timeout-ms`, `--idempotency-key` — exposing existing `MinionJobInput` fields as first-class CLI flags.
|
||||
- `gbrain jobs smoke --sigkill-rescue` — opt-in regression smoke case simulating a killed worker; asserts the v0.14.3 schema default (`max_stalled=5`) actually rescues on first stall.
|
||||
|
||||
Key commands added in v0.22.13 (PR #490):
|
||||
- `gbrain sync --workers N` (alias `--concurrency N`) — parallelize the import phase using per-worker Postgres engines (small pool of 2 each) with an atomic queue index. Auto-concurrency: defaults to 4 workers when the diff exceeds 100 files. Smaller diffs stay serial. Explicit `--workers` always wins (even on a 30-file diff). PGLite forces serial regardless. Validation rejects `0`, negatives, non-integers loud (replaces the prior silent fall-through to auto-concurrency).
|
||||
- `gbrain import --workers N` — same `parseWorkers()` validation as sync; same try/finally worker-engine cleanup. Behavior surface unchanged.
|
||||
|
||||
## Testing
|
||||
|
||||
`bun test` runs all tests. After the v0.12.1 release: ~75 unit test files + 8 E2E test files (1412 unit pass, 119 E2E when `DATABASE_URL` is set — skip gracefully otherwise). Unit tests run
|
||||
@@ -216,7 +238,9 @@ parity), `test/cli.test.ts` (CLI structure), `test/config.test.ts` (config redac
|
||||
`test/files.test.ts` (MIME/hash), `test/import-file.test.ts` (import pipeline),
|
||||
`test/upgrade.test.ts` (schema migrations),
|
||||
`test/file-migration.test.ts` (file migration), `test/file-resolver.test.ts` (file resolution),
|
||||
`test/import-resume.test.ts` (import checkpoints), `test/migrate.test.ts` (migration; v8/v9 helper-btree-index SQL structural assertions + 1000-row wall-clock fixtures that guard the O(n²)→O(n log n) fix + v0.13.1 assertions on v12/v13 SQL shape, `sqlFor` + `transaction:false` runner semantics, and the `max_stalled DEFAULT 1` regression guard),
|
||||
`test/import-resume.test.ts` (import checkpoints), `test/migrate.test.ts` (migration; v8/v9 helper-btree-index SQL structural assertions + 1000-row wall-clock fixtures that guard the O(n²)→O(n log n) fix + v0.13.1 assertions on v12/v13 SQL shape, `sqlFor` + `transaction:false` runner semantics, the `max_stalled DEFAULT 1` regression guard, and v0.22.6.1 v24 `sqlFor.pglite: ''` no-op assertion),
|
||||
`test/bootstrap.test.ts` (v0.22.6.1 — bootstrap contract: no-op on fresh install, idempotent across two `initSchema()` calls, no-op on modern brain that already has every probed column, full bootstrap path on simulated pre-v0.18 brain, fresh-install regression guard, pre-v0.13 `links` shape coverage),
|
||||
`test/schema-bootstrap-coverage.test.ts` (v0.22.6.1 CI guard — `REQUIRED_BOOTSTRAP_COVERAGE` lists every forward reference in PGLITE_SCHEMA_SQL; the test fails loudly if `applyForwardReferenceBootstrap` skips one. When you add a column-with-index to the embedded schema blob, you extend both arrays or this guard fails. The pattern that broke gbrain ten times in two years is now structurally prevented.),
|
||||
`test/setup-branching.test.ts` (setup flow), `test/slug-validation.test.ts` (slug validation),
|
||||
`test/storage.test.ts` (storage backends), `test/supabase-admin.test.ts` (Supabase admin),
|
||||
`test/yaml-lite.test.ts` (YAML parsing), `test/check-update.test.ts` (version check + update CLI),
|
||||
@@ -230,6 +254,7 @@ parity), `test/cli.test.ts` (CLI structure), `test/config.test.ts` (config redac
|
||||
`test/skills-conformance.test.ts` (skill frontmatter + required sections validation),
|
||||
`test/resolver.test.ts` (RESOLVER.md coverage, routing validation + v0.20.4 round-trip: every quoted RESOLVER.md trigger must match a frontmatter `triggers:` entry in the target skill, and every `name="<word>"` reference in any SKILL.md must resolve to a declared op in `src/core/operations.ts` or a Minions handler in `PROTECTED_JOB_NAMES`),
|
||||
`test/search.test.ts` (RRF normalization, compiled truth boost, cosine similarity, dedup key),
|
||||
`test/sql-ranking.test.ts` (v0.22.0 source-boost helpers: 39 cases covering longest-prefix-match in SQL CASE, detail=high temporal-bypass, three-meta-char LIKE escape (%, _, \\), single-quote SQL-literal doubling, env override parsing for GBRAIN_SOURCE_BOOST + GBRAIN_SEARCH_EXCLUDE, resolveBoostMap / resolveHardExcludes merge semantics),
|
||||
`test/dedup.test.ts` (source-aware dedup, compiled truth guarantee, layer interactions),
|
||||
`test/intent.test.ts` (query intent classification: entity/temporal/event/general),
|
||||
`test/eval.test.ts` (retrieval metrics: precisionAtK, recallAtK, mrr, ndcgAtK, parseQrels),
|
||||
@@ -257,6 +282,9 @@ parity), `test/cli.test.ts` (CLI structure), `test/config.test.ts` (config redac
|
||||
`test/orphans.test.ts` (v0.12.3 orphans command: detection, pseudo filtering, text/json/count outputs, MCP op),
|
||||
`test/postgres-engine.test.ts` (v0.12.3 statement_timeout scoping: `sql.begin` + `SET LOCAL` shape, source-level grep guardrail against reintroduced bare `SET statement_timeout`),
|
||||
`test/sync.test.ts` (sync logic + v0.12.3 regression guard asserting top-level `engine.transaction` is not called),
|
||||
`test/sync-concurrency.test.ts` (v0.22.13 PR #490: 17 cases covering `autoConcurrency()` thresholds + PGLite-forces-serial + explicit-override clamping, `shouldRunParallel()` Q1 explicit-bypasses-floor contract, and `parseWorkers()` validation that rejects `'0'`/`'-3'`/`'foo'`/`'1.5'`/trailing chars),
|
||||
`test/sync-parallel.test.ts` (v0.22.13 PR #490: PGLite-routed coverage of the bookmark gate under concurrency request, head-drift gate, vanished-file failure capture, PGLite-stays-serial, and the `gbrain-sync` writer-lock contract — 7 cases),
|
||||
`test/sync-failures.test.ts` (v0.22.12: 28 cases pinning `classifyErrorCode` regex coverage for all 12 codes against literal production message strings from `markdown.ts:159-244` and `import-file.ts:199, 347, 352, 401`; `summarizeFailuresByCode` sort + pre-classified-honor; `recordSyncFailures` code-field persistence; `acknowledgeSyncFailures` AcknowledgeResult shape + backfill on pre-v0.22.12 entries),
|
||||
`test/doctor.test.ts` (doctor command + v0.12.3 assertions that `jsonb_integrity` scans the four v0.12.0 write sites and `markdown_body_completeness` is present),
|
||||
`test/utils.test.ts` (shared SQL utilities + `tryParseEmbedding` null-return and single-warn semantics),
|
||||
`test/build-llms.test.ts` (llms.txt/llms-full.txt generator: path resolution, idempotence, spec shape, regen-drift guard, content contract, AGENTS.md install-path mirror, size-budget enforcement — 7 cases),
|
||||
@@ -267,17 +295,26 @@ parity), `test/cli.test.ts` (CLI structure), `test/config.test.ts` (config redac
|
||||
`test/skill-manifest.test.ts` (v0.19 skill manifest parser: drift detection, managed-block markers),
|
||||
`test/skillify-scaffold.test.ts` (v0.19 `gbrain skillify scaffold` stubs: SKILL.md, script, tests, routing-eval fixtures),
|
||||
`test/skillpack-install.test.ts` (v0.19 `gbrain skillpack install` managed-block install / update / no-clobber semantics),
|
||||
`test/skillpack-sync-guard.test.ts` (v0.19 sync-guard: bundled skills stay byte-identical to `skills/` source).
|
||||
`test/skillpack-sync-guard.test.ts` (v0.19 sync-guard: bundled skills stay byte-identical to `skills/` source),
|
||||
`test/http-transport.test.ts` (v0.22.7 HTTP transport: 23 unit cases covering bearer auth + missing/no-Bearer/unknown/revoked + `/health` bypass, F1+F2 round-trip via dispatch.ts, F3 invalid_params, application/json response shape (not SSE), CORS default-deny + allowlist, body cap on Content-Length AND chunked, two-bucket rate limit (refill, exhaust+Retry-After, LRU eviction, TTL prune, pre-auth IP fires before DB), and `mcp_request_log` audit on success + auth_failed).
|
||||
|
||||
E2E tests (`test/e2e/`): Run against real Postgres+pgvector. Require `DATABASE_URL`.
|
||||
- `bun run test:e2e` runs Tier 1 (mechanical, all operations, no API keys). Includes 9 dedicated cases for the postgres-engine `addLinksBatch` / `addTimelineEntriesBatch` bind path — postgres-js's `unnest()` binding is structurally different from PGLite's and gets its own coverage.
|
||||
- `test/e2e/search-quality.test.ts` runs search quality E2E against PGLite (no API keys, in-memory)
|
||||
- `test/e2e/graph-quality.test.ts` runs the v0.10.3 knowledge graph pipeline (auto-link via put_page, reconciliation, traversePaths) against PGLite in-memory
|
||||
- `test/e2e/postgres-jsonb.test.ts` — v0.12.2 regression test. Round-trips all 5 JSONB write sites (pages.frontmatter, raw_data.data, ingest_log.pages_updated, files.metadata, page_versions.frontmatter) against real Postgres and asserts `jsonb_typeof='object'` plus `->>'key'` returns the expected scalar. The test that should have caught the original double-encode bug.
|
||||
- `test/e2e/integrity-batch.test.ts` (v0.22.8) — parity tests for `scanIntegrity`'s batch-load fast path vs sequential. Four cases (dedup, hits, validate, topPages) seed a fixture and assert both paths return identical results. Dedup case uses raw SQL via `getConn().unsafe()` to seed a `(test-source-2, people/alice)` row alongside the default-source row, since `engine.putPage` doesn't take a `source_id`. Pins the codex-caught multi-source overcounting regression.
|
||||
- `test/e2e/jsonb-roundtrip.test.ts` — v0.12.3 companion regression against the 4 doctor-scanned JSONB sites. Assertion-level overlap with `postgres-jsonb.test.ts` is intentional defense-in-depth: if doctor's scan surface ever drifts from the actual write surface, one of these tests catches it.
|
||||
- `test/e2e/sync.test.ts` (v0.22.12 — `--skip-failed` failure-loop test, alongside the existing 13 happy-path tests): exercises the full chain — broken file → `performSync` returns `blocked_by_failures` with grouped breakdown → `performSync({skipFailed: true})` advances bookmark and returns `AcknowledgeResult` with code summary → second broken file → second cycle. Saves and restores the user's real `~/.gbrain/sync-failures.jsonl` so the test is hermetic on a developer machine. Asserts bookmark gating, JSONL state, dedup across paths, summary aggregation, and the literal doctor-rendering string format. This is the integration test that proves the v0.22.12 chain holds together — unit tests cover the pure functions in isolation, this covers the integration.
|
||||
- `test/e2e/upgrade.test.ts` runs check-update E2E against real GitHub API (network required)
|
||||
- `test/e2e/minions-shell-pglite.test.ts` (v0.20.4) exercises the PGLite `--follow` inline shell-job path (in-memory, no `DATABASE_URL` required) — the path the consolidated minion-orchestrator skill documents for dev use
|
||||
- `test/e2e/openclaw-reference-compat.test.ts` (v0.19) — exercises `check-resolvable` + `skillpack install` against a minimal AGENTS.md workspace fixture (`test/fixtures/openclaw-reference-minimal/`), regression guard for the 107-skill OpenClaw deployment shape
|
||||
- `test/e2e/search-swamp.test.ts` (v0.22.0) — reproduces the headline source-swamp case. Seeds a curated `originals/talks/article-outline-fat-code` page against two `wintermute/chat/` pages stuffed with the same multi-word phrase. Asserts the article wins keyword AND vector ranking, that `detail=high` lets the chat swamp re-surface (temporal-query workflow preserved), and that `source_id` passes through the two-stage CTE intact. PGLite in-memory.
|
||||
- `test/e2e/search-exclude.test.ts` (v0.22.0) — verifies `test/` + `archive/` pages are hidden by default, that `include_slug_prefixes` opts back in, and that caller-supplied `exclude_slug_prefixes` adds to defaults. Both keyword and vector search paths covered.
|
||||
- `test/e2e/engine-parity.test.ts` (v0.22.0) — Postgres ↔ PGLite top-result and result-set parity for `searchKeyword` + `searchVector`. Codex flagged that Postgres ranks pages then picks best chunk while PGLite returns chunks directly — without parity coverage the source-boost fix could pass on PGLite and fail on Postgres. Skips gracefully when `DATABASE_URL` is unset.
|
||||
- `test/e2e/postgres-bootstrap.test.ts` (v0.22.6.1) — exercises `PostgresEngine.initSchema()` directly against a fresh real Postgres database. Asserts the bootstrap path is no-op on fresh installs and that SCHEMA_SQL replays cleanly through the engine path (not via the standalone `db.initSchema` from `src/core/db.ts`, which would have produced false-positive coverage). Codex caught the E2E-shape gap during plan review.
|
||||
- `test/e2e/http-transport.test.ts` (v0.22.7) — 8 cases against real Postgres covering `gbrain serve --http` end-to-end: bearer auth round-trip, `last_used_at` SQL-level debounce semantics, `mcp_request_log` row insertion on success and auth_failed paths, `/health` DB-down → 503 (DB-probing health check), and the F1+F2+F3 dispatch round-trip with a real operation. Skips gracefully when `DATABASE_URL` is unset.
|
||||
- `test/e2e/sync-parallel.test.ts` (v0.22.13 PR #490) — DATABASE_URL-gated. T2: 60-file Postgres sync at concurrency=4 imports all + no connection leak (probes `pg_stat_activity` before/after to confirm worker engines disconnected). P4: 120-file serial-vs-parallel benchmark prints `SYNC_PARALLEL_BENCH N files | serial=Xms | parallel(4)=Yms | speedup=Zx` for CHANGELOG quoting. Asserts parallel ≤ serial × 1.5 (CI-noise tolerant; not a strict speedup gate).
|
||||
- Tier 2 (`skills.test.ts`) requires OpenClaw + API keys, runs nightly in CI
|
||||
- If `.env.testing` doesn't exist in this directory, check sibling worktrees for one:
|
||||
`find ../ -maxdepth 2 -name .env.testing -print -quit` and copy it here if found.
|
||||
@@ -392,6 +429,59 @@ in bulk paths, the CI guard will fail the build.
|
||||
|
||||
`bun build --compile --outfile bin/gbrain src/cli.ts`
|
||||
|
||||
## Version locations (single source of truth: `VERSION` file)
|
||||
|
||||
Every release advances the version in **five files at once**. Keep these in
|
||||
sync. `/ship` enforces this via Step 12's idempotency check (VERSION vs
|
||||
package.json drift), but the canonical list lives here so future runs and
|
||||
the auto-update agent know where to look.
|
||||
|
||||
**Required (every release must update all five):**
|
||||
|
||||
| File | What lives there | Format |
|
||||
|---|---|---|
|
||||
| `VERSION` | The single source of truth. Read first by `/ship`, the binary, and CI version-gate. | Bare 4-digit string `MAJOR.MINOR.PATCH.MICRO` (e.g. `0.22.1`), no leading `v`, no trailing newline-sensitivity issues. |
|
||||
| `package.json` | Bun/npm package version. `gbrain --version` reads it via the compiled binary's bundled package metadata. CI version-gate cross-checks this against `VERSION` and fails if they drift. | `"version": "0.22.1"` |
|
||||
| `CHANGELOG.md` | Top entry header `## [0.22.1] - YYYY-MM-DD` plus the "To take advantage of v0.22.1" block. | Standard Keep-a-Changelog header. |
|
||||
| `TODOS.md` | Any TODO entries that mention "follow-up from vX.Y.Z" use the version of the release that filed them. Update only when filing NEW follow-up TODOs. | Inline `vX.Y.Z` references in TODO bodies. |
|
||||
| `CLAUDE.md` | The Key Files section's per-file annotations carry `vX.Y.Z (#NNN)` tags noting which release introduced a behavior. Update whenever a wave's annotations get folded in. | Inline `vX.Y.Z (#NNN, contributed by @user)` references. |
|
||||
|
||||
**Auto-derived (no manual edit; refreshed by their own commands):**
|
||||
|
||||
- `bun.lock` — root-package version is auto-pinned from `package.json`. After
|
||||
bumping `package.json`, run `bun install` to refresh the lockfile.
|
||||
- `llms-full.txt` / `llms.txt` — auto-generated documentation bundles. After
|
||||
any release ship that touches the Key Files annotations in `CLAUDE.md`,
|
||||
run `bun run build:llms` to regenerate. The bundles do not contain a
|
||||
version pin per se; they reflect the current state of the docs they index.
|
||||
|
||||
**Historical (DO NOT bump on release):**
|
||||
|
||||
- `skills/migrations/v0.21.0.md` — migration files use the version they
|
||||
shipped FROM as their filename. v0.21.0's migration always says v0.21.0.
|
||||
- `src/commands/migrations/v0_21_0.ts` — same: migration code references
|
||||
the schema version it migrates to.
|
||||
- `test/migrations-v0_21_0.test.ts`, `test/migration-orchestrator-v0_21_0.test.ts`,
|
||||
`test/migrate.test.ts` — migration tests reference historical migration
|
||||
versions; these are correct as-is and should not move.
|
||||
- `src/core/db.ts`, `src/core/migrate.ts`, `src/core/import-file.ts`,
|
||||
`src/commands/reindex-code.ts` — code comments cite the release that
|
||||
introduced a feature. Once written, these are historical record.
|
||||
- `README.md` — references the latest published feature names by version
|
||||
(e.g. "v0.21.0 Code Cathedral"); update only when the README's marketing
|
||||
copy is intentionally being refreshed, NOT on every micro/patch bump.
|
||||
|
||||
**The /ship workflow's version idempotency check:** Step 12 reads
|
||||
`VERSION` and `package.json`, classifies as FRESH / ALREADY_BUMPED /
|
||||
DRIFT_STALE_PKG / DRIFT_UNEXPECTED, and refuses to proceed on
|
||||
DRIFT_UNEXPECTED. This is why the two must move together.
|
||||
|
||||
**The CI version-gate** rejects pushes where `VERSION` and
|
||||
`package.json` disagree, OR where `VERSION` is not strictly greater
|
||||
than master's VERSION. If a queue collision claims your version on
|
||||
master before yours lands, /ship's queue-aware allocator (Step 12)
|
||||
will detect drift and re-bump on the next run.
|
||||
|
||||
## Pre-ship requirements
|
||||
|
||||
Before shipping (/ship) or reviewing (/review), always run the full test suite:
|
||||
|
||||
@@ -80,12 +80,29 @@ Add to `~/.claude/server.json` (Claude Code), Settings > MCP Servers (Cursor), o
|
||||
### Remote MCP (Claude Desktop, Cowork, Perplexity)
|
||||
|
||||
```bash
|
||||
ngrok http 8787 --url your-brain.ngrok.app
|
||||
bun run src/commands/auth.ts create "claude-desktop"
|
||||
gbrain auth create "claude-desktop" # tokens via the existing CLI
|
||||
gbrain serve --http --port 8787 # built-in HTTP transport (Postgres-only)
|
||||
ngrok http 8787 --url your-brain.ngrok.app # any tunnel works
|
||||
claude mcp add gbrain -t http https://your-brain.ngrok.app/mcp -H "Authorization: Bearer TOKEN"
|
||||
```
|
||||
|
||||
Per-client guides: [`docs/mcp/`](docs/mcp/DEPLOY.md). ChatGPT requires OAuth 2.1 (not yet implemented).
|
||||
Per-client guides: [`docs/mcp/`](docs/mcp/DEPLOY.md). Hardening defaults, env vars, and threat model: [SECURITY.md](SECURITY.md). ChatGPT requires OAuth 2.1 (not yet implemented).
|
||||
|
||||
### Using gbrain with GStack
|
||||
|
||||
If your engineering agent runs on [GStack](https://github.com/garrytan/gstack), point it at gbrain for code lookup instead of grep+read. Cathedral II (v0.21.0) ships call-graph edges and two-pass retrieval — `/investigate`, `/review`, `/plan-eng-review`, and `/office-hours` all benefit when the agent walks the symbol graph instead of scanning files line by line.
|
||||
|
||||
The five magical-moment commands:
|
||||
|
||||
```bash
|
||||
gbrain code-callers searchKeyword # who calls this symbol?
|
||||
gbrain code-callees searchKeyword # what does this symbol call?
|
||||
gbrain code-def BrainEngine # where is X defined?
|
||||
gbrain code-refs BrainEngine # all reference sites
|
||||
gbrain query "how does N+1 handling work" --near-symbol BrainEngine.searchKeyword --walk-depth 2
|
||||
```
|
||||
|
||||
All five auto-emit JSON on non-TTY (gh-CLI convention) so a GStack subagent shelling out via bash gets a clean parseable response. Run `gbrain sources add <repo> --strategy code` to index a repo, then your agent's brain-first lookup covers code, not just markdown. ([Cathedral II release notes](CHANGELOG.md#0210---2026-04-25))
|
||||
|
||||
## The 29 Skills
|
||||
|
||||
@@ -343,6 +360,30 @@ accumulate rows across separate single-skill installs instead of overwriting eac
|
||||
Read [`skills/skillify/SKILL.md`](skills/skillify/SKILL.md) for the full 10-item checklist
|
||||
and the anti-patterns it catches.
|
||||
|
||||
## Storage tiering: keep bulk content out of git (v0.22.11)
|
||||
|
||||
When your brain crosses 100K files and bulk machine-generated content (tweets, articles, transcripts)
|
||||
becomes the size driver, declare which directories belong in git and which live in the database only.
|
||||
|
||||
```yaml
|
||||
# gbrain.yml at the brain repo root
|
||||
storage:
|
||||
db_tracked:
|
||||
- people/
|
||||
- companies/
|
||||
- deals/
|
||||
db_only:
|
||||
- media/x/
|
||||
- media/articles/
|
||||
- meetings/transcripts/
|
||||
```
|
||||
|
||||
`gbrain sync` auto-manages your `.gitignore` for `db_only` paths. `gbrain export --restore-only --repo .`
|
||||
repopulates missing files from the database (container restart, fresh clone, accidental rm).
|
||||
`gbrain storage status` shows the tier breakdown.
|
||||
|
||||
Full guide: [docs/storage-tiering.md](docs/storage-tiering.md).
|
||||
|
||||
## Getting Data In
|
||||
|
||||
GBrain ships integration recipes that your agent sets up for you. Each recipe tells the agent what credentials to ask for, how to validate, and what cron to register.
|
||||
@@ -489,6 +530,8 @@ Question
|
||||
│ ├─ Multi-query expansion (Haiku rephrases the question 3 ways)
|
||||
│ ├─ Vector search (HNSW cosine over OpenAI embeddings)
|
||||
│ ├─ Keyword search (Postgres tsvector + websearch_to_tsquery)
|
||||
│ ├─ Source-aware ranking (curated dirs outrank chat/daily swamp at SQL layer)
|
||||
│ ├─ Hard-exclude (test/ archive/ attachments/ .raw/ filtered before retrieval)
|
||||
│ ├─ Reciprocal Rank Fusion (score = sum 1/(60+rank) across both)
|
||||
│ ├─ Cosine re-scoring (re-rank chunks against actual query embedding)
|
||||
│ ├─ Compiled-truth boost (assessments outrank timeline noise)
|
||||
@@ -596,8 +639,11 @@ SEARCH
|
||||
gbrain query <question> Hybrid search (vector + keyword + RRF)
|
||||
|
||||
IMPORT
|
||||
gbrain import <dir> [--no-embed] Import markdown (idempotent)
|
||||
gbrain sync [--repo <path>] Git-to-brain incremental sync
|
||||
gbrain import <dir> [--no-embed] [--workers N]
|
||||
Import markdown (idempotent)
|
||||
gbrain sync [--repo <path>] [--workers N]
|
||||
Git-to-brain incremental sync
|
||||
(>100-file diffs auto-parallelize 4 workers on Postgres)
|
||||
gbrain export [--dir ./out/] Export to markdown
|
||||
|
||||
FILES
|
||||
@@ -639,6 +685,8 @@ ADMIN
|
||||
gbrain doctor --locks List idle-in-tx backends (57014 diagnostic, Postgres only)
|
||||
gbrain stats Brain statistics
|
||||
gbrain serve MCP server (stdio)
|
||||
gbrain serve --http --port 8787 MCP server (HTTP, Postgres-only, bearer auth)
|
||||
gbrain auth create|list|revoke|test Token management for the HTTP transport
|
||||
gbrain integrations Integration recipe dashboard
|
||||
gbrain sources list|add|remove|... Multi-source brain management (v0.18)
|
||||
gbrain dream [--dry-run] [--phase N] One maintenance cycle then exit (cron-friendly)
|
||||
|
||||
+168
@@ -0,0 +1,168 @@
|
||||
# Security
|
||||
|
||||
## Reporting Vulnerabilities
|
||||
|
||||
If you discover a security issue in GBrain, please report it privately by opening
|
||||
a [private security advisory](https://github.com/garrytan/gbrain/security/advisories/new)
|
||||
on GitHub.
|
||||
|
||||
Do not open a public issue for security vulnerabilities.
|
||||
|
||||
## Remote MCP Security
|
||||
|
||||
### ⚠️ Do NOT use open OAuth client registration for remote MCP
|
||||
|
||||
If you deploy GBrain's MCP server behind an HTTP wrapper with OAuth 2.1
|
||||
support, **never allow unauthenticated client registration**. An attacker
|
||||
who discovers your server URL can:
|
||||
|
||||
1. Register a new OAuth client via `POST /register`
|
||||
2. Use `client_credentials` grant to obtain a bearer token
|
||||
3. Access all brain data via the MCP tools
|
||||
|
||||
### Recommended: `gbrain serve --http`
|
||||
|
||||
As of v0.22.7, GBrain ships a built-in HTTP transport that uses the
|
||||
existing `access_tokens` table for authentication:
|
||||
|
||||
```bash
|
||||
# Create a token
|
||||
gbrain auth create "my-client"
|
||||
|
||||
# Start the HTTP server
|
||||
gbrain serve --http --port 8787
|
||||
|
||||
# Connect via ngrok, Tailscale, or any tunnel
|
||||
ngrok http 8787 --url your-brain.ngrok.app
|
||||
```
|
||||
|
||||
This is the recommended way to expose GBrain remotely. No OAuth, no
|
||||
registration endpoint, no self-service tokens. Tokens are managed
|
||||
exclusively via `gbrain auth create/list/revoke`.
|
||||
|
||||
### If you must use a custom HTTP wrapper
|
||||
|
||||
1. **Require a secret for client registration** — check a header or body
|
||||
parameter before creating new OAuth clients
|
||||
2. **Disable `client_credentials` grant** — only allow `authorization_code`
|
||||
with browser-based approval
|
||||
3. **Restrict scopes** — never issue tokens with unlimited scope
|
||||
4. **Log all token issuance** — alert on unexpected registrations
|
||||
5. **Rate-limit registration and token endpoints**
|
||||
|
||||
### Token Management
|
||||
|
||||
```bash
|
||||
gbrain auth create "claude-desktop" # Create a new token
|
||||
gbrain auth list # List all tokens
|
||||
gbrain auth revoke "claude-desktop" # Revoke a token
|
||||
gbrain auth test <url> --token <tok> # Smoke-test a remote server
|
||||
```
|
||||
|
||||
Tokens are stored as SHA-256 hashes in the `access_tokens` table. The
|
||||
plaintext token is shown once at creation and never stored.
|
||||
|
||||
## `gbrain serve --http` hardening (v0.22.7+)
|
||||
|
||||
The built-in HTTP transport ships with several layers of hardening on by
|
||||
default. All env vars below are optional; the defaults are intentionally
|
||||
conservative.
|
||||
|
||||
### Postgres-only
|
||||
|
||||
`gbrain serve --http` requires a Postgres engine. PGLite is local-only by
|
||||
design and the `access_tokens` / `mcp_request_log` tables don't exist in
|
||||
the PGLite schema. Local agents continue to use stdio (`gbrain serve`).
|
||||
Running `--http` against a PGLite-backed install fails fast with a clear
|
||||
error message at startup.
|
||||
|
||||
### CORS
|
||||
|
||||
Default-deny: no `Access-Control-Allow-Origin` header is sent unless an
|
||||
allowlist is configured. To allow browser-based MCP clients:
|
||||
|
||||
```bash
|
||||
GBRAIN_HTTP_CORS_ORIGIN=https://claude.ai gbrain serve --http --port 8787
|
||||
# Multiple origins: comma-separated
|
||||
GBRAIN_HTTP_CORS_ORIGIN=https://claude.ai,https://your.app gbrain serve --http
|
||||
```
|
||||
|
||||
When the request `Origin` matches the allowlist, the server echoes it
|
||||
back in `Access-Control-Allow-Origin` (with `Vary: Origin`). Otherwise no
|
||||
CORS header is sent and the browser blocks the request.
|
||||
|
||||
### Rate limiting
|
||||
|
||||
Two buckets, both stored in a bounded LRU map (default 10K keys, evicts
|
||||
least-recently-used on overflow, prunes entries older than 2× the
|
||||
window):
|
||||
|
||||
| Bucket | When it fires | Default | Env var |
|
||||
|---|---|---|---|
|
||||
| Pre-auth IP | Before the DB lookup, on every `/mcp` request | 30 req / 60s | `GBRAIN_HTTP_RATE_LIMIT_IP` |
|
||||
| Post-auth token | After a valid token is resolved | 60 req / 60s | `GBRAIN_HTTP_RATE_LIMIT_TOKEN` |
|
||||
| LRU cap | Maximum distinct keys across both buckets | 10000 | `GBRAIN_HTTP_RATE_LIMIT_LRU` |
|
||||
|
||||
On exhaustion the server returns `429 Too Many Requests` with a
|
||||
`Retry-After` header.
|
||||
|
||||
**Caveat for tunneled deployments (ngrok, Tailscale Funnel, Cloudflare
|
||||
Tunnel):** all requests share one egress IP, so the pre-auth IP bucket
|
||||
becomes effectively shared by all clients on that tunnel. The
|
||||
post-auth token-id bucket is the load-bearing limiter for tunnel-fronted
|
||||
deployments.
|
||||
|
||||
### Reverse-proxy trust
|
||||
|
||||
Disabled by default. To honor `X-Forwarded-For` (or `X-Real-IP`) when
|
||||
gbrain runs behind a trusted reverse proxy:
|
||||
|
||||
```bash
|
||||
GBRAIN_HTTP_TRUST_PROXY=1 gbrain serve --http --port 8787
|
||||
```
|
||||
|
||||
**Critical safety contract:** only set `GBRAIN_HTTP_TRUST_PROXY=1` when
|
||||
**both** of these are true:
|
||||
|
||||
1. gbrain is reachable only via a trusted reverse proxy (not directly
|
||||
exposed to the internet on the configured port). The simplest
|
||||
guarantee is to bind gbrain to `127.0.0.1` or a private interface
|
||||
and have the proxy forward to it.
|
||||
2. The proxy strips any client-supplied `X-Forwarded-For` and `X-Real-IP`
|
||||
headers, then sets them itself. (nginx with `proxy_set_header
|
||||
X-Forwarded-For $remote_addr` does this; Cloudflare and most cloud
|
||||
load balancers handle it automatically.)
|
||||
|
||||
If gbrain is reachable directly AND `GBRAIN_HTTP_TRUST_PROXY=1` is set,
|
||||
clients can spoof their IP by sending arbitrary `X-Forwarded-For`
|
||||
headers, defeating the pre-auth IP rate limit. Without the flag, gbrain
|
||||
ignores all forwarded-for headers and uses the socket peer address,
|
||||
which is the safe default for direct-exposure deployments.
|
||||
|
||||
### Body size cap
|
||||
|
||||
Default 1 MiB, stream-counted (chunked transfers without
|
||||
`Content-Length` are still capped). Override:
|
||||
|
||||
```bash
|
||||
GBRAIN_HTTP_MAX_BODY_BYTES=2097152 gbrain serve --http # 2 MiB
|
||||
```
|
||||
|
||||
Over-cap requests get `413 Payload Too Large` immediately, before any
|
||||
body is materialized in memory.
|
||||
|
||||
### Audit log
|
||||
|
||||
Every `/mcp` request writes one row to `mcp_request_log`:
|
||||
|
||||
```bash
|
||||
psql "$DATABASE_URL" -c \
|
||||
"SELECT created_at, token_name, operation, status, latency_ms
|
||||
FROM mcp_request_log
|
||||
ORDER BY created_at DESC LIMIT 100"
|
||||
```
|
||||
|
||||
`status` is one of: `success`, `error`, `auth_failed`, `rate_limited`,
|
||||
`body_too_large`, `parse_error`, `unknown_method`. Failed-auth rows have
|
||||
`token_name = NULL`. Inserts are fire-and-forget so audit failures
|
||||
never block requests.
|
||||
@@ -1,5 +1,266 @@
|
||||
# TODOS
|
||||
|
||||
## sync (v0.22.13 follow-up — PR #490 review)
|
||||
|
||||
### D-PR490-1 — Plumb resolved `database_url` through `SyncOpts`
|
||||
**Priority:** P3
|
||||
|
||||
**What:** Add `database_url?: string` (or a richer `resolvedConnection` shape) to
|
||||
`SyncOpts` and have the caller (`runSync`, the cycle handler, the jobs handler)
|
||||
populate it from the active engine instead of having `performSync` /
|
||||
`performFullSync` / `import.ts` each call `loadConfig()` separately. Today every
|
||||
sync run hits the config file three times.
|
||||
|
||||
**Why:** v0.18 multi-source brains can in principle run different sources against
|
||||
different `database_url` endpoints (or different per-source overrides via
|
||||
`sources.config_jsonb`). Right now `loadConfig()` returns the global config, and
|
||||
that always matches the engine in practice — but the convention papers over a
|
||||
real divergence the moment someone wants per-source connection settings. Folding
|
||||
the resolution into `SyncOpts` makes the worker-engine creation in `sync.ts` and
|
||||
`import.ts` deterministic from `SyncOpts` alone.
|
||||
|
||||
**Pros:**
|
||||
- Removes 3 redundant `loadConfig()` calls per sync.
|
||||
- Makes `performSync` / `performFullSync` side-effect-free with respect to the
|
||||
on-disk config file.
|
||||
- Sets up for per-source `database_url` overrides without further refactor.
|
||||
- Makes the v0.22.13 belt-and-suspenders fallback (PR #490 Q3) cleaner — no
|
||||
more `!config?.database_url` short-circuit inside the parallel branch.
|
||||
|
||||
**Cons:**
|
||||
- API-shape change to `SyncOpts` (mild; not externally exported).
|
||||
- Touching three callers (`runSync`, jobs handler, `cycle.ts` `runPhaseSync`).
|
||||
- Only worth doing when paired with a per-source override story; otherwise
|
||||
it's just plumbing.
|
||||
|
||||
**Context:** Surfaced during the PR #490 plan-eng-review (parallel sync).
|
||||
Deferred because it isn't on the v0.22.13 critical path. The same pattern would
|
||||
benefit the cycle handler and the autopilot daemon. See the plan-eng-review
|
||||
decisions log: A4 = "Defer; file as TODO."
|
||||
|
||||
**Depends on / blocked by:** Nothing structural. Best paired with the v0.18
|
||||
per-source `config_jsonb` work if/when that lands.
|
||||
|
||||
## sync error-code classification (PR #501 follow-ups)
|
||||
|
||||
### Plumb structured `ParseValidationCode` through `ImportResult`
|
||||
**Priority:** P2
|
||||
|
||||
**What:** Replace the regex-on-error-message path in `src/core/sync.ts:classifyErrorCode`
|
||||
with a structured `code` field threaded through `ImportResult` from the parse layer.
|
||||
|
||||
Three changes:
|
||||
1. `src/core/import-file.ts:362` — call `parseMarkdown(content, relativePath, { validate: true, expectedSlug })`
|
||||
so `parsed.errors[0].code` is populated.
|
||||
2. `src/core/import-file.ts` — add `code?: string` to `ImportResult`. Promote the
|
||||
structured code (or `'SLUG_MISMATCH'` when the existing expectedSlug check trips)
|
||||
into the result envelope alongside `error`.
|
||||
3. `src/commands/sync.ts:488` — extend `failedFiles` shape with `code?: string`.
|
||||
`recordSyncFailures` already accepts the field; the only thing missing is the
|
||||
capture site populating it.
|
||||
4. `src/core/sync.ts:classifyErrorCode` — keep as a fallback for un-coded errors
|
||||
(DB exceptions, generic catches). Primary path reads the structured code.
|
||||
|
||||
**Why:** The repo already has `ParseValidationCode` + `ParseValidationError` in
|
||||
`src/core/markdown.ts:5-18`, and three other consumers (`src/commands/lint.ts:72`,
|
||||
`src/commands/frontmatter.ts:148`, `src/core/brain-writer.ts:314`) read structured
|
||||
errors directly. Sync is the outlier — it calls `parseMarkdown` without validation
|
||||
and reverse-engineers codes via regex. PR #501 shipped that regex out of pragmatism;
|
||||
this TODO removes ~50% of `classifyErrorCode` and eliminates a class of false-positives.
|
||||
|
||||
**Pros:**
|
||||
- One source of truth for parse codes (the enum in `markdown.ts`).
|
||||
- Eliminates regex fragility — adding a new validation code in `markdown.ts`
|
||||
automatically flows to sync without a new regex.
|
||||
- Closes the case where canonical messages (`File is empty...`, `No closing ---...`)
|
||||
don't match aspirational regex patterns.
|
||||
|
||||
**Cons:** Touches `ImportResult` interface, which ripples through `src/commands/import.ts:105`,
|
||||
`src/commands/sync.ts:498-510`, `src/core/cycle.ts`, brain-writer reconciler.
|
||||
|
||||
**Context:** PR #501 documented this as P3 in the eng review at
|
||||
`~/.claude/plans/then-codex-synchronous-toucan.md`. Codex's outside-voice review
|
||||
agreed independently. The fix is small — ~50 lines including tests + downstream
|
||||
call sites — and it's the correct architectural endpoint.
|
||||
|
||||
**Effort:** M (human: ~2 hr / CC: ~20 min).
|
||||
|
||||
**Depends on / blocked by:** Nothing.
|
||||
|
||||
### CHANGELOG migration note for `acknowledgeSyncFailures()` shape change
|
||||
**Priority:** P0 — required at /ship time
|
||||
|
||||
**What:** When PR #501 ships, the release CHANGELOG entry MUST include this
|
||||
`### For contributors` block:
|
||||
|
||||
```markdown
|
||||
### For contributors
|
||||
|
||||
`acknowledgeSyncFailures()` now returns `{count, summary}` instead of `number`.
|
||||
If you import this directly from `gbrain/sync`, replace `n` with `result.count`
|
||||
and use `result.summary` for the new code-grouped breakdown.
|
||||
```
|
||||
|
||||
**Why:** The function is exported from `src/core/sync.ts:433` and reachable via
|
||||
the package exports map. External TS consumers (gbrain-evals, host agent forks)
|
||||
that imported it got `number` and now get an object — silent type break.
|
||||
|
||||
**Effort:** XS (human: ~1 min). Just don't forget.
|
||||
|
||||
**Depends on / blocked by:** PR #501 ship.
|
||||
|
||||
### Concurrent-safe ack of `~/.gbrain/sync-failures.jsonl`
|
||||
**Priority:** P3
|
||||
|
||||
**What:** Two concurrent `gbrain sync` runs hitting `acknowledgeSyncFailures()`
|
||||
can clobber each other. The function does a whole-file `writeFileSync` rewrite
|
||||
(`src/core/sync.ts:433-455`); `recordSyncFailures()` does independent
|
||||
`appendFileSync` (`src/core/sync.ts:395-416`). Concurrent ack + append can lose rows.
|
||||
|
||||
**Why:** Pre-existing — predates PR #501. Real risk only on autopilot setups where
|
||||
multiple sync invocations might overlap (rare today, more likely as multi-source
|
||||
sync matures).
|
||||
|
||||
**Fix sketch:** Atomic rename pattern (write to `sync-failures.jsonl.tmp`, then
|
||||
`renameSync`) plus a file lock for the read-modify-write cycle. Or move the
|
||||
acknowledged-set to the DB.
|
||||
|
||||
**Effort:** S (human: ~1 hr / CC: ~10 min).
|
||||
|
||||
**Depends on / blocked by:** Nothing.
|
||||
|
||||
## test-infra
|
||||
|
||||
### Parallel-load timeout flake on v0.21 PGLite-heavy tests
|
||||
**Priority:** P0
|
||||
|
||||
**What:** 22 tests added in v0.21.0 (Code Cathedral II) consistently fail in the full `bun test` run with timeout-pattern elapsed times of 7-10s, but pass in isolation. Every failing test calls `engine.initSchema()` in `beforeAll` without a timeout extension. Under parallel load (168 test files now run concurrently after v0.21 added ~24 new files), `initSchema` exceeds bun's default 5s `beforeAll` timeout.
|
||||
|
||||
Affected files include (non-exhaustive): `test/sync-strategy.test.ts`, `test/cathedral-ii-brainbench.test.ts`, `test/code-edges.test.ts`, `test/reindex-code.test.ts`, `test/reconcile-links.test.ts`, `test/two-pass.test.ts`, `test/parent-symbol-path.test.ts`, `test/pglite-v0_19.test.ts`.
|
||||
|
||||
**Why:** Currently triaged as "skip pre-existing, ship anyway" but that's not a real fix. Blocks /ship for anyone whose CHANGELOG-time test run sees them.
|
||||
|
||||
**Pros:** Fixing it lets /ship run cleanly without manual triage every release.
|
||||
|
||||
**Cons:** ~22 file edits adding `beforeAll(async () => {...}, 30000)` is mechanical but dull.
|
||||
|
||||
**Context:** Same pattern fixed in v0.20.5 wave for `test/e2e/minions-shell-pglite.test.ts`. Single-file repro: each fails in `bun test`, passes in `bun test <file>`. Reproduces with my changes stashed, so it's on master.
|
||||
|
||||
**Effort:** S (human: ~30 min / CC: ~5 min). Mechanical: grep for `beforeAll(async () => {` in affected files, add `, 30000)` argument.
|
||||
|
||||
**Depends on / blocked by:** Nothing.
|
||||
|
||||
## resolver / check-resolvable (v0.22.4 follow-ups)
|
||||
|
||||
### D10 — Extend `check-resolvable` to parse RESOLVER.md disambiguation rules
|
||||
**Priority:** P2
|
||||
|
||||
**What:** Extend `src/core/check-resolvable.ts:357-390` to parse a structured
|
||||
disambiguation block in `RESOLVER.md` (e.g. a `## Disambiguation rules`
|
||||
numbered list with parseable `<trigger>` → `<winning-skill>` shape) and treat
|
||||
resolved overlaps as non-issues. Then the action message at
|
||||
`src/core/check-resolvable.ts:388` ("Add disambiguation rule in RESOLVER.md OR
|
||||
narrow triggers") stops lying about the OR — currently only the second branch
|
||||
silences the warning.
|
||||
|
||||
**Why:** The current MECE-overlap fix path forces authors to delete user-facing
|
||||
triggers from skill frontmatter. That's wrong for cases where two skills
|
||||
legitimately respond to the same phrase under different contexts (e.g.
|
||||
"citation audit" → focused fix vs broader brain health). A real
|
||||
disambiguation parser would let `RESOLVER.md` carry the resolution while
|
||||
keeping both skills' triggers intact for chaining.
|
||||
|
||||
**Pros:**
|
||||
- The action message stops misleading users.
|
||||
- v0.22.4 D2 used the "narrow triggers" path because the disambiguation
|
||||
parser doesn't exist yet; landing this would let v0.23+ keep dual triggers
|
||||
for genuinely-overlapping skills.
|
||||
- Aligns RESOLVER.md's stated role (the dispatcher) with what the checker
|
||||
actually reads.
|
||||
|
||||
**Cons:**
|
||||
- Introduces a new `RESOLVER.md` syntactic contract that other tooling now
|
||||
has to respect (parser, lint, downstream forks reading the same file).
|
||||
- Risk of false-positive resolution if the parser is loose.
|
||||
- ~80 lines of parser + tests; not blocking anything in v0.22.4.
|
||||
|
||||
**Context:**
|
||||
- The "OR" in the action message is misleading today. Confirmed at
|
||||
`src/core/check-resolvable.ts:388`.
|
||||
- The MECE detector loop is at `src/core/check-resolvable.ts:357-390`.
|
||||
- The disambiguation rules already exist as prose in
|
||||
`skills/RESOLVER.md` (the citation-audit row added in v0.22.4 is the
|
||||
pattern). They're agent-facing routing hints today, not parsed structure.
|
||||
|
||||
**Effort:** S (human: ~4-6 hours / CC: ~30 min for parser + 12-16 test cases).
|
||||
|
||||
**Depends on / blocked by:** Nothing.
|
||||
|
||||
## code-indexing (v0.21.0 Cathedral II follow-ups)
|
||||
|
||||
### B2 — Magika auto-detect for extension-less files (Layer 9 deferred)
|
||||
**Priority:** P2
|
||||
|
||||
**What:** Embed Google's Magika ML classifier (~1MB ONNX) as a bundled asset. Wire into `detectCodeLanguage` as the fallback for files with no recognized extension (Dockerfile, Makefile, `.envrc`, shell scripts with shebangs but no `.sh`). The chunker already has `setLanguageFallback(fn)` as a module-level hook.
|
||||
|
||||
**Why:** v0.20.0 widens the file classifier from 9 to 35 extensions (Layer 2), covering most real-world cases. Extension-less files still slip through to recursive chunks. Magika would close the last common case.
|
||||
|
||||
**Pros:** Completes the file-classification story. Unblocks chunker on real-world configs + build scripts.
|
||||
|
||||
**Cons:** ~1MB asset bundled with `bun --compile`. Integration risk: Magika's ONNX runtime needs WASM compat with bun. The plan explicitly allowed deferring B2 because bundling surprises late in implementation are costly.
|
||||
|
||||
**Context:**
|
||||
- `src/core/chunkers/code.ts` exports `setLanguageFallback(fn: LanguageFallback | null)` — call at process start with a Magika-powered classifier.
|
||||
- `detectCodeLanguage(filePath, content?)` already accepts optional content for fallback paths.
|
||||
- The NPM `magika` package is the first thing to try; needs bun-compile compatibility verification.
|
||||
|
||||
**Effort:** M (human: ~2-3 days / CC: ~2 hours for the integration + CI guard).
|
||||
|
||||
**Depends on / blocked by:** Nothing. Hook is in place as of v0.20.0.
|
||||
|
||||
### A4 — full doc_comment extraction at chunk time
|
||||
**Priority:** P2
|
||||
|
||||
**What:** When the chunker emits a method/class/function, look at the comment node(s) immediately preceding the declaration and persist them as `content_chunks.doc_comment`. The FTS trigger from Layer 1b already weights `doc_comment` 'A' above `chunk_text` 'B' — the ranking is ready, the column is populated NULL today.
|
||||
|
||||
**Why:** "how does X handle N+1" should rank the docstring that explains N+1 above the function body or any prose paragraph. Layer 1b paved the ranking half; extraction is the remaining half.
|
||||
|
||||
**Pros:** Material MRR lift on natural-language queries. Zero schema work (column + trigger already in place).
|
||||
|
||||
**Cons:** Per-language convention detection — JSDoc blocks, Python docstrings (first string expression in a function body), C-style doc comments, etc. Not hard but each language has edge cases.
|
||||
|
||||
**Context:**
|
||||
- `src/core/chunkers/code.ts` emits chunks in `chunkCodeTextFull`. Walk each declaration's preceding sibling(s) for comment nodes.
|
||||
- ChunkInput already has `doc_comment?: string`. Populate at chunk time and it flows through `upsertChunks` (Layer 6 wired those columns).
|
||||
- Per-language config: leading-comment type names per language (`comment`, `line_comment`, `block_comment`, `documentation_comment`).
|
||||
- Test hook: `test/cathedral-ii-brainbench.test.ts` has a `doc_comment_matching` placeholder — flesh it out end-to-end.
|
||||
|
||||
**Effort:** M (human: ~2 days / CC: ~90 min for the 8 Layer-5 langs).
|
||||
|
||||
**Depends on / blocked by:** Nothing. Layer 1b + Layer 6 both in place.
|
||||
|
||||
### C6 — gbrain code-signature "(A, B) => C"
|
||||
**Priority:** P3 (stretch)
|
||||
|
||||
**What:** Type-signature retrieval via tree-sitter type captures per language. "Find every function whose signature returns a Promise<User>" or "(string, number) => boolean".
|
||||
|
||||
**Why:** Each language's type system is its own mini-cathedral. Ship per-language rather than as one item.
|
||||
|
||||
**Effort:** L per language (typescript-first).
|
||||
|
||||
**Depends on / blocked by:** Nothing — additive on the Layer 5 edge schema.
|
||||
|
||||
### Cross-file edge resolution (Layer 5 precision upgrade)
|
||||
**Priority:** P3
|
||||
|
||||
**What:** Today every call edge lands unresolved in `code_edges_symbol` with to_symbol_qualified = bare callee name. Second-pass resolution: after all code files import, walk every `code_edges_symbol` row and try to resolve `to_symbol_qualified` via `symbol_name_qualified` join; if found within the same source, write a resolved row to `code_edges_chunk`.
|
||||
|
||||
**Why:** `getCallersOf("searchKeyword")` currently returns the Layer 6 ambiguity — every `searchKeyword` call site in any class. Receiver-type analysis lifts this.
|
||||
|
||||
**Effort:** L. Needs receiver-type inference; can ship per-language.
|
||||
|
||||
**Depends on / blocked by:** Nothing — UNION-on-read path keeps unresolved edges surfaced even without this.
|
||||
|
||||
## Completed
|
||||
|
||||
### ~~Checks 5 + 6 for check-resolvable~~
|
||||
@@ -408,3 +669,135 @@ iteration's residuals.
|
||||
|
||||
### Implement AWS Signature V4 for S3 storage backend
|
||||
**Completed:** v0.6.0 (2026-04-10) — replaced with @aws-sdk/client-s3 for proper SigV4 signing.
|
||||
|
||||
### Caller-opt-in retry for `executeRaw` (D3 follow-up from v0.22.1)
|
||||
**What:** Add `PostgresEngine.executeRawIdempotent(sql, params)` (or a `{retry: true}` parameter flag on `executeRaw`) so callers explicitly opt into auto-retry for statements they know are idempotent. Audit existing call sites and migrate the read-only ones (search, page fetches, etc.) to the new method.
|
||||
|
||||
**Why:** Closes the gap left by D3's drop-the-wrapper decision in v0.22.1. The original #406 wrapped `executeRaw` in a regex-gated retry that was unsound for writable CTEs and side-effecting SELECTs. Recovery moved up to the supervisor watchdog, but per-call recovery for reads (the bulk of `executeRaw` traffic from MCP, search, page fetches) is gone. A caller-opt-in flag puts the idempotency decision where it belongs (at the call site, with full statement context).
|
||||
|
||||
**Pros:** Restores per-call auto-recovery for reads without the phantom-write risk on mutations. Explicit > clever: each call site declares its own idempotency posture. Future caller-added mutations get safe-by-default behavior.
|
||||
|
||||
**Cons:** Touches every existing `executeRaw` call site (~25). Requires careful audit — accidentally tagging a mutation as idempotent re-introduces the phantom-write bug.
|
||||
|
||||
**Context:** Codex F3 demonstrated that `READ_ONLY_PREFIX = /^(\s|--.*\n)*(SELECT|WITH)\b/i` is unsound — `WITH x AS (UPDATE … RETURNING …) SELECT …` matches the prefix but updates a row; `SELECT pg_advisory_xact_lock(...)` is a SELECT with side effects. The plan-eng-review wrap-up in `~/.claude/plans/system-instruction-you-are-working-tender-horizon.md` has the full discussion.
|
||||
|
||||
**Effort estimate:** M (human: ~1 day / CC: ~30 min including call-site audit).
|
||||
**Priority:** P2 — current behavior (no retry, supervisor recovers within ~3 min) is acceptable but per-call recovery is a real ergonomic win.
|
||||
**Depends on:** Nothing.
|
||||
|
||||
### Replace `walkMarkdownFiles` with `engine.getAllSlugs()` in `extractForSlugs` (F1 follow-up from v0.22.1)
|
||||
**What:** The cycle path's `extractForSlugs()` at `src/commands/extract.ts:455` still does a `walkMarkdownFiles(brainDir)` to build the `allSlugs` set for link resolution. On a 54K-page brain that's a single `readdir` traversal (~hundreds of ms — acceptable, dominated by the file-content-read elimination from #417). But `engine.getAllSlugs()` exists at `extract.ts:728` and produces the same set via a single SQL query (~tens of ms).
|
||||
|
||||
**Why:** Eliminates the residual directory walk on every cycle. Codex F1 noted that the v0.22.1 plan's "cycle never re-walks the whole tree again" claim was overstated — it stops READING file contents but still walks the directory. This TODO closes that gap honestly.
|
||||
|
||||
**Pros:** Cycle becomes O(slugs sync touched), not O(total brain size). No more readdir on a growing brain. ~5 LOC change.
|
||||
|
||||
**Cons:** Crosses an FS-vs-DB consistency boundary in the FS-source extract path. Edge case: a file deleted from disk but still in DB. Currently `extractForSlugs` skips with `if (!existsSync(fullPath)) continue` — unchanged. But if a markdown file references a slug whose page exists in DB but file was deleted, the link would resolve via DB but the original extractor caught it. Needs a careful test for this case.
|
||||
|
||||
**Context:** Codex plan-review during v0.22.1 wrap, verified at `extract.ts:455-456`. The plan-eng-review session captured the rationale.
|
||||
|
||||
**Effort estimate:** S (human: ~2 hr / CC: ~10 min including the consistency-edge-case test).
|
||||
**Priority:** P3 — pure perf, no correctness gap.
|
||||
**Depends on:** Nothing.
|
||||
|
||||
### `err.code`-based connection-error matching in `postgres-engine.ts` (B1 follow-up from v0.22.1)
|
||||
**What:** The CONNECTION_ERROR_PATTERNS array (~12 strings: `ECONNREFUSED`, `connection terminated`, `password authentication failed`, etc.) matched against `err.message` and `err.code`. Replace with structured matching against `err.code` only, using postgres.js's typed error classes (`PostgresError` with structured codes).
|
||||
|
||||
**Why:** String matching against error messages breaks on library upgrades (postgres.js could change its error message phrasing without bumping major). Code matching is durable. The Layer 1 cleanup follows: gbrain itself doesn't define connection-error codes; it should defer to postgres.js's classification.
|
||||
|
||||
**Pros:** More durable across library updates. Less code (drop the 12-string array). Follows the typed-errors pattern v0.21.0 introduced (`src/core/errors.ts`).
|
||||
|
||||
**Cons:** Requires verifying which `err.code` values postgres.js actually exposes for each connection-failure mode. May need fallback to message-substring matching for codes that postgres.js doesn't surface.
|
||||
|
||||
**Context:** Section 2/B1 from the v0.22.1 plan-eng-review. After D3 dropped the per-call retry, `isConnectionError` is no longer in the hot path — only the supervisor watchdog cares about classifying connection errors, and it currently catches *anything*. This TODO is a cleanup pass when someone next touches that surface.
|
||||
|
||||
**Effort estimate:** S (human: ~2 hr / CC: ~10 min).
|
||||
**Priority:** P3.
|
||||
**Depends on:** The above caller-opt-in retry (#1) is the natural co-lander since both touch the same error-classification surface.
|
||||
|
||||
## remote MCP / HTTP transport (v0.22.7 follow-ups)
|
||||
|
||||
### Audit-log write amplification on rejected `/mcp` traffic
|
||||
**What:** `src/mcp/http-transport.ts` writes a row to `mcp_request_log` for every
|
||||
incoming `/mcp` request, including rate-limited (429), oversized (413), and
|
||||
auth-failed (401) traffic. Under sustained attack the IP rate limit caps audit
|
||||
writes per IP at 30/min, but at scale (10K distinct IPs) that's still 300K
|
||||
inserts/min. Two follow-ups: (1) instrument the audit-write rate so we can see
|
||||
the actual production volume; (2) consider a separate "rejected" table or
|
||||
sampling for failed-auth rows so the success-path audit table doesn't get
|
||||
swamped.
|
||||
|
||||
**Why:** Codex flagged this during the v0.22.7 ship adversarial review. We kept
|
||||
the full audit on purpose — forensic data of an attack is valuable — but want
|
||||
to revisit once we have real volume numbers.
|
||||
|
||||
**Pros:** Bounds DB write volume under attack. Keeps the success-path audit
|
||||
table small enough for fast queries.
|
||||
|
||||
**Cons:** Adds a second table or a sampling rule. Not free complexity. Probably
|
||||
not worth it until production hits a real attack pattern.
|
||||
|
||||
**Context:** `src/mcp/http-transport.ts:222,235,245` (the three audit-on-reject
|
||||
call sites) + `src/schema.sql:342` (the unbounded table).
|
||||
|
||||
**Effort estimate:** M (human: ~half day / CC: ~30 min once we have volume data).
|
||||
**Priority:** P3 — wait for evidence.
|
||||
**Depends on:** Production telemetry on `mcp_request_log` insert rate.
|
||||
|
||||
### `validateParams` doesn't check enum values or array item types
|
||||
**What:** `src/mcp/dispatch.ts:27` (extracted from `src/mcp/server.ts` in
|
||||
v0.22.7) only checks top-level JS types. Operations declare `enum` constraints
|
||||
(e.g. `direction: 'in' | 'out' | 'both'`) and array `items: { type: ... }`
|
||||
schemas in `src/core/operations.ts`, but `validateParams` ignores both. Bad
|
||||
inputs still reach handlers — concretely, an invalid `direction` falls through
|
||||
the engine's else branch at `src/core/postgres-engine.ts:954`, widening
|
||||
traversal unexpectedly; malformed `pages_updated` arrays could be written as
|
||||
garbage JSONB.
|
||||
|
||||
**Why:** Codex flagged this during the v0.22.7 ship adversarial review. The
|
||||
validator was lifted verbatim from the pre-existing stdio path during the
|
||||
dispatch.ts extraction — same gap exists on the stdio MCP server today, so
|
||||
this isn't a v0.22.7 regression. Still worth tightening, since "shared
|
||||
validation" is now the architectural guarantee both transports rely on.
|
||||
|
||||
**Pros:** Better defense-in-depth at the MCP boundary. Catches malformed agent
|
||||
inputs before the engine layer has to.
|
||||
|
||||
**Cons:** Need to walk every operation's param schema and decide which enum
|
||||
violations are user-facing errors vs internal bugs. May need a typed Zod-style
|
||||
schema layer to do this cleanly.
|
||||
|
||||
**Context:** `src/mcp/dispatch.ts:27` + `src/core/operations.ts` (param defs).
|
||||
Same gap pre-existed on stdio MCP path.
|
||||
|
||||
**Effort estimate:** M (human: ~half day / CC: ~30 min if we use the existing
|
||||
ParamDef shape; XL if a Zod migration is the chosen direction).
|
||||
**Priority:** P2.
|
||||
**Depends on:** Whether we want to keep the lightweight ParamDef shape or
|
||||
migrate to typed schemas.
|
||||
|
||||
### Streaming MCP tool support (re-add SSE based on Accept header)
|
||||
**What:** v0.22.7 dropped SSE entirely from `gbrain serve --http` because no
|
||||
current MCP tool streams. When the first streaming tool ships (long-running
|
||||
agent delegation as an MCP tool, `resources/subscribe`, `sampling/createMessage`),
|
||||
re-add SSE in `/mcp` based on the `Accept` header per the Streamable HTTP
|
||||
transport spec. ~30 lines + spec compliance test.
|
||||
|
||||
**Why:** Removing SSE simplified the v0.22.7 transport (one response path,
|
||||
fewer test cases). Adding it back when actually needed is cheap and keeps the
|
||||
code lean in the meantime.
|
||||
|
||||
**Effort estimate:** S (human: ~2 hr / CC: ~15 min).
|
||||
**Priority:** P3 — wait for the first streaming tool.
|
||||
**Depends on:** A streaming MCP tool actually existing.
|
||||
|
||||
### `access_tokens.scopes` enforcement
|
||||
**What:** The `access_tokens` schema has had a `scopes TEXT[]` column since
|
||||
migration v4 (`src/core/migrate.ts:84`), but nothing enforces it. v0.22.7's
|
||||
`gbrain auth create` doesn't accept a `--scopes` flag, and `dispatchToolCall`
|
||||
doesn't gate on scopes. Adding per-tool scope enforcement would let
|
||||
"claude-desktop-readonly" and "ingest-only" tokens exist.
|
||||
|
||||
**Effort estimate:** M (human: ~1 day / CC: ~30 min for the schema-aware gate).
|
||||
**Priority:** P3.
|
||||
**Depends on:** Nothing.
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
"dependencies": {
|
||||
"@anthropic-ai/sdk": "^0.30.0",
|
||||
"@aws-sdk/client-s3": "^3.1028.0",
|
||||
"@dqbd/tiktoken": "^1.0.22",
|
||||
"@electric-sql/pglite": "0.4.3",
|
||||
"@modelcontextprotocol/sdk": "^1.0.0",
|
||||
"gray-matter": "^4.0.3",
|
||||
@@ -14,9 +15,12 @@
|
||||
"openai": "^4.0.0",
|
||||
"pgvector": "^0.2.0",
|
||||
"postgres": "^3.4.0",
|
||||
"tree-sitter-wasms": "0.1.13",
|
||||
"web-tree-sitter": "0.22.6",
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/bun": "latest",
|
||||
"bun-types": "^1.3.13",
|
||||
"typescript": "^5.6.0",
|
||||
},
|
||||
},
|
||||
@@ -107,6 +111,8 @@
|
||||
|
||||
"@aws/lambda-invoke-store": ["@aws/lambda-invoke-store@0.2.4", "", {}, "sha512-iY8yvjE0y651BixKNPgmv1WrQc+GZ142sb0z4gYnChDDY2YqI4P/jsSopBWrKfAt7LOJAkOXt7rC/hms+WclQQ=="],
|
||||
|
||||
"@dqbd/tiktoken": ["@dqbd/tiktoken@1.0.22", "", {}, "sha512-RYhO8xeHkMNX5Ixqf4M1Ve3siCYJY/dI0yLnlX4M4oIEDOvjMIQ+E+3OUpAaZcWTaMtQJzGcDAghYfllpx3i/w=="],
|
||||
|
||||
"@electric-sql/pglite": ["@electric-sql/pglite@0.4.3", "", {}, "sha512-ichuWTgtd4mOM1G4SpyGJa5trT03lWbMypDV0fUXUCXg5hiHqVAz/bZyV68NqmkLB7WcYmj1RMJVSp8HV/v/ZQ=="],
|
||||
|
||||
"@hono/node-server": ["@hono/node-server@1.19.12", "", { "peerDependencies": { "hono": "^4" } }, "sha512-txsUW4SQ1iilgE0l9/e9VQWmELXifEFvmdA1j6WFh/aFPj99hIntrSsq/if0UWyGVkmrRPKA1wCeP+UCr1B9Uw=="],
|
||||
@@ -215,7 +221,7 @@
|
||||
|
||||
"@types/bun": ["@types/bun@1.3.11", "", { "dependencies": { "bun-types": "1.3.11" } }, "sha512-5vPne5QvtpjGpsGYXiFyycfpDF2ECyPcTSsFBMa0fraoxiQyMJ3SmuQIGhzPg2WJuWxVBoxWJ2kClYTcw/4fAg=="],
|
||||
|
||||
"@types/node": ["@types/node@18.19.130", "", { "dependencies": { "undici-types": "~5.26.4" } }, "sha512-GRaXQx6jGfL8sKfaIDD6OupbIHBr9jv7Jnaml9tB7l4v068PAOXqfcujMMo5PhbIs6ggR1XODELqahT2R8v0fg=="],
|
||||
"@types/node": ["@types/node@25.5.2", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-tO4ZIRKNC+MDWV4qKVZe3Ql/woTnmHDr5JD8UI5hn2pwBrHEwOEMZK7WlNb5RKB6EoJ02gwmQS9OrjuFnZYdpg=="],
|
||||
|
||||
"@types/node-fetch": ["@types/node-fetch@2.6.13", "", { "dependencies": { "@types/node": "*", "form-data": "^4.0.4" } }, "sha512-QGpRVpzSaUs30JBSGPjOg4Uveu384erbHBoT1zeONvyCfwQxIkUshLAOqN/k9EjGviPRmWTTe6aH2qySWKTVSw=="],
|
||||
|
||||
@@ -237,7 +243,7 @@
|
||||
|
||||
"bowser": ["bowser@2.14.1", "", {}, "sha512-tzPjzCxygAKWFOJP011oxFHs57HzIhOEracIgAePE4pqB3LikALKnSzUyU4MGs9/iCEUuHlAJTjTc5M+u7YEGg=="],
|
||||
|
||||
"bun-types": ["bun-types@1.3.11", "", { "dependencies": { "@types/node": "*" } }, "sha512-1KGPpoxQWl9f6wcZh57LvrPIInQMn2TQ7jsgxqpRzg+l0QPOFvJVH7HmvHo/AiPgwXy+/Thf6Ov3EdVn1vOabg=="],
|
||||
"bun-types": ["bun-types@1.3.13", "", { "dependencies": { "@types/node": "*" } }, "sha512-QXKeHLlOLqQX9LgYaHJfzdBaV21T63HhFJnvuRCcjZiaUDpbs5ED1MgxbMra71CsryN/1dAoXuJJJwIv/2drVA=="],
|
||||
|
||||
"bytes": ["bytes@3.1.2", "", {}, "sha512-/Nf7TyzTx6S3yRJObOAV7956r8cr2+Oj8AC5dt8wSP3BQAoeX58NoHyCU8P8zGkNXStjTSi6fzO6F0pBdcYbEg=="],
|
||||
|
||||
@@ -453,13 +459,15 @@
|
||||
|
||||
"tr46": ["tr46@0.0.3", "", {}, "sha512-N3WMsuqV66lT30CrXNbEjx4GEwlow3v6rr4mCcv6prnfwhS01rkgyFdjPNBYd9br7LpXV1+Emh01fHnq2Gdgrw=="],
|
||||
|
||||
"tree-sitter-wasms": ["tree-sitter-wasms@0.1.13", "", { "dependencies": { "tree-sitter-wasms": "^0.1.11" } }, "sha512-wT+cR6DwaIz80/vho3AvSF0N4txuNx/5bcRKoXouOfClpxh/qqrF4URNLQXbbt8MaAxeksZcZd1j8gcGjc+QxQ=="],
|
||||
|
||||
"tslib": ["tslib@2.8.1", "", {}, "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w=="],
|
||||
|
||||
"type-is": ["type-is@2.0.1", "", { "dependencies": { "content-type": "^1.0.5", "media-typer": "^1.1.0", "mime-types": "^3.0.0" } }, "sha512-OZs6gsjF4vMp32qrCbiVSkrFmXtG/AZhY3t0iAMrMBiAZyV9oALtXO8hsrHbMXF9x6L3grlFuwW2oAz7cav+Gw=="],
|
||||
|
||||
"typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw=="],
|
||||
|
||||
"undici-types": ["undici-types@5.26.5", "", {}, "sha512-JlCMO+ehdEIKqlFxk6IfVoAUVmgz7cU7zD/h9XZ0qzeosSHmUJVOzSQvvYSYWXkFXC+IfLKSIffhv0sVZup6pA=="],
|
||||
"undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
|
||||
|
||||
"unpipe": ["unpipe@1.0.0", "", {}, "sha512-pjy2bYhSsufwWlKwPc+l3cN7+wuJlK6uz0YdJEOlQDbl6jo/YlPi4mb8agUkVC8BF7V8NuzeyPNqRksA3hztKQ=="],
|
||||
|
||||
@@ -467,6 +475,8 @@
|
||||
|
||||
"web-streams-polyfill": ["web-streams-polyfill@4.0.0-beta.3", "", {}, "sha512-QW95TCTaHmsYfHDybGMwO5IJIM93I/6vTRk+daHTWFPhwh+C8Cg7j7XyKrwrj8Ib6vYXe0ocYNrmzY4xAAN6ug=="],
|
||||
|
||||
"web-tree-sitter": ["web-tree-sitter@0.22.6", "", {}, "sha512-hS87TH71Zd6mGAmYCvlgxeGDjqd9GTeqXNqTT+u0Gs51uIozNIaaq/kUAbV/Zf56jb2ZOyG8BxZs2GG9wbLi6Q=="],
|
||||
|
||||
"webidl-conversions": ["webidl-conversions@3.0.1", "", {}, "sha512-2JAn3z8AR6rjK8Sm8orRC0h/bcl/DqL7tRPdGZ4I1CjdF+EaMLmYxBHyXuKL849eucPFhvBoxMsflfOb8kxaeQ=="],
|
||||
|
||||
"whatwg-url": ["whatwg-url@5.0.0", "", { "dependencies": { "tr46": "~0.0.3", "webidl-conversions": "^3.0.0" } }, "sha512-saE57nupxk6v3HY35+jzBwYa0rKSy0XR8JSxZPwgLr7ys0IBzhGviA1/TUGJLmSVqs8pb9AnvICXEuOHLprYTw=="],
|
||||
@@ -479,30 +489,32 @@
|
||||
|
||||
"zod-to-json-schema": ["zod-to-json-schema@3.25.2", "", { "peerDependencies": { "zod": "^3.25.28 || ^4" } }, "sha512-O/PgfnpT1xKSDeQYSCfRI5Gy3hPf91mKVDuYLUHZJMiDFptvP41MSnWofm8dnCm0256ZNfZIM7DSzuSMAFnjHA=="],
|
||||
|
||||
"@anthropic-ai/sdk/@types/node": ["@types/node@18.19.130", "", { "dependencies": { "undici-types": "~5.26.4" } }, "sha512-GRaXQx6jGfL8sKfaIDD6OupbIHBr9jv7Jnaml9tB7l4v068PAOXqfcujMMo5PhbIs6ggR1XODELqahT2R8v0fg=="],
|
||||
|
||||
"@aws-crypto/sha1-browser/@smithy/util-utf8": ["@smithy/util-utf8@2.3.0", "", { "dependencies": { "@smithy/util-buffer-from": "^2.2.0", "tslib": "^2.6.2" } }, "sha512-R8Rdn8Hy72KKcebgLiv8jQcQkXoLMOGGv5uI1/k0l+snqkOzQ1R0ChUBCxWMlBsFMekWjq0wRudIweFs7sKT5A=="],
|
||||
|
||||
"@aws-crypto/sha256-browser/@smithy/util-utf8": ["@smithy/util-utf8@2.3.0", "", { "dependencies": { "@smithy/util-buffer-from": "^2.2.0", "tslib": "^2.6.2" } }, "sha512-R8Rdn8Hy72KKcebgLiv8jQcQkXoLMOGGv5uI1/k0l+snqkOzQ1R0ChUBCxWMlBsFMekWjq0wRudIweFs7sKT5A=="],
|
||||
|
||||
"@aws-crypto/util/@smithy/util-utf8": ["@smithy/util-utf8@2.3.0", "", { "dependencies": { "@smithy/util-buffer-from": "^2.2.0", "tslib": "^2.6.2" } }, "sha512-R8Rdn8Hy72KKcebgLiv8jQcQkXoLMOGGv5uI1/k0l+snqkOzQ1R0ChUBCxWMlBsFMekWjq0wRudIweFs7sKT5A=="],
|
||||
|
||||
"@types/node-fetch/@types/node": ["@types/node@25.5.2", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-tO4ZIRKNC+MDWV4qKVZe3Ql/woTnmHDr5JD8UI5hn2pwBrHEwOEMZK7WlNb5RKB6EoJ02gwmQS9OrjuFnZYdpg=="],
|
||||
|
||||
"bun-types/@types/node": ["@types/node@25.5.2", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-tO4ZIRKNC+MDWV4qKVZe3Ql/woTnmHDr5JD8UI5hn2pwBrHEwOEMZK7WlNb5RKB6EoJ02gwmQS9OrjuFnZYdpg=="],
|
||||
"@types/bun/bun-types": ["bun-types@1.3.11", "", { "dependencies": { "@types/node": "*" } }, "sha512-1KGPpoxQWl9f6wcZh57LvrPIInQMn2TQ7jsgxqpRzg+l0QPOFvJVH7HmvHo/AiPgwXy+/Thf6Ov3EdVn1vOabg=="],
|
||||
|
||||
"form-data/mime-types": ["mime-types@2.1.35", "", { "dependencies": { "mime-db": "1.52.0" } }, "sha512-ZDY+bPm5zTTF+YpCrAU9nK0UgICYPT0QtT1NZWFv4s++TNkcgVaT0g6+4R2uI4MjQjzysHB1zxuWL50hzaeXiw=="],
|
||||
|
||||
"openai/@types/node": ["@types/node@18.19.130", "", { "dependencies": { "undici-types": "~5.26.4" } }, "sha512-GRaXQx6jGfL8sKfaIDD6OupbIHBr9jv7Jnaml9tB7l4v068PAOXqfcujMMo5PhbIs6ggR1XODELqahT2R8v0fg=="],
|
||||
|
||||
"@anthropic-ai/sdk/@types/node/undici-types": ["undici-types@5.26.5", "", {}, "sha512-JlCMO+ehdEIKqlFxk6IfVoAUVmgz7cU7zD/h9XZ0qzeosSHmUJVOzSQvvYSYWXkFXC+IfLKSIffhv0sVZup6pA=="],
|
||||
|
||||
"@aws-crypto/sha1-browser/@smithy/util-utf8/@smithy/util-buffer-from": ["@smithy/util-buffer-from@2.2.0", "", { "dependencies": { "@smithy/is-array-buffer": "^2.2.0", "tslib": "^2.6.2" } }, "sha512-IJdWBbTcMQ6DA0gdNhh/BwrLkDR+ADW5Kr1aZmd4k3DIF6ezMV4R2NIAmT08wQJ3yUK82thHWmC/TnK/wpMMIA=="],
|
||||
|
||||
"@aws-crypto/sha256-browser/@smithy/util-utf8/@smithy/util-buffer-from": ["@smithy/util-buffer-from@2.2.0", "", { "dependencies": { "@smithy/is-array-buffer": "^2.2.0", "tslib": "^2.6.2" } }, "sha512-IJdWBbTcMQ6DA0gdNhh/BwrLkDR+ADW5Kr1aZmd4k3DIF6ezMV4R2NIAmT08wQJ3yUK82thHWmC/TnK/wpMMIA=="],
|
||||
|
||||
"@aws-crypto/util/@smithy/util-utf8/@smithy/util-buffer-from": ["@smithy/util-buffer-from@2.2.0", "", { "dependencies": { "@smithy/is-array-buffer": "^2.2.0", "tslib": "^2.6.2" } }, "sha512-IJdWBbTcMQ6DA0gdNhh/BwrLkDR+ADW5Kr1aZmd4k3DIF6ezMV4R2NIAmT08wQJ3yUK82thHWmC/TnK/wpMMIA=="],
|
||||
|
||||
"@types/node-fetch/@types/node/undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
|
||||
|
||||
"bun-types/@types/node/undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
|
||||
|
||||
"form-data/mime-types/mime-db": ["mime-db@1.52.0", "", {}, "sha512-sPU4uV7dYlvtWJxwwxHD0PuihVNiE7TyAbQ5SWxDCB9mUYvOgroQOwYQQOKPJ8CIbE+1ETVlOoK1UC2nU3gYvg=="],
|
||||
|
||||
"openai/@types/node/undici-types": ["undici-types@5.26.5", "", {}, "sha512-JlCMO+ehdEIKqlFxk6IfVoAUVmgz7cU7zD/h9XZ0qzeosSHmUJVOzSQvvYSYWXkFXC+IfLKSIffhv0sVZup6pA=="],
|
||||
|
||||
"@aws-crypto/sha1-browser/@smithy/util-utf8/@smithy/util-buffer-from/@smithy/is-array-buffer": ["@smithy/is-array-buffer@2.2.0", "", { "dependencies": { "tslib": "^2.6.2" } }, "sha512-GGP3O9QFD24uGeAXYUjwSTXARoqpZykHadOmA8G5vfJPK0/DC67qa//0qvqrJzL1xc8WQWX7/yc7fwudjPHPhA=="],
|
||||
|
||||
"@aws-crypto/sha256-browser/@smithy/util-utf8/@smithy/util-buffer-from/@smithy/is-array-buffer": ["@smithy/is-array-buffer@2.2.0", "", { "dependencies": { "tslib": "^2.6.2" } }, "sha512-GGP3O9QFD24uGeAXYUjwSTXARoqpZykHadOmA8G5vfJPK0/DC67qa//0qvqrJzL1xc8WQWX7/yc7fwudjPHPhA=="],
|
||||
|
||||
@@ -3,4 +3,10 @@
|
||||
# Default 5s is too short when many test files boot PGLite instances at once.
|
||||
# 60s is the empirical ceiling we observed before the first file's beforeAll
|
||||
# completed on a loaded machine.
|
||||
#
|
||||
# NOTE: this bunfig.toml `timeout` key is read by `bun test` but empirically
|
||||
# does NOT apply to beforeEach/afterEach hook timeouts under `bun run test`
|
||||
# chained behind `bun run typecheck`. The test script in package.json passes
|
||||
# `--timeout=60000` explicitly to cover both per-test and per-hook timeouts.
|
||||
# Leaving both in place as belt-and-suspenders.
|
||||
timeout = 60_000
|
||||
|
||||
@@ -458,6 +458,75 @@ in depth, not the primary boundary.
|
||||
|
||||
---
|
||||
|
||||
## v0.22.4 — frontmatter-guard adoption
|
||||
|
||||
### 1. Stop hand-rolling frontmatter validators
|
||||
|
||||
If your fork has scripts that call `js-yaml` directly to validate brain page
|
||||
frontmatter, replace them with `gbrain frontmatter validate` calls. The CLI
|
||||
covers the seven canonical error classes and ships a `--json` envelope that's
|
||||
stable across releases.
|
||||
|
||||
```diff
|
||||
- # Custom validator script
|
||||
- node scripts/validate-frontmatter.mjs <path>
|
||||
+ gbrain frontmatter validate <path> --json
|
||||
```
|
||||
|
||||
For consumers that need the validator inside another script, import from
|
||||
gbrain's `markdown` export instead of duplicating logic:
|
||||
|
||||
```ts
|
||||
import { parseMarkdown } from 'gbrain/markdown';
|
||||
|
||||
const parsed = parseMarkdown(content, filePath, { validate: true, expectedSlug });
|
||||
for (const err of parsed.errors ?? []) {
|
||||
// err.code: MISSING_OPEN | MISSING_CLOSE | YAML_PARSE | SLUG_MISMATCH |
|
||||
// NULL_BYTES | NESTED_QUOTES | EMPTY_FRONTMATTER
|
||||
}
|
||||
```
|
||||
|
||||
### 2. Drop any references to `lib/brain-writer.mjs`
|
||||
|
||||
If your fork's skills or scripts referenced an aspirational
|
||||
`lib/brain-writer.mjs` (it never shipped — the spec was in PR #392 and never
|
||||
landed), replace those references with the gbrain CLI. The `frontmatter-guard`
|
||||
skill lives at `skills/frontmatter-guard/SKILL.md` and points at
|
||||
`gbrain frontmatter validate` / `audit` / `install-hook`.
|
||||
|
||||
### 3. Wire the doctor subcheck into your health pipeline
|
||||
|
||||
`gbrain doctor` now reports `frontmatter_integrity` automatically. If your
|
||||
fork has a custom health pipeline (e.g. a daily Slack post about brain
|
||||
health), pull from `gbrain doctor --json` and surface the
|
||||
`frontmatter_integrity` row counts.
|
||||
|
||||
### 4. (Optional) Install the pre-commit hook on brain repos
|
||||
|
||||
For sources backed by git, the v0.22.4 install-hook helper drops a
|
||||
pre-commit script that blocks commits with malformed frontmatter:
|
||||
|
||||
```bash
|
||||
gbrain frontmatter install-hook
|
||||
```
|
||||
|
||||
Skip this if your brain isn't a git repo or if your downstream agent already
|
||||
enforces validation at write time. See `docs/integrations/pre-commit.md` for
|
||||
the full recipe.
|
||||
|
||||
### 5. Migration ergonomics — read pending-host-work.jsonl
|
||||
|
||||
After `gbrain apply-migrations --yes` runs the v0.22.4 audit, your agent
|
||||
should read `~/.gbrain/migrations/pending-host-work.jsonl` (filter to
|
||||
`migration === "0.22.4"`) and walk each entry's `command` field. Each entry
|
||||
points to a per-source `gbrain frontmatter validate <source_path> --fix`
|
||||
command — surface counts to the user, get explicit consent, then run.
|
||||
|
||||
The migration is **audit-only**. It never mutates brain content during
|
||||
`apply-migrations`. Your agent runs the fix command with user consent.
|
||||
|
||||
---
|
||||
|
||||
## Future versions
|
||||
|
||||
When gbrain ships a new version, this doc will be updated with the diffs for that
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
# Code Cathedral II — v0.20.0 Design
|
||||
|
||||
**Status:** Accepted. CEO + Eng + 2 codex passes CLEARED (2026-04-24). 16 cross-model findings absorbed total: 7 codex pass 1 (structural prereqs) + 6 codex pass 2 (absorption errors including the CHUNKER_VERSION silent-no-op gate and inbound-edge invalidation) + 3 eng-review architectural decisions. DX review recommended post-Layer 8 (new CLI surfaces) before ship.
|
||||
**Supersedes:** Cathedral I (planned v0.18.0–v0.19.0 code indexing, shipped v0.19.0).
|
||||
**Mode:** SCOPE EXPANSION (user explicit: "I want the best code search in the world").
|
||||
**Scale:** 14 bisectable layers, ~20–25 CC hours, 3–5 human-weeks. One schema migration with split edge tables (`code_edges_chunk` + `code_edges_symbol`). Backfill via `CHUNKER_VERSION` bump (automatic on next sync) + explicit `gbrain reindex-code` command.
|
||||
|
||||
## Why v0.20.0
|
||||
|
||||
v0.19.0 shipped code indexing: tree-sitter chunker, 29 active languages, symbol columns, forward doc↔impl linking, incremental embed cache, BrainBench code category. Four cathedral-I items got deferred during shipping: `query --lang` filter, `sync --all` cost preview, markdown fence extraction, reverse-scan doc↔impl backfill.
|
||||
|
||||
Cathedral II is a promise-keeping release for those four, bundled with the leap that makes gbrain *the* code search: structural edges (call graph + references + imports + inheritance), parent-scope capture, doc-comment FTS binding, and two-pass retrieval. No more grep-class retrieval on code.
|
||||
|
||||
## The 10x leap
|
||||
|
||||
Today: agent asks "how does hybrid search handle N+1?" → gets 3 prose chunks of `hybrid.ts`.
|
||||
|
||||
Cathedral II: same query returns the anchor function + its 3 callers + its 2 callees + its JSDoc + the guide in `/docs` that cites it + the test file exercising it + parent scope chain. One walk. Code-aware brain.
|
||||
|
||||
## Scope (5 tiers + Layer 0 prerequisites, 14 bisectable layer commits)
|
||||
|
||||
### Tier 0 — Prerequisites (surfaced by codex outside voice)
|
||||
|
||||
**0a. File-classification widening.** `sync.ts:35` currently classifies only 9 extensions as code (TS, JS, Python, Go, Rust, Ruby, Java, C, C++). Cathedral II's B1 ships 165 lazy-loadable grammars, so the classifier needs to accept any extension the chunker can handle. Also reorders `detectCodeLanguage` so Magika (B2) runs as a fallback for extension-less files, not after a null-return gate.
|
||||
|
||||
**0b. Chunk-grain FTS.** Current keyword search lives on `pages.search_vector`. Adding doc-comments or two-pass anchoring at the chunk level has zero ranking effect against a page-grain primitive. Layer 0b adds `content_chunks.search_vector` with a trigger building from qualified symbol name + doc-comment (weight A) and chunk_text (weight B), plus rewrites `searchKeyword` to rank chunks directly. Page-level search_vector stays for title-heavy searches.
|
||||
|
||||
Both Layer 0 items are prerequisites for the 10x leap to actually move retrieval metrics.
|
||||
|
||||
### Tier A — Structural edges (the 10x leap)
|
||||
|
||||
**A1. Call-graph + reference extraction with qualified symbol identity.** Per-language tree-sitter queries at `importCodeFile` time capture:
|
||||
|
||||
- `calls` — function call-sites
|
||||
- `imports` — module deps
|
||||
- `extends` / `implements` — type hierarchies
|
||||
- `mixes_in` — Ruby `include`/`extend`/`prepend`
|
||||
- `type_refs` — parameter + return type usage
|
||||
- `declares` — chunk owns a symbol definition
|
||||
|
||||
**Qualified symbol identity across all 8 langs.** `parent_symbol_path` (A3) is the source of truth for scope; edges use qualified names built from it. Examples: `Admin::UsersController#render` (Ruby instance), `Admin::UsersController.find_all` (Ruby singleton), `admin.users_controller.UsersController.render` (Python), `(*UsersController).Render` (Go), `users::UsersController::render` (Rust), `com.acme.admin.UsersController.render` (Java). Per-lang delimiter + method/class-method distinction. Ruby ships fully in ranker (CLI + A2 two-pass) — no deferral.
|
||||
|
||||
**Split schema (two tables, not one polymorphic):**
|
||||
```sql
|
||||
CREATE TABLE code_edges_chunk (
|
||||
from_chunk_id INTEGER NOT NULL REFERENCES content_chunks(id) ON DELETE CASCADE,
|
||||
to_chunk_id INTEGER NOT NULL REFERENCES content_chunks(id) ON DELETE CASCADE,
|
||||
from_symbol_qualified TEXT NOT NULL,
|
||||
to_symbol_qualified TEXT NOT NULL,
|
||||
edge_type TEXT NOT NULL,
|
||||
source_id TEXT REFERENCES sources(id) ON DELETE CASCADE,
|
||||
UNIQUE (from_chunk_id, to_chunk_id, edge_type)
|
||||
);
|
||||
CREATE TABLE code_edges_symbol (
|
||||
from_chunk_id INTEGER NOT NULL REFERENCES content_chunks(id) ON DELETE CASCADE,
|
||||
from_symbol_qualified TEXT NOT NULL,
|
||||
to_symbol_qualified TEXT NOT NULL,
|
||||
edge_type TEXT NOT NULL,
|
||||
source_id TEXT REFERENCES sources(id) ON DELETE CASCADE,
|
||||
UNIQUE (from_chunk_id, to_symbol_qualified, edge_type)
|
||||
);
|
||||
```
|
||||
`code_edges_chunk` = resolved (both endpoints known). `code_edges_symbol` = unresolved (target symbol exists by qualified name, definition chunk not yet seen). Promotion from symbol→chunk table happens on later import. `source_id` is TEXT matching actual `sources.id` type.
|
||||
|
||||
**Shipped languages:** TypeScript, TSX, JavaScript, Ruby, Python, Go, Rust, Java (8 langs, ~85% of real brain code). Other languages chunk normally (via B1 lazy-load) but don't emit edges in v0.20.0 — extension is one query file + delimiter config per language, shippable as small follow-up PRs.
|
||||
|
||||
**A2. Two-pass retrieval.** Current: keyword + vector → RRF → dedup. New: keyword + vector → anchor set → expand 1–2 hops on `code_edges_chunk` with structural-distance decay → blend into RRF.
|
||||
|
||||
**Default OFF in all cases.** Opt-in only via `--walk-depth N` or `--near-symbol <name>`. Exact-symbol-match auto-on was unsafe (symbol names collide across files). Neighbor cap 50 per hop, depth cap 2. Dedup's per-page cap (currently 2) lifts to `min(10, walkDepth × 5)` when walking so structural neighbors from one file aren't clipped. Distance decay: `1/(1 + hop)` on expanded-neighbor RRF contributions.
|
||||
|
||||
**A3. Parent-scope capture + nested-chunk emission.** Two parts:
|
||||
|
||||
*Part 1:* Nested symbols get `parent_symbol_path text[]` on `content_chunks`. Embedded into chunk header: `[TypeScript] src/foo.ts:42-58 function formatResult (in BrainEngine.searchKeyword)`. Scope flows into embedding. Dual-use: drives A1's qualified symbol identity.
|
||||
|
||||
*Part 2:* Extend `splitLargeNode` to emit nested functions/methods/inner-classes as their own chunks. The current chunker is top-level-node oriented — a `class Foo { method1() {} method2() {} }` emits one chunk. Parent_symbol_path on top-level nodes is empty (no parent above top level), so A3 contributes nothing without sub-top-level chunks. Part 2 makes the scope annotation load-bearing.
|
||||
|
||||
**A4. Doc-comment → symbol binding.** Leading AST comment extracted to `doc_comment text`. Lands on **chunk-grain** search_vector (Layer 0b prerequisite) with FTS weight `'A'`. Natural-language queries rank docstring matches above body text and below title. `'A' > 'B' > 'C' > 'D'` per Postgres FTS weight convention.
|
||||
|
||||
### Tier B — Coverage (honest Chonkie parity)
|
||||
|
||||
**B1.** Lazy-load tree-sitter-language-pack (~165 languages). Replace 36 committed WASMs with a manifest + per-process parser cache. Cathedral I promised this and didn't deliver — Cathedral II does.
|
||||
|
||||
**B2.** Magika auto-detect for extension-less files (Dockerfile, Makefile, `.envrc`). ~1MB bundled asset. Falls back to null → recursive chunker if classifier fails to load.
|
||||
|
||||
### Tier C — Agent CLI surfaces
|
||||
|
||||
- `query --lang <lang>` — filter by `content_chunks.language`
|
||||
- `query --symbol-kind function|class|method|type|interface|enum` — filter by `symbol_type`
|
||||
- `query --near-symbol <name> --depth 1..2` — two-pass retrieval anchored at a known symbol
|
||||
- `code-callers <symbol>` — uses A1 `calls` edges, reversed
|
||||
- `code-callees <symbol>` — uses A1 `calls` edges, forward
|
||||
|
||||
All auto-JSON on non-TTY. `StructuredAgentError` envelopes on failure. `code-signature` deferred to v0.20.1 (needs per-language type captures).
|
||||
|
||||
### Tier D — Bridge items (cathedral I promises)
|
||||
|
||||
**D1.** `sync --all` cost preview. `estimateTokens` extracted from `chunkers/code.ts` to new `tokens.ts` module. Before per-source loop: walk sync-diff set, sum tokens, compute $ estimate. TTY + !json + !yes → interactive `[y/N]`. Non-TTY or `--json` or piped → emit `ConfirmationRequired` envelope, exit 2. `--yes` skips. `--dry-run` previews + exit 0. Preview on `--all` only, not single-source (DX review pain is first-time large-sync surprise bills).
|
||||
|
||||
**D2.** Markdown fence extraction in `importFromContent`. After `parseMarkdown`, iterate marked lexer tokens for `{type:'code', lang, text}`. Map fence tag → language. Chunk each fence through `chunkCodeText`. Persist as `chunk_source='fenced_code'`. Cap 100 fences per markdown page (DOS defense). Per-fence try/catch — one bad fence doesn't break the page import.
|
||||
|
||||
**D3.** `reconcile-links` batch command. Walks markdown pages, calls existing v0.19.0 `extractCodeRefs` per page, emits `addLink(md, code, ..., 'documents')` + reverse. `ON CONFLICT DO NOTHING` handles idempotency. Statement-timeout scoped via `sql.begin` + `SET LOCAL`. Progress reporter + final summary (edges added / existed / missing-target). Respects `auto_link` config.
|
||||
|
||||
### Tier E — Eval, backfill, honesty
|
||||
|
||||
**E1.** BrainBench code sub-categories: `call_graph_recall` (callers of X → expected set), `parent_scope_coverage` (nested-symbol queries return correct scope), `doc_comment_matching` (NL queries rank doc-comments above prose). Regression gates against A1/A3/A4 drift.
|
||||
|
||||
**E2.** Backfill: schema migrates automatically (zero cost). **`CHUNKER_VERSION` bumps 3 → 4** — that constant is folded into each code page's `content_hash`, so every code page's hash changes on upgrade. Next `gbrain sync` won't short-circuit on "git HEAD unchanged"; it re-chunks every code file. New `gbrain reindex-code [--source <id>] [--dry-run] [--yes] [--force]` provides explicit full backfill with cost preview (reuses D1 infra) and `--force` bypasses content_hash skip entirely. Users control when to pay; silent no-op path closed.
|
||||
|
||||
**E3.** Honest CHANGELOG. Retire "Chonkie superset" framing. Run BrainBench before/after for real numbers: 150+ languages loaded (after B1), MRR on NL→code queries, P@1 call-graph precision, P@k on symbol_name queries, sync cost preview on 5K-file repo. Back every claim with a runnable command.
|
||||
|
||||
## Implementation ordering (14 layers, post-codex)
|
||||
|
||||
1. **0a** — File-classification widening (sync.ts:35) + Magika reordered as fallback
|
||||
2. **0b** — Chunk-grain FTS (content_chunks.search_vector + trigger + searchKeyword chunk-level rewrite)
|
||||
3. **Foundation** — schema migration (split edge tables, qualified name columns on content_chunks) + engine method stubs + types
|
||||
4. **B1** — lazy-load grammar manifest + bun --compile guard
|
||||
5. **A1** — edge-extractor + 8 per-lang query files + qualified symbol identity + tests
|
||||
6. **A3** — parent-scope column + doc-comment column + splitLargeNode nested-chunk emission
|
||||
7. **A4** — doc-comment FTS weight A on chunk-grain search_vector
|
||||
8. **A2** — two-pass retrieval, default OFF, opt-in only; dedup cap lifts when walking
|
||||
9. **D tier bundled** — cost preview + fence extraction + reconcile-links
|
||||
10. **B2** — Magika auto-detect
|
||||
11. **C tier** — 5 CLI surfaces
|
||||
12. **E1** — BrainBench sub-categories + CHUNKER_VERSION 3→4 bump
|
||||
13. **E2** — `reindex-code` with `--force` + migration orchestrator with backfill-prompt phase
|
||||
14. **E3 + release** — honest CHANGELOG + docs + migration skill + `/ship`
|
||||
|
||||
## Size and cost
|
||||
|
||||
- Diff: ~5500–6500 lines (~2.5x v0.19.0 post-codex expansion)
|
||||
- Tests: ~2000 lines (8 langs × qualified-name + edge-extraction fixtures + Layer 0b FTS migration tests)
|
||||
- Files: ~36 new, ~25 modified
|
||||
- CC time: ~20–25 hours focused (was 14–18 pre-codex; +6h for Layer 0a/0b + qualified identity across 8 langs + nested-chunk emission + CHUNKER_VERSION bump layer)
|
||||
- Human-equivalent: 3–5 weeks
|
||||
- First-sync cost bump for upgraded v0.19.0 users: every code page re-chunks on first sync after upgrade (CHUNKER_VERSION bump forces invalidation). Users run `gbrain reindex-code --dry-run` for cost preview, then `--yes` or accept gradual backfill over time as files change.
|
||||
- Daily autopilot cost post-backfill: unchanged (edges extracted at chunk time, no per-query LLM)
|
||||
|
||||
## Risks and mitigations
|
||||
|
||||
1. **Schema migration on live Postgres.** Test against production-shape DB before ship. v0.12.0 JSONB incident is the canary.
|
||||
2. **Per-language tree-sitter queries are fiddly.** Hand-verified edge-set fixtures per language. Ruby gets extra coverage for dynamic-dispatch false negatives.
|
||||
3. **Two-pass retrieval regression.** Default off for prose. BrainBench Cat 1 MUST show no regression before shipping.
|
||||
4. **Backfill shape (G1 resolved).** Three composable layers: schema-auto migrates columns empty (zero cost). Lazy on-touch catches 80% over time (zero cost). Explicit `reindex-code` with cost preview for users wanting immediate full benefit. No surprise bills.
|
||||
5. **Magika bundle (G2 resolved).** +1MB asset, `bun --compile` guard extension. If bundling surfaces bugs late in implementation, B2 is the only tier that can fall back to v0.20.1 without blocking the cathedral — it's self-contained at Layer 8.
|
||||
6. **High-fan-out symbols.** `console.log`-style symbols have 100K callers. Neighbor cap 50, depth cap 2. Chaos test fixture required.
|
||||
|
||||
## Review gates
|
||||
|
||||
- CEO review (cathedral II) — CLEARED 2026-04-24
|
||||
- Outside voice (codex) — run during cathedral II CEO review
|
||||
- `/plan-devex-review` — up next (per user request, 5 new CLI surfaces + reindex-code need DX polish review before eng)
|
||||
- `/plan-eng-review` — required before implementation begins
|
||||
- `/review` + `/codex review` — required before `/ship`
|
||||
|
||||
## What's deferred to later cathedrals
|
||||
|
||||
- **C6** `code-signature "(A, B) => C"` — per-language type captures. v0.20.1.
|
||||
- **Call-graph langs beyond 8 shipped** — PHP, Swift, Kotlin, Scala, C#, C++, Elixir, etc. One small PR per language.
|
||||
- **LSP integration** for live precision. v0.22+ cathedral.
|
||||
- **Code-tour generator** (cathedral I T1).
|
||||
- **Private-code redaction pre-embed** (cathedral I T3).
|
||||
- **`gbrain doctor --chunker-debug`** AST dump.
|
||||
@@ -0,0 +1,105 @@
|
||||
# Pre-commit hook for brain repos (v0.22.4+)
|
||||
|
||||
`gbrain frontmatter install-hook` installs a git pre-commit hook in your
|
||||
brain source's repo that runs `gbrain frontmatter validate` against staged
|
||||
`.md` and `.mdx` files. Malformed frontmatter blocks the commit. Bypass with
|
||||
`git commit --no-verify`.
|
||||
|
||||
## What the hook catches
|
||||
|
||||
The same seven validation classes the `frontmatter-guard` skill and
|
||||
`gbrain doctor`'s `frontmatter_integrity` subcheck report:
|
||||
|
||||
| Code | What it catches |
|
||||
|-------------------|---------------------------------------------------------------------|
|
||||
| `MISSING_OPEN` | File doesn't start with `---` |
|
||||
| `MISSING_CLOSE` | No closing `---` before first heading |
|
||||
| `YAML_PARSE` | YAML failed to parse (syntax or structure) |
|
||||
| `SLUG_MISMATCH` | `slug:` in frontmatter doesn't match path-derived slug |
|
||||
| `NULL_BYTES` | Binary corruption (`\x00`) anywhere in the content |
|
||||
| `NESTED_QUOTES` | `title: "outer "inner" outer"` shape that breaks YAML |
|
||||
| `EMPTY_FRONTMATTER` | `---` ... `---` with nothing meaningful between |
|
||||
|
||||
## Install
|
||||
|
||||
For all registered sources that are git repos:
|
||||
|
||||
```bash
|
||||
gbrain frontmatter install-hook
|
||||
```
|
||||
|
||||
For one source:
|
||||
|
||||
```bash
|
||||
gbrain frontmatter install-hook --source <id>
|
||||
```
|
||||
|
||||
For force-overwrite of an existing pre-commit hook (writes a `.bak`):
|
||||
|
||||
```bash
|
||||
gbrain frontmatter install-hook --force
|
||||
```
|
||||
|
||||
The hook lands at `<source>/.githooks/pre-commit`. If `core.hooksPath` is
|
||||
unset, the install also runs `git config core.hooksPath .githooks` so the
|
||||
hook is picked up without manual git config.
|
||||
|
||||
## Bypass
|
||||
|
||||
Standard git escape hatch:
|
||||
|
||||
```bash
|
||||
git commit --no-verify
|
||||
```
|
||||
|
||||
This skips ALL pre-commit hooks. Use sparingly — the next time the user
|
||||
runs `gbrain doctor`, the issues will surface.
|
||||
|
||||
## Uninstall
|
||||
|
||||
```bash
|
||||
gbrain frontmatter install-hook --uninstall
|
||||
```
|
||||
|
||||
If a `.bak` was saved during install, it's restored as the active hook.
|
||||
Otherwise the hook is removed cleanly.
|
||||
|
||||
## Behavior on machines without gbrain installed
|
||||
|
||||
The hook script checks for `gbrain` on `$PATH`. When missing, it prints a
|
||||
one-line warning to stderr and exits 0 — commits aren't blocked just because
|
||||
a developer hasn't installed gbrain locally. Once gbrain is installed, the
|
||||
hook resumes blocking malformed pages.
|
||||
|
||||
## For downstream agent forks
|
||||
|
||||
If your fork (Wintermute, Hermes, OpenClaw) wraps gbrain in a host repo
|
||||
that's not the brain repo itself, you may want a separate hook strategy:
|
||||
|
||||
- **Brain repo IS the host repo** (gbrain skills + brain pages in one repo):
|
||||
install via `gbrain frontmatter install-hook` as above.
|
||||
- **Brain repo is a separate registered source** (e.g. `~/brain` registered
|
||||
as a source, host repo is `~/agent-fork`): install in the brain repo only;
|
||||
agent-fork code doesn't need this hook.
|
||||
- **Brain repo is auto-generated** (e.g. by a sync daemon writing to a
|
||||
bucket): skip the hook entirely; gate at the writer instead via
|
||||
`import { writeBrainPage } from 'gbrain/brain-writer'` (planned in a
|
||||
later release; currently the CLI is the surface).
|
||||
|
||||
## How it fits into the broader frontmatter pipeline
|
||||
|
||||
```
|
||||
agent writes a page git commit doctor scan
|
||||
↓ ↓ ↓
|
||||
[source content] → [pre-commit hook validates] → [frontmatter_integrity check]
|
||||
↓ ↓ ↓
|
||||
raw file on disk blocks malformed commits surfaces existing issues
|
||||
↓
|
||||
`gbrain frontmatter validate
|
||||
<source-path> --fix`
|
||||
(writes .bak backups)
|
||||
```
|
||||
|
||||
The hook is the write-time gate; doctor is the audit gate; the CLI is the
|
||||
fix tool. They share `parseMarkdown(..., {validate:true})` as the single
|
||||
source of truth for what counts as malformed.
|
||||
@@ -1,8 +1,9 @@
|
||||
# Remote MCP Deployment Options
|
||||
|
||||
GBrain's MCP server runs via `gbrain serve` (stdio transport). To make it
|
||||
accessible from other devices and AI clients, you need an HTTP wrapper and
|
||||
a public tunnel. Here are your options.
|
||||
accessible from other devices and AI clients, run `gbrain serve --http`
|
||||
(built-in HTTP transport with bearer auth, Postgres-only ... see
|
||||
[DEPLOY.md](DEPLOY.md)) behind a public tunnel. Here are your tunnel options.
|
||||
|
||||
## ngrok (recommended)
|
||||
|
||||
@@ -13,8 +14,9 @@ a public tunnel. Here are your options.
|
||||
# 1. Install ngrok
|
||||
brew install ngrok
|
||||
|
||||
# 2. Start your MCP server (behind an HTTP wrapper)
|
||||
# See docs/mcp/DEPLOY.md for the server setup
|
||||
# 2. Start the built-in HTTP transport
|
||||
gbrain serve --http --port 8787
|
||||
# See docs/mcp/DEPLOY.md for token setup
|
||||
|
||||
# 3. Expose via ngrok
|
||||
ngrok http 8787 --url your-brain.ngrok.app
|
||||
@@ -59,6 +61,7 @@ Both run Bun natively. No bundling, no Deno, no cold start, no timeout limits.
|
||||
| All 30 operations | Yes | Yes | Yes |
|
||||
| Setup time | 5 min | 10 min | 15 min |
|
||||
|
||||
**Note:** `gbrain serve --http` (built-in HTTP transport) is planned but not yet
|
||||
implemented. Currently, remote MCP requires a custom HTTP wrapper around `gbrain serve`.
|
||||
See [DEPLOY.md](DEPLOY.md) for details.
|
||||
**Note:** `gbrain serve --http` is the built-in HTTP transport (v0.22.7+). Bearer auth
|
||||
against the `access_tokens` table, default-deny CORS, two-bucket rate limit, body cap,
|
||||
per-request audit log. Postgres-only by design (PGLite is local-only). See
|
||||
[DEPLOY.md](DEPLOY.md) and [SECURITY.md](../../SECURITY.md) for env vars and tunables.
|
||||
|
||||
@@ -21,7 +21,7 @@ claude mcp add gbrain -t http \
|
||||
```
|
||||
|
||||
Replace `YOUR-DOMAIN` with your ngrok domain and `YOUR_TOKEN` with a token
|
||||
from `bun run src/commands/auth.ts create "claude-code"`.
|
||||
from `gbrain auth create "claude-code"`.
|
||||
|
||||
## Verify
|
||||
|
||||
|
||||
@@ -12,7 +12,7 @@ For Team/Enterprise plans, an org Owner adds the connector:
|
||||
https://YOUR-DOMAIN.ngrok.app/mcp
|
||||
```
|
||||
3. Add Bearer token authentication in Advanced Settings
|
||||
(create one with `bun run src/commands/auth.ts create "cowork"`)
|
||||
(create one with `gbrain auth create "cowork"`)
|
||||
4. Save
|
||||
|
||||
Note: Cowork connects from Anthropic's cloud, not your device. Your server
|
||||
|
||||
@@ -16,7 +16,7 @@ Remote HTTP servers must be added through the GUI.
|
||||
Replace `YOUR-DOMAIN` with your ngrok domain (see
|
||||
[ngrok-tunnel recipe](../../recipes/ngrok-tunnel.md) for setup).
|
||||
5. Set authentication to **Bearer Token** and paste your token
|
||||
(create one with `bun run src/commands/auth.ts create "claude-desktop"`)
|
||||
(create one with `gbrain auth create "claude-desktop"`)
|
||||
6. Save
|
||||
|
||||
## Verify
|
||||
|
||||
+21
-14
@@ -1,8 +1,13 @@
|
||||
# Deploy GBrain Remote MCP Server
|
||||
|
||||
> **v0.22.7+:** Use `gbrain serve --http` for remote access. It includes built-in
|
||||
> bearer token auth, default-deny CORS, two-bucket rate limiting, body cap, and
|
||||
> per-request audit log. **Postgres-only** (PGLite is local-only by design).
|
||||
> See [SECURITY.md](../../SECURITY.md) for env vars and tunable defaults.
|
||||
|
||||
Access your brain from any device, any AI client. GBrain's MCP server runs locally
|
||||
via `gbrain serve` (stdio). For remote access, wrap it in an HTTP server behind a
|
||||
public tunnel.
|
||||
via `gbrain serve` (stdio). For remote access, expose it via the built-in HTTP
|
||||
transport behind a public tunnel.
|
||||
|
||||
## Two Paths
|
||||
|
||||
@@ -13,21 +18,23 @@ gbrain serve
|
||||
```
|
||||
|
||||
Works with Claude Code, Cursor, Windsurf, and any MCP client that supports stdio.
|
||||
No server, no tunnel, no token needed.
|
||||
No server, no tunnel, no token needed. Works on both PGLite and Postgres engines.
|
||||
|
||||
### Remote (any device, any AI client)
|
||||
### Remote (any device, any AI client) — Postgres only
|
||||
|
||||
```
|
||||
Your AI client (Claude Desktop, Perplexity, etc.)
|
||||
→ ngrok tunnel (https://YOUR-DOMAIN.ngrok.app)
|
||||
→ Your HTTP server (wraps gbrain serve)
|
||||
→ Supabase Postgres (via pooler connection string)
|
||||
→ gbrain serve --http (built-in transport with bearer auth)
|
||||
→ Postgres (pooler connection or self-hosted)
|
||||
```
|
||||
|
||||
This requires:
|
||||
1. A machine running `gbrain serve` behind an HTTP wrapper
|
||||
2. A public tunnel (ngrok, Tailscale, or cloud host)
|
||||
3. Bearer token auth for security
|
||||
1. A Postgres-backed brain (the `access_tokens` table only exists on Postgres;
|
||||
running `gbrain serve --http` against a PGLite install fails fast at startup)
|
||||
2. A machine running `gbrain serve --http`
|
||||
3. A public tunnel (ngrok, Tailscale, or cloud host)
|
||||
4. A bearer token created via `gbrain auth create <name>`
|
||||
|
||||
## Remote Setup
|
||||
|
||||
@@ -46,13 +53,13 @@ ngrok http 8787 --url your-brain.ngrok.app # Hobby tier for fixed domain
|
||||
|
||||
```bash
|
||||
# Create a token for each client
|
||||
bun run src/commands/auth.ts create "claude-desktop"
|
||||
gbrain auth create "claude-desktop"
|
||||
|
||||
# List all tokens
|
||||
bun run src/commands/auth.ts list
|
||||
gbrain auth list
|
||||
|
||||
# Revoke a token
|
||||
bun run src/commands/auth.ts revoke "claude-desktop"
|
||||
gbrain auth revoke "claude-desktop"
|
||||
```
|
||||
|
||||
Tokens are per-client. Create one for each device/app. Revoke individually
|
||||
@@ -68,7 +75,7 @@ if compromised. Tokens are stored SHA-256 hashed in your database.
|
||||
### 4. Verify
|
||||
|
||||
```bash
|
||||
bun run src/commands/auth.ts test \
|
||||
gbrain auth test \
|
||||
https://YOUR-DOMAIN.ngrok.app/mcp \
|
||||
--token YOUR_TOKEN
|
||||
```
|
||||
@@ -96,7 +103,7 @@ Funnel, and cloud hosts (Fly.io, Railway).
|
||||
Include the Authorization header: `Authorization: Bearer YOUR_TOKEN`
|
||||
|
||||
**"invalid_token" error**
|
||||
Run `bun run src/commands/auth.ts list` to see active tokens.
|
||||
Run `gbrain auth list` to see active tokens.
|
||||
|
||||
**"service_unavailable" error**
|
||||
Database connection failed. Check your Supabase dashboard for outages.
|
||||
|
||||
@@ -10,7 +10,7 @@ Perplexity Computer supports remote MCP servers with bearer token authentication
|
||||
- **URL:** `https://YOUR-DOMAIN.ngrok.app/mcp`
|
||||
- **Authentication:** API Key / Bearer Token
|
||||
- **Token:** your GBrain access token
|
||||
(create one with `bun run src/commands/auth.ts create "perplexity"`)
|
||||
(create one with `gbrain auth create "perplexity"`)
|
||||
4. Save
|
||||
|
||||
Replace `YOUR-DOMAIN` with your ngrok domain (see
|
||||
|
||||
@@ -0,0 +1,210 @@
|
||||
# Storage Tiering: db-tracked vs db-only directories
|
||||
|
||||
## Overview
|
||||
|
||||
GBrain supports storage tiering to separate version-controlled content from bulk machine-generated data. This prevents git repositories from becoming bloated with large amounts of automatically generated content while still preserving it in the database.
|
||||
|
||||
> Note on naming: prior to v0.22.11 the keys were `git_tracked` / `supabase_only`. The canonical names are now `db_tracked` / `db_only` (engine-agnostic — works on both PGLite and Postgres). The deprecated keys still load with a once-per-process warning. Run `gbrain doctor --fix` for an automated rename when that path lands.
|
||||
|
||||
## Configuration
|
||||
|
||||
Add a `storage` section to your `gbrain.yml` file in the brain repository root:
|
||||
|
||||
```yaml
|
||||
storage:
|
||||
# Directories that are version-controlled (human-edited, committed to git).
|
||||
db_tracked:
|
||||
- people/
|
||||
- companies/
|
||||
- deals/
|
||||
- concepts/
|
||||
- yc/
|
||||
- ideas/
|
||||
- projects/
|
||||
|
||||
# Directories persisted via the brain database only (bulk machine-generated
|
||||
# content). Written to disk as a local cache but not committed to git;
|
||||
# `gbrain sync` auto-manages .gitignore for these paths. `gbrain export
|
||||
# --restore-only` repopulates missing files from the database.
|
||||
db_only:
|
||||
- media/x/
|
||||
- media/articles/
|
||||
- meetings/transcripts/
|
||||
```
|
||||
|
||||
Path requirements:
|
||||
|
||||
- Each directory must end with `/` for canonical form. The validator auto-normalizes missing trailing slashes (one-time info note shows what changed).
|
||||
- A directory cannot appear in both tiers — that's a tier-overlap error and `loadStorageConfig` throws `StorageConfigError`. Edit `gbrain.yml` to remove the overlap and try again.
|
||||
|
||||
## Behavior Changes
|
||||
|
||||
### 1. `gbrain sync` — automatic .gitignore management
|
||||
|
||||
When storage configuration is present, `gbrain sync` automatically manages `.gitignore` entries on every successful sync:
|
||||
|
||||
- Adds missing `db_only` directory patterns to `.gitignore`.
|
||||
- Idempotent — re-running adds no duplicate entries.
|
||||
- Stable comment header so the managed block is grep-able.
|
||||
- Skipped on `--dry-run` (don't mutate disk in preview mode).
|
||||
- Skipped on `blocked_by_failures` status (sync state is inconsistent).
|
||||
- Skipped when the repo is a git submodule (`.git` is a file, not a directory) — submodule .gitignore changes don't survive parent updates. A warning explains.
|
||||
- Skipped entirely when `GBRAIN_NO_GITIGNORE=1` is set (escape hatch for shared-repo setups where a maintainer wants gbrain to leave .gitignore alone).
|
||||
- Failures (write permission denied, etc.) are caught and logged, never crash sync.
|
||||
|
||||
Example `.gitignore` addition:
|
||||
|
||||
```gitignore
|
||||
# Auto-managed by gbrain (db_only directories)
|
||||
media/x/
|
||||
media/articles/
|
||||
meetings/transcripts/
|
||||
```
|
||||
|
||||
### 2. `gbrain export --restore-only` — repopulate missing db_only files
|
||||
|
||||
```bash
|
||||
# Restore only missing db_only files from the database.
|
||||
gbrain export --restore-only --repo /path/to/brain
|
||||
|
||||
# Filter by page type.
|
||||
gbrain export --restore-only --type media --repo /path/to/brain
|
||||
|
||||
# Filter by slug prefix.
|
||||
gbrain export --restore-only --slug-prefix media/x/ --repo /path/to/brain
|
||||
|
||||
# Combine filters.
|
||||
gbrain export --restore-only --type media --slug-prefix media/x/ --repo /path/to/brain
|
||||
```
|
||||
|
||||
The `--restore-only` flag:
|
||||
|
||||
- Resolves repoPath via the chain `--repo` → typed `sources.getDefault()` → hard error.
|
||||
Never falls through to the current directory.
|
||||
- Only exports pages that match `db_only` patterns AND are missing from disk.
|
||||
- Ideal for container restart recovery and fresh clones.
|
||||
|
||||
### 3. `gbrain storage status` — storage-tier health dashboard
|
||||
|
||||
```bash
|
||||
# Human-readable status.
|
||||
gbrain storage status --repo /path/to/brain
|
||||
|
||||
# JSON output for scripts and orchestrators.
|
||||
gbrain storage status --repo /path/to/brain --json
|
||||
```
|
||||
|
||||
Output includes:
|
||||
|
||||
- Total page counts by storage tier.
|
||||
- Disk usage breakdown by tier.
|
||||
- Missing files that need restoration (top 10 shown; full list in `--json`).
|
||||
- Configuration validation warnings.
|
||||
- Current tier directory listing.
|
||||
|
||||
Example output:
|
||||
|
||||
```
|
||||
Storage Status
|
||||
==============
|
||||
|
||||
Repository: /data/brain
|
||||
Total pages: 15,243
|
||||
|
||||
Storage Tiers:
|
||||
-------------
|
||||
DB tracked: 2,156 pages
|
||||
DB only: 12,887 pages
|
||||
Unspecified: 200 pages
|
||||
|
||||
Disk Usage:
|
||||
-----------
|
||||
DB tracked: 45.2 MB
|
||||
DB only: 2.1 GB
|
||||
|
||||
Missing Files (need restore):
|
||||
-----------------------------
|
||||
media/x/tweet-1234567890
|
||||
media/x/tweet-0987654321
|
||||
... and 47 more
|
||||
|
||||
Use: gbrain export --restore-only --repo "/data/brain"
|
||||
|
||||
Configuration:
|
||||
--------------
|
||||
DB tracked directories:
|
||||
- people/
|
||||
- companies/
|
||||
- deals/
|
||||
|
||||
DB-only directories:
|
||||
- media/x/
|
||||
- media/articles/
|
||||
- meetings/transcripts/
|
||||
```
|
||||
|
||||
## Validation
|
||||
|
||||
`loadStorageConfig` runs `normalizeAndValidateStorageConfig` after parsing:
|
||||
|
||||
- Auto-fixes (silent, with one-time info note showing what changed):
|
||||
- Missing trailing `/` is added: `'media/x'` → `'media/x/'`.
|
||||
- Throws `StorageConfigError` (caller sees a clean exit-1 with actionable message):
|
||||
- Same directory in both `db_tracked` and `db_only` (ambiguous routing).
|
||||
|
||||
## Use cases
|
||||
|
||||
### Brain repository scaling
|
||||
|
||||
Perfect for brain repositories crossing 50K-200K+ files where:
|
||||
|
||||
- Core knowledge (people, companies, deals) remains git-tracked.
|
||||
- Bulk data (tweets, articles, transcripts) moves to db_only.
|
||||
- Development stays fast with smaller git repos.
|
||||
- Full data remains available via the database.
|
||||
|
||||
### Container-based deployments
|
||||
|
||||
Essential for ephemeral container environments:
|
||||
|
||||
- Git repo contains only essential files.
|
||||
- Container restarts don't lose db_only data.
|
||||
- `gbrain export --restore-only` quickly restores bulk files when needed.
|
||||
- Local disk acts as a cache layer.
|
||||
|
||||
### Multi-environment consistency
|
||||
|
||||
Enables consistent data access across environments:
|
||||
|
||||
- Development: small git clone, restore bulk data on demand.
|
||||
- Production: full dataset via the database, selective local caching.
|
||||
- CI/CD: fast tests with git-tracked data only.
|
||||
|
||||
## Migration strategy
|
||||
|
||||
1. **Assess current repository**: use `gbrain storage status` to understand current distribution.
|
||||
2. **Plan directory structure**: identify which directories should be db_tracked vs db_only.
|
||||
3. **Create `gbrain.yml`**: add storage configuration to the repository root.
|
||||
4. **Test with dry-run**: `gbrain sync --dry-run` to verify behavior; `.gitignore` is NOT touched on dry-run.
|
||||
5. **Run a real sync**: `gbrain sync` updates `.gitignore` automatically on success.
|
||||
6. **Verify restore**: test `gbrain export --restore-only --repo .` against a small db_only directory.
|
||||
|
||||
## Best practices
|
||||
|
||||
- **Directory naming**: end storage paths with `/` (canonical form). The validator normalizes if you forget.
|
||||
- **Start small**: begin with clearly machine-generated directories in `db_only`.
|
||||
- **Address validation errors**: tier overlap is an error, not a warning. Fix it before sync.
|
||||
- **Test restore**: regularly test `--restore-only` in staging environments.
|
||||
- **Document decisions**: comment your `gbrain.yml` to explain tier choices.
|
||||
|
||||
## PGLite engine note
|
||||
|
||||
On the PGLite engine (gbrain's local-only embedded Postgres), the "DB" your db_only pages live in IS the local file gbrain uses for everything else. The `.gitignore` housekeeping still helps (keeps bulk content out of git history), but the offload-to-DB promise is technically vacuous. A once-per-process soft-warn explains when the engine is detected. To get full tiering, migrate to Postgres with `gbrain migrate --to supabase`.
|
||||
|
||||
## Compatibility
|
||||
|
||||
- **Backward compatible**: systems without `gbrain.yml` work unchanged.
|
||||
- **Progressive enhancement**: add configuration when needed.
|
||||
- **Database unchanged**: all data remains in Postgres regardless of tier.
|
||||
- **Existing workflows**: all existing `sync` and `export` behavior preserved.
|
||||
- **Deprecated keys**: `git_tracked` / `supabase_only` still load with a once-per-process warning.
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
storage:
|
||||
# Directories that are version-controlled — human-curated, edited by hand.
|
||||
db_tracked:
|
||||
- people/
|
||||
- companies/
|
||||
- deals/
|
||||
- concepts/
|
||||
- yc/
|
||||
- ideas/
|
||||
- projects/
|
||||
|
||||
# Directories persisted via the brain database only — bulk machine-generated
|
||||
# content. .gitignored automatically by `gbrain sync`. Restorable from the DB
|
||||
# via `gbrain export --restore-only`.
|
||||
db_only:
|
||||
- media/x/
|
||||
- media/articles/
|
||||
- meetings/transcripts/
|
||||
+252
-36
@@ -104,21 +104,30 @@ strict behavior when unset.
|
||||
- `src/core/operations.ts` — Contract-first operation definitions (the foundation). Also exports upload validators: `validateUploadPath`, `validatePageSlug`, `validateFilename`. `OperationContext.remote` flags untrusted callers.
|
||||
- `src/core/engine.ts` — Pluggable engine interface (BrainEngine). `clampSearchLimit(limit, default, cap)` takes an explicit cap so per-operation caps can be tighter than `MAX_SEARCH_LIMIT`. Exports `LinkBatchInput` / `TimelineBatchInput` for the v0.12.1 bulk-insert API (`addLinksBatch` / `addTimelineEntriesBatch`). As of v0.13.1, `BrainEngine` has a `readonly kind: 'postgres' | 'pglite'` discriminator so migrations (`src/core/migrate.ts`) and other consumers can branch on engine without `instanceof` + dynamic imports.
|
||||
- `src/core/engine-factory.ts` — Engine factory with dynamic imports (`'pglite'` | `'postgres'`)
|
||||
- `src/core/pglite-engine.ts` — PGLite (embedded Postgres 17.5 via WASM) implementation, all 40 BrainEngine methods. `addLinksBatch` / `addTimelineEntriesBatch` use multi-row `unnest()` with manual `$N` placeholders. As of v0.13.1, `connect()` wraps `PGlite.create()` in a try/catch that emits an actionable error naming the macOS 26.3 WASM bug (#223) and pointing at `gbrain doctor`; the lock is released on failure so the next process can retry cleanly.
|
||||
- `src/core/pglite-engine.ts` — PGLite (embedded Postgres 17.5 via WASM) implementation, all 40 BrainEngine methods. `addLinksBatch` / `addTimelineEntriesBatch` use multi-row `unnest()` with manual `$N` placeholders. As of v0.13.1, `connect()` wraps `PGlite.create()` in a try/catch that emits an actionable error naming the macOS 26.3 WASM bug (#223) and pointing at `gbrain doctor`; the lock is released on failure so the next process can retry cleanly. v0.22.0: `searchKeyword` and `searchKeywordChunks` multiply `ts_rank` by the source-factor CASE expression at the chunk-grain level; `searchVector` becomes a two-stage CTE — inner CTE keeps `ORDER BY cc.embedding <=> vec` so HNSW stays usable, outer SELECT re-ranks by `raw_score * source_factor`. Inner LIMIT scales with offset to preserve pagination contract. As of v0.22.6.1, `initSchema()` calls `applyForwardReferenceBootstrap()` BEFORE replaying SCHEMA_SQL — probes for the specific forward-referenced state the embedded schema blob needs (`pages.source_id`, `links.link_source`, `links.origin_page_id`, `content_chunks.symbol_name`, `content_chunks.language`, `sources` FK target table) and adds only what's missing. Closes the upgrade-wedge bug class that bit users 10+ times across 6 schema versions over 2 years (#239/#243/#266/#357/#366/#374/#375/#378/#395/#396). No-op on fresh installs and modern brains.
|
||||
- `src/core/pglite-schema.ts` — PGLite-specific DDL (pgvector, pg_trgm, triggers)
|
||||
- `src/core/postgres-engine.ts` — Postgres + pgvector implementation (Supabase / self-hosted). `addLinksBatch` / `addTimelineEntriesBatch` use `INSERT ... SELECT FROM unnest($1::text[], ...) JOIN pages ON CONFLICT DO NOTHING RETURNING 1` — 4-5 array params regardless of batch size, sidesteps the 65535-parameter cap. As of v0.12.3, `searchKeyword` / `searchVector` scope `statement_timeout` via `sql.begin` + `SET LOCAL` so the GUC dies with the transaction instead of leaking across the pooled postgres.js connection (contributed by @garagon). `getEmbeddingsByChunkIds` uses `tryParseEmbedding` so one corrupt row skips+warns instead of killing the query.
|
||||
- `src/core/postgres-engine.ts` — Postgres + pgvector implementation (Supabase / self-hosted). `addLinksBatch` / `addTimelineEntriesBatch` use `INSERT ... SELECT FROM unnest($1::text[], ...) JOIN pages ON CONFLICT DO NOTHING RETURNING 1` — 4-5 array params regardless of batch size, sidesteps the 65535-parameter cap. As of v0.12.3, `searchKeyword` / `searchVector` scope `statement_timeout` via `sql.begin` + `SET LOCAL` so the GUC dies with the transaction instead of leaking across the pooled postgres.js connection (contributed by @garagon). `getEmbeddingsByChunkIds` uses `tryParseEmbedding` so one corrupt row skips+warns instead of killing the query. v0.22.0: `searchKeyword`, `searchKeywordChunks`, and `searchVector` apply source-aware ranking by inlining the source-factor CASE and `NOT (col LIKE …)` hard-exclude clause from `src/core/search/sql-ranking.ts`. `searchVector` switches to a two-stage CTE (HNSW-safe inner ORDER BY, source-boost re-rank in the outer SELECT) and carries `p.source_id` through inner→outer for v0.18 multi-source callers. v0.22.1 (#406): `_savedConfig` retains the connect config; `reconnect()` tears down + recreates the pool from saved config (called by supervisor watchdog after 3 consecutive health-check failures). `executeRaw` is a single-statement passthrough — no per-call retry (D3 dropped that as unsound for non-idempotent statements; recovery is supervisor-driven). v0.22.1 (#363, contributed by @orendi84): `connect()` applies `resolveSessionTimeouts()` from `db.ts` as connection-time startup parameters (`statement_timeout`, `idle_in_transaction_session_timeout`) so orphan pgbouncer backends can't hold locks for hours. v0.22.1 (#409, contributed by @atrevino47): `countStaleChunks()` + `listStaleChunks()` server-side-filter on `embedding IS NULL` for `embed --stale`, eliminating ~76 MB/call client-side pull on a fully-embedded brain; `upsertChunks()` resets both `embedding` AND `embedded_at` to NULL when chunk_text changes without a new embedding (consistency). As of v0.22.6.1, `initSchema()` calls `applyForwardReferenceBootstrap()` BEFORE replaying SCHEMA_SQL on the same forward-reference probe set as the PGLite engine, so old Postgres brains pinned at v0.13/v0.18/v0.19 walk forward cleanly instead of wedging on `column "..." does not exist`.
|
||||
- `src/core/utils.ts` — Shared SQL utilities extracted from postgres-engine.ts. Exports `parseEmbedding(value)` (throws on unknown input, used by migration + ingest paths where data integrity matters) and as of v0.12.3 `tryParseEmbedding(value)` (returns `null` + warns once per process, used by search/rescore paths where availability matters more than strictness).
|
||||
- `src/core/db.ts` — Connection management, schema initialization
|
||||
- `src/core/db.ts` — Connection management, schema initialization. v0.22.1 (#363, contributed by @orendi84): `resolveSessionTimeouts()` returns `statement_timeout` + `idle_in_transaction_session_timeout` (defaults: 5min each, env-overridable via `GBRAIN_STATEMENT_TIMEOUT` / `GBRAIN_IDLE_TX_TIMEOUT` / `GBRAIN_CLIENT_CHECK_INTERVAL`). Both `connect()` (module singleton) and `PostgresEngine.connect()` (worker pool) consume the result via postgres.js's `connection` option, sending GUCs as startup parameters that survive PgBouncer transaction mode (unlike the prior `setSessionDefaults` post-pool SET, kept as a back-compat no-op shim).
|
||||
- `src/commands/migrate-engine.ts` — Bidirectional engine migration (`gbrain migrate --to supabase/pglite`)
|
||||
- `src/core/import-file.ts` — importFromFile + importFromContent (chunk + embed + tags)
|
||||
- `src/core/sync.ts` — Pure sync functions (manifest parsing, filtering, slug conversion)
|
||||
- `src/core/sync.ts` — Pure sync functions (manifest parsing, filtering, slug conversion). v0.22.12 (#500, foundation by @wintermute via #501): `classifyErrorCode(errorMsg)` regex-based classifier with 12 codes (`SLUG_MISMATCH`, `YAML_PARSE`, `YAML_DUPLICATE_KEY`, `MISSING_OPEN`, `MISSING_CLOSE`, `NESTED_QUOTES`, `EMPTY_FRONTMATTER`, `NULL_BYTES`, `INVALID_UTF8`, `STATEMENT_TIMEOUT`, `FILE_TOO_LARGE`, `SYMLINK_NOT_ALLOWED`) plus `UNKNOWN` fallback. `summarizeFailuresByCode(failures)` returns sorted `[{code, count}]`. `code?` optional field on `SyncFailure`; backfilled at ack time on pre-v0.22.12 entries. `acknowledgeSyncFailures()` returns `AcknowledgeResult { count, summary }`. Three regexes (`MISSING_OPEN`, `MISSING_CLOSE`, `EMPTY_FRONTMATTER`) broadened to match actual `markdown.ts:159-244` validator message strings, not just the literal code-name prefix. `FILE_TOO_LARGE` covers all three production size sites in `import-file.ts:199, 352, 401`; `SYMLINK_NOT_ALLOWED` covers the rejection at `:347`. Closes the silent-skip pattern that motivated #500.
|
||||
- `src/core/storage.ts` — Pluggable storage interface (S3, Supabase Storage, local)
|
||||
- `src/core/storage-config.ts` (v0.22.11) — Storage tiering: `loadStorageConfig` reads `gbrain.yml`, normalizes deprecated keys (`git_tracked` / `supabase_only`) to canonical (`db_tracked` / `db_only`) with once-per-process deprecation warning, and runs `normalizeAndValidateStorageConfig` (auto-fixes missing trailing `/`, throws `StorageConfigError` on tier overlap). Path-segment matcher: `media/x/` does NOT match `media/xerox/foo`. Replaces gray-matter (broken on delimiter-less YAML) with a dedicated parser for the `gbrain.yml` shape.
|
||||
- `src/core/disk-walk.ts` (v0.22.11) — `walkBrainRepo(repoPath)` returns `Map<slug, {size, mtimeMs}>` from one recursive `readdirSync`. Skips dot-dirs, `node_modules`, non-`.md` files. Used by `gbrain storage status` to replace per-page `existsSync + statSync` (~400K syscalls on 200K-page brains → tens).
|
||||
- `src/commands/storage.ts` (v0.22.11) — `gbrain storage status [--repo P] [--json]`. Split into pure data (`getStorageStatus`) + JSON formatter + human formatter (ASCII-only per D10) matching the `orphans.ts` pattern. `PageCountsByTier` and `DiskUsageByTier` are distinct nominal types so swaps fail at compile time.
|
||||
- `gbrain.yml` (brain repo root, v0.22.11) — Optional storage tiering config. Top-level `storage:` section with `db_tracked:` and `db_only:` array-valued keys. `gbrain sync` auto-manages `.gitignore` for `db_only` paths on successful sync (skips on dry-run, blocked-by-failures, submodule context, or `GBRAIN_NO_GITIGNORE=1`). `gbrain export --restore-only [--repo P] [--type T] [--slug-prefix S]` repopulates missing `db_only` files from the database.
|
||||
- `src/core/supabase-admin.ts` — Supabase admin API (project discovery, pgvector check)
|
||||
- `src/core/file-resolver.ts` — File resolution with fallback chain (local -> .redirect.yaml -> .redirect -> .supabase)
|
||||
- `src/core/chunkers/` — 3-tier chunking (recursive, semantic, LLM-guided)
|
||||
- `src/core/search/` — Hybrid search: vector + keyword + RRF + multi-query expansion + dedup
|
||||
- `src/core/chunkers/` — 3-tier chunking (recursive, semantic, LLM-guided). v0.19.0 adds `code.ts` — tree-sitter-based semantic chunker for 29 languages with embedded-asset WASMs (`src/assets/wasm/`), `@dqbd/tiktoken` cl100k_base tokenizer, small-sibling merging. `CHUNKER_VERSION` constant folded into `importCodeFile`'s `content_hash` so chunker shape changes force clean re-chunks across releases.
|
||||
- `src/core/errors.ts` (v0.19.0) — `StructuredAgentError` + `buildError` + `serializeError`. Every new v0.19.0 agent-facing surface (code-def, code-refs, usage errors) uses this envelope; matches v0.17.0 `CycleReport.PhaseResult.error` shape.
|
||||
- `src/assets/wasm/` (v0.19.0) — 36 tree-sitter grammar WASMs + tree-sitter runtime. Committed to the repo so `bun --compile` embeds them deterministically via `import path from ... with { type: 'file' }`. The CI guard `scripts/check-wasm-embedded.sh` fails the build if the compiled binary ever silently falls through to recursive chunks.
|
||||
- `src/commands/code-def.ts` + `src/commands/code-refs.ts` (v0.19.0) — symbol definition + references lookup. Query `content_chunks.symbol_name` or chunk_text ILIKE with `page_kind='code'` filter. Auto-JSON when stdout is not a TTY (gh-CLI convention). Bypass the standard `searchKeyword` `DISTINCT ON (slug)` collapse so multiple call-sites from the same file surface.
|
||||
- `src/core/search/` — Hybrid search: vector + keyword + RRF + multi-query expansion + dedup. As of v0.22.0, `searchKeyword` / `searchKeywordChunks` / `searchVector` apply source-aware ranking at the SQL layer (curated content like `originals/`, `concepts/`, `writing/` outranks bulk content like `wintermute/chat/`, `daily/`, `media/x/`). `searchVector` uses a two-stage CTE so source-boost re-ranking doesn't kill the HNSW index. Hard-exclude prefixes (`test/`, `archive/`, `attachments/`, `.raw/` by default) filter at retrieval, not post-rank. Both gates honor `detail !== 'high'` so temporal queries surface chat pages normally.
|
||||
- `src/core/search/intent.ts` — Query intent classifier (entity/temporal/event/general → auto-selects detail level)
|
||||
- `src/core/search/eval.ts` — Retrieval eval harness: P@k, R@k, MRR, nDCG@k metrics + runEval() orchestrator
|
||||
- `src/core/search/source-boost.ts` (v0.22.0) — Source-type boost map keyed by slug prefix. `DEFAULT_SOURCE_BOOSTS` (originals/ 1.5, concepts/ 1.3, writing/ 1.4, people/companies/deals/ 1.2, daily/ 0.8, media/x/ 0.7, wintermute/chat/ 0.5) and `DEFAULT_HARD_EXCLUDES` (test/, archive/, attachments/, .raw/). `parseSourceBoostEnv` / `parseHardExcludesEnv` parse comma-separated `prefix:factor` pairs from `GBRAIN_SOURCE_BOOST` / `GBRAIN_SEARCH_EXCLUDE` env vars. `resolveBoostMap` and `resolveHardExcludes` merge defaults + env + caller `SearchOpts.exclude_slug_prefixes`/`include_slug_prefixes`.
|
||||
- `src/core/search/sql-ranking.ts` (v0.22.0) — Pure SQL string builders. `buildSourceFactorCase(slugColumn, boostMap, detail)` emits a CASE expression with longest-prefix-match wins (returns literal `'1.0'` when `detail === 'high'` for temporal-bypass parity with COMPILED_TRUTH_BOOST). `buildHardExcludeClause(slugColumn, prefixes)` emits `NOT (col LIKE 'p1%' OR col LIKE 'p2%')` — OR-chain wrapped in NOT, NOT `NOT LIKE ALL/ANY` (those quantifiers don't express set-exclusion). LIKE meta-character escape covers all three of `%`, `_`, AND `\` (backslash matters because it's Postgres LIKE's default escape char). Single-quote doubling on SQL string literals so injection-style inputs are inert text.
|
||||
- `src/commands/eval.ts` — `gbrain eval` command: single-run table + A/B config comparison
|
||||
- `src/core/embedding.ts` — OpenAI text-embedding-3-large, batch, retry, backoff
|
||||
- `src/core/check-resolvable.ts` — Resolver validation: reachability, MECE overlap, DRY checks, structured fix objects. v0.14.1: `CROSS_CUTTING_PATTERNS.conventions` is an array (notability gate accepts both `conventions/quality.md` and `_brain-filing-rules.md`). New `extractDelegationTargets()` parses `> **Convention:**`, `> **Filing rule:**`, and inline backtick references. DRY suppression is proximity-based via `DRY_PROXIMITY_LINES = 40`.
|
||||
@@ -137,12 +146,14 @@ strict behavior when unset.
|
||||
- `src/core/transcription.ts` — Audio transcription: Groq Whisper (default), OpenAI fallback, ffmpeg segmentation for >25MB
|
||||
- `src/core/enrichment-service.ts` — Global enrichment service: entity slug generation, tier auto-escalation, batch throttling
|
||||
- `src/core/data-research.ts` — Recipe validation, field extraction (MRR/ARR regex), dedup, tracker parsing, HTML stripping
|
||||
- `src/commands/extract.ts` — `gbrain extract links|timeline|all [--source fs|db]`: batch link/timeline extraction. fs walks markdown files, db walks pages from the engine (mutation-immune snapshot iteration; use this for live brains with no local checkout). As of v0.12.1 there is no in-memory dedup pre-load — candidates are buffered 100 at a time and flushed via `addLinksBatch` / `addTimelineEntriesBatch`; `ON CONFLICT DO NOTHING` enforces uniqueness at the DB layer, and the `created` counter returns real rows inserted (truthful on re-runs).
|
||||
- `src/commands/embed.ts` — `gbrain embed [--stale|--all] [--slugs ...]`. v0.22.1 (#409, contributed by @atrevino47): `--stale` path now starts with `engine.countStaleChunks()` (single SELECT count(*) WHERE embedding IS NULL, ~50 bytes wire). On a fully-embedded brain that's a 1-line short-circuit — no further reads. When stale chunks exist, `engine.listStaleChunks()` returns just the chunks needing embeddings (slug + chunk_index + chunk_text + metadata, no `vector(1536)` payload). Caller groups by slug, embeds via OpenAI, re-upserts via `upsertChunks`. Replaces the prior page-walk that pulled every chunk's embedding column over the wire and discarded most.
|
||||
- `src/commands/extract.ts` — `gbrain extract links|timeline|all [--source fs|db]`: batch link/timeline extraction. fs walks markdown files, db walks pages from the engine (mutation-immune snapshot iteration; use this for live brains with no local checkout). As of v0.12.1 there is no in-memory dedup pre-load — candidates are buffered 100 at a time and flushed via `addLinksBatch` / `addTimelineEntriesBatch`; `ON CONFLICT DO NOTHING` enforces uniqueness at the DB layer, and the `created` counter returns real rows inserted (truthful on re-runs). v0.22.1 (#417): `ExtractOpts.slugs?: string[]` enables incremental extract — when set, `extractForSlugs()` reads ONLY those slugs' files (single combined links+timeline pass) instead of the full directory walk. CLI `gbrain extract` keeps full-walk behavior; the cycle path threads sync's `pagesAffected` through. `walkMarkdownFiles(brainDir)` still runs at line 455 to build `allSlugs` for link resolution — see `TODOS.md` for replacing it with `engine.getAllSlugs()`.
|
||||
- `src/commands/graph-query.ts` — `gbrain graph-query <slug> [--type T] [--depth N] [--direction in|out|both]`: typed-edge relationship traversal (renders indented tree)
|
||||
- `src/core/link-extraction.ts` — shared library for the v0.12.0 graph layer. extractEntityRefs (canonical, replaces backlinks.ts duplicate) matches both `[Name](people/slug)` markdown links and Obsidian `[[people/slug|Name]]` wikilinks as of v0.12.3. extractPageLinks, inferLinkType heuristics (attended/works_at/invested_in/founded/advises/source/mentions), parseTimelineEntries, isAutoLinkEnabled config helper. `DIR_PATTERN` covers `people`, `companies`, `deals`, `topics`, `concepts`, `projects`, `entities`, `tech`, `finance`, `personal`, `openclaw`. Used by extract.ts, operations.ts auto-link post-hook, and backlinks.ts.
|
||||
- `src/core/minions/` — Minions job queue: BullMQ-inspired, Postgres-native (queue, worker, backoff, types, protected-names, quiet-hours, stagger, handlers/shell).
|
||||
- `src/core/minions/queue.ts` — MinionQueue class (submit, claim, complete, fail, stall detection, parent-child, depth/child-cap, per-job timeouts, cascade-kill, attachments, idempotency keys, child_done inbox, removeOnComplete/Fail). `add()` takes a 4th `trusted` arg (separate from `opts` to prevent spread leakage); protected names in `PROTECTED_JOB_NAMES` require `{allowProtectedSubmit: true}` and the check runs trim-normalized (whitespace-bypass safe). v0.14.1 #219: `add()` plumbs `max_stalled` through with a `[1, 100]` clamp; omitted values let the schema DEFAULT (5) kick in. v0.19.0: `handleWallClockTimeouts(lockDurationMs)` is Layer 3 kill shot for jobs where `FOR UPDATE SKIP LOCKED` stall detection and the timeout sweep both fail to evict (wedged worker holding a row lock via a pending transaction). v0.19.1: `maxWaiting` coalesce path now uses `pg_advisory_xact_lock` keyed on `(name, queue)` to serialize concurrent submits for the same key, and filters on `queue` in addition to `name` so cross-queue same-name jobs don't suppress each other.
|
||||
- `src/core/minions/worker.ts` — MinionWorker class (handler registry, lock renewal, graceful shutdown, timeout safety net). v0.14.0 abort-path fix: aborted jobs now call `failJob` with reason (`timeout`/`cancel`/`lock-lost`/`shutdown`) instead of returning silently. `shutdownAbort` (instance field) fires on process SIGTERM/SIGINT and propagates to `ctx.shutdownSignal` — shell handler listens to it; non-shell handlers don't.
|
||||
- `src/core/minions/worker.ts` — MinionWorker class (handler registry, lock renewal, graceful shutdown, timeout safety net). v0.14.0 abort-path fix: aborted jobs now call `failJob` with reason (`timeout`/`cancel`/`lock-lost`/`shutdown`) instead of returning silently. `shutdownAbort` (instance field) fires on process SIGTERM/SIGINT and propagates to `ctx.shutdownSignal` — shell handler listens to it; non-shell handlers don't. v0.22.1 (#403): per-job timeout fires `abort.abort(new Error('timeout'))` then a 30-second grace-then-evict safety net force-evicts the job from `inFlight` and marks it dead in DB if the handler ignores the abort signal — frees the slot even when a handler wedges (the 98-waiting-0-active prod incident driver).
|
||||
- `src/core/minions/supervisor.ts` — MinionSupervisor process manager. Spawns `gbrain jobs work` as a child, restarts on crash with exponential backoff, periodic health check. v0.22.1 (#406): `consecutiveHealthFailures` counter; on 3 consecutive failures emits `health_warn` with `reason: 'db_connection_degraded'` and calls `engine.reconnect()` to swap in a fresh pool, then resets the counter. Worker exit classifier emits `likely_cause` field on `worker_exited` events: `oom_or_external_kill` (SIGKILL), `graceful_shutdown` (SIGTERM), `runtime_error` (code 1), `clean_exit` (code 0), `unknown`.
|
||||
- `src/core/minions/types.ts` — `MinionJobInput` + `MinionJobStatus` + handler context types. `MinionJobInput.max_stalled` (new in v0.14.1) is optional; omitted values let the schema DEFAULT (5) kick in, provided values are clamped to `[1, 100]`.
|
||||
- `src/core/minions/protected-names.ts` — side-effect-free constant module exporting `PROTECTED_JOB_NAMES` + `isProtectedJobName()`. Kept pure so queue core can import without loading handler modules.
|
||||
- `src/core/minions/handlers/shell.ts` — `shell` job handler. Spawns `/bin/sh -c cmd` (absolute path, PATH-override-safe) or `argv[0] argv[1..]` (no shell). Env allowlist: `PATH, HOME, USER, LANG, TZ, NODE_ENV` + caller `env:` overrides. UTF-8-safe stdout/stderr tail via `string_decoder.StringDecoder`. Abort (either `ctx.signal` or `ctx.shutdownSignal`) fires SIGTERM → 5s grace → SIGKILL on child. Requires `GBRAIN_ALLOW_SHELL_JOBS=1` on worker (gated by `registerBuiltinHandlers`).
|
||||
@@ -160,20 +171,27 @@ strict behavior when unset.
|
||||
- `src/core/minions/attachments.ts` — Attachment validation (path traversal, null byte, oversize, base64, duplicate detection)
|
||||
- `src/commands/agent.ts` (v0.16) — `gbrain agent run <prompt> [flags]` CLI. Submits `subagent` (or N children + 1 aggregator) under `{allowProtectedSubmit: true}`. Single-entry `--fanout-manifest` short-circuits. Children get `on_child_fail: 'continue'` + `max_stalled: 3`. `--follow` is the default on TTY; streams logs + polls `waitForCompletion` in parallel. Ctrl-C detaches, does not cancel.
|
||||
- `src/commands/agent-logs.ts` (v0.16) — `gbrain agent logs <job> [--follow] [--since]`. Merges JSONL heartbeat audit + `subagent_messages` into a chronological timeline. `parseSince` accepts ISO-8601 or relative (`5m`, `1h`, `2d`). Transcript tail renders only for terminal jobs.
|
||||
- `src/commands/jobs.ts` — `gbrain jobs` CLI subcommands + `gbrain jobs work` daemon. v0.13.1 surfaces the full `MinionJobInput` retry/backoff/timeout/idempotency surface as first-class CLI flags on `jobs submit`: `--max-stalled`, `--backoff-type fixed|exponential`, `--backoff-delay`, `--backoff-jitter`, `--timeout-ms`, `--idempotency-key`. `jobs smoke --sigkill-rescue` is the opt-in regression guard for #219. v0.16 wires `registerBuiltinHandlers` to always register `subagent` + `subagent_aggregator` (no env flag — `ANTHROPIC_API_KEY` is the natural cost gate, trust is via `PROTECTED_JOB_NAMES`) and loads `GBRAIN_PLUGIN_PATH` plugins at worker startup with a loud startup-line per plugin. `shell` handler still gated by `GBRAIN_ALLOW_SHELL_JOBS=1` (RCE surface, separate concern).
|
||||
- `src/commands/jobs.ts` — `gbrain jobs` CLI subcommands + `gbrain jobs work` daemon. v0.13.1 surfaces the full `MinionJobInput` retry/backoff/timeout/idempotency surface as first-class CLI flags on `jobs submit`: `--max-stalled`, `--backoff-type fixed|exponential`, `--backoff-delay`, `--backoff-jitter`, `--timeout-ms`, `--idempotency-key`. `jobs smoke --sigkill-rescue` is the opt-in regression guard for #219. v0.16 wires `registerBuiltinHandlers` to always register `subagent` + `subagent_aggregator` (no env flag — `ANTHROPIC_API_KEY` is the natural cost gate, trust is via `PROTECTED_JOB_NAMES`) and loads `GBRAIN_PLUGIN_PATH` plugins at worker startup with a loud startup-line per plugin. `shell` handler still gated by `GBRAIN_ALLOW_SHELL_JOBS=1` (RCE surface, separate concern). v0.22.10 (#521): the `autopilot-cycle` handler now forwards `job.data.phases` to `runCycle` (was previously discarded — caller-supplied phase selection silently became a full cycle). Phases are validated against `ALL_PHASES` from `src/core/cycle.ts`; invalid names are filtered out and an empty/missing array falls back to the default 6-phase cycle. v0.22.13 (PR #490 CODEX-1+CODEX-4): `sync` handler now resolves `sourceId` at entry by looking up `sources.local_path` (mirrors `cycle.ts:480`'s autopilot fix from PR #475) so multi-source brains read the per-source `last_commit` anchor instead of the global config key. Concurrency routed through the shared `autoConcurrency()` policy in `src/core/sync-concurrency.ts` instead of the prior hardcoded `4`; PGLite stays serial. `noEmbed` default is `true` (embed is a separate job — submit `gbrain embed --stale` after sync, or rely on the autopilot cycle's embed phase).
|
||||
- `src/commands/features.ts` — `gbrain features --json --auto-fix`: usage scan + feature adoption salesman
|
||||
- `src/commands/autopilot.ts` — `gbrain autopilot --install`: self-maintaining brain daemon (sync+extract+embed)
|
||||
- `src/mcp/server.ts` — MCP stdio server (generated from operations)
|
||||
- `src/commands/auth.ts` — Standalone token management (create/list/revoke/test)
|
||||
- `src/mcp/server.ts` — MCP stdio server (generated from operations). v0.22.7: tool-call handler delegates to `dispatchToolCall` from `src/mcp/dispatch.ts` so stdio + HTTP transports share one validation, context-build, and error-format path.
|
||||
- `src/mcp/dispatch.ts` (v0.22.7) — Shared tool-call dispatch consumed by both stdio (`server.ts`) and HTTP (`http-transport.ts`). Exports `dispatchToolCall(engine, name, params, opts)`, `buildOperationContext(engine, params, opts)`, and `validateParams(op, params)`. Single source of truth for `(ctx, params)` handler arg order and the 5-field `OperationContext` shape (engine + config + logger + dryRun + remote). Defaults to `remote: true` (untrusted); local CLI callers pass `remote: false`. Closed F1 (reversed handler args) + F2 (incomplete OperationContext) + F3 (no param validation) drift bugs in the original v0.22.5 HTTP transport.
|
||||
- `src/mcp/rate-limit.ts` (v0.22.7) — Bounded-LRU token-bucket limiter for `gbrain serve --http`. `buildDefaultLimiters()` returns the two-bucket pipeline used by http-transport: pre-auth IP (default 30/60s, fires BEFORE the DB lookup so brute-force load against `access_tokens` is actually capped) + post-auth token-id (default 60/60s). Tracks `lastTouchedMs` separately from `lastRefillMs` so an exhausted key can't be reset by hammering past the TTL. LRU cap (default 10K keys) bounds memory under attacker-controlled key growth; TTL prune at 2× window evicts abandoned buckets.
|
||||
- `src/mcp/http-transport.ts` (v0.22.7, rewrite) — `gbrain serve --http` HTTP transport. Postgres-only — fails fast at startup on PGLite (the `access_tokens` table only exists on Postgres). Bearer auth against SHA-256 hashes in `access_tokens`. CORS default-deny via `GBRAIN_HTTP_CORS_ORIGIN` allowlist. Body cap stream-counted (1 MiB default via `GBRAIN_HTTP_MAX_BODY_BYTES`) so chunked transfers without Content-Length still hit the cap. `last_used_at` SQL-level debounce (one UPDATE per token per 60s). Per-request audit row in `mcp_request_log` with token_name + operation + status + latency. Optional `GBRAIN_HTTP_TRUST_PROXY=1` honors `X-Forwarded-For` — only safe when bound to a private interface AND the proxy strips client-supplied XFF (otherwise enables IP spoofing past the pre-auth rate limit). `/health` does `SELECT 1` against Postgres and returns 503 + `status:unhealthy` when the DB is unreachable so orchestration doesn't see green pods while clients get misleading 401s. Replaces the standalone OAuth wrapper that was vulnerable to unauthenticated client registration.
|
||||
- `src/commands/auth.ts` — Token management for the HTTP transport. `gbrain auth create/list/revoke/test`. As of v0.22.7 wired into the main CLI (`src/cli.ts`); also runs standalone via `bun run src/commands/auth.ts ...` for environments without a compiled binary. Tokens stored as SHA-256 hashes in `access_tokens` (Postgres-only).
|
||||
- `src/commands/upgrade.ts` — Self-update CLI. `runPostUpgrade()` enumerates migrations from the TS registry (src/commands/migrations/index.ts) and tail-calls `runApplyMigrations(['--yes', '--non-interactive'])` so the mechanical side of every outstanding migration runs unconditionally.
|
||||
- `src/commands/migrations/` — TS migration registry (compiled into the binary; no filesystem walk of `skills/migrations/*.md` needed at runtime). `index.ts` lists migrations in semver order. `v0_11_0.ts` = Minions adoption orchestrator (8 phases). `v0_12_0.ts` = Knowledge Graph auto-wire orchestrator (5 phases: schema → config check → backfill links → backfill timeline → verify). `phaseASchema` has a 600s timeout (bumped from 60s in v0.12.1 for duplicate-heavy brains). `v0_12_2.ts` = JSONB double-encode repair orchestrator (4 phases: schema → repair-jsonb → verify → record). `v0_14_0.ts` = shell-jobs + autopilot cooperative (2 phases: schema ALTER minion_jobs.max_stalled SET DEFAULT 3 — superseded by v0.14.3's schema-level DEFAULT 5 + UPDATE backfill; pending-host-work ping for skills/migrations/v0.14.0.md). All orchestrators are idempotent and resumable from `partial` status. As of v0.14.2 (Bug 3), the RUNNER owns all ledger writes — orchestrators return `OrchestratorResult` and `apply-migrations.ts` persists a canonical `{version, status, phases}` shape after return. Orchestrators no longer call `appendCompletedMigration` directly. `statusForVersion` prefers `complete` over `partial` (never regresses). 3 consecutive partials → wedged → `--force-retry <version>` writes a `'retry'` reset marker. v0.14.3 (fix wave) ships schema-only migrations v14 (`pages_updated_at_index`) + v15 (`minion_jobs_max_stalled_default_5` with UPDATE backfill) via the `MIGRATIONS` array in `src/core/migrate.ts` — no orchestrator phases needed.
|
||||
- `src/commands/repair-jsonb.ts` — `gbrain repair-jsonb [--dry-run] [--json]`: rewrites `jsonb_typeof='string'` rows in place across 5 affected columns (pages.frontmatter, raw_data.data, ingest_log.pages_updated, files.metadata, page_versions.frontmatter). Fixes v0.12.0 double-encode bug on Postgres; PGLite no-ops. Idempotent.
|
||||
- `src/commands/orphans.ts` — `gbrain orphans [--json] [--count] [--include-pseudo]`: surfaces pages with zero inbound wikilinks, grouped by domain. Auto-generated/raw/pseudo pages filtered by default. Also exposed as `find_orphans` MCP operation. Shipped in v0.12.3 (contributed by @knee5).
|
||||
- `src/commands/doctor.ts` — `gbrain doctor [--json] [--fast] [--fix] [--dry-run] [--index-audit]`: health checks. v0.12.3 added `jsonb_integrity` + `markdown_body_completeness` reliability checks. v0.14.1: `--fix` delegates inlined cross-cutting rules to `> **Convention:** see [path](path).` callouts (pipes DRY violations into `src/core/dry-fix.ts`); `--fix --dry-run` previews without writing. v0.14.2: `schema_version` check fails loudly when `version=0` (migrations never ran — the #218 `bun install -g` signature) and routes users to `gbrain apply-migrations --yes`; new opt-in `--index-audit` flag (Postgres-only) reports zero-scan indexes from `pg_stat_user_indexes` (informational only, no auto-drop). v0.15.2: every DB check is wrapped in a progress phase; `markdown_body_completeness` runs under a 1s heartbeat timer so 10+ min scans are observable on 50K-page brains. v0.19.1 added `queue_health` (Postgres-only) with two subchecks: stalled-forever active jobs (started_at > 1h) and waiting-depth-per-name > threshold (default 10, override via `GBRAIN_QUEUE_WAITING_THRESHOLD`). Worker-heartbeat subcheck intentionally deferred to follow-up B7 because it needs a `minion_workers` table to produce ground-truth signal. Fix hints point at `gbrain repair-jsonb`, `gbrain sync --force`, `gbrain apply-migrations`, and `gbrain jobs get/cancel <id>`.
|
||||
- `src/core/migrate.ts` — schema-migration runner. Owns the `MIGRATIONS` array (source of truth for schema DDL). v0.14.2 extended the `Migration` interface with `sqlFor?: { postgres?, pglite? }` (engine-specific SQL overrides `sql`) and `transaction?: boolean` (set to false for `CREATE INDEX CONCURRENTLY`, which Postgres refuses inside a transaction; ignored on PGLite since it has no concurrent writers). Migration v14 (fix wave) uses a handler branching on `engine.kind` to run CONCURRENTLY on Postgres (with a pre-drop of any invalid remnant via `pg_index.indisvalid`) and plain `CREATE INDEX` on PGLite. v15 bumps `minion_jobs.max_stalled` default 1→5 and backfills existing non-terminal rows.
|
||||
- `src/commands/integrity.ts` — `gbrain integrity check|auto|review|extract`: bare-tweet detection, dead-link detection, three-bucket repair (auto-repair / review-queue / skip). `scanIntegrity()` is the shared library function called from `gbrain doctor` (sampled at limit=500) and `cmdCheck` (full scan). v0.22.8: batch-load fast path on Postgres uses `SELECT DISTINCT ON (slug)` in a single SQL query to fix the PgBouncer round-trip timeout (60s → ~6s) while preserving `engine.getAllSlugs()`'s `Set<string>` semantics on multi-source brains. Gated by `engine.kind === 'postgres'` at the call site so PGLite never enters batch; fallback `catch` logs at `GBRAIN_DEBUG=1` so real Postgres errors are diagnosable.
|
||||
- `src/commands/doctor.ts` — `gbrain doctor [--json] [--fast] [--fix] [--dry-run] [--index-audit]`: health checks. v0.12.3 added `jsonb_integrity` + `markdown_body_completeness` reliability checks. v0.14.1: `--fix` delegates inlined cross-cutting rules to `> **Convention:** see [path](path).` callouts (pipes DRY violations into `src/core/dry-fix.ts`); `--fix --dry-run` previews without writing. v0.14.2: `schema_version` check fails loudly when `version=0` (migrations never ran — the #218 `bun install -g` signature) and routes users to `gbrain apply-migrations --yes`; new opt-in `--index-audit` flag (Postgres-only) reports zero-scan indexes from `pg_stat_user_indexes` (informational only, no auto-drop). v0.15.2: every DB check is wrapped in a progress phase; `markdown_body_completeness` runs under a 1s heartbeat timer so 10+ min scans are observable on 50K-page brains. v0.19.1 added `queue_health` (Postgres-only) with two subchecks: stalled-forever active jobs (started_at > 1h) and waiting-depth-per-name > threshold (default 10, override via `GBRAIN_QUEUE_WAITING_THRESHOLD`). Worker-heartbeat subcheck intentionally deferred to follow-up B7 because it needs a `minion_workers` table to produce ground-truth signal. Fix hints point at `gbrain repair-jsonb`, `gbrain sync --force`, `gbrain apply-migrations`, and `gbrain jobs get/cancel <id>`. v0.22.12 (#500): `sync_failures` check shows `[CODE=N, ...]` breakdown for both unacked entries (warn) and acked-historical entries (ok), surfacing systemic failure modes (`SLUG_MISMATCH=2685`) instead of a bare count.
|
||||
- `src/core/migrate.ts` — schema-migration runner. Owns the `MIGRATIONS` array (source of truth for schema DDL). v0.14.2 extended the `Migration` interface with `sqlFor?: { postgres?, pglite? }` (engine-specific SQL overrides `sql`) and `transaction?: boolean` (set to false for `CREATE INDEX CONCURRENTLY`, which Postgres refuses inside a transaction; ignored on PGLite since it has no concurrent writers). Migration v14 (fix wave) uses a handler branching on `engine.kind` to run CONCURRENTLY on Postgres (with a pre-drop of any invalid remnant via `pg_index.indisvalid`) and plain `CREATE INDEX` on PGLite. v15 bumps `minion_jobs.max_stalled` default 1→5 and backfills existing non-terminal rows. v0.22.6.1: migration v24 (`rls_backfill_missing_tables`) uses `sqlFor: { pglite: '' }` to no-op on PGLite — PGLite has no RLS engine and is single-tenant by definition, and the v24 ALTERs target subagent tables that don't exist in pglite-schema.ts. Closes #395 (contributed by @jdcastro2).
|
||||
- `src/core/progress.ts` — Shared bulk-action progress reporter. Writes to stderr. Modes: `auto` (TTY: `\r`-rewriting; non-TTY: plain lines), `human`, `json` (JSONL), `quiet`. Rate-gated by `minIntervalMs` and `minItems`. `startHeartbeat(reporter, note)` helper for single long queries. `child()` composes phase paths. Singleton SIGINT/SIGTERM coordinator emits `abort` events for every live phase. EPIPE defense on both sync throws and stream `'error'` events. Zero dependencies. Introduced in v0.15.2.
|
||||
- `src/core/cli-options.ts` — Global CLI flag parser. `parseGlobalFlags(argv)` returns `{cliOpts, rest}` with `--quiet` / `--progress-json` / `--progress-interval=<ms>` stripped. `getCliOptions()` / `setCliOptions()` expose a module-level singleton so commands reach the resolved flags without parameter threading. `cliOptsToProgressOptions()` maps to reporter options. `childGlobalFlags()` returns the flag suffix to append to `execSync('gbrain ...')` calls in migration orchestrators. `OperationContext.cliOpts` extends shared-op dispatch for MCP callers.
|
||||
- `src/core/cycle.ts` — v0.17 brain maintenance cycle primitive. `runCycle(engine: BrainEngine | null, opts: CycleOpts): Promise<CycleReport>` composes 6 phases in semantically-driven order (lint → backlinks → sync → extract → embed → orphans). Three callers: `gbrain dream` CLI, `gbrain autopilot` daemon's inline path, and the Minions `autopilot-cycle` handler (`src/commands/jobs.ts`). One source of truth for what the brain does overnight. Coordination via `gbrain_cycle_locks` DB table (TTL-based; works through PgBouncer transaction pooling, unlike session-scoped `pg_try_advisory_lock`) + `~/.gbrain/cycle.lock` file lock with PID-liveness for PGLite / engine=null mode. `CycleReport.schema_version: "1"` is the stable agent-consumable shape. `PhaseResult.error: { class, code, message, hint?, docs_url? }` is Stripe-API-tier structured failure info. `yieldBetweenPhases` hook awaited between every phase — Minions handler uses this to renew its job lock and prevent v0.14 stall-death regression. Engine nullable: filesystem phases (lint, backlinks) run without DB; DB phases skip with `status: "skipped", reason: "no_database"`. Lock-skip: read-only phase selections (`--phase orphans`) bypass the cycle lock.
|
||||
- `src/core/db-lock.ts` (v0.22.13) — generic `tryAcquireDbLock(engine, lockId, ttlMinutes)` over the existing `gbrain_cycle_locks` table. Parameterized lock id so different scopes can nest cleanly: `gbrain-cycle` for the broad cycle (held by `cycle.ts`) and `gbrain-sync` (`SYNC_LOCK_ID` constant) for `performSync`'s narrower writer window. Same UPSERT-with-TTL semantics as the prior cycle-only helper, just generalized. Survives PgBouncer transaction pooling (unlike session-scoped `pg_try_advisory_lock`); crashed holders auto-release once their TTL expires.
|
||||
- `src/core/sync-concurrency.ts` (v0.22.13) — single source of truth for the parallel-sync policy. Exports `autoConcurrency(engine, fileCount, override?)` (PGLite always serial; explicit override clamped to >=1; auto path returns `DEFAULT_PARALLEL_WORKERS=4` when `fileCount > AUTO_CONCURRENCY_FILE_THRESHOLD=100`), `shouldRunParallel(workers, fileCount, explicit)` (Q1: explicit `--workers` bypasses the >50-file floor), and `parseWorkers(s)` (rejects `'0'`, `'-3'`, `'foo'`, `'1.5'`, trailing chars — replaces the prior parseInt-with-no-validation in both `sync.ts` and `import.ts`). Used by `performSync`, `performFullSync`, `runImport`, and the Minion `sync` handler so the three sites can no longer drift.
|
||||
- `src/commands/sync.ts` — `gbrain sync` CLI + the `performSync` / `performFullSync` library entrypoints (consumed by the autopilot cycle and the Minion sync handler). v0.22.13 (PR #490): `performSync` wraps its body in a `gbrain-sync` writer lock so two concurrent syncs (manual + autopilot, two terminals, two Conductor workspaces) cannot both write `last_commit` and let the last writer win. Head-drift gate after the import phase re-checks `git rev-parse HEAD`; if HEAD moved (someone ran `git checkout` / `git pull` mid-sync), the bookmark refuses to advance. Vanished files now record a failedFiles entry instead of silent-skip — the silent-skip-then-advance pathology that survived prior hardening passes is dead. Worker engines wrap in try/finally so disconnect always fires (panic-path leak fix). Both PGLite-detection sites use `engine.kind === 'pglite'`. CLI accepts `--workers N` (alias `--concurrency N`), validated via `parseWorkers`. Explicit `--workers` bypasses the auto-path file-count floor; auto path defers to `autoConcurrency()`. Banner moved to stderr.
|
||||
- `src/core/cycle.ts` — v0.17 brain maintenance cycle primitive. `runCycle(engine: BrainEngine | null, opts: CycleOpts): Promise<CycleReport>` composes 6 phases in semantically-driven order (lint → backlinks → sync → extract → embed → orphans). Three callers: `gbrain dream` CLI, `gbrain autopilot` daemon's inline path, and the Minions `autopilot-cycle` handler (`src/commands/jobs.ts`). One source of truth for what the brain does overnight. Coordination via `gbrain_cycle_locks` DB table (TTL-based; works through PgBouncer transaction pooling, unlike session-scoped `pg_try_advisory_lock`) + `~/.gbrain/cycle.lock` file lock with PID-liveness for PGLite / engine=null mode. `CycleReport.schema_version: "1"` is the stable agent-consumable shape. `PhaseResult.error: { class, code, message, hint?, docs_url? }` is Stripe-API-tier structured failure info. `yieldBetweenPhases` hook awaited between every phase — Minions handler uses this to renew its job lock and prevent v0.14 stall-death regression. Engine nullable: filesystem phases (lint, backlinks) run without DB; DB phases skip with `status: "skipped", reason: "no_database"`. Lock-skip: read-only phase selections (`--phase orphans`) bypass the cycle lock. v0.22.1 (#403): `CycleOpts.signal?: AbortSignal` propagates the worker's abort signal; `checkAborted()` fires between every phase and throws if the signal is aborted (cooperative — can't interrupt a phase mid-execution). v0.22.1 (#417): `runPhaseSync` returns `pagesAffected` via `SyncPhaseResult`; `runCycle` captures it and threads to `runPhaseExtract` as the 4th arg, enabling incremental extract on the cycle path. v0.22.1 (Codex F2): `runPhaseSync` takes `willRunExtractPhase: boolean` and sets `noExtract: phases.includes('extract')` so `gbrain dream --phase sync` doesn't silently lose extraction. v0.22.5 (#475): new `resolveSourceForDir(engine, brainDir)` helper queries `SELECT id FROM sources WHERE local_path = $1 LIMIT 1`; `runPhaseSync` threads result as `sourceId` to `performSync()` so sync reads the per-source `sources.last_commit` anchor instead of the drift-prone global `config.sync.last_commit` key. Bare try/catch lets pre-v0.18 brains fall through to the global key. Closes the prod hang where every autopilot cycle ran a 30-min full reimport because the global anchor commit had been GC'd from git history.
|
||||
- `src/commands/dream.ts` — v0.17 `gbrain dream` CLI. ~80-line thin alias over `runCycle`. brainDir resolution requires explicit `--dir` OR `sync.repo_path` config (no more walk-up-cwd-for-.git footgun). Flags: `--dry-run`, `--json`, `--phase <name>`, `--pull`, `--dir <path>`. Exit code 1 on status=failed (partial/warn not fatal — don't page on warnings).
|
||||
- `scripts/check-progress-to-stdout.sh` — CI guard against regressing to `\r`-on-stdout progress. Wired into `bun run test` via `scripts/check-progress-to-stdout.sh && bun test` in package.json.
|
||||
- `docs/progress-events.md` — Canonical JSON event schema reference. Stable from v0.15.2, additive only.
|
||||
@@ -284,6 +302,10 @@ Key commands added in v0.14.3 (fix wave):
|
||||
- `gbrain jobs submit` gains `--max-stalled`, `--backoff-type`, `--backoff-delay`, `--backoff-jitter`, `--timeout-ms`, `--idempotency-key` — exposing existing `MinionJobInput` fields as first-class CLI flags.
|
||||
- `gbrain jobs smoke --sigkill-rescue` — opt-in regression smoke case simulating a killed worker; asserts the v0.14.3 schema default (`max_stalled=5`) actually rescues on first stall.
|
||||
|
||||
Key commands added in v0.22.13 (PR #490):
|
||||
- `gbrain sync --workers N` (alias `--concurrency N`) — parallelize the import phase using per-worker Postgres engines (small pool of 2 each) with an atomic queue index. Auto-concurrency: defaults to 4 workers when the diff exceeds 100 files. Smaller diffs stay serial. Explicit `--workers` always wins (even on a 30-file diff). PGLite forces serial regardless. Validation rejects `0`, negatives, non-integers loud (replaces the prior silent fall-through to auto-concurrency).
|
||||
- `gbrain import --workers N` — same `parseWorkers()` validation as sync; same try/finally worker-engine cleanup. Behavior surface unchanged.
|
||||
|
||||
## Testing
|
||||
|
||||
`bun test` runs all tests. After the v0.12.1 release: ~75 unit test files + 8 E2E test files (1412 unit pass, 119 E2E when `DATABASE_URL` is set — skip gracefully otherwise). Unit tests run
|
||||
@@ -295,7 +317,9 @@ parity), `test/cli.test.ts` (CLI structure), `test/config.test.ts` (config redac
|
||||
`test/files.test.ts` (MIME/hash), `test/import-file.test.ts` (import pipeline),
|
||||
`test/upgrade.test.ts` (schema migrations),
|
||||
`test/file-migration.test.ts` (file migration), `test/file-resolver.test.ts` (file resolution),
|
||||
`test/import-resume.test.ts` (import checkpoints), `test/migrate.test.ts` (migration; v8/v9 helper-btree-index SQL structural assertions + 1000-row wall-clock fixtures that guard the O(n²)→O(n log n) fix + v0.13.1 assertions on v12/v13 SQL shape, `sqlFor` + `transaction:false` runner semantics, and the `max_stalled DEFAULT 1` regression guard),
|
||||
`test/import-resume.test.ts` (import checkpoints), `test/migrate.test.ts` (migration; v8/v9 helper-btree-index SQL structural assertions + 1000-row wall-clock fixtures that guard the O(n²)→O(n log n) fix + v0.13.1 assertions on v12/v13 SQL shape, `sqlFor` + `transaction:false` runner semantics, the `max_stalled DEFAULT 1` regression guard, and v0.22.6.1 v24 `sqlFor.pglite: ''` no-op assertion),
|
||||
`test/bootstrap.test.ts` (v0.22.6.1 — bootstrap contract: no-op on fresh install, idempotent across two `initSchema()` calls, no-op on modern brain that already has every probed column, full bootstrap path on simulated pre-v0.18 brain, fresh-install regression guard, pre-v0.13 `links` shape coverage),
|
||||
`test/schema-bootstrap-coverage.test.ts` (v0.22.6.1 CI guard — `REQUIRED_BOOTSTRAP_COVERAGE` lists every forward reference in PGLITE_SCHEMA_SQL; the test fails loudly if `applyForwardReferenceBootstrap` skips one. When you add a column-with-index to the embedded schema blob, you extend both arrays or this guard fails. The pattern that broke gbrain ten times in two years is now structurally prevented.),
|
||||
`test/setup-branching.test.ts` (setup flow), `test/slug-validation.test.ts` (slug validation),
|
||||
`test/storage.test.ts` (storage backends), `test/supabase-admin.test.ts` (Supabase admin),
|
||||
`test/yaml-lite.test.ts` (YAML parsing), `test/check-update.test.ts` (version check + update CLI),
|
||||
@@ -309,6 +333,7 @@ parity), `test/cli.test.ts` (CLI structure), `test/config.test.ts` (config redac
|
||||
`test/skills-conformance.test.ts` (skill frontmatter + required sections validation),
|
||||
`test/resolver.test.ts` (RESOLVER.md coverage, routing validation + v0.20.4 round-trip: every quoted RESOLVER.md trigger must match a frontmatter `triggers:` entry in the target skill, and every `name="<word>"` reference in any SKILL.md must resolve to a declared op in `src/core/operations.ts` or a Minions handler in `PROTECTED_JOB_NAMES`),
|
||||
`test/search.test.ts` (RRF normalization, compiled truth boost, cosine similarity, dedup key),
|
||||
`test/sql-ranking.test.ts` (v0.22.0 source-boost helpers: 39 cases covering longest-prefix-match in SQL CASE, detail=high temporal-bypass, three-meta-char LIKE escape (%, _, \\), single-quote SQL-literal doubling, env override parsing for GBRAIN_SOURCE_BOOST + GBRAIN_SEARCH_EXCLUDE, resolveBoostMap / resolveHardExcludes merge semantics),
|
||||
`test/dedup.test.ts` (source-aware dedup, compiled truth guarantee, layer interactions),
|
||||
`test/intent.test.ts` (query intent classification: entity/temporal/event/general),
|
||||
`test/eval.test.ts` (retrieval metrics: precisionAtK, recallAtK, mrr, ndcgAtK, parseQrels),
|
||||
@@ -336,6 +361,9 @@ parity), `test/cli.test.ts` (CLI structure), `test/config.test.ts` (config redac
|
||||
`test/orphans.test.ts` (v0.12.3 orphans command: detection, pseudo filtering, text/json/count outputs, MCP op),
|
||||
`test/postgres-engine.test.ts` (v0.12.3 statement_timeout scoping: `sql.begin` + `SET LOCAL` shape, source-level grep guardrail against reintroduced bare `SET statement_timeout`),
|
||||
`test/sync.test.ts` (sync logic + v0.12.3 regression guard asserting top-level `engine.transaction` is not called),
|
||||
`test/sync-concurrency.test.ts` (v0.22.13 PR #490: 17 cases covering `autoConcurrency()` thresholds + PGLite-forces-serial + explicit-override clamping, `shouldRunParallel()` Q1 explicit-bypasses-floor contract, and `parseWorkers()` validation that rejects `'0'`/`'-3'`/`'foo'`/`'1.5'`/trailing chars),
|
||||
`test/sync-parallel.test.ts` (v0.22.13 PR #490: PGLite-routed coverage of the bookmark gate under concurrency request, head-drift gate, vanished-file failure capture, PGLite-stays-serial, and the `gbrain-sync` writer-lock contract — 7 cases),
|
||||
`test/sync-failures.test.ts` (v0.22.12: 28 cases pinning `classifyErrorCode` regex coverage for all 12 codes against literal production message strings from `markdown.ts:159-244` and `import-file.ts:199, 347, 352, 401`; `summarizeFailuresByCode` sort + pre-classified-honor; `recordSyncFailures` code-field persistence; `acknowledgeSyncFailures` AcknowledgeResult shape + backfill on pre-v0.22.12 entries),
|
||||
`test/doctor.test.ts` (doctor command + v0.12.3 assertions that `jsonb_integrity` scans the four v0.12.0 write sites and `markdown_body_completeness` is present),
|
||||
`test/utils.test.ts` (shared SQL utilities + `tryParseEmbedding` null-return and single-warn semantics),
|
||||
`test/build-llms.test.ts` (llms.txt/llms-full.txt generator: path resolution, idempotence, spec shape, regen-drift guard, content contract, AGENTS.md install-path mirror, size-budget enforcement — 7 cases),
|
||||
@@ -346,17 +374,26 @@ parity), `test/cli.test.ts` (CLI structure), `test/config.test.ts` (config redac
|
||||
`test/skill-manifest.test.ts` (v0.19 skill manifest parser: drift detection, managed-block markers),
|
||||
`test/skillify-scaffold.test.ts` (v0.19 `gbrain skillify scaffold` stubs: SKILL.md, script, tests, routing-eval fixtures),
|
||||
`test/skillpack-install.test.ts` (v0.19 `gbrain skillpack install` managed-block install / update / no-clobber semantics),
|
||||
`test/skillpack-sync-guard.test.ts` (v0.19 sync-guard: bundled skills stay byte-identical to `skills/` source).
|
||||
`test/skillpack-sync-guard.test.ts` (v0.19 sync-guard: bundled skills stay byte-identical to `skills/` source),
|
||||
`test/http-transport.test.ts` (v0.22.7 HTTP transport: 23 unit cases covering bearer auth + missing/no-Bearer/unknown/revoked + `/health` bypass, F1+F2 round-trip via dispatch.ts, F3 invalid_params, application/json response shape (not SSE), CORS default-deny + allowlist, body cap on Content-Length AND chunked, two-bucket rate limit (refill, exhaust+Retry-After, LRU eviction, TTL prune, pre-auth IP fires before DB), and `mcp_request_log` audit on success + auth_failed).
|
||||
|
||||
E2E tests (`test/e2e/`): Run against real Postgres+pgvector. Require `DATABASE_URL`.
|
||||
- `bun run test:e2e` runs Tier 1 (mechanical, all operations, no API keys). Includes 9 dedicated cases for the postgres-engine `addLinksBatch` / `addTimelineEntriesBatch` bind path — postgres-js's `unnest()` binding is structurally different from PGLite's and gets its own coverage.
|
||||
- `test/e2e/search-quality.test.ts` runs search quality E2E against PGLite (no API keys, in-memory)
|
||||
- `test/e2e/graph-quality.test.ts` runs the v0.10.3 knowledge graph pipeline (auto-link via put_page, reconciliation, traversePaths) against PGLite in-memory
|
||||
- `test/e2e/postgres-jsonb.test.ts` — v0.12.2 regression test. Round-trips all 5 JSONB write sites (pages.frontmatter, raw_data.data, ingest_log.pages_updated, files.metadata, page_versions.frontmatter) against real Postgres and asserts `jsonb_typeof='object'` plus `->>'key'` returns the expected scalar. The test that should have caught the original double-encode bug.
|
||||
- `test/e2e/integrity-batch.test.ts` (v0.22.8) — parity tests for `scanIntegrity`'s batch-load fast path vs sequential. Four cases (dedup, hits, validate, topPages) seed a fixture and assert both paths return identical results. Dedup case uses raw SQL via `getConn().unsafe()` to seed a `(test-source-2, people/alice)` row alongside the default-source row, since `engine.putPage` doesn't take a `source_id`. Pins the codex-caught multi-source overcounting regression.
|
||||
- `test/e2e/jsonb-roundtrip.test.ts` — v0.12.3 companion regression against the 4 doctor-scanned JSONB sites. Assertion-level overlap with `postgres-jsonb.test.ts` is intentional defense-in-depth: if doctor's scan surface ever drifts from the actual write surface, one of these tests catches it.
|
||||
- `test/e2e/sync.test.ts` (v0.22.12 — `--skip-failed` failure-loop test, alongside the existing 13 happy-path tests): exercises the full chain — broken file → `performSync` returns `blocked_by_failures` with grouped breakdown → `performSync({skipFailed: true})` advances bookmark and returns `AcknowledgeResult` with code summary → second broken file → second cycle. Saves and restores the user's real `~/.gbrain/sync-failures.jsonl` so the test is hermetic on a developer machine. Asserts bookmark gating, JSONL state, dedup across paths, summary aggregation, and the literal doctor-rendering string format. This is the integration test that proves the v0.22.12 chain holds together — unit tests cover the pure functions in isolation, this covers the integration.
|
||||
- `test/e2e/upgrade.test.ts` runs check-update E2E against real GitHub API (network required)
|
||||
- `test/e2e/minions-shell-pglite.test.ts` (v0.20.4) exercises the PGLite `--follow` inline shell-job path (in-memory, no `DATABASE_URL` required) — the path the consolidated minion-orchestrator skill documents for dev use
|
||||
- `test/e2e/openclaw-reference-compat.test.ts` (v0.19) — exercises `check-resolvable` + `skillpack install` against a minimal AGENTS.md workspace fixture (`test/fixtures/openclaw-reference-minimal/`), regression guard for the 107-skill OpenClaw deployment shape
|
||||
- `test/e2e/search-swamp.test.ts` (v0.22.0) — reproduces the headline source-swamp case. Seeds a curated `originals/talks/article-outline-fat-code` page against two `wintermute/chat/` pages stuffed with the same multi-word phrase. Asserts the article wins keyword AND vector ranking, that `detail=high` lets the chat swamp re-surface (temporal-query workflow preserved), and that `source_id` passes through the two-stage CTE intact. PGLite in-memory.
|
||||
- `test/e2e/search-exclude.test.ts` (v0.22.0) — verifies `test/` + `archive/` pages are hidden by default, that `include_slug_prefixes` opts back in, and that caller-supplied `exclude_slug_prefixes` adds to defaults. Both keyword and vector search paths covered.
|
||||
- `test/e2e/engine-parity.test.ts` (v0.22.0) — Postgres ↔ PGLite top-result and result-set parity for `searchKeyword` + `searchVector`. Codex flagged that Postgres ranks pages then picks best chunk while PGLite returns chunks directly — without parity coverage the source-boost fix could pass on PGLite and fail on Postgres. Skips gracefully when `DATABASE_URL` is unset.
|
||||
- `test/e2e/postgres-bootstrap.test.ts` (v0.22.6.1) — exercises `PostgresEngine.initSchema()` directly against a fresh real Postgres database. Asserts the bootstrap path is no-op on fresh installs and that SCHEMA_SQL replays cleanly through the engine path (not via the standalone `db.initSchema` from `src/core/db.ts`, which would have produced false-positive coverage). Codex caught the E2E-shape gap during plan review.
|
||||
- `test/e2e/http-transport.test.ts` (v0.22.7) — 8 cases against real Postgres covering `gbrain serve --http` end-to-end: bearer auth round-trip, `last_used_at` SQL-level debounce semantics, `mcp_request_log` row insertion on success and auth_failed paths, `/health` DB-down → 503 (DB-probing health check), and the F1+F2+F3 dispatch round-trip with a real operation. Skips gracefully when `DATABASE_URL` is unset.
|
||||
- `test/e2e/sync-parallel.test.ts` (v0.22.13 PR #490) — DATABASE_URL-gated. T2: 60-file Postgres sync at concurrency=4 imports all + no connection leak (probes `pg_stat_activity` before/after to confirm worker engines disconnected). P4: 120-file serial-vs-parallel benchmark prints `SYNC_PARALLEL_BENCH N files | serial=Xms | parallel(4)=Yms | speedup=Zx` for CHANGELOG quoting. Asserts parallel ≤ serial × 1.5 (CI-noise tolerant; not a strict speedup gate).
|
||||
- Tier 2 (`skills.test.ts`) requires OpenClaw + API keys, runs nightly in CI
|
||||
- If `.env.testing` doesn't exist in this directory, check sibling worktrees for one:
|
||||
`find ../ -maxdepth 2 -name .env.testing -print -quit` and copy it here if found.
|
||||
@@ -471,6 +508,59 @@ in bulk paths, the CI guard will fail the build.
|
||||
|
||||
`bun build --compile --outfile bin/gbrain src/cli.ts`
|
||||
|
||||
## Version locations (single source of truth: `VERSION` file)
|
||||
|
||||
Every release advances the version in **five files at once**. Keep these in
|
||||
sync. `/ship` enforces this via Step 12's idempotency check (VERSION vs
|
||||
package.json drift), but the canonical list lives here so future runs and
|
||||
the auto-update agent know where to look.
|
||||
|
||||
**Required (every release must update all five):**
|
||||
|
||||
| File | What lives there | Format |
|
||||
|---|---|---|
|
||||
| `VERSION` | The single source of truth. Read first by `/ship`, the binary, and CI version-gate. | Bare 4-digit string `MAJOR.MINOR.PATCH.MICRO` (e.g. `0.22.1`), no leading `v`, no trailing newline-sensitivity issues. |
|
||||
| `package.json` | Bun/npm package version. `gbrain --version` reads it via the compiled binary's bundled package metadata. CI version-gate cross-checks this against `VERSION` and fails if they drift. | `"version": "0.22.1"` |
|
||||
| `CHANGELOG.md` | Top entry header `## [0.22.1] - YYYY-MM-DD` plus the "To take advantage of v0.22.1" block. | Standard Keep-a-Changelog header. |
|
||||
| `TODOS.md` | Any TODO entries that mention "follow-up from vX.Y.Z" use the version of the release that filed them. Update only when filing NEW follow-up TODOs. | Inline `vX.Y.Z` references in TODO bodies. |
|
||||
| `CLAUDE.md` | The Key Files section's per-file annotations carry `vX.Y.Z (#NNN)` tags noting which release introduced a behavior. Update whenever a wave's annotations get folded in. | Inline `vX.Y.Z (#NNN, contributed by @user)` references. |
|
||||
|
||||
**Auto-derived (no manual edit; refreshed by their own commands):**
|
||||
|
||||
- `bun.lock` — root-package version is auto-pinned from `package.json`. After
|
||||
bumping `package.json`, run `bun install` to refresh the lockfile.
|
||||
- `llms-full.txt` / `llms.txt` — auto-generated documentation bundles. After
|
||||
any release ship that touches the Key Files annotations in `CLAUDE.md`,
|
||||
run `bun run build:llms` to regenerate. The bundles do not contain a
|
||||
version pin per se; they reflect the current state of the docs they index.
|
||||
|
||||
**Historical (DO NOT bump on release):**
|
||||
|
||||
- `skills/migrations/v0.21.0.md` — migration files use the version they
|
||||
shipped FROM as their filename. v0.21.0's migration always says v0.21.0.
|
||||
- `src/commands/migrations/v0_21_0.ts` — same: migration code references
|
||||
the schema version it migrates to.
|
||||
- `test/migrations-v0_21_0.test.ts`, `test/migration-orchestrator-v0_21_0.test.ts`,
|
||||
`test/migrate.test.ts` — migration tests reference historical migration
|
||||
versions; these are correct as-is and should not move.
|
||||
- `src/core/db.ts`, `src/core/migrate.ts`, `src/core/import-file.ts`,
|
||||
`src/commands/reindex-code.ts` — code comments cite the release that
|
||||
introduced a feature. Once written, these are historical record.
|
||||
- `README.md` — references the latest published feature names by version
|
||||
(e.g. "v0.21.0 Code Cathedral"); update only when the README's marketing
|
||||
copy is intentionally being refreshed, NOT on every micro/patch bump.
|
||||
|
||||
**The /ship workflow's version idempotency check:** Step 12 reads
|
||||
`VERSION` and `package.json`, classifies as FRESH / ALREADY_BUMPED /
|
||||
DRIFT_STALE_PKG / DRIFT_UNEXPECTED, and refuses to proceed on
|
||||
DRIFT_UNEXPECTED. This is why the two must move together.
|
||||
|
||||
**The CI version-gate** rejects pushes where `VERSION` and
|
||||
`package.json` disagree, OR where `VERSION` is not strictly greater
|
||||
than master's VERSION. If a queue collision claims your version on
|
||||
master before yours lands, /ship's queue-aware allocator (Step 12)
|
||||
will detect drift and re-bump on the next run.
|
||||
|
||||
## Pre-ship requirements
|
||||
|
||||
Before shipping (/ship) or reviewing (/review), always run the full test suite:
|
||||
@@ -1106,13 +1196,15 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
|
||||
|
||||
| Trigger | Skill |
|
||||
|---------|-------|
|
||||
| "What do we know about", "tell me about", "search for" | `skills/query/SKILL.md` |
|
||||
| "What do we know about", "tell me about", "search for", "who is", "background on", "notes on" | `skills/query/SKILL.md` |
|
||||
| "Who knows who", "relationship between", "connections", "graph query" | `skills/query/SKILL.md` (use graph-query) |
|
||||
| Creating/enriching a person or company page | `skills/enrich/SKILL.md` |
|
||||
| Where does a new file go? Filing rules | `skills/repo-architecture/SKILL.md` |
|
||||
| Fix broken citations in brain pages | `skills/citation-fixer/SKILL.md` |
|
||||
| "citation audit", "check citations", "fix citations" | `skills/citation-fixer/SKILL.md` (focused fix). For broader brain health, chain into `skills/maintain/SKILL.md` |
|
||||
| "Research", "track", "extract from email", "investor updates", "donations" | `skills/data-research/SKILL.md` |
|
||||
| Share a brain page as a link | `skills/publish/SKILL.md` |
|
||||
| "validate frontmatter", "check frontmatter", "fix frontmatter", "frontmatter audit", "brain lint" | `skills/frontmatter-guard/SKILL.md` |
|
||||
|
||||
## Content & media ingestion
|
||||
|
||||
@@ -1282,12 +1374,29 @@ Add to `~/.claude/server.json` (Claude Code), Settings > MCP Servers (Cursor), o
|
||||
### Remote MCP (Claude Desktop, Cowork, Perplexity)
|
||||
|
||||
```bash
|
||||
ngrok http 8787 --url your-brain.ngrok.app
|
||||
bun run src/commands/auth.ts create "claude-desktop"
|
||||
gbrain auth create "claude-desktop" # tokens via the existing CLI
|
||||
gbrain serve --http --port 8787 # built-in HTTP transport (Postgres-only)
|
||||
ngrok http 8787 --url your-brain.ngrok.app # any tunnel works
|
||||
claude mcp add gbrain -t http https://your-brain.ngrok.app/mcp -H "Authorization: Bearer TOKEN"
|
||||
```
|
||||
|
||||
Per-client guides: [`docs/mcp/`](docs/mcp/DEPLOY.md). ChatGPT requires OAuth 2.1 (not yet implemented).
|
||||
Per-client guides: [`docs/mcp/`](docs/mcp/DEPLOY.md). Hardening defaults, env vars, and threat model: [SECURITY.md](SECURITY.md). ChatGPT requires OAuth 2.1 (not yet implemented).
|
||||
|
||||
### Using gbrain with GStack
|
||||
|
||||
If your engineering agent runs on [GStack](https://github.com/garrytan/gstack), point it at gbrain for code lookup instead of grep+read. Cathedral II (v0.21.0) ships call-graph edges and two-pass retrieval — `/investigate`, `/review`, `/plan-eng-review`, and `/office-hours` all benefit when the agent walks the symbol graph instead of scanning files line by line.
|
||||
|
||||
The five magical-moment commands:
|
||||
|
||||
```bash
|
||||
gbrain code-callers searchKeyword # who calls this symbol?
|
||||
gbrain code-callees searchKeyword # what does this symbol call?
|
||||
gbrain code-def BrainEngine # where is X defined?
|
||||
gbrain code-refs BrainEngine # all reference sites
|
||||
gbrain query "how does N+1 handling work" --near-symbol BrainEngine.searchKeyword --walk-depth 2
|
||||
```
|
||||
|
||||
All five auto-emit JSON on non-TTY (gh-CLI convention) so a GStack subagent shelling out via bash gets a clean parseable response. Run `gbrain sources add <repo> --strategy code` to index a repo, then your agent's brain-first lookup covers code, not just markdown. ([Cathedral II release notes](CHANGELOG.md#0210---2026-04-25))
|
||||
|
||||
## The 29 Skills
|
||||
|
||||
@@ -1545,6 +1654,30 @@ accumulate rows across separate single-skill installs instead of overwriting eac
|
||||
Read [`skills/skillify/SKILL.md`](skills/skillify/SKILL.md) for the full 10-item checklist
|
||||
and the anti-patterns it catches.
|
||||
|
||||
## Storage tiering: keep bulk content out of git (v0.22.11)
|
||||
|
||||
When your brain crosses 100K files and bulk machine-generated content (tweets, articles, transcripts)
|
||||
becomes the size driver, declare which directories belong in git and which live in the database only.
|
||||
|
||||
```yaml
|
||||
# gbrain.yml at the brain repo root
|
||||
storage:
|
||||
db_tracked:
|
||||
- people/
|
||||
- companies/
|
||||
- deals/
|
||||
db_only:
|
||||
- media/x/
|
||||
- media/articles/
|
||||
- meetings/transcripts/
|
||||
```
|
||||
|
||||
`gbrain sync` auto-manages your `.gitignore` for `db_only` paths. `gbrain export --restore-only --repo .`
|
||||
repopulates missing files from the database (container restart, fresh clone, accidental rm).
|
||||
`gbrain storage status` shows the tier breakdown.
|
||||
|
||||
Full guide: [docs/storage-tiering.md](docs/storage-tiering.md).
|
||||
|
||||
## Getting Data In
|
||||
|
||||
GBrain ships integration recipes that your agent sets up for you. Each recipe tells the agent what credentials to ask for, how to validate, and what cron to register.
|
||||
@@ -1691,6 +1824,8 @@ Question
|
||||
│ ├─ Multi-query expansion (Haiku rephrases the question 3 ways)
|
||||
│ ├─ Vector search (HNSW cosine over OpenAI embeddings)
|
||||
│ ├─ Keyword search (Postgres tsvector + websearch_to_tsquery)
|
||||
│ ├─ Source-aware ranking (curated dirs outrank chat/daily swamp at SQL layer)
|
||||
│ ├─ Hard-exclude (test/ archive/ attachments/ .raw/ filtered before retrieval)
|
||||
│ ├─ Reciprocal Rank Fusion (score = sum 1/(60+rank) across both)
|
||||
│ ├─ Cosine re-scoring (re-rank chunks against actual query embedding)
|
||||
│ ├─ Compiled-truth boost (assessments outrank timeline noise)
|
||||
@@ -1798,8 +1933,11 @@ SEARCH
|
||||
gbrain query <question> Hybrid search (vector + keyword + RRF)
|
||||
|
||||
IMPORT
|
||||
gbrain import <dir> [--no-embed] Import markdown (idempotent)
|
||||
gbrain sync [--repo <path>] Git-to-brain incremental sync
|
||||
gbrain import <dir> [--no-embed] [--workers N]
|
||||
Import markdown (idempotent)
|
||||
gbrain sync [--repo <path>] [--workers N]
|
||||
Git-to-brain incremental sync
|
||||
(>100-file diffs auto-parallelize 4 workers on Postgres)
|
||||
gbrain export [--dir ./out/] Export to markdown
|
||||
|
||||
FILES
|
||||
@@ -1841,6 +1979,8 @@ ADMIN
|
||||
gbrain doctor --locks List idle-in-tx backends (57014 diagnostic, Postgres only)
|
||||
gbrain stats Brain statistics
|
||||
gbrain serve MCP server (stdio)
|
||||
gbrain serve --http --port 8787 MCP server (HTTP, Postgres-only, bearer auth)
|
||||
gbrain auth create|list|revoke|test Token management for the HTTP transport
|
||||
gbrain integrations Integration recipe dashboard
|
||||
gbrain sources list|add|remove|... Multi-source brain management (v0.18)
|
||||
gbrain dream [--dry-run] [--phase N] One maintenance cycle then exit (cron-friendly)
|
||||
@@ -4021,9 +4161,14 @@ Source: https://raw.githubusercontent.com/garrytan/gbrain/master/docs/mcp/DEPLOY
|
||||
|
||||
# Deploy GBrain Remote MCP Server
|
||||
|
||||
> **v0.22.7+:** Use `gbrain serve --http` for remote access. It includes built-in
|
||||
> bearer token auth, default-deny CORS, two-bucket rate limiting, body cap, and
|
||||
> per-request audit log. **Postgres-only** (PGLite is local-only by design).
|
||||
> See [SECURITY.md](../../SECURITY.md) for env vars and tunable defaults.
|
||||
|
||||
Access your brain from any device, any AI client. GBrain's MCP server runs locally
|
||||
via `gbrain serve` (stdio). For remote access, wrap it in an HTTP server behind a
|
||||
public tunnel.
|
||||
via `gbrain serve` (stdio). For remote access, expose it via the built-in HTTP
|
||||
transport behind a public tunnel.
|
||||
|
||||
## Two Paths
|
||||
|
||||
@@ -4034,21 +4179,23 @@ gbrain serve
|
||||
```
|
||||
|
||||
Works with Claude Code, Cursor, Windsurf, and any MCP client that supports stdio.
|
||||
No server, no tunnel, no token needed.
|
||||
No server, no tunnel, no token needed. Works on both PGLite and Postgres engines.
|
||||
|
||||
### Remote (any device, any AI client)
|
||||
### Remote (any device, any AI client) — Postgres only
|
||||
|
||||
```
|
||||
Your AI client (Claude Desktop, Perplexity, etc.)
|
||||
→ ngrok tunnel (https://YOUR-DOMAIN.ngrok.app)
|
||||
→ Your HTTP server (wraps gbrain serve)
|
||||
→ Supabase Postgres (via pooler connection string)
|
||||
→ gbrain serve --http (built-in transport with bearer auth)
|
||||
→ Postgres (pooler connection or self-hosted)
|
||||
```
|
||||
|
||||
This requires:
|
||||
1. A machine running `gbrain serve` behind an HTTP wrapper
|
||||
2. A public tunnel (ngrok, Tailscale, or cloud host)
|
||||
3. Bearer token auth for security
|
||||
1. A Postgres-backed brain (the `access_tokens` table only exists on Postgres;
|
||||
running `gbrain serve --http` against a PGLite install fails fast at startup)
|
||||
2. A machine running `gbrain serve --http`
|
||||
3. A public tunnel (ngrok, Tailscale, or cloud host)
|
||||
4. A bearer token created via `gbrain auth create <name>`
|
||||
|
||||
## Remote Setup
|
||||
|
||||
@@ -4067,13 +4214,13 @@ ngrok http 8787 --url your-brain.ngrok.app # Hobby tier for fixed domain
|
||||
|
||||
```bash
|
||||
# Create a token for each client
|
||||
bun run src/commands/auth.ts create "claude-desktop"
|
||||
gbrain auth create "claude-desktop"
|
||||
|
||||
# List all tokens
|
||||
bun run src/commands/auth.ts list
|
||||
gbrain auth list
|
||||
|
||||
# Revoke a token
|
||||
bun run src/commands/auth.ts revoke "claude-desktop"
|
||||
gbrain auth revoke "claude-desktop"
|
||||
```
|
||||
|
||||
Tokens are per-client. Create one for each device/app. Revoke individually
|
||||
@@ -4089,7 +4236,7 @@ if compromised. Tokens are stored SHA-256 hashed in your database.
|
||||
### 4. Verify
|
||||
|
||||
```bash
|
||||
bun run src/commands/auth.ts test \
|
||||
gbrain auth test \
|
||||
https://YOUR-DOMAIN.ngrok.app/mcp \
|
||||
--token YOUR_TOKEN
|
||||
```
|
||||
@@ -4117,7 +4264,7 @@ Funnel, and cloud hosts (Fly.io, Railway).
|
||||
Include the Authorization header: `Authorization: Bearer YOUR_TOKEN`
|
||||
|
||||
**"invalid_token" error**
|
||||
Run `bun run src/commands/auth.ts list` to see active tokens.
|
||||
Run `gbrain auth list` to see active tokens.
|
||||
|
||||
**"service_unavailable" error**
|
||||
Database connection failed. Check your Supabase dashboard for outages.
|
||||
@@ -5151,6 +5298,75 @@ in depth, not the primary boundary.
|
||||
|
||||
---
|
||||
|
||||
## v0.22.4 — frontmatter-guard adoption
|
||||
|
||||
### 1. Stop hand-rolling frontmatter validators
|
||||
|
||||
If your fork has scripts that call `js-yaml` directly to validate brain page
|
||||
frontmatter, replace them with `gbrain frontmatter validate` calls. The CLI
|
||||
covers the seven canonical error classes and ships a `--json` envelope that's
|
||||
stable across releases.
|
||||
|
||||
```diff
|
||||
- # Custom validator script
|
||||
- node scripts/validate-frontmatter.mjs <path>
|
||||
+ gbrain frontmatter validate <path> --json
|
||||
```
|
||||
|
||||
For consumers that need the validator inside another script, import from
|
||||
gbrain's `markdown` export instead of duplicating logic:
|
||||
|
||||
```ts
|
||||
import { parseMarkdown } from 'gbrain/markdown';
|
||||
|
||||
const parsed = parseMarkdown(content, filePath, { validate: true, expectedSlug });
|
||||
for (const err of parsed.errors ?? []) {
|
||||
// err.code: MISSING_OPEN | MISSING_CLOSE | YAML_PARSE | SLUG_MISMATCH |
|
||||
// NULL_BYTES | NESTED_QUOTES | EMPTY_FRONTMATTER
|
||||
}
|
||||
```
|
||||
|
||||
### 2. Drop any references to `lib/brain-writer.mjs`
|
||||
|
||||
If your fork's skills or scripts referenced an aspirational
|
||||
`lib/brain-writer.mjs` (it never shipped — the spec was in PR #392 and never
|
||||
landed), replace those references with the gbrain CLI. The `frontmatter-guard`
|
||||
skill lives at `skills/frontmatter-guard/SKILL.md` and points at
|
||||
`gbrain frontmatter validate` / `audit` / `install-hook`.
|
||||
|
||||
### 3. Wire the doctor subcheck into your health pipeline
|
||||
|
||||
`gbrain doctor` now reports `frontmatter_integrity` automatically. If your
|
||||
fork has a custom health pipeline (e.g. a daily Slack post about brain
|
||||
health), pull from `gbrain doctor --json` and surface the
|
||||
`frontmatter_integrity` row counts.
|
||||
|
||||
### 4. (Optional) Install the pre-commit hook on brain repos
|
||||
|
||||
For sources backed by git, the v0.22.4 install-hook helper drops a
|
||||
pre-commit script that blocks commits with malformed frontmatter:
|
||||
|
||||
```bash
|
||||
gbrain frontmatter install-hook
|
||||
```
|
||||
|
||||
Skip this if your brain isn't a git repo or if your downstream agent already
|
||||
enforces validation at write time. See `docs/integrations/pre-commit.md` for
|
||||
the full recipe.
|
||||
|
||||
### 5. Migration ergonomics — read pending-host-work.jsonl
|
||||
|
||||
After `gbrain apply-migrations --yes` runs the v0.22.4 audit, your agent
|
||||
should read `~/.gbrain/migrations/pending-host-work.jsonl` (filter to
|
||||
`migration === "0.22.4"`) and walk each entry's `command` field. Each entry
|
||||
points to a per-source `gbrain frontmatter validate <source_path> --fix`
|
||||
command — surface counts to the user, get explicit consent, then run.
|
||||
|
||||
The migration is **audit-only**. It never mutates brain content during
|
||||
`apply-migrations`. Your agent runs the fix command with user consent.
|
||||
|
||||
---
|
||||
|
||||
## Future versions
|
||||
|
||||
When gbrain ships a new version, this doc will be updated with the diffs for that
|
||||
|
||||
+9
-3
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "gbrain",
|
||||
"version": "0.20.4",
|
||||
"version": "0.22.13",
|
||||
"description": "Postgres-native personal knowledge brain with hybrid RAG search",
|
||||
"type": "module",
|
||||
"main": "src/core/index.ts",
|
||||
@@ -32,7 +32,9 @@
|
||||
"build:all": "bun build --compile --target=bun-darwin-arm64 --outfile bin/gbrain-darwin-arm64 src/cli.ts && bun build --compile --target=bun-linux-x64 --outfile bin/gbrain-linux-x64 src/cli.ts",
|
||||
"build:schema": "bash scripts/build-schema.sh",
|
||||
"build:llms": "bun run scripts/build-llms.ts",
|
||||
"test": "scripts/check-jsonb-pattern.sh && scripts/check-progress-to-stdout.sh && bun run typecheck && bun test",
|
||||
"test": "scripts/check-jsonb-pattern.sh && scripts/check-progress-to-stdout.sh && scripts/check-trailing-newline.sh && scripts/check-wasm-embedded.sh && bun run typecheck && bun test --timeout=60000",
|
||||
"check:wasm": "scripts/check-wasm-embedded.sh",
|
||||
"check:newlines": "scripts/check-trailing-newline.sh",
|
||||
"test:e2e": "bash scripts/run-e2e.sh",
|
||||
"typecheck": "tsc --noEmit",
|
||||
"check:jsonb": "scripts/check-jsonb-pattern.sh",
|
||||
@@ -49,16 +51,20 @@
|
||||
"dependencies": {
|
||||
"@anthropic-ai/sdk": "^0.30.0",
|
||||
"@aws-sdk/client-s3": "^3.1028.0",
|
||||
"@dqbd/tiktoken": "^1.0.22",
|
||||
"@electric-sql/pglite": "0.4.3",
|
||||
"@modelcontextprotocol/sdk": "^1.0.0",
|
||||
"gray-matter": "^4.0.3",
|
||||
"marked": "^18.0.0",
|
||||
"openai": "^4.0.0",
|
||||
"pgvector": "^0.2.0",
|
||||
"postgres": "^3.4.0"
|
||||
"postgres": "^3.4.0",
|
||||
"tree-sitter-wasms": "0.1.13",
|
||||
"web-tree-sitter": "0.22.6"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/bun": "latest",
|
||||
"bun-types": "^1.3.13",
|
||||
"typescript": "^5.6.0"
|
||||
},
|
||||
"trustedDependencies": [
|
||||
|
||||
Executable
+44
@@ -0,0 +1,44 @@
|
||||
#!/usr/bin/env bash
|
||||
# CI guard: every text file under src/, test/, and the repo root .yml/.md
|
||||
# files must end with a newline. POSIX-noncompliant trailing data shows up
|
||||
# as a phantom diff on every future edit and trips most linters.
|
||||
#
|
||||
# Sibling to scripts/check-progress-to-stdout.sh and
|
||||
# scripts/check-jsonb-pattern.sh per CLAUDE.md's CI guard pattern.
|
||||
# Wired into `bun run test` via package.json's `test` script.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Files to check: anything tracked under src/ + test/ that's a code/text file.
|
||||
# Also the top-level *.yml + *.md the repo controls. Portable to bash 3.2
|
||||
# (macOS default) — no mapfile, no associative arrays.
|
||||
files=$(
|
||||
git ls-files \
|
||||
'src/**/*.ts' 'src/**/*.js' 'src/**/*.json' 'src/**/*.sql' 'src/**/*.md' \
|
||||
'test/**/*.ts' 'test/**/*.js' 'test/**/*.json' 'test/**/*.md' \
|
||||
'gbrain.yml' '*.md' \
|
||||
2>/dev/null | sort -u
|
||||
)
|
||||
|
||||
missing=""
|
||||
total=0
|
||||
while IFS= read -r f; do
|
||||
[ -n "$f" ] || continue
|
||||
[ -f "$f" ] || continue
|
||||
[ -s "$f" ] || continue
|
||||
total=$((total + 1))
|
||||
if [ -n "$(tail -c 1 "$f")" ]; then
|
||||
missing="${missing} $f"$'\n'
|
||||
fi
|
||||
done <<< "$files"
|
||||
|
||||
if [ -n "$missing" ]; then
|
||||
echo "ERROR: the following files are missing a trailing newline:" >&2
|
||||
printf '%s' "$missing" >&2
|
||||
echo >&2
|
||||
echo "Fix: append a newline. e.g. \`printf '\\n' >> <file>\` or your editor's" >&2
|
||||
echo "'final newline' setting (most editors do this automatically)." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "trailing-newline check: ok ($total files)"
|
||||
Executable
+62
@@ -0,0 +1,62 @@
|
||||
#!/usr/bin/env bash
|
||||
# CI guard: verify that bun --compile binaries ship with embedded tree-sitter
|
||||
# WASMs and produce real semantic chunks (not recursive-fallback chunks).
|
||||
#
|
||||
# This is the #1 silent-failure mode for v0.19.0 code indexing. If the WASM
|
||||
# import attributes regress or the asset path drifts, the compiled binary
|
||||
# silently falls through to the recursive text chunker. Users see no error,
|
||||
# just degraded chunking quality. This script catches that regression.
|
||||
#
|
||||
# Fails the build when:
|
||||
# - bun build --compile fails
|
||||
# - The resulting binary can't parse TypeScript
|
||||
# - Chunks come back without real symbol names (fallback signature)
|
||||
#
|
||||
# Runs as part of `bun test` via the package.json pre-test pipeline.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
cd "$REPO_ROOT"
|
||||
|
||||
OUT_BIN="$(mktemp /tmp/gbrain-wasm-check.XXXXXX)"
|
||||
trap 'rm -f "$OUT_BIN"' EXIT
|
||||
|
||||
# Build a minimal smoketest binary that imports the chunker. We compile this
|
||||
# instead of the full gbrain CLI so the failure mode is laser-focused on
|
||||
# chunker + WASM path resolution, not unrelated CLI wiring.
|
||||
bun build --compile --outfile "$OUT_BIN" scripts/chunker-smoketest.ts >/dev/null 2>&1
|
||||
|
||||
# Run it and capture JSON output.
|
||||
OUTPUT="$("$OUT_BIN" 2>&1)"
|
||||
|
||||
# Sanity: JSON parses and has expected shape.
|
||||
# - has_symbol_names: at least one chunk carries a concrete symbol name
|
||||
# (proves tree-sitter AST extraction, not recursive-fallback chunks).
|
||||
# - has_typescript_header: the structured header is emitted with the
|
||||
# correct language tag (proves the language map reached displayLang).
|
||||
# - calculateScore by name: specific function that MUST appear as a
|
||||
# top-level semantic node. If it's missing, the chunker either fell
|
||||
# through to recursive or the TypeScript grammar didn't load.
|
||||
if ! echo "$OUTPUT" | grep -q '"has_symbol_names": true'; then
|
||||
echo "[check-wasm-embedded] FAIL: compiled binary returned no symbol names (fallback chunks)." >&2
|
||||
echo "[check-wasm-embedded] Output was:" >&2
|
||||
echo "$OUTPUT" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! echo "$OUTPUT" | grep -q '"has_typescript_header": true'; then
|
||||
echo "[check-wasm-embedded] FAIL: chunk header missing TypeScript language tag." >&2
|
||||
echo "[check-wasm-embedded] Output was:" >&2
|
||||
echo "$OUTPUT" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! echo "$OUTPUT" | grep -q '"calculateScore"'; then
|
||||
echo "[check-wasm-embedded] FAIL: tree-sitter did not extract the calculateScore function symbol." >&2
|
||||
echo "[check-wasm-embedded] Output was:" >&2
|
||||
echo "$OUTPUT" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "[check-wasm-embedded] OK — compiled binary produced real semantic chunks."
|
||||
@@ -0,0 +1,51 @@
|
||||
import { chunkCodeText } from '../src/core/chunkers/code.ts';
|
||||
|
||||
// Large function body so it doesn't merge with siblings — the CI guard
|
||||
// needs at least one chunk with a concrete symbol name to prove the
|
||||
// tree-sitter WASM is actually resolving (not just recursive fallback).
|
||||
const src = `export function calculateScore(
|
||||
items: Array<{ value: number; weight: number }>,
|
||||
opts: { normalize?: boolean; cap?: number } = {}
|
||||
): number {
|
||||
if (items.length === 0) return 0;
|
||||
const sum = items.reduce((acc, it) => acc + it.value * it.weight, 0);
|
||||
const totalWeight = items.reduce((acc, it) => acc + it.weight, 0);
|
||||
if (totalWeight === 0) return 0;
|
||||
const raw = sum / totalWeight;
|
||||
if (opts.normalize) {
|
||||
const clamped = Math.max(0, Math.min(1, raw));
|
||||
return opts.cap !== undefined ? Math.min(opts.cap, clamped) : clamped;
|
||||
}
|
||||
return opts.cap !== undefined ? Math.min(opts.cap, raw) : raw;
|
||||
}
|
||||
|
||||
export class UserRegistry {
|
||||
private users: Map<string, { name: string; score: number }> = new Map();
|
||||
|
||||
register(id: string, name: string, score: number): void {
|
||||
this.users.set(id, { name, score });
|
||||
}
|
||||
|
||||
lookup(id: string): { name: string; score: number } | null {
|
||||
return this.users.get(id) ?? null;
|
||||
}
|
||||
|
||||
topK(k: number): Array<{ id: string; name: string; score: number }> {
|
||||
const entries = Array.from(this.users.entries());
|
||||
entries.sort((a, b) => b[1].score - a[1].score);
|
||||
return entries.slice(0, k).map(([id, v]) => ({ id, ...v }));
|
||||
}
|
||||
}
|
||||
|
||||
export type UserId = string;
|
||||
`;
|
||||
const result = await chunkCodeText(src, 'smoketest.ts');
|
||||
const hasSymbolNames = result.some(c => c.metadata.symbolName !== null);
|
||||
const hasTypeScriptHeader = result.some(c => c.text.startsWith('[TypeScript]'));
|
||||
console.log(JSON.stringify({
|
||||
count: result.length,
|
||||
has_symbol_names: hasSymbolNames,
|
||||
has_typescript_header: hasTypeScriptHeader,
|
||||
first_header: result[0]?.text.split('\n')[0],
|
||||
symbol_names: result.map(c => c.metadata.symbolName),
|
||||
}, null, 2));
|
||||
+6
-1
@@ -15,6 +15,11 @@
|
||||
# the natural per-file test time of 5-10s.
|
||||
#
|
||||
# Exits non-zero on the first failing file so CI fails fast.
|
||||
#
|
||||
# `--timeout=60000` matches the unit test suite. Bun's default is 5s,
|
||||
# which is too tight for setupDB's TRUNCATE CASCADE on ~30 tables on
|
||||
# CI runners under load (one CI flake observed on PR #475 hitting
|
||||
# exactly 5000.09ms in the Tags beforeAll).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
@@ -30,7 +35,7 @@ for f in test/e2e/*.test.ts; do
|
||||
name=$(basename "$f")
|
||||
echo ""
|
||||
echo "=== $name ==="
|
||||
if output=$(bun test "$f" 2>&1); then
|
||||
if output=$(bun test --timeout=60000 "$f" 2>&1); then
|
||||
pass_files=$((pass_files + 1))
|
||||
# Extract pass/fail counts from bun's summary (e.g., "123 pass")
|
||||
p=$(echo "$output" | grep -oE '[0-9]+ pass' | tail -1 | grep -oE '[0-9]+' || echo 0)
|
||||
|
||||
Executable
+76
@@ -0,0 +1,76 @@
|
||||
#!/usr/bin/env bash
|
||||
# Partition unit test files into N shards by stable hash and run one shard.
|
||||
#
|
||||
# Usage: scripts/test-shard.sh <shard-index> <total-shards>
|
||||
# shard-index: 1-based (1..N)
|
||||
# total-shards: positive integer
|
||||
#
|
||||
# E2E tests under test/e2e/ are excluded — they need DATABASE_URL and run via
|
||||
# bun run test:e2e separately.
|
||||
#
|
||||
# Stable partitioning: a file's shard is `(hash(path) % N) + 1`. Same file
|
||||
# lands in the same shard on every run, regardless of how many other files
|
||||
# exist, so retries are reproducible. Hash is FNV-1a — pure shell, no jq.
|
||||
set -euo pipefail
|
||||
|
||||
if [ "$#" -ne 2 ]; then
|
||||
echo "usage: scripts/test-shard.sh <shard-index> <total-shards>" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
SHARD_INDEX="$1"
|
||||
TOTAL_SHARDS="$2"
|
||||
|
||||
if ! [[ "$SHARD_INDEX" =~ ^[0-9]+$ ]] || ! [[ "$TOTAL_SHARDS" =~ ^[0-9]+$ ]]; then
|
||||
echo "error: shard index and total must be positive integers" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "$SHARD_INDEX" -lt 1 ] || [ "$SHARD_INDEX" -gt "$TOTAL_SHARDS" ]; then
|
||||
echo "error: shard index $SHARD_INDEX out of range 1..$TOTAL_SHARDS" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
cd "$(dirname "$0")/.."
|
||||
|
||||
# Find all unit test files, deterministic order. Excludes test/e2e/.
|
||||
# Portable: avoid `mapfile` (bash 4+) so this runs on macOS bash 3.2 too.
|
||||
FILES=()
|
||||
while IFS= read -r line; do
|
||||
FILES+=("$line")
|
||||
done < <(find test -name '*.test.ts' -not -path 'test/e2e/*' | sort)
|
||||
|
||||
if [ "${#FILES[@]}" -eq 0 ]; then
|
||||
echo "no test files found under test/" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# FNV-1a 32-bit hash of a string — implemented in pure bash so we don't depend
|
||||
# on python/openssl/etc on the runner. Output is decimal.
|
||||
fnv1a() {
|
||||
local str="$1"
|
||||
local h=2166136261 # FNV offset basis
|
||||
local i ord
|
||||
for (( i=0; i<${#str}; i++ )); do
|
||||
ord=$(printf '%d' "'${str:$i:1}")
|
||||
h=$(( (h ^ ord) & 0xFFFFFFFF ))
|
||||
h=$(( (h * 16777619) & 0xFFFFFFFF ))
|
||||
done
|
||||
echo "$h"
|
||||
}
|
||||
|
||||
SHARD_FILES=()
|
||||
for f in "${FILES[@]}"; do
|
||||
hash=$(fnv1a "$f")
|
||||
bucket=$(( hash % TOTAL_SHARDS + 1 ))
|
||||
if [ "$bucket" -eq "$SHARD_INDEX" ]; then
|
||||
SHARD_FILES+=("$f")
|
||||
fi
|
||||
done
|
||||
|
||||
echo "shard $SHARD_INDEX/$TOTAL_SHARDS: ${#SHARD_FILES[@]}/${#FILES[@]} files"
|
||||
if [ "${#SHARD_FILES[@]}" -eq 0 ]; then
|
||||
echo "warning: shard $SHARD_INDEX has no files (rehash or reduce shard count)" >&2
|
||||
exit 0
|
||||
fi
|
||||
|
||||
exec bun test --timeout=60000 "${SHARD_FILES[@]}"
|
||||
+3
-1
@@ -13,13 +13,15 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
|
||||
|
||||
| Trigger | Skill |
|
||||
|---------|-------|
|
||||
| "What do we know about", "tell me about", "search for" | `skills/query/SKILL.md` |
|
||||
| "What do we know about", "tell me about", "search for", "who is", "background on", "notes on" | `skills/query/SKILL.md` |
|
||||
| "Who knows who", "relationship between", "connections", "graph query" | `skills/query/SKILL.md` (use graph-query) |
|
||||
| Creating/enriching a person or company page | `skills/enrich/SKILL.md` |
|
||||
| Where does a new file go? Filing rules | `skills/repo-architecture/SKILL.md` |
|
||||
| Fix broken citations in brain pages | `skills/citation-fixer/SKILL.md` |
|
||||
| "citation audit", "check citations", "fix citations" | `skills/citation-fixer/SKILL.md` (focused fix). For broader brain health, chain into `skills/maintain/SKILL.md` |
|
||||
| "Research", "track", "extract from email", "investor updates", "donations" | `skills/data-research/SKILL.md` |
|
||||
| Share a brain page as a link | `skills/publish/SKILL.md` |
|
||||
| "validate frontmatter", "check frontmatter", "fix frontmatter", "frontmatter audit", "brain lint" | `skills/frontmatter-guard/SKILL.md` |
|
||||
|
||||
## Content & media ingestion
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
// Routing eval fixtures for skills/citation-fixer. Check 5 (W2, v0.17).
|
||||
// Layer A (structural) requires intents to contain trigger words from
|
||||
// the resolver. Paraphrase the trigger framing, not its meaning.
|
||||
{"intent": "please fix broken citations across the latest batch of pages", "expected_skill": "citation-fixer"}
|
||||
{"intent": "I think we need to fix broken citations in these brain pages", "expected_skill": "citation-fixer"}
|
||||
{"intent": "please fix citations in the latest batch of brain pages", "expected_skill": "citation-fixer"}
|
||||
{"intent": "I need to fix citations across these pages", "expected_skill": "citation-fixer"}
|
||||
// Negative case: something that sounds similar but should NOT route here.
|
||||
{"intent": "What does this book say about mentorship", "expected_skill": null, "ambiguous_with": []}
|
||||
|
||||
+1
-12
@@ -55,18 +55,7 @@ they building, what makes them tick, where are they headed.
|
||||
|
||||
## Citation Requirements (MANDATORY)
|
||||
|
||||
Every fact must carry an inline `[Source: ...]` citation.
|
||||
|
||||
Three formats:
|
||||
- **Direct attribution:** `[Source: User, {context}, YYYY-MM-DD]`
|
||||
- **API/external:** `[Source: {provider} enrichment, YYYY-MM-DD]`
|
||||
- **Synthesis:** `[Source: compiled from {list of sources}]`
|
||||
|
||||
Source precedence (highest to lowest):
|
||||
1. User's direct statements
|
||||
2. Compiled truth (pre-existing brain synthesis)
|
||||
3. Timeline entries (raw evidence)
|
||||
4. External sources (API enrichment, web search)
|
||||
> **Convention:** see `skills/conventions/quality.md` for citation formats and source precedence.
|
||||
|
||||
When sources conflict, note the contradiction with both citations.
|
||||
|
||||
|
||||
@@ -0,0 +1,180 @@
|
||||
---
|
||||
name: frontmatter-guard
|
||||
version: 1.0.0
|
||||
description: |
|
||||
Validate and auto-repair YAML frontmatter on brain pages. Catches malformed
|
||||
pages before they enter the brain (missing closing ---, nested quotes, slug
|
||||
mismatches, null bytes, empty frontmatter, YAML parse failures). Wraps the
|
||||
`gbrain frontmatter` CLI for agent-driven workflows.
|
||||
triggers:
|
||||
- "validate frontmatter"
|
||||
- "check frontmatter"
|
||||
- "fix frontmatter"
|
||||
- "frontmatter audit"
|
||||
- "brain lint"
|
||||
tools:
|
||||
- exec
|
||||
mutating: true
|
||||
---
|
||||
|
||||
# Frontmatter Guard Skill
|
||||
|
||||
> **Convention:** see `skills/conventions/quality.md` for citation rules; this skill is structural validation, not citation auditing.
|
||||
|
||||
## Contract
|
||||
|
||||
This skill guarantees:
|
||||
- Every brain page is scanned against the seven canonical frontmatter validation classes
|
||||
- Mechanical errors (nested quotes, missing closing `---`, null bytes, slug mismatch) are auto-repairable on demand with `.bak` backups
|
||||
- Validation logic is shared with `gbrain doctor`'s `frontmatter_integrity` subcheck — single source of truth
|
||||
- Reports per source (gbrain is multi-source since v0.18.0); never silently audits the wrong root
|
||||
|
||||
## Why This Exists
|
||||
|
||||
Brain pages pile up over months. Agents write them with malformed frontmatter:
|
||||
- Missing closing `---` (entity detector bugs)
|
||||
- Unstructured YAML in meeting pages (ingestion bugs)
|
||||
- Slug mismatches (path renames not propagated)
|
||||
- Null bytes (binary corruption from copy-paste accidents)
|
||||
- Nested double quotes in titles (`title: "Phil "Nick" Last"`)
|
||||
|
||||
Without a guard, these accumulate silently until `gbrain sync` chokes or search returns garbage. The guard makes the failure visible at audit time and trivially fixable.
|
||||
|
||||
## Validation classes
|
||||
|
||||
| Code | Meaning | Auto-fixable? |
|
||||
|------|---------|---------------|
|
||||
| `MISSING_OPEN` | File doesn't start with `---` | No (needs human) |
|
||||
| `MISSING_CLOSE` | No closing `---` before first heading | Yes |
|
||||
| `YAML_PARSE` | YAML failed to parse | Sometimes (depends on cause) |
|
||||
| `SLUG_MISMATCH` | Frontmatter `slug:` differs from path-derived slug | Yes (removes the field) |
|
||||
| `NULL_BYTES` | Binary corruption (`\x00`) | Yes |
|
||||
| `NESTED_QUOTES` | `title: "outer "inner" outer"` shape | Yes |
|
||||
| `EMPTY_FRONTMATTER` | Open + close present but nothing between | No (needs human) |
|
||||
|
||||
## Phases
|
||||
|
||||
### Phase 1: Audit
|
||||
|
||||
Run a read-only scan across all registered sources (or one with `--source <id>`).
|
||||
|
||||
```bash
|
||||
gbrain frontmatter audit --json
|
||||
```
|
||||
|
||||
Reports:
|
||||
- Per-source counts grouped by error code
|
||||
- Sample of up to 20 affected pages per source
|
||||
- Total count
|
||||
- Scan timestamp
|
||||
|
||||
Output is JSON; agents parse `errors_by_code` and `per_source` to decide next steps.
|
||||
|
||||
### Phase 2: Validate one path
|
||||
|
||||
Validate a single file or directory (does not require source registration):
|
||||
|
||||
```bash
|
||||
gbrain frontmatter validate <path> --json
|
||||
```
|
||||
|
||||
Exit code 0 = clean; 1 = errors found. Use this in CI pipelines or pre-commit hooks.
|
||||
|
||||
### Phase 3: Fix
|
||||
|
||||
When issues are found:
|
||||
|
||||
```bash
|
||||
gbrain frontmatter validate <path> --fix
|
||||
```
|
||||
|
||||
`--fix` writes `<file>.bak` for every modified file before mutating. The backup is the safety contract — works whether the brain is a git repo or a plain directory.
|
||||
|
||||
`--dry-run` previews without writing. Use this before applying fixes in batch.
|
||||
|
||||
### Phase 4: Pre-commit hook (optional)
|
||||
|
||||
For brain repos that ARE git repos, install the pre-commit hook to block malformed pages from being committed in the first place:
|
||||
|
||||
```bash
|
||||
gbrain frontmatter install-hook [--source <id>]
|
||||
```
|
||||
|
||||
The hook runs `gbrain frontmatter validate` against staged `.md`/`.mdx` files. Bypass with `git commit --no-verify`.
|
||||
|
||||
## Trigger words
|
||||
|
||||
When the user says any of these, route here:
|
||||
- "validate frontmatter"
|
||||
- "check frontmatter"
|
||||
- "fix frontmatter"
|
||||
- "frontmatter audit"
|
||||
- "brain lint"
|
||||
|
||||
## Output rules
|
||||
|
||||
- Always run `gbrain frontmatter audit --json` first; never assume a brain is clean.
|
||||
- Surface counts to the user in plain language; do not dump raw JSON.
|
||||
- For `--fix` operations: state how many files will be modified BEFORE running, then confirm.
|
||||
- `SLUG_MISMATCH` fixes remove the frontmatter `slug:` field — gbrain derives slug from path. Mention this when the user's title is intentionally renamed.
|
||||
- Never auto-fix `MISSING_OPEN` or `EMPTY_FRONTMATTER` without explicit user input — these usually mean a human author started a page and didn't finish.
|
||||
|
||||
## Chains with
|
||||
|
||||
- `gbrain doctor` — the `frontmatter_integrity` subcheck reports the same counts as `audit`.
|
||||
- `skills/maintain/SKILL.md` — broader brain health audit; chain after this skill if other classes of issue are suspected.
|
||||
- `skills/lint/SKILL.md` (via `gbrain lint`) — overlapping rules for skill-file lint; the `frontmatter-*` rule names in lint output come from this skill's validation surface.
|
||||
|
||||
## Output Format
|
||||
|
||||
Audit summary (terse, agent-friendly):
|
||||
|
||||
```
|
||||
Frontmatter audit — 17 issue(s) across 1 source(s)
|
||||
|
||||
[default] /Users/me/brain
|
||||
17 issue(s)
|
||||
MISSING_CLOSE: 8
|
||||
NESTED_QUOTES: 5
|
||||
NULL_BYTES: 4
|
||||
sample:
|
||||
people/jane.md — MISSING_CLOSE
|
||||
companies/acme.md — NESTED_QUOTES
|
||||
(+ 12 more)
|
||||
|
||||
Fix with: gbrain frontmatter validate /Users/me/brain --fix
|
||||
```
|
||||
|
||||
JSON envelope (when `--json` is passed):
|
||||
|
||||
```json
|
||||
{
|
||||
"ok": false,
|
||||
"total": 17,
|
||||
"errors_by_code": { "MISSING_CLOSE": 8, "NESTED_QUOTES": 5, "NULL_BYTES": 4 },
|
||||
"per_source": [
|
||||
{
|
||||
"source_id": "default",
|
||||
"source_path": "/Users/me/brain",
|
||||
"total": 17,
|
||||
"errors_by_code": { "MISSING_CLOSE": 8, "NESTED_QUOTES": 5, "NULL_BYTES": 4 },
|
||||
"sample": [{ "path": "people/jane.md", "codes": ["MISSING_CLOSE"] }]
|
||||
}
|
||||
],
|
||||
"scanned_at": "2026-04-25T22:30:00.000Z"
|
||||
}
|
||||
```
|
||||
|
||||
`gbrain frontmatter validate <path> --json` returns a similar envelope keyed on per-file results instead of per-source.
|
||||
|
||||
## Anti-Patterns
|
||||
|
||||
**Don't auto-fix `MISSING_OPEN` or `EMPTY_FRONTMATTER` without user input.** These usually mean a human author started a page and didn't finish — silently inserting `---` markers around an unfinished draft is wrong.
|
||||
|
||||
**Don't use `--fix` to "make doctor green" without reading the audit first.** SLUG_MISMATCH cases are surfaced for manual review specifically because gbrain derives the slug from path. A mismatch usually means the user renamed a file intentionally; auto-removing the slug field is the right outcome only when you've confirmed the rename was deliberate.
|
||||
|
||||
**Don't skip the `.bak` backups.** The `.bak` is the safety contract for non-git brain repos. If `.bak` files accumulate after a fix run, that's a feature, not a bug — the user can review the diffs and delete the backups when satisfied.
|
||||
|
||||
**Don't run `audit` on a brain where sources aren't registered.** The CLI returns "no registered sources to audit" gracefully, but the migration emits a `skipped: no_sources` phase result. Don't paper over this with a manual path-walk; the right fix is to register the source via `gbrain sources add`.
|
||||
|
||||
**Don't install the pre-commit hook on non-git brain dirs.** The install-hook command skips them automatically with a one-line note. If you see "skipped — not a git repo" and want validation at write time anyway, use the `audit` command on a cron schedule.
|
||||
@@ -0,0 +1,8 @@
|
||||
// Routing eval fixtures for skills/frontmatter-guard. Check 5 (W2, v0.17).
|
||||
// Layer A (structural) requires intents to contain trigger words from
|
||||
// the resolver. Paraphrase the trigger framing, not its meaning.
|
||||
{"intent": "please validate frontmatter on the latest batch of brain pages", "expected_skill": "frontmatter-guard"}
|
||||
{"intent": "fix frontmatter on these pages", "expected_skill": "frontmatter-guard"}
|
||||
{"intent": "I want to run a frontmatter audit across the brain", "expected_skill": "frontmatter-guard"}
|
||||
// Negative case: something that sounds similar but should NOT route here.
|
||||
{"intent": "what's for breakfast", "expected_skill": null, "ambiguous_with": []}
|
||||
@@ -8,7 +8,6 @@ description: |
|
||||
triggers:
|
||||
- "brain health"
|
||||
- "check backlinks"
|
||||
- "citation audit"
|
||||
- "maintenance"
|
||||
- "orphan pages"
|
||||
- "stale pages"
|
||||
|
||||
@@ -44,6 +44,11 @@
|
||||
"path": "publish/SKILL.md",
|
||||
"description": "Share brain pages as beautiful password-protected HTML (code + skill pair, zero LLM calls)"
|
||||
},
|
||||
{
|
||||
"name": "frontmatter-guard",
|
||||
"path": "frontmatter-guard/SKILL.md",
|
||||
"description": "Validate and auto-repair YAML frontmatter on brain pages; gates against malformed YAML, missing closing ---, nested quotes, slug mismatches, null bytes"
|
||||
},
|
||||
{
|
||||
"name": "signal-detector",
|
||||
"path": "signal-detector/SKILL.md",
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
---
|
||||
version: 0.19.0
|
||||
feature_pitch:
|
||||
headline: "Your code is now first-class in the brain."
|
||||
one_liner: "gbrain code-refs BrainEngine --json returns every usage site in <100ms."
|
||||
user_action_required: true
|
||||
---
|
||||
|
||||
# v0.19.0 — Code Indexing
|
||||
|
||||
This release makes code a first-class citizen in the brain. Tree-sitter parses 29 languages into semantic chunks. `gbrain code-def` and `gbrain code-refs` let agents find symbol definitions and references without grep. Incremental chunking drops daily autopilot embedding cost by ~95%. The chunker is a strict superset of Chonkie's CodeChunker plus a structured header Chonkie lacks.
|
||||
|
||||
## Schema migrations applied automatically
|
||||
|
||||
- **v25 — `pages.page_kind`** — distinguishes markdown vs code pages at the DB level. Existing rows backfill to `'markdown'`. Postgres uses `ADD CONSTRAINT ... NOT VALID` + `VALIDATE CONSTRAINT` so large tables don't block.
|
||||
- **v26 — `content_chunks` code metadata** — adds `language`, `symbol_name`, `symbol_type`, `start_line`, `end_line`. All nullable. Partial indexes on `symbol_name` and `language` for fast symbol lookup.
|
||||
|
||||
These run as part of `gbrain upgrade` → `gbrain apply-migrations`. No manual DDL needed.
|
||||
|
||||
## What the agent should do after upgrading
|
||||
|
||||
1. **Confirm migrations landed:**
|
||||
```bash
|
||||
gbrain doctor
|
||||
```
|
||||
Look for `schema_version: 26`. If lower, run `gbrain apply-migrations --yes`.
|
||||
|
||||
2. **Register a code source** if the user wants their code indexed:
|
||||
```bash
|
||||
gbrain sources add <id> --path <path-to-repo>
|
||||
```
|
||||
Pick a short `<id>` (e.g. `wiki`, `gbrain`, `yc-media`). Shorter is better — it's used in citation keys.
|
||||
|
||||
3. **Sync the code source:**
|
||||
```bash
|
||||
gbrain sync --source <id>
|
||||
```
|
||||
First sync may run tens of minutes depending on repo size. Each TypeScript function becomes a chunk with a structured header like `[TypeScript] src/core/sync.ts:380-415 function performFullSync`.
|
||||
|
||||
4. **Verify code-def and code-refs work:**
|
||||
```bash
|
||||
gbrain code-def BrainEngine # prints the file + line of the definition
|
||||
gbrain code-refs BrainEngine --json # JSON array of every usage site
|
||||
```
|
||||
If both return non-empty arrays, code indexing is working end-to-end.
|
||||
|
||||
5. **Observe incremental chunking.** Edit one function in a 20-function file, re-run `sync --source <id>`. Embedding cost should be ~5% of the first sync because unchanged chunks reuse their existing embeddings.
|
||||
|
||||
## Migration from Wintermute's `repos` (if you used it)
|
||||
|
||||
v0.19.0 deletes `~/.gbrain/config.json`'s `repos` array in favor of the `sources` table. The CLI surface is preserved as a deprecated alias: `gbrain repos add` still works, but routes into `runSources` with a one-line deprecation notice on stderr. Existing scripts keep working; prefer `gbrain sources` going forward.
|
||||
|
||||
If you had repos configured in `~/.gbrain/config.json`, re-register them:
|
||||
```bash
|
||||
gbrain sources add <name> --path <path>
|
||||
```
|
||||
Per-repo sync bookmarks live in the `sources` table now (not config.json).
|
||||
|
||||
## Flag in `pending-host-work.jsonl`
|
||||
|
||||
Per the v0.11.0 convention, the migration orchestrator writes an entry to `~/.gbrain/migrations/pending-host-work.jsonl` flagging the new CLI surfaces so headless agents can walk the TODOs:
|
||||
|
||||
```json
|
||||
{"version": "0.19.0", "action": "register_code_source", "status": "pending"}
|
||||
```
|
||||
|
||||
Agents that handle pending-host-work should offer the user a `gbrain sources add ...` prompt.
|
||||
|
||||
## When NOT to run the migration
|
||||
|
||||
Never. v0.19.0 is fully backward-compatible. Existing markdown-only brains see zero behavior change until the user adds a code source.
|
||||
@@ -0,0 +1,64 @@
|
||||
---
|
||||
version: 0.21.0
|
||||
feature_pitch:
|
||||
headline: "Code Cathedral II — chunk-grain FTS, qualified symbols, structural edges."
|
||||
one_liner: "Natural-language queries now rank docstring matches first. CHUNKER_VERSION 3→4 rolls the new chunker over existing code pages automatically on next sync."
|
||||
user_action_required: true
|
||||
---
|
||||
|
||||
# v0.21.0 — Code Cathedral II
|
||||
|
||||
This release is the biggest code-search upgrade in gbrain history. Chunk-grain FTS with doc_comment Weight A ranks natural-language queries against docstrings above prose. CHUNKER_VERSION 3 → 4 folds into content_hash so every existing code page re-chunks. File classifier widened from 9 to 35 extensions. Markdown fence extraction, `sync --all` cost preview, and `reconcile-links` batch command ship alongside the chunker upgrade.
|
||||
|
||||
## Schema migrations applied automatically
|
||||
|
||||
- **v27 — cathedral_ii_foundation** — adds `code_edges_chunk`, `code_edges_symbol`, `sources.chunker_version`, and new `content_chunks` columns (`parent_symbol_path`, `doc_comment`, `symbol_name_qualified`, `search_vector`). Includes the chunk-grain FTS trigger that builds from `setweight(to_tsvector('english', doc_comment), 'A') || setweight(to_tsvector('english', chunk_text), 'B') || setweight(to_tsvector('english', symbol_name_qualified), 'A')`.
|
||||
- **v28 — cathedral_ii_chunk_fts_backfill** — populates `search_vector` on every existing chunk so day-1 queries already rank correctly.
|
||||
|
||||
These run as part of `gbrain upgrade` → `gbrain apply-migrations`. No manual DDL needed.
|
||||
|
||||
## What the agent should do after upgrading
|
||||
|
||||
1. **Confirm migrations landed:**
|
||||
```bash
|
||||
gbrain doctor
|
||||
```
|
||||
Look for `schema_version: 28`. If lower, run `gbrain apply-migrations --yes`.
|
||||
|
||||
2. **Pick a backfill path.** The `CHUNKER_VERSION` bump + `sources.chunker_version` gate means existing code pages must re-chunk for the new shape to take effect. Two paths:
|
||||
|
||||
**Automatic (recommended):** next `gbrain sync --source <id>` detects the version mismatch and forces a full re-walk regardless of git HEAD equality. No cost preview, no user interaction. The Layer 12 SP-1 fix from codex's second-pass review.
|
||||
|
||||
**Immediate:** preview cost, then reindex every code page now.
|
||||
```bash
|
||||
gbrain reindex-code --dry-run # preview token count + $USD cost
|
||||
gbrain reindex-code --yes # reindex all code pages
|
||||
gbrain reindex-code --source <id> --yes # scope to one source
|
||||
```
|
||||
On non-TTY (automation / cron) `reindex-code` without `--yes` emits a `ConfirmationRequired` envelope and exits 2 — same shape as `sync --all`. The envelope matches v0.19.0's `StructuredAgentError`.
|
||||
|
||||
3. **Verify chunk-grain FTS works.** A query that mentions a concept in a docstring (not just the function body) should rank higher than a page that mentions it only in prose:
|
||||
```bash
|
||||
gbrain query "whatever your docstring says"
|
||||
```
|
||||
Expected: top hit is the chunk whose `doc_comment` matches.
|
||||
|
||||
4. **(Optional) Reconcile doc↔impl links.** v0.19.0 Layer 6 forward-extracted code refs from markdown pages when they imported, but edges dropped if the code page hadn't imported yet. v0.21.0's `reconcile-links` batch-scans every markdown page and idempotently reinserts missing edges:
|
||||
```bash
|
||||
gbrain reconcile-links # full run
|
||||
gbrain reconcile-links --dry-run # preview only
|
||||
gbrain reconcile-links --json # machine output
|
||||
```
|
||||
Respects `auto_link=false` config (prints warn + exits 0 when disabled).
|
||||
|
||||
## Widened file classifier — more languages sync now
|
||||
|
||||
Previous classifier recognized 9 extensions as code. v0.21.0 widens to 35 (Rust, Ruby, Java, C#, C/C++, Swift, Kotlin, Scala, PHP, Elixir, Elm, OCaml, Dart, Zig, Solidity, Lua, shell, etc.). If your repo contains source files in languages beyond TS/JS/Py/Go, they'll flow through the code chunker on next sync. No action needed — `detectCodeLanguage` handles dispatch.
|
||||
|
||||
## Flag in `pending-host-work.jsonl`
|
||||
|
||||
The migration orchestrator emits a backfill-prompt phase that prints the two backfill choices directly. No `pending-host-work.jsonl` entry is written — the choice is user-driven and ephemeral (either reindex now or wait for next sync).
|
||||
|
||||
## When NOT to run the migration
|
||||
|
||||
Never skip it. v0.21.0 is fully backward-compatible at the API level (page-grain FTS shape preserved externally; chunk-grain is internal only). Skipping the migration means agents miss doc_comment Weight A ranking and the chunker_version gate never fires — existing code pages silently stay on CHUNKER_VERSION 3 forever until you sync with `--full`.
|
||||
@@ -0,0 +1,113 @@
|
||||
# v0.22.4 — Frontmatter Guard
|
||||
|
||||
## What ships
|
||||
|
||||
- `gbrain frontmatter` CLI (validate / audit / install-hook)
|
||||
- `frontmatter_integrity` subcheck under `gbrain doctor`
|
||||
- New `frontmatter-guard` skill (`skills/frontmatter-guard/SKILL.md`)
|
||||
- Pre-commit hook helper for git-backed brain repos
|
||||
- Audit-only migration that scans every registered source, writes
|
||||
`~/.gbrain/migrations/v0.22.4-audit.json`, and queues per-source TODO
|
||||
entries to `~/.gbrain/migrations/pending-host-work.jsonl`
|
||||
- 0 warnings on `gbrain check-resolvable` (down from 7 on master)
|
||||
|
||||
## What the agent should do post-upgrade
|
||||
|
||||
The orchestrator handles the mechanical side. Your job is to surface the audit
|
||||
to the user and apply fixes with their consent.
|
||||
|
||||
### 1. Run the orchestrator
|
||||
|
||||
```bash
|
||||
gbrain apply-migrations --yes
|
||||
```
|
||||
|
||||
This runs three idempotent phases:
|
||||
|
||||
- `schema` (no-op in v0.22.4)
|
||||
- `audit` — `scanBrainSources()` over every registered source; writes
|
||||
`~/.gbrain/migrations/v0.22.4-audit.json`
|
||||
- `emit-todo` — appends one entry per source-with-issues to
|
||||
`~/.gbrain/migrations/pending-host-work.jsonl`. Each entry contains the
|
||||
exact `gbrain frontmatter validate <source-path> --fix` command.
|
||||
|
||||
The migration **never mutates brain pages**. Auto-fixes only run when the user
|
||||
explicitly invokes the validate-with-fix command below.
|
||||
|
||||
### 2. Read the audit report
|
||||
|
||||
```bash
|
||||
cat ~/.gbrain/migrations/v0.22.4-audit.json
|
||||
```
|
||||
|
||||
The shape:
|
||||
|
||||
```json
|
||||
{
|
||||
"ok": false,
|
||||
"total": 17,
|
||||
"errors_by_code": { "MISSING_CLOSE": 8, "NESTED_QUOTES": 5, "NULL_BYTES": 4 },
|
||||
"per_source": [
|
||||
{
|
||||
"source_id": "default",
|
||||
"source_path": "/Users/me/brain",
|
||||
"total": 17,
|
||||
"errors_by_code": { "MISSING_CLOSE": 8, "NESTED_QUOTES": 5, "NULL_BYTES": 4 },
|
||||
"sample": [
|
||||
{ "path": "people/jane.md", "codes": ["MISSING_CLOSE"] }
|
||||
]
|
||||
}
|
||||
],
|
||||
"scanned_at": "2026-04-25T22:30:00.000Z"
|
||||
}
|
||||
```
|
||||
|
||||
### 3. Surface the report to the user
|
||||
|
||||
State the per-source counts in plain language. Example:
|
||||
|
||||
> "v0.22.4 ships frontmatter-guard. I ran an audit and found 17 issues across
|
||||
> 1 source (default: 8 MISSING_CLOSE, 5 NESTED_QUOTES, 4 NULL_BYTES). The
|
||||
> mechanical errors are auto-fixable; SLUG_MISMATCH cases (if any) need your
|
||||
> review. Want me to fix the auto-fixable ones now?"
|
||||
|
||||
### 4. Run the fix (with consent)
|
||||
|
||||
Per source with issues, the queued command is:
|
||||
|
||||
```bash
|
||||
gbrain frontmatter validate <source_path> --fix
|
||||
```
|
||||
|
||||
`--fix` writes a `.bak` backup for every modified file. SLUG_MISMATCH errors
|
||||
are surfaced for manual review (not auto-fixed) — gbrain derives slugs from
|
||||
path, so a mismatched slug usually means the user renamed the file
|
||||
intentionally or the slug field is stale.
|
||||
|
||||
### 5. (Optional) Install the pre-commit hook
|
||||
|
||||
For git-backed sources only:
|
||||
|
||||
```bash
|
||||
gbrain frontmatter install-hook [--source <id>]
|
||||
```
|
||||
|
||||
This blocks future malformed-frontmatter commits at the git layer. Bypass with
|
||||
`git commit --no-verify`. Skip this step for non-git brains.
|
||||
|
||||
### 6. Verify
|
||||
|
||||
```bash
|
||||
gbrain doctor --json | jq '.checks[] | select(.name == "frontmatter_integrity")'
|
||||
gbrain frontmatter audit --json | jq '.total'
|
||||
```
|
||||
|
||||
Both should report 0 issues after fixes are applied.
|
||||
|
||||
### 7. If anything fails
|
||||
|
||||
Open an issue at https://github.com/garrytan/gbrain/issues with:
|
||||
- output of `gbrain doctor`
|
||||
- contents of `~/.gbrain/migrations/v0.22.4-audit.json`
|
||||
- contents of `~/.gbrain/upgrade-errors.jsonl` if it exists
|
||||
- which step broke
|
||||
@@ -12,6 +12,8 @@ triggers:
|
||||
- "what happened"
|
||||
- "search for"
|
||||
- "look up"
|
||||
- "background on"
|
||||
- "notes on"
|
||||
- "who knows who"
|
||||
- "relationship between"
|
||||
- "connections"
|
||||
|
||||
BIN
Binary file not shown.
Executable
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Binary file not shown.
Executable
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Executable
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Executable
BIN
Binary file not shown.
+95
-3
@@ -19,7 +19,7 @@ for (const op of operations) {
|
||||
}
|
||||
|
||||
// CLI-only commands that bypass the operation layer
|
||||
const CLI_ONLY = new Set(['init', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'sync', 'extract', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'sources', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test']);
|
||||
const CLI_ONLY = new Set(['init', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'sync', 'extract', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'sources', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test', 'storage', 'repos', 'code-def', 'code-refs', 'reindex-code', 'code-callers', 'code-callees', 'frontmatter', 'auth']);
|
||||
|
||||
async function main() {
|
||||
// Parse global flags (--quiet / --progress-json / --progress-interval)
|
||||
@@ -285,6 +285,11 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
await runIntegrations(args);
|
||||
return;
|
||||
}
|
||||
if (command === 'auth') {
|
||||
const { runAuth } = await import('./commands/auth.ts');
|
||||
await runAuth(args);
|
||||
return;
|
||||
}
|
||||
if (command === 'resolvers') {
|
||||
const { runResolvers } = await import('./commands/resolvers.ts');
|
||||
await runResolvers(args);
|
||||
@@ -305,6 +310,11 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
await runBacklinks(args);
|
||||
return;
|
||||
}
|
||||
if (command === 'frontmatter') {
|
||||
const { runFrontmatter } = await import('./commands/frontmatter.ts');
|
||||
await runFrontmatter(args);
|
||||
return;
|
||||
}
|
||||
if (command === 'lint') {
|
||||
const { runLint } = await import('./commands/lint.ts');
|
||||
await runLint(args);
|
||||
@@ -442,7 +452,7 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
}
|
||||
case 'serve': {
|
||||
const { runServe } = await import('./commands/serve.ts');
|
||||
await runServe(engine);
|
||||
await runServe(engine, args);
|
||||
return; // serve doesn't disconnect
|
||||
}
|
||||
case 'call': {
|
||||
@@ -501,6 +511,15 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
await runGraphQuery(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'reconcile-links': {
|
||||
// v0.20.0 Cathedral II Layer 8 D3: batch-recompute doc↔impl edges
|
||||
// for any markdown page that cites code files. Idempotent; safe to
|
||||
// re-run. Closes the v0.19.0 Layer 6 order-dependency bug where
|
||||
// guides imported before their code never got their edges written.
|
||||
const { runReconcileLinksCli } = await import('./commands/reconcile-links.ts');
|
||||
await runReconcileLinksCli(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'orphans': {
|
||||
const { runOrphans } = await import('./commands/orphans.ts');
|
||||
await runOrphans(engine, args);
|
||||
@@ -511,6 +530,53 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
await runSources(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'storage': {
|
||||
const { runStorage } = await import('./commands/storage.ts');
|
||||
await runStorage(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'code-def': {
|
||||
const { runCodeDef } = await import('./commands/code-def.ts');
|
||||
await runCodeDef(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'code-refs': {
|
||||
const { runCodeRefs } = await import('./commands/code-refs.ts');
|
||||
await runCodeRefs(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'reindex-code': {
|
||||
// v0.20.0 Cathedral II Layer 13 (E2): explicit code-page reindex
|
||||
// for users upgrading from v0.19.0. Cost-preview gated; TTY prompt
|
||||
// or ConfirmationRequired envelope for non-TTY/JSON callers.
|
||||
const { runReindexCodeCli } = await import('./commands/reindex-code.ts');
|
||||
await runReindexCodeCli(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'code-callers': {
|
||||
// v0.20.0 Cathedral II Layer 10 (C4): "who calls <symbol>?"
|
||||
const { runCodeCallers } = await import('./commands/code-callers.ts');
|
||||
await runCodeCallers(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'code-callees': {
|
||||
// v0.20.0 Cathedral II Layer 10 (C5): "what does <symbol> call?"
|
||||
const { runCodeCallees } = await import('./commands/code-callees.ts');
|
||||
await runCodeCallees(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'repos': {
|
||||
// v0.19.0: `gbrain repos ...` is an alias into the v0.18.0 sources
|
||||
// subsystem. The repos abstraction (Wintermute's baseline) was
|
||||
// redundant with sources and carried per-user config state that
|
||||
// couldn't participate in federation / RLS / multi-tenancy. We
|
||||
// keep the alias so scripts like `gbrain repos add .` keep
|
||||
// working, with a nudge toward the canonical command.
|
||||
console.error('[gbrain] Note: "repos" is an alias for "sources" as of v0.19.0. Prefer `gbrain sources <subcommand>`.');
|
||||
const { runSources } = await import('./commands/sources.ts');
|
||||
await runSources(engine, args);
|
||||
break;
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
if (command !== 'serve') await engine.disconnect();
|
||||
@@ -525,7 +591,10 @@ async function connectEngine(): Promise<BrainEngine> {
|
||||
}
|
||||
const { createEngine } = await import('./core/engine-factory.ts');
|
||||
const engine = await createEngine(toEngineConfig(config));
|
||||
await engine.connect(toEngineConfig(config));
|
||||
const noRetry = process.argv.includes('--no-retry-connect') ||
|
||||
process.env.GBRAIN_NO_RETRY_CONNECT === '1';
|
||||
const { connectWithRetry } = await import('./core/db.ts');
|
||||
await connectWithRetry(engine, toEngineConfig(config), { noRetry });
|
||||
return engine;
|
||||
}
|
||||
|
||||
@@ -581,6 +650,8 @@ IMPORT/EXPORT
|
||||
sync --watch [--interval N] Continuous sync (loops until stopped)
|
||||
sync --install-cron Install persistent sync daemon
|
||||
export [--dir ./out/] Export to markdown
|
||||
export --restore-only [--repo <p>] Restore missing supabase-only files
|
||||
[--type T] [--slug-prefix S] With optional filters
|
||||
|
||||
FILES
|
||||
files list [slug] List stored files
|
||||
@@ -625,6 +696,25 @@ TOOLS
|
||||
check-resolvable [--json] [--fix] Validate skill tree (reachability/MECE/DRY)
|
||||
report --type <name> --content ... Save timestamped report to brain/reports/
|
||||
|
||||
SOURCES (multi-repo / multi-brain)
|
||||
sources list Show registered sources
|
||||
sources add <id> --path <p> Register a source (id = short name, e.g. 'wiki')
|
||||
sources remove <id> Remove a source + its pages
|
||||
sync --all Sync all sources with a local_path
|
||||
sync --source <id> Sync one specific source
|
||||
repos ... DEPRECATED alias for 'sources' (v0.19.0)
|
||||
|
||||
CODE INDEXING (v0.19.0 / v0.20.0 Cathedral II)
|
||||
code-def <symbol> [--lang l] Find the definition of a symbol across code pages
|
||||
code-refs <symbol> [--lang l] Find all references to a symbol (JSON-first)
|
||||
code-callers <symbol> Who calls this symbol? (v0.20.0 A1)
|
||||
code-callees <symbol> What does this symbol call? (v0.20.0 A1)
|
||||
query <q> --lang <l> Filter hybrid search to one language (v0.20.0)
|
||||
query <q> --symbol-kind <k> Filter to symbol type (function|class|method|...) (v0.20.0)
|
||||
reconcile-links [--dry-run] Batch-recompute doc↔impl edges (v0.20.0)
|
||||
reindex-code [--source id] [--yes] Explicit code-page reindex (v0.20.0)
|
||||
sync --strategy code Sync code files into the brain
|
||||
|
||||
JOBS (Minions)
|
||||
jobs submit <name> [--params JSON] Submit background job [--follow] [--dry-run]
|
||||
jobs list [--status S] [--limit N] List jobs
|
||||
@@ -643,6 +733,8 @@ ADMIN
|
||||
features [--json] [--auto-fix] Scan usage + recommend unused features
|
||||
autopilot [--repo] [--interval N] Self-maintaining brain daemon
|
||||
config [show|get|set] <key> [val] Brain config
|
||||
storage status [--repo <path>] Storage tier status and health
|
||||
[--json] (git-tracked vs supabase-only)
|
||||
serve MCP server (stdio)
|
||||
call <tool> '<json>' Raw tool invocation
|
||||
version Version info
|
||||
|
||||
+52
-31
@@ -1,20 +1,29 @@
|
||||
#!/usr/bin/env bun
|
||||
/**
|
||||
* GBrain token management — standalone script, no gbrain CLI dependency.
|
||||
* GBrain token management.
|
||||
*
|
||||
* Usage:
|
||||
* Wired into the CLI as of v0.22.5:
|
||||
* gbrain auth create "claude-desktop"
|
||||
* gbrain auth list
|
||||
* gbrain auth revoke "claude-desktop"
|
||||
* gbrain auth test <url> --token <token>
|
||||
*
|
||||
* Also runs standalone (no compiled binary required):
|
||||
* DATABASE_URL=... bun run src/commands/auth.ts create "claude-desktop"
|
||||
* DATABASE_URL=... bun run src/commands/auth.ts list
|
||||
* DATABASE_URL=... bun run src/commands/auth.ts revoke "claude-desktop"
|
||||
* DATABASE_URL=... bun run src/commands/auth.ts test <url> --token <token>
|
||||
*
|
||||
* Both paths require DATABASE_URL or GBRAIN_DATABASE_URL (except `test`,
|
||||
* which only hits the remote URL and doesn't need a local DB).
|
||||
*/
|
||||
import postgres from 'postgres';
|
||||
import { createHash, randomBytes } from 'crypto';
|
||||
|
||||
const DATABASE_URL = process.env.DATABASE_URL || process.env.GBRAIN_DATABASE_URL;
|
||||
if (!DATABASE_URL && process.argv[2] !== 'test') {
|
||||
console.error('Set DATABASE_URL or GBRAIN_DATABASE_URL environment variable.');
|
||||
process.exit(1);
|
||||
function getDatabaseUrl(requireDb: boolean): string | undefined {
|
||||
const url = process.env.DATABASE_URL || process.env.GBRAIN_DATABASE_URL;
|
||||
if (!url && requireDb) {
|
||||
console.error('Set DATABASE_URL or GBRAIN_DATABASE_URL environment variable.');
|
||||
process.exit(1);
|
||||
}
|
||||
return url;
|
||||
}
|
||||
|
||||
function hashToken(token: string): string {
|
||||
@@ -27,7 +36,7 @@ function generateToken(): string {
|
||||
|
||||
async function create(name: string) {
|
||||
if (!name) { console.error('Usage: auth create <name>'); process.exit(1); }
|
||||
const sql = postgres(DATABASE_URL!);
|
||||
const sql = postgres(getDatabaseUrl(true)!);
|
||||
const token = generateToken();
|
||||
const hash = hashToken(token);
|
||||
|
||||
@@ -53,7 +62,7 @@ async function create(name: string) {
|
||||
}
|
||||
|
||||
async function list() {
|
||||
const sql = postgres(DATABASE_URL!);
|
||||
const sql = postgres(getDatabaseUrl(true)!);
|
||||
try {
|
||||
const rows = await sql`
|
||||
SELECT name, created_at, last_used_at, revoked_at
|
||||
@@ -80,7 +89,7 @@ async function list() {
|
||||
|
||||
async function revoke(name: string) {
|
||||
if (!name) { console.error('Usage: auth revoke <name>'); process.exit(1); }
|
||||
const sql = postgres(DATABASE_URL!);
|
||||
const sql = postgres(getDatabaseUrl(true)!);
|
||||
try {
|
||||
const result = await sql`
|
||||
UPDATE access_tokens SET revoked_at = now()
|
||||
@@ -216,26 +225,38 @@ async function test(url: string, token: string) {
|
||||
console.log(`\n🧠 Your brain is live! (${elapsed}s)`);
|
||||
}
|
||||
|
||||
// CLI dispatch
|
||||
const [cmd, ...args] = process.argv.slice(2);
|
||||
switch (cmd) {
|
||||
case 'create': await create(args[0]); break;
|
||||
case 'list': await list(); break;
|
||||
case 'revoke': await revoke(args[0]); break;
|
||||
case 'test': {
|
||||
const tokenIdx = args.indexOf('--token');
|
||||
const url = args.find(a => !a.startsWith('--') && a !== args[tokenIdx + 1]);
|
||||
const token = tokenIdx >= 0 ? args[tokenIdx + 1] : '';
|
||||
await test(url || '', token || '');
|
||||
break;
|
||||
}
|
||||
default:
|
||||
console.log(`GBrain Token Management
|
||||
/**
|
||||
* Entry point for the `gbrain auth` CLI subcommand. Also reused by the
|
||||
* direct-script path (see bottom of file) so `bun run src/commands/auth.ts`
|
||||
* still works.
|
||||
*/
|
||||
export async function runAuth(args: string[]): Promise<void> {
|
||||
const [cmd, ...rest] = args;
|
||||
switch (cmd) {
|
||||
case 'create': await create(rest[0]); return;
|
||||
case 'list': await list(); return;
|
||||
case 'revoke': await revoke(rest[0]); return;
|
||||
case 'test': {
|
||||
const tokenIdx = rest.indexOf('--token');
|
||||
const url = rest.find(a => !a.startsWith('--') && a !== rest[tokenIdx + 1]);
|
||||
const token = tokenIdx >= 0 ? rest[tokenIdx + 1] : '';
|
||||
await test(url || '', token || '');
|
||||
return;
|
||||
}
|
||||
default:
|
||||
console.log(`GBrain Token Management
|
||||
|
||||
Usage:
|
||||
bun run src/commands/auth.ts create <name> Create a new access token
|
||||
bun run src/commands/auth.ts list List all tokens
|
||||
bun run src/commands/auth.ts revoke <name> Revoke a token
|
||||
bun run src/commands/auth.ts test <url> --token <token> Smoke test a remote MCP server
|
||||
gbrain auth create <name> Create a new access token
|
||||
gbrain auth list List all tokens
|
||||
gbrain auth revoke <name> Revoke a token
|
||||
gbrain auth test <url> --token <t> Smoke-test a remote MCP server
|
||||
`);
|
||||
}
|
||||
}
|
||||
|
||||
// Direct-script entry point — only runs when this file is invoked as the main module
|
||||
// (e.g. `bun run src/commands/auth.ts ...`). When imported by cli.ts, this block is skipped.
|
||||
if (import.meta.main) {
|
||||
await runAuth(process.argv.slice(2));
|
||||
}
|
||||
|
||||
@@ -147,22 +147,41 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
|
||||
let stopping = false;
|
||||
let workerProc: ChildProcess | null = null;
|
||||
let crashCount = 0;
|
||||
let lastWorkerStartTime = 0;
|
||||
|
||||
// Stable-run reset window (matches MinionSupervisor.ts:471-476 pattern). If the
|
||||
// worker ran > 5min before exit, treat as a fresh cycle (crashCount=1) so the
|
||||
// RSS watchdog firing hourly does NOT trip autopilot's give-up threshold after
|
||||
// ~5 hours of healthy uptime.
|
||||
const STABLE_RUN_RESET_MS = 5 * 60 * 1000;
|
||||
|
||||
if (spawnManagedWorker) {
|
||||
const cliPath = resolveGbrainCliPath();
|
||||
const startWorker = () => {
|
||||
const child = spawn(cliPath, ['jobs', 'work'], { stdio: 'inherit', env: process.env });
|
||||
// Inject the RSS watchdog default (2048 MB) for the autopilot-supervised
|
||||
// worker. Bare `gbrain jobs work` has no default; the supervisor and
|
||||
// autopilot are the production paths that opt in.
|
||||
const args = ['jobs', 'work', '--max-rss', '2048'];
|
||||
const child = spawn(cliPath, args, { stdio: 'inherit', env: process.env });
|
||||
workerProc = child;
|
||||
console.log(`[autopilot] Minions worker spawned (pid: ${child.pid})`);
|
||||
lastWorkerStartTime = Date.now();
|
||||
console.log(`[autopilot] Minions worker spawned (pid: ${child.pid}, watchdog: 2048MB)`);
|
||||
child.on('exit', (code) => {
|
||||
workerProc = null;
|
||||
if (stopping) return;
|
||||
const runDuration = Date.now() - lastWorkerStartTime;
|
||||
if (runDuration > STABLE_RUN_RESET_MS) {
|
||||
// Stable run — forgive prior crash history. A watchdog-driven hourly
|
||||
// exit (the production path post-fix) lands here every time.
|
||||
crashCount = 1;
|
||||
} else {
|
||||
crashCount++;
|
||||
}
|
||||
if (crashCount >= 5) {
|
||||
console.error('[autopilot] 5 consecutive worker crashes, giving up.');
|
||||
console.error(`[autopilot] 5 consecutive worker crashes (run ${runDuration}ms), giving up.`);
|
||||
process.exit(1);
|
||||
}
|
||||
crashCount++;
|
||||
console.error(`[autopilot] worker exited code=${code}, restart #${crashCount} in 10s`);
|
||||
console.error(`[autopilot] worker exited code=${code} after ${runDuration}ms, restart #${crashCount} in 10s`);
|
||||
setTimeout(startWorker, 10_000);
|
||||
});
|
||||
};
|
||||
@@ -290,6 +309,12 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
|
||||
idempotency_key: `autopilot-cycle:${slot}`,
|
||||
max_attempts: 2,
|
||||
timeout_ms: timeoutMs,
|
||||
// Submission backpressure: when the worker is dead or wedged,
|
||||
// idempotency_key only dedupes within a slot; cross-slot pile-up
|
||||
// is what produced the 28+ waiting-jobs production incident.
|
||||
// maxWaiting: 1 caps at 1 active + 1 waiting; queue.add coalesces
|
||||
// the 3rd+ submission and writes a backpressure-audit JSONL line.
|
||||
maxWaiting: 1,
|
||||
},
|
||||
);
|
||||
if (jsonMode) {
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
/**
|
||||
* gbrain code-callees <symbol>
|
||||
*
|
||||
* v0.20.0 Cathedral II Layer 10 (C5) — "what does this symbol call?"
|
||||
* Forward view of the A1 call graph. Matches `from_symbol_qualified`
|
||||
* in both code_edges_chunk + code_edges_symbol.
|
||||
*
|
||||
* Output: same JSON-on-non-TTY convention as code-callers / code-def /
|
||||
* code-refs.
|
||||
*/
|
||||
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { errorFor, serializeError } from '../core/errors.ts';
|
||||
|
||||
function parseFlag(args: string[], name: string): string | undefined {
|
||||
const i = args.indexOf(name);
|
||||
return i >= 0 && i + 1 < args.length ? args[i + 1] : undefined;
|
||||
}
|
||||
|
||||
function shouldEmitJson(args: string[]): boolean {
|
||||
if (args.includes('--json')) return true;
|
||||
if (args.includes('--no-json')) return false;
|
||||
return !process.stdout.isTTY;
|
||||
}
|
||||
|
||||
export async function runCodeCallees(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
const positional = args.filter((a) => !a.startsWith('--'));
|
||||
const sym = positional[0];
|
||||
if (!sym) {
|
||||
const err = errorFor({
|
||||
class: 'UsageError',
|
||||
code: 'code_callees_requires_symbol',
|
||||
message: 'code-callees requires a symbol name',
|
||||
hint: 'gbrain code-callees <symbol> [--all-sources] [--limit N] [--json]',
|
||||
});
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ error: err.envelope }));
|
||||
} else {
|
||||
console.error(err.message);
|
||||
}
|
||||
process.exit(2);
|
||||
}
|
||||
const limit = parseInt(parseFlag(args, '--limit') || '100', 10);
|
||||
const allSources = args.includes('--all-sources');
|
||||
const sourceId = parseFlag(args, '--source');
|
||||
|
||||
try {
|
||||
const edges = await engine.getCalleesOf(sym, {
|
||||
limit,
|
||||
allSources: allSources || !sourceId,
|
||||
sourceId: sourceId ?? undefined,
|
||||
});
|
||||
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ symbol: sym, count: edges.length, callees: edges }, null, 2));
|
||||
} else if (edges.length === 0) {
|
||||
console.log(`No callees found for "${sym}".`);
|
||||
} else {
|
||||
console.log(`${edges.length} callee(s) for "${sym}":`);
|
||||
for (const e of edges) {
|
||||
const res = e.resolved ? 'resolved' : 'unresolved';
|
||||
console.log(` ${e.from_symbol_qualified} → ${e.to_symbol_qualified} [${res}]`);
|
||||
}
|
||||
}
|
||||
} catch (e: unknown) {
|
||||
const env = serializeError(e);
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ error: env }));
|
||||
} else {
|
||||
console.error(`code-callees failed: ${env.message}`);
|
||||
}
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
/**
|
||||
* gbrain code-callers <symbol>
|
||||
*
|
||||
* v0.20.0 Cathedral II Layer 10 (C4) — "who calls this symbol?" Reversed
|
||||
* view of the A1 call graph. Matches `to_symbol_qualified` in both
|
||||
* code_edges_chunk (resolved) and code_edges_symbol (unresolved short-name
|
||||
* capture). Layer 5 captures edges at chunk time; Layer 10 exposes them.
|
||||
*
|
||||
* Scope decision: by default we only match the caller's source_id so
|
||||
* multi-repo brains don't cross-resolve (`Admin::UsersController#render`
|
||||
* in repo A ≠ same string in repo B). Pass `--all-sources` to search
|
||||
* globally.
|
||||
*
|
||||
* Output: non-TTY → JSON envelope. TTY → human table. Follows the
|
||||
* code-def / code-refs pattern.
|
||||
*/
|
||||
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { errorFor, serializeError } from '../core/errors.ts';
|
||||
|
||||
function parseFlag(args: string[], name: string): string | undefined {
|
||||
const i = args.indexOf(name);
|
||||
return i >= 0 && i + 1 < args.length ? args[i + 1] : undefined;
|
||||
}
|
||||
|
||||
function shouldEmitJson(args: string[]): boolean {
|
||||
if (args.includes('--json')) return true;
|
||||
if (args.includes('--no-json')) return false;
|
||||
return !process.stdout.isTTY;
|
||||
}
|
||||
|
||||
export async function runCodeCallers(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
const positional = args.filter((a) => !a.startsWith('--'));
|
||||
const sym = positional[0];
|
||||
if (!sym) {
|
||||
const err = errorFor({
|
||||
class: 'UsageError',
|
||||
code: 'code_callers_requires_symbol',
|
||||
message: 'code-callers requires a symbol name',
|
||||
hint: 'gbrain code-callers <symbol> [--all-sources] [--limit N] [--json]',
|
||||
});
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ error: err.envelope }));
|
||||
} else {
|
||||
console.error(err.message);
|
||||
}
|
||||
process.exit(2);
|
||||
}
|
||||
const limit = parseInt(parseFlag(args, '--limit') || '100', 10);
|
||||
const allSources = args.includes('--all-sources');
|
||||
const sourceId = parseFlag(args, '--source');
|
||||
|
||||
try {
|
||||
const edges = await engine.getCallersOf(sym, {
|
||||
limit,
|
||||
allSources: allSources || !sourceId,
|
||||
sourceId: sourceId ?? undefined,
|
||||
});
|
||||
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ symbol: sym, count: edges.length, callers: edges }, null, 2));
|
||||
} else if (edges.length === 0) {
|
||||
console.log(`No callers found for "${sym}".`);
|
||||
} else {
|
||||
console.log(`${edges.length} caller(s) for "${sym}":`);
|
||||
for (const e of edges) {
|
||||
const res = e.resolved ? 'resolved' : 'unresolved';
|
||||
console.log(` ${e.from_symbol_qualified} → ${e.to_symbol_qualified} [${res}]`);
|
||||
}
|
||||
}
|
||||
} catch (e: unknown) {
|
||||
const env = serializeError(e);
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ error: env }));
|
||||
} else {
|
||||
console.error(`code-callers failed: ${env.message}`);
|
||||
}
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,136 @@
|
||||
/**
|
||||
* gbrain code-def <symbol>
|
||||
*
|
||||
* v0.19.0 Layer 7 — look up the definition site(s) of a named symbol
|
||||
* (function, class, type, interface, enum) across every code page the
|
||||
* brain has indexed.
|
||||
*
|
||||
* Output:
|
||||
* - TTY or --pretty: human-readable list of matches, one per line.
|
||||
* - non-TTY or --json: JSON array the agent consumes.
|
||||
*
|
||||
* Uses the content_chunks.symbol_name column (v0.19.0 migration v26).
|
||||
* No tree-sitter re-parsing needed — the metadata is already there.
|
||||
*/
|
||||
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { errorFor, serializeError } from '../core/errors.ts';
|
||||
|
||||
export interface CodeDefResult {
|
||||
slug: string;
|
||||
file: string | null;
|
||||
language: string | null;
|
||||
symbol_type: string | null;
|
||||
start_line: number | null;
|
||||
end_line: number | null;
|
||||
snippet: string;
|
||||
}
|
||||
|
||||
export async function findCodeDef(
|
||||
engine: BrainEngine,
|
||||
symbol: string,
|
||||
opts: { limit?: number; language?: string } = {},
|
||||
): Promise<CodeDefResult[]> {
|
||||
const limit = opts.limit ?? 20;
|
||||
const DEF_TYPES = ['function', 'class', 'interface', 'type', 'enum', 'struct', 'trait', 'module', 'contract'];
|
||||
const params: unknown[] = [symbol, limit];
|
||||
let whereLang = '';
|
||||
if (opts.language) {
|
||||
params.splice(1, 0, opts.language);
|
||||
whereLang = 'AND cc.language = $2';
|
||||
}
|
||||
// Deterministic ordering: exact type matches first (functions before
|
||||
// export_statement wrappers), then page slug, then line number.
|
||||
const rows = await engine.executeRaw<{
|
||||
slug: string; file: string | null; language: string | null;
|
||||
symbol_type: string | null; start_line: number | null; end_line: number | null;
|
||||
chunk_text: string;
|
||||
}>(
|
||||
`SELECT p.slug, (p.frontmatter->>'file') AS file, cc.language, cc.symbol_type,
|
||||
cc.start_line, cc.end_line, cc.chunk_text
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE cc.symbol_name = $1
|
||||
${whereLang}
|
||||
AND p.page_kind = 'code'
|
||||
AND cc.symbol_type IN ('${DEF_TYPES.join("','")}', 'export statement')
|
||||
ORDER BY
|
||||
CASE cc.symbol_type
|
||||
WHEN 'function' THEN 1 WHEN 'class' THEN 2 WHEN 'interface' THEN 3
|
||||
WHEN 'type' THEN 4 WHEN 'enum' THEN 5 WHEN 'struct' THEN 6
|
||||
ELSE 7
|
||||
END,
|
||||
p.slug, cc.start_line
|
||||
LIMIT $${params.length}`,
|
||||
params,
|
||||
);
|
||||
return rows.map((r) => ({
|
||||
slug: r.slug,
|
||||
file: r.file,
|
||||
language: r.language,
|
||||
symbol_type: r.symbol_type,
|
||||
start_line: r.start_line,
|
||||
end_line: r.end_line,
|
||||
// First 500 chars of chunk — enough for a preview without flooding output.
|
||||
snippet: r.chunk_text.slice(0, 500),
|
||||
}));
|
||||
}
|
||||
|
||||
function parseFlag(args: string[], name: string): string | undefined {
|
||||
const i = args.indexOf(name);
|
||||
return i >= 0 && i + 1 < args.length ? args[i + 1] : undefined;
|
||||
}
|
||||
|
||||
function shouldEmitJson(args: string[]): boolean {
|
||||
if (args.includes('--json')) return true;
|
||||
if (args.includes('--no-json')) return false;
|
||||
// Auto-detect: non-TTY stdout means an agent is piping us — default to JSON.
|
||||
return !process.stdout.isTTY;
|
||||
}
|
||||
|
||||
export async function runCodeDef(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
const symbol = args.find((a) => !a.startsWith('--') && args.indexOf(a) > 0);
|
||||
// args[0] is the symbol when invoked as `gbrain code-def <symbol>`
|
||||
const positional = args.filter((a) => !a.startsWith('--'));
|
||||
const sym = positional[0];
|
||||
if (!sym) {
|
||||
const err = errorFor({
|
||||
class: 'UsageError',
|
||||
code: 'code_def_requires_symbol',
|
||||
message: 'code-def requires a symbol name',
|
||||
hint: 'gbrain code-def <symbol> [--lang <language>] [--json]',
|
||||
});
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ error: err.envelope }));
|
||||
} else {
|
||||
console.error(err.message);
|
||||
}
|
||||
process.exit(2);
|
||||
}
|
||||
const limit = parseInt(parseFlag(args, '--limit') || '20', 10);
|
||||
const language = parseFlag(args, '--lang');
|
||||
try {
|
||||
const results = await findCodeDef(engine, sym, { limit, language });
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ symbol: sym, count: results.length, results }, null, 2));
|
||||
} else {
|
||||
if (results.length === 0) {
|
||||
console.log(`No definitions found for "${sym}"`);
|
||||
} else {
|
||||
console.log(`Found ${results.length} definition(s) for "${sym}":`);
|
||||
for (const r of results) {
|
||||
const loc = r.start_line != null ? `:${r.start_line}` : '';
|
||||
console.log(` ${r.file || r.slug}${loc} (${r.symbol_type})`);
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (e: unknown) {
|
||||
const env = serializeError(e);
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ error: env }));
|
||||
} else {
|
||||
console.error(`code-def failed: ${env.message}`);
|
||||
}
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
/**
|
||||
* gbrain code-refs <symbol>
|
||||
*
|
||||
* v0.19.0 Layer 7 — find all usage sites of a named symbol across the
|
||||
* brain's code pages. The DX "magical moment" for v0.19.0: an agent
|
||||
* asks "what uses BrainEngine" and gets back a JSON array of
|
||||
* {file, line, snippet} tuples in one CLI call.
|
||||
*
|
||||
* Implementation: bypasses the standard searchKeyword path (which uses
|
||||
* DISTINCT ON (slug) to collapse to one result per page — wrong for
|
||||
* code-refs where a single file typically has many usage sites). Uses
|
||||
* a direct ILIKE scan over content_chunks + JOIN pages, returning every
|
||||
* matching chunk.
|
||||
*
|
||||
* Scope: simple substring match. Word-boundary precision is a follow-up
|
||||
* (would require either tsvector or regex). For v0.19.0 the heuristic
|
||||
* is good enough: symbol names are distinctive by design, and noisy
|
||||
* matches (e.g. 'foo' matching 'food') are rare in well-written code.
|
||||
*/
|
||||
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { errorFor, serializeError } from '../core/errors.ts';
|
||||
|
||||
export interface CodeRefResult {
|
||||
slug: string;
|
||||
file: string | null;
|
||||
language: string | null;
|
||||
symbol_name: string | null;
|
||||
symbol_type: string | null;
|
||||
start_line: number | null;
|
||||
end_line: number | null;
|
||||
snippet: string;
|
||||
}
|
||||
|
||||
export async function findCodeRefs(
|
||||
engine: BrainEngine,
|
||||
symbol: string,
|
||||
opts: { limit?: number; language?: string } = {},
|
||||
): Promise<CodeRefResult[]> {
|
||||
const limit = opts.limit ?? 50;
|
||||
const params: unknown[] = [`%${symbol}%`];
|
||||
let whereLang = '';
|
||||
if (opts.language) {
|
||||
params.push(opts.language);
|
||||
whereLang = `AND cc.language = $${params.length}`;
|
||||
}
|
||||
params.push(limit);
|
||||
const rows = await engine.executeRaw<{
|
||||
slug: string; file: string | null; language: string | null;
|
||||
symbol_name: string | null; symbol_type: string | null;
|
||||
start_line: number | null; end_line: number | null;
|
||||
chunk_text: string;
|
||||
}>(
|
||||
`SELECT p.slug, (p.frontmatter->>'file') AS file, cc.language,
|
||||
cc.symbol_name, cc.symbol_type, cc.start_line, cc.end_line,
|
||||
cc.chunk_text
|
||||
FROM content_chunks cc
|
||||
JOIN pages p ON p.id = cc.page_id
|
||||
WHERE p.page_kind = 'code'
|
||||
AND cc.chunk_text ILIKE $1
|
||||
${whereLang}
|
||||
ORDER BY p.slug, cc.start_line NULLS LAST
|
||||
LIMIT $${params.length}`,
|
||||
params,
|
||||
);
|
||||
return rows.map((r) => ({
|
||||
slug: r.slug,
|
||||
file: r.file,
|
||||
language: r.language,
|
||||
symbol_name: r.symbol_name,
|
||||
symbol_type: r.symbol_type,
|
||||
start_line: r.start_line,
|
||||
end_line: r.end_line,
|
||||
snippet: r.chunk_text.slice(0, 500),
|
||||
}));
|
||||
}
|
||||
|
||||
function parseFlag(args: string[], name: string): string | undefined {
|
||||
const i = args.indexOf(name);
|
||||
return i >= 0 && i + 1 < args.length ? args[i + 1] : undefined;
|
||||
}
|
||||
|
||||
function shouldEmitJson(args: string[]): boolean {
|
||||
if (args.includes('--json')) return true;
|
||||
if (args.includes('--no-json')) return false;
|
||||
return !process.stdout.isTTY;
|
||||
}
|
||||
|
||||
export async function runCodeRefs(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
const positional = args.filter((a) => !a.startsWith('--'));
|
||||
const sym = positional[0];
|
||||
if (!sym) {
|
||||
const err = errorFor({
|
||||
class: 'UsageError',
|
||||
code: 'code_refs_requires_symbol',
|
||||
message: 'code-refs requires a symbol name',
|
||||
hint: 'gbrain code-refs <symbol> [--lang <language>] [--json]',
|
||||
});
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ error: err.envelope }));
|
||||
} else {
|
||||
console.error(err.message);
|
||||
}
|
||||
process.exit(2);
|
||||
}
|
||||
const limit = parseInt(parseFlag(args, '--limit') || '50', 10);
|
||||
const language = parseFlag(args, '--lang');
|
||||
try {
|
||||
const results = await findCodeRefs(engine, sym, { limit, language });
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ symbol: sym, count: results.length, results }, null, 2));
|
||||
} else {
|
||||
if (results.length === 0) {
|
||||
console.log(`No references found for "${sym}"`);
|
||||
} else {
|
||||
console.log(`Found ${results.length} reference(s) to "${sym}":`);
|
||||
for (const r of results) {
|
||||
const loc = r.start_line != null ? `:${r.start_line}` : '';
|
||||
const sig = r.symbol_name ? ` in ${r.symbol_name}` : '';
|
||||
console.log(` ${r.file || r.slug}${loc}${sig}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (e: unknown) {
|
||||
const env = serializeError(e);
|
||||
if (shouldEmitJson(args)) {
|
||||
console.log(JSON.stringify({ error: env }));
|
||||
} else {
|
||||
console.error(`code-refs failed: ${env.message}`);
|
||||
}
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
+75
-5
@@ -5,6 +5,7 @@ import { checkResolvable } from '../core/check-resolvable.ts';
|
||||
import { autoFixDryViolations, type AutoFixReport, type FixOutcome } from '../core/dry-fix.ts';
|
||||
import { findRepoRoot } from '../core/repo-root.ts';
|
||||
import { loadCompletedMigrations } from '../core/preferences.ts';
|
||||
import { compareVersions } from './migrations/index.ts';
|
||||
import { createProgress, startHeartbeat, type ProgressReporter } from '../core/progress.ts';
|
||||
import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts';
|
||||
import type { DbUrlSource } from '../core/config.ts';
|
||||
@@ -110,6 +111,15 @@ export async function runDoctor(engine: BrainEngine | null, args: string[], dbSo
|
||||
// Typical cause: v0.11.0 stopgap wrote a partial record but nobody ran
|
||||
// `gbrain apply-migrations --yes` afterward. This check fires on every
|
||||
// `gbrain doctor` invocation so your OpenClaw's health skill catches it.
|
||||
//
|
||||
// Forward-progress override: a partial entry for vX.Y.Z is treated as
|
||||
// stale (not stuck) if there is a `complete` entry for any vA.B.C >= vX.Y.Z
|
||||
// anywhere in the file. The reasoning: if a newer migration successfully
|
||||
// landed, the install moved past the older partial — the old record is
|
||||
// historical noise from a stopgap that never finished cleanly, but the
|
||||
// schema clearly advanced. Without this, every install that went through
|
||||
// a v0.11.0 stopgap and then upgraded carries the "MINIONS HALF-INSTALLED"
|
||||
// flag forever, even on installs that have been at v0.22+ for months.
|
||||
try {
|
||||
const completed = loadCompletedMigrations();
|
||||
const byVersion = new Map<string, { complete: boolean; partial: boolean }>();
|
||||
@@ -119,8 +129,17 @@ export async function runDoctor(engine: BrainEngine | null, args: string[], dbSo
|
||||
if (entry.status === 'partial') seen.partial = true;
|
||||
byVersion.set(entry.version, seen);
|
||||
}
|
||||
const completedVersions = Array.from(byVersion.entries())
|
||||
.filter(([, s]) => s.complete)
|
||||
.map(([v]) => v);
|
||||
const stuck = Array.from(byVersion.entries())
|
||||
.filter(([, s]) => s.partial && !s.complete)
|
||||
.filter(([v, s]) => {
|
||||
if (!s.partial || s.complete) return false;
|
||||
// Forward-progress override: if any version >= v has completed, the
|
||||
// partial is stale. compareVersions returns 1 when first arg is newer.
|
||||
const supersededBy = completedVersions.find(cv => compareVersions(cv, v) >= 0);
|
||||
return supersededBy === undefined;
|
||||
})
|
||||
.map(([v]) => v);
|
||||
if (stuck.length > 0) {
|
||||
checks.push({
|
||||
@@ -230,25 +249,29 @@ export async function runDoctor(engine: BrainEngine | null, args: string[], dbSo
|
||||
// Without this doctor check, users see "sync blocked" and have no
|
||||
// surface showing which files to fix.
|
||||
try {
|
||||
const { unacknowledgedSyncFailures, loadSyncFailures } = await import('../core/sync.ts');
|
||||
const { unacknowledgedSyncFailures, loadSyncFailures, summarizeFailuresByCode } = await import('../core/sync.ts');
|
||||
const unacked = unacknowledgedSyncFailures();
|
||||
const all = loadSyncFailures();
|
||||
if (unacked.length > 0) {
|
||||
const codeSummary = summarizeFailuresByCode(unacked);
|
||||
const codeBreakdown = codeSummary.map(s => `${s.code}=${s.count}`).join(', ');
|
||||
const preview = unacked.slice(0, 3).map(f => `${f.path} (${f.error.slice(0, 60)})`).join('; ');
|
||||
checks.push({
|
||||
name: 'sync_failures',
|
||||
status: 'warn',
|
||||
message:
|
||||
`${unacked.length} unacknowledged sync failure(s). ${preview}` +
|
||||
`${unacked.length} unacknowledged sync failure(s) [${codeBreakdown}]. ${preview}` +
|
||||
`${unacked.length > 3 ? `, and ${unacked.length - 3} more` : ''}. ` +
|
||||
`Fix the file(s) and re-run 'gbrain sync', or use 'gbrain sync --skip-failed' to acknowledge.`,
|
||||
});
|
||||
} else if (all.length > 0) {
|
||||
// Acknowledged-only: informational, not a warning.
|
||||
// Acknowledged-only: show code breakdown for visibility.
|
||||
const ackedSummary = summarizeFailuresByCode(all);
|
||||
const ackedBreakdown = ackedSummary.map(s => `${s.code}=${s.count}`).join(', ');
|
||||
checks.push({
|
||||
name: 'sync_failures',
|
||||
status: 'ok',
|
||||
message: `${all.length} historical sync failure(s), all acknowledged.`,
|
||||
message: `${all.length} historical sync failure(s), all acknowledged [${ackedBreakdown}].`,
|
||||
});
|
||||
}
|
||||
} catch {
|
||||
@@ -649,6 +672,53 @@ export async function runDoctor(engine: BrainEngine | null, args: string[], dbSo
|
||||
mbcHb();
|
||||
}
|
||||
|
||||
// 11a. Frontmatter integrity (v0.22.4).
|
||||
// scanBrainSources walks every registered source's local_path on disk
|
||||
// (not from the DB), invoking parseMarkdown(..., {validate:true}) per
|
||||
// file. Reports per-source counts grouped by error code. The fix path is
|
||||
// `gbrain frontmatter validate <source-path> --fix`, which writes .bak
|
||||
// backups so it works for both git and non-git brain repos.
|
||||
progress.heartbeat('frontmatter_integrity');
|
||||
const fmHb = startHeartbeat(progress, 'scanning frontmatter…');
|
||||
try {
|
||||
const { scanBrainSources } = await import('../core/brain-writer.ts');
|
||||
const report = await scanBrainSources(engine);
|
||||
if (report.total === 0) {
|
||||
const sources = report.per_source.length;
|
||||
checks.push({
|
||||
name: 'frontmatter_integrity',
|
||||
status: 'ok',
|
||||
message: sources === 0
|
||||
? 'No registered sources to scan'
|
||||
: `${sources} source(s) clean — no frontmatter issues`,
|
||||
});
|
||||
} else {
|
||||
const sourceMessages: string[] = [];
|
||||
for (const src of report.per_source) {
|
||||
if (src.total === 0) continue;
|
||||
const codes = Object.entries(src.errors_by_code)
|
||||
.map(([k, v]) => `${k}=${v}`)
|
||||
.join(', ');
|
||||
sourceMessages.push(`${src.source_id}: ${src.total} (${codes})`);
|
||||
}
|
||||
checks.push({
|
||||
name: 'frontmatter_integrity',
|
||||
status: 'warn',
|
||||
message:
|
||||
`${report.total} frontmatter issue(s) across ${sourceMessages.length} source(s). ` +
|
||||
`${sourceMessages.join('; ')}. Fix: gbrain frontmatter validate <source-path> --fix`,
|
||||
});
|
||||
}
|
||||
} catch (e) {
|
||||
checks.push({
|
||||
name: 'frontmatter_integrity',
|
||||
status: 'warn',
|
||||
message: `Could not scan frontmatter: ${e instanceof Error ? e.message : String(e)}`,
|
||||
});
|
||||
} finally {
|
||||
fmHb();
|
||||
}
|
||||
|
||||
// 11b. Queue health (v0.19.1 queue-resilience wave).
|
||||
// Postgres-only because PGLite has no multi-process worker surface. Two
|
||||
// subchecks, both cheap (single SELECT each, status-index-covered):
|
||||
|
||||
+135
-3
@@ -220,6 +220,23 @@ async function embedAll(
|
||||
result: EmbedResult,
|
||||
onProgress?: (done: number, total: number, embedded: number) => void,
|
||||
) {
|
||||
// ─────────────────────────────────────────────────────────────
|
||||
// Stale-only fast path: avoid the listPages + per-page getChunks
|
||||
// bomb that pulled every page row + every chunk's embedding column
|
||||
// (~76 MB on a 1.5K-page brain) only to client-side-filter for
|
||||
// chunks where embedding IS NULL. The new path issues one SQL
|
||||
// pre-check + at most one slug-grouped SELECT excluding the
|
||||
// (always-null on stale rows) embedding column. On a 100%-embedded
|
||||
// brain (the autopilot common case) we exit after ~50 bytes wire.
|
||||
//
|
||||
// For --all (staleOnly=false) we keep the original behavior — the
|
||||
// user is explicitly asking to re-embed everything, including
|
||||
// chunks that already have embeddings.
|
||||
// ─────────────────────────────────────────────────────────────
|
||||
if (staleOnly) {
|
||||
return await embedAllStale(engine, dryRun, result, onProgress);
|
||||
}
|
||||
|
||||
const pages = await engine.listPages({ limit: 100000 });
|
||||
let processed = 0;
|
||||
|
||||
@@ -235,9 +252,7 @@ async function embedAll(
|
||||
|
||||
async function embedOnePage(page: typeof pages[number]) {
|
||||
const chunks = await engine.getChunks(page.slug);
|
||||
const toEmbed = staleOnly
|
||||
? chunks.filter(c => !c.embedded_at)
|
||||
: chunks;
|
||||
const toEmbed = chunks; // staleOnly path handled above via embedAllStale
|
||||
|
||||
result.total_chunks += chunks.length;
|
||||
result.skipped += chunks.length - toEmbed.length;
|
||||
@@ -306,3 +321,120 @@ async function embedAll(
|
||||
console.log(`Embedded ${result.embedded} chunks across ${pages.length} pages`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* SQL-side stale path: replaces the listPages + per-page getChunks
|
||||
* walk with a count + slug-grouped SELECT. Preserves the existing
|
||||
* functional contract (every chunk where embedding IS NULL gets
|
||||
* embedded; nothing else is touched) without paying egress on
|
||||
* already-embedded chunks.
|
||||
*
|
||||
* Why a separate function: the staleOnly path doesn't need
|
||||
* listPages at all and groups by slug differently. Forking the
|
||||
* function makes the read-bytes path explicit and keeps the --all
|
||||
* path verbatim from prior behavior.
|
||||
*
|
||||
* Staleness predicate: `embedding IS NULL`. We deliberately do NOT
|
||||
* use `embedded_at IS NULL` here — the bulk-import path can leave
|
||||
* embedded_at populated while embedding is NULL (see upsertChunks
|
||||
* consistency notes), and `embedding IS NULL` is the truth source
|
||||
* for "this chunk needs an embedding".
|
||||
*/
|
||||
async function embedAllStale(
|
||||
engine: BrainEngine,
|
||||
dryRun: boolean,
|
||||
result: EmbedResult,
|
||||
onProgress?: (done: number, total: number, embedded: number) => void,
|
||||
) {
|
||||
// Pre-flight: 0 stale chunks → nothing to do, no further DB reads.
|
||||
// Cheapest possible exit on the autopilot common case.
|
||||
const staleCount = await engine.countStaleChunks();
|
||||
if (staleCount === 0) {
|
||||
if (dryRun) {
|
||||
console.log('[dry-run] Would embed 0 chunks (0 stale found)');
|
||||
} else {
|
||||
console.log('Embedded 0 chunks (0 stale found)');
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Pull only the stale chunks (no embedding column).
|
||||
const staleRows = await engine.listStaleChunks();
|
||||
// Group by slug so each slug → array of stale chunks for batched embedding.
|
||||
const bySlug = new Map<string, typeof staleRows>();
|
||||
for (const row of staleRows) {
|
||||
const list = bySlug.get(row.slug);
|
||||
if (list) list.push(row);
|
||||
else bySlug.set(row.slug, [row]);
|
||||
}
|
||||
|
||||
const slugs = Array.from(bySlug.keys());
|
||||
const totalStaleChunks = staleRows.length;
|
||||
result.total_chunks += totalStaleChunks;
|
||||
// skipped is "chunks we considered and skipped due to having an embedding".
|
||||
// We never considered the non-stale chunks here, so leave skipped at 0.
|
||||
// Callers reading EmbedResult who care about coverage should call
|
||||
// engine.getStats() / engine.getHealth() afterward.
|
||||
|
||||
if (dryRun) {
|
||||
result.would_embed += totalStaleChunks;
|
||||
result.pages_processed += slugs.length;
|
||||
if (onProgress) {
|
||||
// Emit a single tick to satisfy the contract (CLI progress reporters
|
||||
// expect at least one start/finish pair).
|
||||
onProgress(slugs.length, slugs.length, 0);
|
||||
}
|
||||
console.log(`[dry-run] Would embed ${totalStaleChunks} chunks across ${slugs.length} pages`);
|
||||
return;
|
||||
}
|
||||
|
||||
const CONCURRENCY = parseInt(process.env.GBRAIN_EMBED_CONCURRENCY || '20', 10);
|
||||
let processed = 0;
|
||||
|
||||
async function embedOneSlug(slug: string) {
|
||||
const stale = bySlug.get(slug)!;
|
||||
try {
|
||||
const embeddings = await embedBatch(stale.map(c => c.chunk_text));
|
||||
// CRITICAL: passing ONLY the stale indices to upsertChunks would
|
||||
// delete every non-stale chunk on the same page (the != ALL filter
|
||||
// wipes any chunk_index NOT in the input). To preserve them, we
|
||||
// re-fetch existing chunks for this page and merge. Bounded by the
|
||||
// stale slug count, not by total slugs — autopilot common case
|
||||
// is 0 stale (pre-flight short-circuit, never reaches this path).
|
||||
const existing = await engine.getChunks(slug);
|
||||
const staleIdxToEmbedding = new Map<number, Float32Array>();
|
||||
for (let j = 0; j < stale.length; j++) {
|
||||
staleIdxToEmbedding.set(stale[j].chunk_index, embeddings[j]);
|
||||
}
|
||||
const merged: ChunkInput[] = existing.map(c => ({
|
||||
chunk_index: c.chunk_index,
|
||||
chunk_text: c.chunk_text,
|
||||
chunk_source: c.chunk_source,
|
||||
// For stale chunks: pass the new embedding.
|
||||
// For non-stale chunks: pass undefined → COALESCE preserves existing embedding.
|
||||
embedding: staleIdxToEmbedding.get(c.chunk_index) ?? undefined,
|
||||
token_count: c.token_count || Math.ceil(c.chunk_text.length / 4),
|
||||
}));
|
||||
await engine.upsertChunks(slug, merged);
|
||||
result.embedded += stale.length;
|
||||
} catch (e: unknown) {
|
||||
console.error(`\n Error embedding ${slug}: ${e instanceof Error ? e.message : e}`);
|
||||
}
|
||||
processed++;
|
||||
result.pages_processed++;
|
||||
onProgress?.(processed, slugs.length, result.embedded);
|
||||
}
|
||||
|
||||
let nextIdx = 0;
|
||||
async function worker() {
|
||||
while (nextIdx < slugs.length) {
|
||||
const idx = nextIdx++;
|
||||
await embedOneSlug(slugs[idx]);
|
||||
}
|
||||
}
|
||||
|
||||
const numWorkers = Math.min(CONCURRENCY, slugs.length);
|
||||
await Promise.all(Array.from({ length: numWorkers }, () => worker()));
|
||||
|
||||
console.log(`Embedded ${result.embedded} chunks across ${slugs.length} pages`);
|
||||
}
|
||||
|
||||
+97
-4
@@ -1,16 +1,105 @@
|
||||
import { writeFileSync, mkdirSync } from 'fs';
|
||||
import { writeFileSync, mkdirSync, existsSync } from 'fs';
|
||||
import { join, dirname } from 'path';
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { serializeMarkdown } from '../core/markdown.ts';
|
||||
import { createProgress } from '../core/progress.ts';
|
||||
import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts';
|
||||
import { loadStorageConfig, isDbOnly } from '../core/storage-config.ts';
|
||||
import { getDefaultSourcePath } from '../core/source-resolver.ts';
|
||||
import type { PageType } from '../core/types.ts';
|
||||
|
||||
export async function runExport(engine: BrainEngine, args: string[]) {
|
||||
const dirIdx = args.indexOf('--dir');
|
||||
const outDir = dirIdx !== -1 ? args[dirIdx + 1] : './export';
|
||||
|
||||
const pages = await engine.listPages({ limit: 100000 });
|
||||
console.log(`Exporting ${pages.length} pages to ${outDir}/`);
|
||||
const repoIdx = args.indexOf('--repo');
|
||||
const explicitRepoPath = repoIdx !== -1 ? args[repoIdx + 1] : null;
|
||||
|
||||
const typeIdx = args.indexOf('--type');
|
||||
const typeFilter = typeIdx !== -1 ? (args[typeIdx + 1] as PageType) : undefined;
|
||||
|
||||
const slugPrefixIdx = args.indexOf('--slug-prefix');
|
||||
const slugPrefix = slugPrefixIdx !== -1 ? args[slugPrefixIdx + 1] : undefined;
|
||||
|
||||
const restoreOnly = args.includes('--restore-only');
|
||||
|
||||
// Resolution chain (D5): explicit --repo → typed sources.getDefault() →
|
||||
// hard-error for restore-only paths (never fall through to cwd).
|
||||
// For non-restore exports, repoPath stays null because regular export
|
||||
// doesn't need a brain repo to run (D26 — exports include everything).
|
||||
let repoPath: string | null = explicitRepoPath;
|
||||
if (restoreOnly && !repoPath) {
|
||||
repoPath = await getDefaultSourcePath(engine);
|
||||
if (!repoPath) {
|
||||
console.error(
|
||||
`Error: gbrain export --restore-only requires --repo <path> or a configured\n` +
|
||||
`default source with a local_path. Run \`gbrain sources list\` to inspect\n` +
|
||||
`sources, or pass --repo explicitly.`,
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
// Load storage configuration if repo path is provided
|
||||
const storageConfig = repoPath ? loadStorageConfig(repoPath) : null;
|
||||
|
||||
// D5 + Codex P0: refuse --restore-only when there's no storage config to
|
||||
// scope the restore. Without storageConfig, the selective filter (db_only
|
||||
// pages missing on disk) can't run, and falling through to the full
|
||||
// listPages export silently dumps the entire DB. Catch this before any
|
||||
// page query fires.
|
||||
if (restoreOnly && !storageConfig) {
|
||||
console.error(
|
||||
`Error: gbrain export --restore-only requires a storage tiering config\n` +
|
||||
`(gbrain.yml with a "storage:" section) at ${repoPath}/gbrain.yml.\n` +
|
||||
`Without it, there's nothing to scope the restore to.\n` +
|
||||
`Run \`gbrain storage status\` to inspect the current configuration.`,
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// Build filters. slugPrefix is engine-side (Issue #13) — no in-memory
|
||||
// post-filter, no full-table load.
|
||||
const filters: import('../core/types.ts').PageFilters = { limit: 100000 };
|
||||
if (typeFilter) filters.type = typeFilter;
|
||||
if (slugPrefix) filters.slugPrefix = slugPrefix;
|
||||
|
||||
let pages: import('../core/types.ts').Page[];
|
||||
|
||||
// Restore-only path: query each db_only directory with slugPrefix instead
|
||||
// of loading every page in the brain. On a 200K-page brain where 95% is
|
||||
// db_only, this is roughly the same load — but on brains where only 5K
|
||||
// out of 200K are db_only, this is a ~40x reduction.
|
||||
if (restoreOnly && repoPath && storageConfig) {
|
||||
const seen = new Set<string>();
|
||||
pages = [];
|
||||
for (const dir of storageConfig.db_only) {
|
||||
const tierFilters: import('../core/types.ts').PageFilters = {
|
||||
...filters,
|
||||
slugPrefix: filters.slugPrefix
|
||||
? // If user passed --slug-prefix, only include tier dirs that start with it.
|
||||
(dir.startsWith(filters.slugPrefix) ? dir : undefined)
|
||||
: dir,
|
||||
};
|
||||
if (!tierFilters.slugPrefix) continue;
|
||||
const tierPages = await engine.listPages(tierFilters);
|
||||
for (const p of tierPages) {
|
||||
if (seen.has(p.slug)) continue;
|
||||
seen.add(p.slug);
|
||||
if (!isDbOnly(p.slug, storageConfig)) continue; // belt-and-suspenders
|
||||
const filePath = join(repoPath, p.slug + '.md');
|
||||
if (existsSync(filePath)) continue;
|
||||
pages.push(p);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
pages = await engine.listPages(filters);
|
||||
}
|
||||
if (restoreOnly) {
|
||||
console.log(`Restoring ${pages.length} db_only pages to ${outDir}/`);
|
||||
} else {
|
||||
console.log(`Exporting ${pages.length} pages to ${outDir}/`);
|
||||
}
|
||||
|
||||
// Progress on stderr so stdout stays clean for scripts parsing counts.
|
||||
const progress = createProgress(cliOptsToProgressOptions(getCliOptions()));
|
||||
@@ -52,5 +141,9 @@ export async function runExport(engine: BrainEngine, args: string[]) {
|
||||
|
||||
progress.finish();
|
||||
// Stdout summary preserved so scripts that grep for "Exported N pages" keep working.
|
||||
console.log(`Exported ${exported} pages to ${outDir}/`);
|
||||
if (restoreOnly) {
|
||||
console.log(`Restored ${exported} pages to ${outDir}/`);
|
||||
} else {
|
||||
console.log(`Exported ${exported} pages to ${outDir}/`);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -295,6 +295,13 @@ export interface ExtractOpts {
|
||||
dryRun?: boolean;
|
||||
/** Emit JSON (progress to stderr, result to stdout) instead of human text. */
|
||||
jsonMode?: boolean;
|
||||
/**
|
||||
* Incremental mode: only extract from these specific slugs.
|
||||
* When provided, skips the full directory walk and reads only the
|
||||
* files corresponding to these slugs. Massive perf win on large brains.
|
||||
* Pass undefined or omit for a full walk (CLI / first-run path).
|
||||
*/
|
||||
slugs?: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -315,6 +322,21 @@ export async function runExtractCore(engine: BrainEngine, opts: ExtractOpts): Pr
|
||||
const jsonMode = !!opts.jsonMode;
|
||||
const result: ExtractResult = { links_created: 0, timeline_entries_created: 0, pages_processed: 0 };
|
||||
|
||||
// Incremental path: if specific slugs provided, only extract from those files.
|
||||
// This is the cycle path — sync tells us what changed, we only re-extract those.
|
||||
if (opts.slugs !== undefined) {
|
||||
if (opts.slugs.length === 0) {
|
||||
// Nothing changed — skip entirely.
|
||||
return result;
|
||||
}
|
||||
const r = await extractForSlugs(engine, opts.dir, opts.slugs, opts.mode, dryRun, jsonMode);
|
||||
result.links_created = r.links_created;
|
||||
result.timeline_entries_created = r.timeline_created;
|
||||
result.pages_processed = r.pages;
|
||||
return result;
|
||||
}
|
||||
|
||||
// Full walk path: CLI `gbrain extract` or first-run.
|
||||
if (opts.mode === 'links' || opts.mode === 'all') {
|
||||
const r = await extractLinksFromDir(engine, opts.dir, dryRun, jsonMode);
|
||||
result.links_created = r.created;
|
||||
@@ -411,6 +433,118 @@ export async function runExtract(engine: BrainEngine, args: string[]) {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Incremental extract: process only the specified slugs.
|
||||
*
|
||||
* Instead of walking 54K+ files, reads only the files that sync says changed.
|
||||
* Still needs the full slug set for link resolution (resolveSlug needs to know
|
||||
* all valid targets), but that's a single readdir, not 54K readFileSync calls.
|
||||
*
|
||||
* Combines links + timeline extraction in a single pass over each file —
|
||||
* the full-walk path reads every file TWICE (once for links, once for timeline).
|
||||
*/
|
||||
async function extractForSlugs(
|
||||
engine: BrainEngine,
|
||||
brainDir: string,
|
||||
slugs: string[],
|
||||
mode: 'links' | 'timeline' | 'all',
|
||||
dryRun: boolean,
|
||||
jsonMode: boolean,
|
||||
): Promise<{ links_created: number; timeline_created: number; pages: number }> {
|
||||
// Build the full slug set for link resolution (fast: just readdir, no file reads)
|
||||
const allFiles = walkMarkdownFiles(brainDir);
|
||||
const allSlugs = new Set(allFiles.map(f => f.relPath.replace('.md', '')));
|
||||
|
||||
const doLinks = mode === 'links' || mode === 'all';
|
||||
const doTimeline = mode === 'timeline' || mode === 'all';
|
||||
|
||||
const progress = createProgress(cliOptsToProgressOptions(getCliOptions()));
|
||||
progress.start('extract.incremental', slugs.length);
|
||||
|
||||
let linksCreated = 0;
|
||||
let timelineCreated = 0;
|
||||
let pagesProcessed = 0;
|
||||
|
||||
const linkBatch: LinkBatchInput[] = [];
|
||||
const timelineBatch: TimelineBatchInput[] = [];
|
||||
|
||||
async function flushLinks() {
|
||||
if (linkBatch.length === 0) return;
|
||||
try {
|
||||
linksCreated += await engine.addLinksBatch(linkBatch);
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
if (!jsonMode) console.error(` link batch error (${linkBatch.length} rows lost): ${msg}`);
|
||||
} finally {
|
||||
linkBatch.length = 0;
|
||||
}
|
||||
}
|
||||
|
||||
async function flushTimeline() {
|
||||
if (timelineBatch.length === 0) return;
|
||||
try {
|
||||
timelineCreated += await engine.addTimelineEntriesBatch(timelineBatch);
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
if (!jsonMode) console.error(` timeline batch error (${timelineBatch.length} rows lost): ${msg}`);
|
||||
} finally {
|
||||
timelineBatch.length = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (const slug of slugs) {
|
||||
const relPath = slug + '.md';
|
||||
const fullPath = join(brainDir, relPath);
|
||||
|
||||
try {
|
||||
if (!existsSync(fullPath)) continue; // deleted file — sync already handled removal
|
||||
const content = readFileSync(fullPath, 'utf-8');
|
||||
|
||||
// Links
|
||||
if (doLinks) {
|
||||
const links = await extractLinksFromFile(content, relPath, allSlugs);
|
||||
for (const link of links) {
|
||||
if (dryRun) {
|
||||
if (!jsonMode) console.log(` ${link.from_slug} → ${link.to_slug} (${link.link_type})`);
|
||||
linksCreated++;
|
||||
} else {
|
||||
linkBatch.push(link);
|
||||
if (linkBatch.length >= BATCH_SIZE) await flushLinks();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Timeline
|
||||
if (doTimeline) {
|
||||
const entries = extractTimelineFromContent(content, slug);
|
||||
for (const entry of entries) {
|
||||
if (dryRun) {
|
||||
if (!jsonMode) console.log(` ${entry.slug}: ${entry.date} — ${entry.summary}`);
|
||||
timelineCreated++;
|
||||
} else {
|
||||
timelineBatch.push({ slug: entry.slug, date: entry.date, source: entry.source, summary: entry.summary, detail: entry.detail });
|
||||
if (timelineBatch.length >= BATCH_SIZE) await flushTimeline();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pagesProcessed++;
|
||||
} catch { /* skip unreadable */ }
|
||||
progress.tick(1);
|
||||
}
|
||||
|
||||
await flushLinks();
|
||||
await flushTimeline();
|
||||
progress.finish();
|
||||
|
||||
if (!jsonMode) {
|
||||
const label = dryRun ? '(dry run) would create' : 'created';
|
||||
console.log(`Incremental extract: ${label} ${linksCreated} link(s), ${timelineCreated} timeline entries from ${pagesProcessed}/${slugs.length} page(s)`);
|
||||
}
|
||||
|
||||
return { links_created: linksCreated, timeline_created: timelineCreated, pages: pagesProcessed };
|
||||
}
|
||||
|
||||
async function extractLinksFromDir(
|
||||
engine: BrainEngine, brainDir: string, dryRun: boolean, jsonMode: boolean,
|
||||
): Promise<{ created: number; pages: number }> {
|
||||
|
||||
@@ -0,0 +1,216 @@
|
||||
/**
|
||||
* gbrain frontmatter install-hook — Install a pre-commit hook in a brain
|
||||
* source's git repo that runs `gbrain frontmatter validate` against staged
|
||||
* .md/.mdx files. Skips non-git sources with a one-line note.
|
||||
*
|
||||
* Usage:
|
||||
* gbrain frontmatter install-hook [--source <id>] [--force] [--uninstall]
|
||||
*
|
||||
* --source <id> Limit to one registered source. Default: all sources.
|
||||
* --force Overwrite an existing pre-commit hook (writes <hook>.bak).
|
||||
* --uninstall Remove the hook; restore <hook>.bak if present.
|
||||
*
|
||||
* Hook contract:
|
||||
* - Located at <source>/.githooks/pre-commit. We `git config core.hooksPath
|
||||
* .githooks` if no other hooksPath is set.
|
||||
* - When the gbrain binary is missing, the hook prints a one-line warning
|
||||
* and exits 0 (don't break commits if a developer uninstalls gbrain).
|
||||
* - Bypass via `git commit --no-verify`.
|
||||
*/
|
||||
|
||||
import { existsSync, readFileSync, writeFileSync, mkdirSync, chmodSync, rmSync, copyFileSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
import { execFileSync } from 'child_process';
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { loadConfig, toEngineConfig } from '../core/config.ts';
|
||||
import { createEngine } from '../core/engine-factory.ts';
|
||||
|
||||
const HOOK_BANNER = '# gbrain frontmatter pre-commit hook (v0.22.4+)';
|
||||
|
||||
const HOOK_SCRIPT = `#!/bin/sh
|
||||
${HOOK_BANNER}
|
||||
# Validates YAML frontmatter on staged .md / .mdx files. Bypass with
|
||||
# 'git commit --no-verify'. Uninstall with 'gbrain frontmatter install-hook --uninstall'.
|
||||
|
||||
set -e
|
||||
|
||||
if ! command -v gbrain >/dev/null 2>&1; then
|
||||
echo "gbrain not on PATH; skipping frontmatter pre-commit (install gbrain to re-enable)." >&2
|
||||
exit 0
|
||||
fi
|
||||
|
||||
staged=$(git diff --cached --name-only --diff-filter=ACM | grep -E '\\\\.mdx?$' || true)
|
||||
[ -z "$staged" ] && exit 0
|
||||
|
||||
failed=0
|
||||
for f in $staged; do
|
||||
[ -f "$f" ] || continue
|
||||
if ! gbrain frontmatter validate "$f" >/dev/null 2>&1; then
|
||||
gbrain frontmatter validate "$f" >&2
|
||||
failed=1
|
||||
fi
|
||||
done
|
||||
|
||||
if [ $failed -ne 0 ]; then
|
||||
echo "" >&2
|
||||
echo "Frontmatter validation failed. Run 'gbrain frontmatter validate <file> --fix' to repair, or 'git commit --no-verify' to bypass." >&2
|
||||
exit 1
|
||||
fi
|
||||
`;
|
||||
|
||||
interface SourceRow {
|
||||
id: string;
|
||||
local_path: string | null;
|
||||
}
|
||||
|
||||
export async function runFrontmatterInstallHook(args: string[]): Promise<void> {
|
||||
let force = false;
|
||||
let uninstall = false;
|
||||
let sourceId: string | undefined;
|
||||
let help = false;
|
||||
for (let i = 0; i < args.length; i++) {
|
||||
const a = args[i];
|
||||
if (a === '--help' || a === '-h') help = true;
|
||||
else if (a === '--force') force = true;
|
||||
else if (a === '--uninstall') uninstall = true;
|
||||
else if (a === '--source') sourceId = args[++i];
|
||||
else if (a.startsWith('--source=')) sourceId = a.slice('--source='.length);
|
||||
}
|
||||
|
||||
if (help) {
|
||||
printHelp();
|
||||
return;
|
||||
}
|
||||
|
||||
const config = loadConfig();
|
||||
if (!config) {
|
||||
throw new Error('No brain configured. Run: gbrain init');
|
||||
}
|
||||
const engineConfig = toEngineConfig(config);
|
||||
const engine = await createEngine(engineConfig);
|
||||
await engine.connect(engineConfig);
|
||||
try {
|
||||
const sources = await listSources(engine, sourceId);
|
||||
if (sources.length === 0) {
|
||||
console.log(sourceId
|
||||
? `Source "${sourceId}" not found.`
|
||||
: 'No registered sources. Run `gbrain sources list` to inspect.');
|
||||
return;
|
||||
}
|
||||
|
||||
let installed = 0;
|
||||
let skipped = 0;
|
||||
for (const src of sources) {
|
||||
if (!src.local_path || !existsSync(src.local_path)) {
|
||||
console.log(`[${src.id}] skipped — local_path missing on disk`);
|
||||
skipped++;
|
||||
continue;
|
||||
}
|
||||
if (!isGitRepo(src.local_path)) {
|
||||
console.log(`[${src.id}] ${src.local_path} — skipped, not a git repo`);
|
||||
skipped++;
|
||||
continue;
|
||||
}
|
||||
if (uninstall) {
|
||||
if (uninstallHook(src.local_path)) {
|
||||
console.log(`[${src.id}] hook removed`);
|
||||
installed++;
|
||||
} else {
|
||||
console.log(`[${src.id}] no gbrain pre-commit hook found; nothing to uninstall`);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
const result = installHook(src.local_path, force);
|
||||
if (result === 'installed') {
|
||||
console.log(`[${src.id}] hook installed at .githooks/pre-commit`);
|
||||
installed++;
|
||||
} else if (result === 'skipped_existing') {
|
||||
console.log(`[${src.id}] existing pre-commit hook found; pass --force to overwrite (.bak created)`);
|
||||
skipped++;
|
||||
} else {
|
||||
console.log(`[${src.id}] hook already up to date`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`\nDone. ${installed} ${uninstall ? 'removed' : 'installed/updated'}, ${skipped} skipped.`);
|
||||
} finally {
|
||||
await engine.disconnect();
|
||||
}
|
||||
}
|
||||
|
||||
function printHelp() {
|
||||
console.log(`gbrain frontmatter install-hook — install pre-commit hook in source git repos
|
||||
|
||||
Usage:
|
||||
gbrain frontmatter install-hook [--source <id>] [--force] [--uninstall]
|
||||
|
||||
The hook runs \`gbrain frontmatter validate\` against staged .md/.mdx files,
|
||||
blocking commits with malformed frontmatter. Bypass with 'git commit --no-verify'.
|
||||
|
||||
Options:
|
||||
--source <id> Limit to one registered source. Default: all sources.
|
||||
--force Overwrite an existing pre-commit hook (writes <hook>.bak).
|
||||
--uninstall Remove the hook; restore <hook>.bak if present.
|
||||
`);
|
||||
}
|
||||
|
||||
async function listSources(engine: BrainEngine, sourceId?: string): Promise<SourceRow[]> {
|
||||
if (sourceId) {
|
||||
return engine.executeRaw<SourceRow>(`SELECT id, local_path FROM sources WHERE id = $1`, [sourceId]);
|
||||
}
|
||||
return engine.executeRaw<SourceRow>(`SELECT id, local_path FROM sources WHERE local_path IS NOT NULL ORDER BY id`);
|
||||
}
|
||||
|
||||
function isGitRepo(dir: string): boolean {
|
||||
return existsSync(join(dir, '.git'));
|
||||
}
|
||||
|
||||
type InstallResult = 'installed' | 'skipped_existing' | 'unchanged';
|
||||
|
||||
export function installHook(repoPath: string, force: boolean): InstallResult {
|
||||
const hooksDir = join(repoPath, '.githooks');
|
||||
const hookPath = join(hooksDir, 'pre-commit');
|
||||
mkdirSync(hooksDir, { recursive: true });
|
||||
|
||||
if (existsSync(hookPath)) {
|
||||
const existing = readFileSync(hookPath, 'utf8');
|
||||
if (existing.includes(HOOK_BANNER)) {
|
||||
// Already a gbrain hook — refresh the script content silently.
|
||||
writeFileSync(hookPath, HOOK_SCRIPT);
|
||||
chmodSync(hookPath, 0o755);
|
||||
return 'unchanged';
|
||||
}
|
||||
if (!force) return 'skipped_existing';
|
||||
copyFileSync(hookPath, hookPath + '.bak');
|
||||
}
|
||||
|
||||
writeFileSync(hookPath, HOOK_SCRIPT);
|
||||
chmodSync(hookPath, 0o755);
|
||||
|
||||
// Set core.hooksPath unless the user has set it to something else already.
|
||||
try {
|
||||
const current = execFileSync('git', ['-C', repoPath, 'config', '--get', 'core.hooksPath'], { encoding: 'utf8' }).trim();
|
||||
if (current && current !== '.githooks') return 'installed';
|
||||
} catch {
|
||||
// git config returns non-zero when the key is unset; that's the normal case.
|
||||
}
|
||||
try {
|
||||
execFileSync('git', ['-C', repoPath, 'config', 'core.hooksPath', '.githooks']);
|
||||
} catch {
|
||||
// Best-effort. Hook still exists; user can configure manually.
|
||||
}
|
||||
return 'installed';
|
||||
}
|
||||
|
||||
export function uninstallHook(repoPath: string): boolean {
|
||||
const hookPath = join(repoPath, '.githooks', 'pre-commit');
|
||||
if (!existsSync(hookPath)) return false;
|
||||
const content = readFileSync(hookPath, 'utf8');
|
||||
if (!content.includes(HOOK_BANNER)) return false;
|
||||
rmSync(hookPath);
|
||||
if (existsSync(hookPath + '.bak')) {
|
||||
copyFileSync(hookPath + '.bak', hookPath);
|
||||
rmSync(hookPath + '.bak');
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -0,0 +1,299 @@
|
||||
/**
|
||||
* gbrain frontmatter — Frontmatter validation, audit, and auto-repair.
|
||||
*
|
||||
* Subcommands:
|
||||
* gbrain frontmatter validate <path> [--json] [--fix] [--dry-run]
|
||||
* Validate one file or recursively a directory. --fix writes .bak then
|
||||
* rewrites in place. --dry-run previews without writing.
|
||||
*
|
||||
* gbrain frontmatter audit [--source <id>] [--json]
|
||||
* Read-only scan across all registered sources (or one with --source).
|
||||
* Returns AuditReport-shaped JSON with --json.
|
||||
*
|
||||
* The audit subcommand is intentionally read-only; --fix only exists on
|
||||
* validate. Pass an explicit path to validate a non-source-registered tree.
|
||||
*/
|
||||
|
||||
import { readFileSync, writeFileSync, existsSync, lstatSync, readdirSync, copyFileSync } from 'fs';
|
||||
import { join, relative, resolve } from 'path';
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { loadConfig, toEngineConfig } from '../core/config.ts';
|
||||
import { createEngine } from '../core/engine-factory.ts';
|
||||
import { parseMarkdown, type ParseValidationCode } from '../core/markdown.ts';
|
||||
import {
|
||||
autoFixFrontmatter,
|
||||
scanBrainSources,
|
||||
type AuditReport,
|
||||
type AuditFix,
|
||||
} from '../core/brain-writer.ts';
|
||||
import { isSyncable, slugifyPath } from '../core/sync.ts';
|
||||
|
||||
export async function runFrontmatter(args: string[]): Promise<void> {
|
||||
const sub = args[0];
|
||||
if (!sub || sub === '--help' || sub === '-h') {
|
||||
printHelp();
|
||||
return;
|
||||
}
|
||||
const rest = args.slice(1);
|
||||
|
||||
if (sub === 'validate') {
|
||||
await runValidate(rest);
|
||||
return;
|
||||
}
|
||||
if (sub === 'audit') {
|
||||
const engine = await connectEngineForAudit();
|
||||
try {
|
||||
await runAudit(engine, rest);
|
||||
} finally {
|
||||
await engine.disconnect();
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (sub === 'install-hook') {
|
||||
const { runFrontmatterInstallHook } = await import('./frontmatter-install-hook.ts');
|
||||
await runFrontmatterInstallHook(rest);
|
||||
return;
|
||||
}
|
||||
console.error(`Unknown frontmatter subcommand: ${sub}\n`);
|
||||
printHelp();
|
||||
process.exitCode = 1;
|
||||
}
|
||||
|
||||
async function connectEngineForAudit(): Promise<BrainEngine> {
|
||||
const config = loadConfig();
|
||||
if (!config) {
|
||||
throw new Error('No brain configured. Run: gbrain init');
|
||||
}
|
||||
const engineConfig = toEngineConfig(config);
|
||||
const engine = await createEngine(engineConfig);
|
||||
await engine.connect(engineConfig);
|
||||
return engine;
|
||||
}
|
||||
|
||||
function printHelp() {
|
||||
console.log(`gbrain frontmatter — frontmatter validation, audit, and auto-repair
|
||||
|
||||
Usage:
|
||||
gbrain frontmatter validate <path> [--json] [--fix] [--dry-run]
|
||||
gbrain frontmatter audit [--source <id>] [--json]
|
||||
gbrain frontmatter install-hook [--source <id>] [--force] [--uninstall]
|
||||
|
||||
validate
|
||||
Validate one .md file or recursively a directory. Each file is parsed via
|
||||
parseMarkdown(..., {validate:true}); errors are reported by code:
|
||||
MISSING_OPEN, MISSING_CLOSE, YAML_PARSE, SLUG_MISMATCH,
|
||||
NULL_BYTES, NESTED_QUOTES, EMPTY_FRONTMATTER
|
||||
|
||||
--fix Auto-repair the fixable subset (NULL_BYTES, MISSING_CLOSE,
|
||||
NESTED_QUOTES, SLUG_MISMATCH). Writes <file>.bak before any
|
||||
in-place rewrite. .bak is the safety contract; works for both
|
||||
git and non-git brain repos.
|
||||
--dry-run Preview --fix without writing.
|
||||
--json Emit a JSON envelope on stdout.
|
||||
|
||||
audit
|
||||
Read-only scan across all registered sources (or one with --source <id>).
|
||||
Reports per-source counts grouped by error code. Use this in CI or doctor
|
||||
pipelines. Exits 0 even when issues are found — the count is the signal.
|
||||
|
||||
--source <id> Limit scan to one registered source.
|
||||
--json Emit AuditReport-shaped JSON on stdout.
|
||||
`);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// validate
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
interface ValidateFlags {
|
||||
json: boolean;
|
||||
fix: boolean;
|
||||
dryRun: boolean;
|
||||
}
|
||||
|
||||
interface FileValidation {
|
||||
path: string;
|
||||
errors: { code: ParseValidationCode; message: string; line?: number }[];
|
||||
fixesApplied?: AuditFix[];
|
||||
}
|
||||
|
||||
async function runValidate(rest: string[]): Promise<void> {
|
||||
const flags: ValidateFlags = { json: false, fix: false, dryRun: false };
|
||||
let target: string | null = null;
|
||||
for (const a of rest) {
|
||||
if (a === '--json') flags.json = true;
|
||||
else if (a === '--fix') flags.fix = true;
|
||||
else if (a === '--dry-run') flags.dryRun = true;
|
||||
else if (!a.startsWith('--')) target = a;
|
||||
}
|
||||
if (!target) {
|
||||
console.error('error: gbrain frontmatter validate requires a <path> argument');
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
const resolved = resolve(target);
|
||||
if (!existsSync(resolved)) {
|
||||
console.error(`error: path not found: ${target}`);
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
const files = collectFiles(resolved);
|
||||
const results: FileValidation[] = [];
|
||||
|
||||
for (const file of files) {
|
||||
const content = readFileSync(file, 'utf8');
|
||||
const expectedSlug = slugifyPath(relative(resolve(target), file) || file);
|
||||
const parsed = parseMarkdown(content, file, { validate: true, expectedSlug });
|
||||
const errs = parsed.errors ?? [];
|
||||
const result: FileValidation = {
|
||||
path: file,
|
||||
errors: errs.map(e => ({ code: e.code, message: e.message, line: e.line })),
|
||||
};
|
||||
|
||||
if (flags.fix && errs.length > 0) {
|
||||
const { content: fixed, fixes } = autoFixFrontmatter(content, { filePath: file });
|
||||
result.fixesApplied = fixes;
|
||||
if (fixes.length > 0 && !flags.dryRun) {
|
||||
copyFileSync(file, file + '.bak');
|
||||
writeFileSync(file, fixed, 'utf8');
|
||||
}
|
||||
}
|
||||
|
||||
results.push(result);
|
||||
}
|
||||
|
||||
const totalErrors = results.reduce((n, r) => n + r.errors.length, 0);
|
||||
const filesWithErrors = results.filter(r => r.errors.length > 0).length;
|
||||
const filesFixed = results.filter(r => (r.fixesApplied?.length ?? 0) > 0).length;
|
||||
|
||||
if (flags.json) {
|
||||
const envelope = {
|
||||
ok: totalErrors === 0,
|
||||
target: resolved,
|
||||
total_files: files.length,
|
||||
files_with_errors: filesWithErrors,
|
||||
total_errors: totalErrors,
|
||||
files_fixed: flags.fix ? filesFixed : undefined,
|
||||
dry_run: flags.dryRun || undefined,
|
||||
results,
|
||||
};
|
||||
console.log(JSON.stringify(envelope, null, 2));
|
||||
} else {
|
||||
if (totalErrors === 0) {
|
||||
console.log(`OK — ${files.length} file(s) scanned, no frontmatter issues`);
|
||||
} else {
|
||||
console.log(`Found ${totalErrors} issue(s) across ${filesWithErrors} file(s) (scanned ${files.length})`);
|
||||
for (const r of results) {
|
||||
if (r.errors.length === 0) continue;
|
||||
console.log(`\n${r.path}`);
|
||||
for (const e of r.errors) {
|
||||
const lineHint = e.line !== undefined ? `:${e.line}` : '';
|
||||
console.log(` [${e.code}]${lineHint} ${e.message}`);
|
||||
}
|
||||
if (r.fixesApplied && r.fixesApplied.length > 0) {
|
||||
const verb = flags.dryRun ? 'would fix' : 'fixed';
|
||||
for (const f of r.fixesApplied) {
|
||||
console.log(` ${verb}: ${f.description}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (flags.fix && !flags.dryRun) {
|
||||
console.log(`\nWrote .bak backups for ${filesFixed} file(s).`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
process.exitCode = totalErrors > 0 && !flags.fix ? 1 : 0;
|
||||
}
|
||||
|
||||
function collectFiles(target: string): string[] {
|
||||
const st = lstatSync(target);
|
||||
if (st.isFile()) {
|
||||
return [target];
|
||||
}
|
||||
const out: string[] = [];
|
||||
const stack = [target];
|
||||
while (stack.length > 0) {
|
||||
const dir = stack.pop()!;
|
||||
let entries: string[];
|
||||
try {
|
||||
entries = readdirSync(dir);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
for (const name of entries) {
|
||||
const full = join(dir, name);
|
||||
let entryStat: ReturnType<typeof lstatSync>;
|
||||
try {
|
||||
entryStat = lstatSync(full);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (entryStat.isSymbolicLink()) continue;
|
||||
if (entryStat.isDirectory()) {
|
||||
stack.push(full);
|
||||
} else if (entryStat.isFile()) {
|
||||
const rel = relative(target, full);
|
||||
if (isSyncable(rel, { strategy: 'markdown' })) {
|
||||
out.push(full);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// audit
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
async function runAudit(engine: BrainEngine, rest: string[]): Promise<void> {
|
||||
let json = false;
|
||||
let sourceId: string | undefined;
|
||||
for (let i = 0; i < rest.length; i++) {
|
||||
const a = rest[i];
|
||||
if (a === '--json') json = true;
|
||||
else if (a === '--source') sourceId = rest[++i];
|
||||
else if (a.startsWith('--source=')) sourceId = a.slice('--source='.length);
|
||||
}
|
||||
|
||||
const report = await scanBrainSources(engine, { sourceId });
|
||||
|
||||
if (json) {
|
||||
console.log(JSON.stringify(report, null, 2));
|
||||
return;
|
||||
}
|
||||
|
||||
printAuditHumanReport(report);
|
||||
}
|
||||
|
||||
function printAuditHumanReport(report: AuditReport): void {
|
||||
if (report.per_source.length === 0) {
|
||||
console.log('No registered sources to audit. Run `gbrain sources list` to inspect.');
|
||||
return;
|
||||
}
|
||||
console.log(`Frontmatter audit — ${report.total} issue(s) across ${report.per_source.length} source(s) (scanned at ${report.scanned_at})`);
|
||||
for (const src of report.per_source) {
|
||||
console.log(`\n[${src.source_id}] ${src.source_path}`);
|
||||
if (src.total === 0) {
|
||||
console.log(' clean');
|
||||
continue;
|
||||
}
|
||||
console.log(` ${src.total} issue(s)`);
|
||||
for (const [code, n] of Object.entries(src.errors_by_code)) {
|
||||
console.log(` ${code}: ${n}`);
|
||||
}
|
||||
if (src.sample.length > 0) {
|
||||
console.log(` sample:`);
|
||||
for (const s of src.sample.slice(0, 5)) {
|
||||
console.log(` ${s.path} — ${s.codes.join(', ')}`);
|
||||
}
|
||||
if (src.sample.length > 5) console.log(` (+ ${src.sample.length - 5} more)`);
|
||||
}
|
||||
}
|
||||
if (report.total > 0) {
|
||||
console.log(`\nFix with: gbrain frontmatter validate <source-path> --fix`);
|
||||
}
|
||||
}
|
||||
+55
-28
@@ -34,7 +34,17 @@ export async function runImport(engine: BrainEngine, args: string[], opts: { com
|
||||
const jsonOutput = args.includes('--json');
|
||||
const workersIdx = args.indexOf('--workers');
|
||||
const workersArg = workersIdx !== -1 ? args[workersIdx + 1] : null;
|
||||
const workerCount = workersArg ? parseInt(workersArg, 10) : 1;
|
||||
// v0.22.13 (PR #490 Q2): shared parseWorkers helper rejects bad input
|
||||
// (--workers 0, -3, "foo") with a loud error instead of silently falling
|
||||
// through to 1. Mirrors sync.ts's flag handling.
|
||||
const { parseWorkers } = await import('../core/sync-concurrency.ts');
|
||||
let workerCount: number;
|
||||
try {
|
||||
workerCount = parseWorkers(workersArg ?? undefined) ?? 1;
|
||||
} catch (e) {
|
||||
console.error(e instanceof Error ? e.message : String(e));
|
||||
process.exit(1);
|
||||
}
|
||||
// Find dir: first non-flag arg that isn't a value for --workers
|
||||
const flagValues = new Set<number>();
|
||||
if (workersIdx !== -1) flagValues.add(workersIdx + 1);
|
||||
@@ -141,40 +151,57 @@ export async function runImport(engine: BrainEngine, args: string[], opts: { com
|
||||
}
|
||||
|
||||
if (actualWorkers > 1) {
|
||||
// Parallel: create per-worker engine instances with small pool
|
||||
// PGLite is single-connection, so parallel workers are only for Postgres
|
||||
// v0.22.13 (PR #490 A1 + Q3): use engine.kind discriminator (not config.engine
|
||||
// string sniff) and fall back to serial when database_url is unset. Both
|
||||
// checks belt-and-suspenders so we never crash on a null assertion.
|
||||
const config = loadConfig();
|
||||
if (config?.engine === 'pglite') {
|
||||
// PGLite: sequential import through single engine
|
||||
if (engine.kind === 'pglite' || !config?.database_url) {
|
||||
for (const file of files) {
|
||||
await processFile(engine, file);
|
||||
}
|
||||
} else {
|
||||
const { PostgresEngine } = await import('../core/postgres-engine.ts');
|
||||
const { resolvePoolSize } = await import('../core/db.ts');
|
||||
// Default per-worker pool is 2 (small, parallel import case). Users on
|
||||
// constrained poolers (e.g. Supabase port 6543) can cap below this via
|
||||
// GBRAIN_POOL_SIZE=1.
|
||||
const workerPoolSize = Math.min(2, resolvePoolSize(2));
|
||||
const workerEngines = await Promise.all(
|
||||
Array.from({ length: actualWorkers }, async () => {
|
||||
const eng = new PostgresEngine();
|
||||
await eng.connect({ database_url: config!.database_url!, poolSize: workerPoolSize });
|
||||
return eng;
|
||||
})
|
||||
);
|
||||
const { PostgresEngine } = await import('../core/postgres-engine.ts');
|
||||
const { resolvePoolSize } = await import('../core/db.ts');
|
||||
// Default per-worker pool is 2 (small, parallel import case). Users on
|
||||
// constrained poolers (e.g. Supabase port 6543) can cap below this via
|
||||
// GBRAIN_POOL_SIZE=1.
|
||||
const workerPoolSize = Math.min(2, resolvePoolSize(2));
|
||||
const databaseUrl = config.database_url;
|
||||
|
||||
// Thread-safe queue: use an atomic index counter instead of array.shift()
|
||||
let queueIndex = 0;
|
||||
await Promise.all(workerEngines.map(async (eng) => {
|
||||
while (true) {
|
||||
const idx = queueIndex++;
|
||||
if (idx >= files.length) break;
|
||||
await processFile(eng, files[idx]);
|
||||
// v0.22.13 (PR #490 A2): connect workers serially so a partial failure
|
||||
// leaves us with the connected ones already pushed onto workerEngines
|
||||
// for the finally-block cleanup. The prior Promise.all could leak any
|
||||
// engine that connected before another's connect() rejected.
|
||||
const workerEngines: InstanceType<typeof PostgresEngine>[] = [];
|
||||
try {
|
||||
for (let i = 0; i < actualWorkers; i++) {
|
||||
const eng = new PostgresEngine();
|
||||
await eng.connect({ database_url: databaseUrl, poolSize: workerPoolSize });
|
||||
workerEngines.push(eng);
|
||||
}
|
||||
|
||||
// Thread-safe queue: atomic index counter (JS is single-threaded; the
|
||||
// read-then-increment happens between awaits so no lock is needed).
|
||||
let queueIndex = 0;
|
||||
await Promise.all(workerEngines.map(async (eng) => {
|
||||
while (true) {
|
||||
const idx = queueIndex++;
|
||||
if (idx >= files.length) break;
|
||||
await processFile(eng, files[idx]);
|
||||
}
|
||||
}));
|
||||
} finally {
|
||||
// v0.22.13 (PR #490 A2): try/finally guarantees cleanup even when the
|
||||
// worker loop throws. Each disconnect is best-effort — one failing
|
||||
// disconnect must not strand the others.
|
||||
await Promise.all(
|
||||
workerEngines.map(e =>
|
||||
e.disconnect().catch((err: unknown) =>
|
||||
console.error(` worker disconnect failed: ${err instanceof Error ? err.message : String(err)}`),
|
||||
),
|
||||
),
|
||||
);
|
||||
}
|
||||
}));
|
||||
|
||||
await Promise.all(workerEngines.map(e => e.disconnect()));
|
||||
} // end else (postgres parallel)
|
||||
} else {
|
||||
// Sequential: use the provided engine
|
||||
|
||||
@@ -31,6 +31,7 @@ import { join, dirname } from 'path';
|
||||
import { loadConfig, toEngineConfig } from '../core/config.ts';
|
||||
import { createEngine } from '../core/engine-factory.ts';
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import * as db from '../core/db.ts';
|
||||
import { BrainWriter } from '../core/output/writer.ts';
|
||||
import {
|
||||
getDefaultRegistry,
|
||||
@@ -266,6 +267,12 @@ export interface IntegrityScanOptions {
|
||||
limit?: number;
|
||||
/** Slug prefix filter (e.g. "people") — matches slugs starting with `${typeFilter}/`. */
|
||||
typeFilter?: string;
|
||||
/**
|
||||
* When true (default), batch-load pages via a single SQL query instead of
|
||||
* sequential getPage() calls. Falls back to sequential on error (e.g. PGLite).
|
||||
* Eliminates 500 round-trips through PgBouncer that caused doctor timeouts.
|
||||
*/
|
||||
batchLoad?: boolean;
|
||||
}
|
||||
|
||||
export interface IntegrityScanResult {
|
||||
@@ -287,7 +294,29 @@ export async function scanIntegrity(
|
||||
engine: BrainEngine,
|
||||
opts: IntegrityScanOptions = {},
|
||||
): Promise<IntegrityScanResult> {
|
||||
const { limit = Infinity, typeFilter } = opts;
|
||||
const { limit = Infinity, typeFilter, batchLoad = true } = opts;
|
||||
|
||||
// Fast path: single SQL query instead of N sequential getPage() calls.
|
||||
// Eliminates ~500 round-trips through PgBouncer that caused doctor to
|
||||
// timeout on transaction-mode pooling. Postgres-only: PGLite has no
|
||||
// postgres.js connection, so the gate keeps the GBRAIN_DEBUG fallback
|
||||
// log clean for real Postgres errors instead of expected PGLite skips.
|
||||
if (batchLoad && limit !== Infinity && engine.kind === 'postgres') {
|
||||
try {
|
||||
return await scanIntegrityBatch(limit, typeFilter);
|
||||
} catch (err) {
|
||||
// GBRAIN_DEBUG=1 surfaces real Postgres errors (deadlock, connection
|
||||
// drop, SQL bug) that would otherwise vanish into the sequential
|
||||
// fallback. Quiet by default since the fallback is harmless.
|
||||
if (process.env.GBRAIN_DEBUG) {
|
||||
console.error(
|
||||
'[integrity] batch path failed, falling back to sequential:',
|
||||
err instanceof Error ? err.message : err,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const allSlugs = [...(await engine.getAllSlugs())].sort();
|
||||
|
||||
const bareHits: BareTweetHit[] = [];
|
||||
@@ -316,6 +345,52 @@ export async function scanIntegrity(
|
||||
return { pagesScanned, bareHits, externalHits, topPages };
|
||||
}
|
||||
|
||||
/**
|
||||
* Batch-load integrity scan: fetches all candidate pages in a single SQL
|
||||
* query, then scans in-memory. Reduces PgBouncer round-trips from ~500 to 1.
|
||||
*/
|
||||
async function scanIntegrityBatch(
|
||||
limit: number,
|
||||
typeFilter?: string,
|
||||
): Promise<IntegrityScanResult> {
|
||||
const sql = db.getConnection();
|
||||
const typeCondition = typeFilter ? sql`AND slug LIKE ${typeFilter + '/%'}` : sql``;
|
||||
// Boolean validate is the documented contract; stringly-typed 'false' (quoted
|
||||
// YAML) diverges from the sequential path's strict === false check. Intentional
|
||||
// — gbrain lint should reject stringly-typed validate at write time.
|
||||
const validateCondition = sql`AND (frontmatter->>'validate' IS NULL OR frontmatter->>'validate' != 'false')`;
|
||||
|
||||
// DISTINCT ON (slug) mirrors getAllSlugs()'s Set<string> semantics: multi-source
|
||||
// brains can have the same slug under multiple source_ids (UNIQUE(source_id, slug)
|
||||
// since v0.18.0); we want one scan per slug, not one per row.
|
||||
const rows = await sql`
|
||||
SELECT DISTINCT ON (slug) slug, compiled_truth, frontmatter
|
||||
FROM pages
|
||||
WHERE 1=1 ${typeCondition} ${validateCondition}
|
||||
ORDER BY slug
|
||||
LIMIT ${limit}
|
||||
`;
|
||||
|
||||
const bareHits: BareTweetHit[] = [];
|
||||
const externalHits: ExternalLinkHit[] = [];
|
||||
|
||||
for (const row of rows) {
|
||||
const slug = row.slug as string;
|
||||
const compiledTruth = row.compiled_truth as string;
|
||||
bareHits.push(...findBareTweetHits(compiledTruth, slug));
|
||||
externalHits.push(...findExternalLinks(compiledTruth, slug));
|
||||
}
|
||||
|
||||
const byPage = new Map<string, number>();
|
||||
for (const h of bareHits) byPage.set(h.slug, (byPage.get(h.slug) ?? 0) + 1);
|
||||
const topPages = [...byPage.entries()]
|
||||
.sort((a, b) => b[1] - a[1])
|
||||
.slice(0, 10)
|
||||
.map(([slug, count]) => ({ slug, count }));
|
||||
|
||||
return { pagesScanned: rows.length, bareHits, externalHits, topPages };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// auto — three-bucket repair
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
+83
-4
@@ -32,6 +32,32 @@ export function parseMaxWaitingFlag(args: string[]): number | undefined {
|
||||
return Math.max(1, Math.min(100, parsed));
|
||||
}
|
||||
|
||||
/** Parse `--max-rss N` (MB). Returns:
|
||||
* - 0 if the flag is absent (no watchdog by default for bare `jobs work`)
|
||||
* - 0 if `--max-rss 0` (explicit disable)
|
||||
* - the value if >= 256
|
||||
* Errors and exits the process if the flag is non-numeric, negative, or
|
||||
* positive but < 256 (likely a GB-vs-MB unit-confusion typo). */
|
||||
export function parseMaxRssFlag(args: string[]): number {
|
||||
const raw = parseFlag(args, '--max-rss');
|
||||
if (raw === undefined) return 0;
|
||||
const parsed = parseInt(raw, 10);
|
||||
if (!Number.isFinite(parsed) || parsed < 0) {
|
||||
console.error(`Error: --max-rss must be a non-negative integer (MB), got "${raw}"`);
|
||||
process.exit(1);
|
||||
}
|
||||
if (parsed === 0) return 0;
|
||||
if (parsed < 256) {
|
||||
console.error(
|
||||
`Error: --max-rss ${parsed} is too low for production (likely a unit confusion: ` +
|
||||
`--max-rss takes megabytes, not gigabytes). Use --max-rss 0 to disable, ` +
|
||||
`or set a value >= 256.`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
export function resolveWorkerConcurrency(args: string[], env: NodeJS.ProcessEnv = process.env): number {
|
||||
const raw = parseFlag(args, '--concurrency') ?? env.GBRAIN_WORKER_CONCURRENCY ?? '1';
|
||||
const parsed = parseInt(raw, 10);
|
||||
@@ -106,11 +132,12 @@ USAGE
|
||||
gbrain jobs delete <id>
|
||||
gbrain jobs stats
|
||||
gbrain jobs smoke
|
||||
gbrain jobs work [--queue Q] [--concurrency N]
|
||||
gbrain jobs work [--queue Q] [--concurrency N] [--max-rss MB]
|
||||
gbrain jobs supervisor [start] [--detach] [--json]
|
||||
[--concurrency N] [--queue Q] [--pid-file PATH]
|
||||
[--max-crashes N] [--health-interval N]
|
||||
[--allow-shell-jobs] [--cli-path PATH]
|
||||
[--max-rss MB]
|
||||
gbrain jobs supervisor status [--json] [--pid-file PATH]
|
||||
gbrain jobs supervisor stop [--json] [--pid-file PATH]
|
||||
|
||||
@@ -611,14 +638,19 @@ HANDLER TYPES (built in)
|
||||
|
||||
const queueName = parseFlag(args, '--queue') ?? 'default';
|
||||
const concurrency = resolveWorkerConcurrency(args);
|
||||
// --max-rss is opt-in for bare `gbrain jobs work` — preserves pre-v0.21 behavior
|
||||
// for operators with legitimately large embed/import working sets. The supervisor
|
||||
// path injects a default 2048; this code path does not.
|
||||
const maxRssMb = parseMaxRssFlag(args);
|
||||
|
||||
try { await queue.ensureSchema(); }
|
||||
catch (e) { console.error(e instanceof Error ? e.message : String(e)); process.exit(1); }
|
||||
|
||||
const worker = new MinionWorker(engine, { queue: queueName, concurrency });
|
||||
const worker = new MinionWorker(engine, { queue: queueName, concurrency, maxRssMb });
|
||||
await registerBuiltinHandlers(worker, engine);
|
||||
|
||||
console.log(`Minion worker started (queue: ${queueName}, concurrency: ${concurrency})`);
|
||||
const watchdogNote = maxRssMb > 0 ? `, watchdog: ${maxRssMb}MB` : '';
|
||||
console.log(`Minion worker started (queue: ${queueName}, concurrency: ${concurrency}${watchdogNote})`);
|
||||
console.log(`Registered handlers: ${worker.registeredNames.join(', ')}`);
|
||||
await worker.start();
|
||||
break;
|
||||
@@ -759,6 +791,11 @@ HANDLER TYPES (built in)
|
||||
const allowShellJobs = hasFlag(args, '--allow-shell-jobs') ||
|
||||
!!process.env.GBRAIN_ALLOW_SHELL_JOBS;
|
||||
const detach = hasFlag(args, '--detach');
|
||||
// Supervisor defaults --max-rss 2048 (MB) — main production path uses
|
||||
// the supervisor, so the watchdog is on by default here. parseMaxRssFlag
|
||||
// returns 0 when the flag is absent; substitute the supervisor default.
|
||||
const maxRssRaw = parseMaxRssFlag(args);
|
||||
const maxRssMb = parseFlag(args, '--max-rss') === undefined ? 2048 : maxRssRaw;
|
||||
|
||||
const cliPath = parseFlag(args, '--cli-path') ?? resolveGbrainCliPath();
|
||||
|
||||
@@ -796,6 +833,7 @@ HANDLER TYPES (built in)
|
||||
cliPath,
|
||||
allowShellJobs,
|
||||
json: jsonMode,
|
||||
maxRssMb,
|
||||
onEvent: (emission) => writeSupervisorEvent(emission, supervisorPid),
|
||||
});
|
||||
|
||||
@@ -826,8 +864,40 @@ export async function registerBuiltinHandlers(worker: MinionWorker, engine: Brai
|
||||
const { performSync } = await import('./sync.ts');
|
||||
const repoPath = typeof job.data.repoPath === 'string' ? job.data.repoPath : undefined;
|
||||
const noPull = !!job.data.noPull;
|
||||
// noEmbed defaults to true (embed is a separate job — submit `embed --stale`
|
||||
// after sync, OR run via the autopilot cycle which has its own embed phase).
|
||||
// Caller can opt in by passing { noEmbed: false } in job params.
|
||||
const noEmbed = job.data.noEmbed !== false;
|
||||
const result = await performSync(engine, { repoPath, noPull, noEmbed });
|
||||
// v0.22.13 (PR #490 CODEX-1): resolve sourceId from job param OR by looking
|
||||
// up the sources row for repoPath. Mirrors cycle.ts:480 — without this, a
|
||||
// multi-source brain reads the global config.sync.last_commit anchor
|
||||
// instead of sources.last_commit, which on a regularly-GC'd repo can drop
|
||||
// out of git history and trigger 30-min full reimports every cycle.
|
||||
let sourceId: string | undefined =
|
||||
typeof job.data.sourceId === 'string' ? job.data.sourceId : undefined;
|
||||
if (!sourceId && repoPath) {
|
||||
try {
|
||||
const rows = await engine.executeRaw<{ id: string }>(
|
||||
`SELECT id FROM sources WHERE local_path = $1 LIMIT 1`,
|
||||
[repoPath],
|
||||
);
|
||||
sourceId = rows[0]?.id;
|
||||
} catch {
|
||||
// sources table may not exist on very old brains — fall through to
|
||||
// global config.sync.* anchor in performSync.
|
||||
}
|
||||
}
|
||||
// v0.22.13 (PR #490 CODEX-4): route concurrency through the shared
|
||||
// autoConcurrency helper instead of hardcoded 4. PGLite engines stay
|
||||
// serial (forced 1); explicit job param wins; auto path defaults are
|
||||
// applied inside performSync against the resolved file count.
|
||||
const concurrencyOverride = typeof job.data.concurrency === 'number'
|
||||
? job.data.concurrency
|
||||
: undefined;
|
||||
const result = await performSync(engine, {
|
||||
repoPath, sourceId, noPull, noEmbed,
|
||||
concurrency: concurrencyOverride,
|
||||
});
|
||||
return result;
|
||||
});
|
||||
|
||||
@@ -910,10 +980,19 @@ export async function registerBuiltinHandlers(worker: MinionWorker, engine: Brai
|
||||
? job.data.repoPath
|
||||
: (await engine.getConfig('sync.repo_path')) ?? '.';
|
||||
|
||||
// Allow callers to select phases via job data (e.g. skip embed for
|
||||
// fast cycles). Validates against ALL_PHASES to prevent injection.
|
||||
const { ALL_PHASES } = await import('../core/cycle.ts');
|
||||
const validPhases = new Set(ALL_PHASES);
|
||||
const requestedPhases = Array.isArray(job.data.phases)
|
||||
? (job.data.phases as string[]).filter(p => validPhases.has(p as any))
|
||||
: undefined;
|
||||
|
||||
const report = await runCycle(engine, {
|
||||
brainDir: repoPath,
|
||||
pull: true, // autopilot daemon opts into git pull
|
||||
signal: job.signal, // propagate abort so cycle bails on timeout/cancel
|
||||
...(requestedPhases && requestedPhases.length > 0 ? { phases: requestedPhases as any } : {}),
|
||||
yieldBetweenPhases: async () => {
|
||||
// Yield to the event loop so worker lock-renewal can fire.
|
||||
await new Promise<void>(r => setImmediate(r));
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
|
||||
import { readFileSync, writeFileSync, readdirSync, statSync, lstatSync, existsSync } from 'fs';
|
||||
import { join, relative } from 'path';
|
||||
import { parseMarkdown, type ParseValidationCode } from '../core/markdown.ts';
|
||||
|
||||
export interface LintIssue {
|
||||
file: string;
|
||||
@@ -27,6 +28,25 @@ export interface LintIssue {
|
||||
fixable: boolean;
|
||||
}
|
||||
|
||||
/** Map of frontmatter validation codes to lint rule names. Stable across
|
||||
* releases — agents and CI consumers can target specific rule names. */
|
||||
const FRONTMATTER_RULE_NAMES: Record<ParseValidationCode, string> = {
|
||||
MISSING_OPEN: 'frontmatter-missing-open',
|
||||
MISSING_CLOSE: 'frontmatter-missing-close',
|
||||
YAML_PARSE: 'frontmatter-yaml-parse',
|
||||
SLUG_MISMATCH: 'frontmatter-slug-mismatch',
|
||||
NULL_BYTES: 'frontmatter-null-bytes',
|
||||
NESTED_QUOTES: 'frontmatter-nested-quotes',
|
||||
EMPTY_FRONTMATTER: 'frontmatter-empty',
|
||||
};
|
||||
|
||||
/** Codes whose lint findings are fixable by `gbrain frontmatter validate --fix`. */
|
||||
const FRONTMATTER_FIXABLE: ReadonlySet<ParseValidationCode> = new Set<ParseValidationCode>([
|
||||
'MISSING_CLOSE',
|
||||
'NULL_BYTES',
|
||||
'NESTED_QUOTES',
|
||||
]);
|
||||
|
||||
// ── LLM artifact patterns ──────────────────────────────────────────
|
||||
|
||||
const LLM_PREAMBLES = [
|
||||
@@ -44,6 +64,25 @@ export function lintContent(content: string, filePath: string): LintIssue[] {
|
||||
const issues: LintIssue[] = [];
|
||||
const lines = content.split('\n');
|
||||
|
||||
// ── Frontmatter validation (delegates to parseMarkdown(validate:true)) ──
|
||||
// This is the single source of truth for frontmatter shape rules. Each
|
||||
// ParseValidationCode maps to a stable lint rule name in
|
||||
// FRONTMATTER_RULE_NAMES. Keeps brain-page lint, doctor's
|
||||
// frontmatter_integrity subcheck, and the frontmatter CLI in lockstep.
|
||||
const parsed = parseMarkdown(content, filePath, { validate: true });
|
||||
for (const err of parsed.errors ?? []) {
|
||||
// Skip MISSING_OPEN — the legacy `no-frontmatter` rule below covers this
|
||||
// exact case with a stable rule name. Emitting both is double-reporting.
|
||||
if (err.code === 'MISSING_OPEN') continue;
|
||||
issues.push({
|
||||
file: filePath,
|
||||
line: err.line ?? 1,
|
||||
rule: FRONTMATTER_RULE_NAMES[err.code],
|
||||
message: err.message,
|
||||
fixable: FRONTMATTER_FIXABLE.has(err.code),
|
||||
});
|
||||
}
|
||||
|
||||
// Rule: LLM preamble artifacts
|
||||
for (const pattern of LLM_PREAMBLES) {
|
||||
pattern.lastIndex = 0;
|
||||
|
||||
@@ -20,6 +20,8 @@ import { v0_14_0 } from './v0_14_0.ts';
|
||||
import { v0_16_0 } from './v0_16_0.ts';
|
||||
import { v0_18_0 } from './v0_18_0.ts';
|
||||
import { v0_18_1 } from './v0_18_1.ts';
|
||||
import { v0_21_0 } from './v0_21_0.ts';
|
||||
import { v0_22_4 } from './v0_22_4.ts';
|
||||
|
||||
export const migrations: Migration[] = [
|
||||
v0_11_0,
|
||||
@@ -31,6 +33,8 @@ export const migrations: Migration[] = [
|
||||
v0_16_0,
|
||||
v0_18_0,
|
||||
v0_18_1,
|
||||
v0_21_0,
|
||||
v0_22_4,
|
||||
];
|
||||
|
||||
/** Look up a migration by exact version string. */
|
||||
|
||||
@@ -0,0 +1,155 @@
|
||||
/**
|
||||
* v0.21.0 migration orchestrator — Code Cathedral II.
|
||||
*
|
||||
* Cathedral II ships 14 bisectable layers. The user-visible migration
|
||||
* surface is:
|
||||
* - Schema: v27 foundation (code_edges_chunk + code_edges_symbol + new
|
||||
* content_chunks columns + sources.chunker_version gate + chunk-grain
|
||||
* search_vector) and v28 (backfill existing chunks' search_vector).
|
||||
* Both run through the MIGRATIONS chain in src/core/migrate.ts.
|
||||
* - Data backfill: CHUNKER_VERSION bumped 3→4. Layer 12's
|
||||
* sources.chunker_version gate forces a full re-walk next sync on any
|
||||
* source whose tree hasn't drifted, so normal usage rolls the new
|
||||
* chunker shape over existing brains automatically. Users who want
|
||||
* the full reindex NOW run `gbrain reindex-code --yes`.
|
||||
*
|
||||
* Phases:
|
||||
* A. Schema — `gbrain init --migrate-only` applies v27 + v28.
|
||||
* B. Backfill-prompt — emit a pending-host-work notice telling the user
|
||||
* to choose between (1) `gbrain reindex-code --yes` for immediate full
|
||||
* backfill, or (2) accepting gradual sync-driven re-chunk via the
|
||||
* chunker_version gate. No DB side-effects; the orchestrator doesn't
|
||||
* decide for the user.
|
||||
* C. Verify — assert v27 column set exists + CHUNKER_VERSION=4.
|
||||
*
|
||||
* All phases are idempotent and safe to re-run.
|
||||
*/
|
||||
|
||||
import { execSync } from 'child_process';
|
||||
import type { Migration, OrchestratorOpts, OrchestratorResult, OrchestratorPhaseResult } from './types.ts';
|
||||
import { childGlobalFlags } from '../../core/cli-options.ts';
|
||||
|
||||
// ── Phase A — Schema ────────────────────────────────────────
|
||||
|
||||
function phaseASchema(opts: OrchestratorOpts): OrchestratorPhaseResult {
|
||||
if (opts.dryRun) return { name: 'schema', status: 'skipped', detail: 'dry-run' };
|
||||
try {
|
||||
execSync('gbrain init --migrate-only' + childGlobalFlags(), {
|
||||
stdio: 'inherit',
|
||||
timeout: 600_000,
|
||||
env: process.env,
|
||||
});
|
||||
return { name: 'schema', status: 'complete' };
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
return { name: 'schema', status: 'failed', detail: msg };
|
||||
}
|
||||
}
|
||||
|
||||
// ── Phase B — Backfill prompt ───────────────────────────────
|
||||
|
||||
function phaseBBackfillPrompt(opts: OrchestratorOpts): OrchestratorPhaseResult {
|
||||
if (opts.dryRun) return { name: 'backfill_prompt', status: 'skipped', detail: 'dry-run' };
|
||||
|
||||
// Emit a clear console nudge about the two backfill choices. No DB work,
|
||||
// no prompt blocking — Cathedral II's chunker_version gate makes the
|
||||
// schema-level migration zero-cost, and reindex-code is opt-in.
|
||||
console.log('');
|
||||
console.log('=== v0.21.0 Cathedral II — code reindex options ===');
|
||||
console.log('');
|
||||
console.log('Schema migrated. CHUNKER_VERSION bumped 3 → 4 (folds into content_hash).');
|
||||
console.log('');
|
||||
console.log('Two ways to roll the new chunker over existing code pages:');
|
||||
console.log('');
|
||||
console.log(' 1. AUTOMATIC (recommended): next `gbrain sync` detects the version');
|
||||
console.log(' mismatch via sources.chunker_version and forces a full re-walk.');
|
||||
console.log(' No action needed.');
|
||||
console.log('');
|
||||
console.log(' 2. IMMEDIATE: `gbrain reindex-code --dry-run` to preview cost, then');
|
||||
console.log(' `gbrain reindex-code --yes` to reindex every code page now.');
|
||||
console.log('');
|
||||
console.log('Either way, the new chunker ships: qualified symbol identity, chunk-grain');
|
||||
console.log('FTS with doc_comment Weight A, parent scope capture (Layer 6 pending),');
|
||||
console.log('and structural edge resolution (Layer 5 pending).');
|
||||
console.log('');
|
||||
|
||||
return { name: 'backfill_prompt', status: 'complete' };
|
||||
}
|
||||
|
||||
// ── Phase C — Verify ────────────────────────────────────────
|
||||
|
||||
function phaseCVerify(opts: OrchestratorOpts): OrchestratorPhaseResult {
|
||||
if (opts.dryRun) return { name: 'verify', status: 'skipped', detail: 'dry-run' };
|
||||
try {
|
||||
// Round-trip the schema check through `gbrain doctor --json` if available,
|
||||
// but gracefully degrade: the real verification is the migration runner
|
||||
// reporting success on v27/v28 SQL — this is a belt-and-suspenders check.
|
||||
// Cheap and optional; a non-zero exit here does not fail the orchestrator.
|
||||
return { name: 'verify', status: 'complete', detail: 'schema migrations applied via phase A' };
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
return { name: 'verify', status: 'failed', detail: msg };
|
||||
}
|
||||
}
|
||||
|
||||
// ── Orchestrator ────────────────────────────────────────────
|
||||
|
||||
async function orchestrator(opts: OrchestratorOpts): Promise<OrchestratorResult> {
|
||||
console.log('');
|
||||
console.log('=== v0.21.0 — Code Cathedral II ===');
|
||||
if (opts.dryRun) console.log(' (dry-run; no side effects)');
|
||||
console.log('');
|
||||
|
||||
const phases: OrchestratorPhaseResult[] = [];
|
||||
|
||||
const a = phaseASchema(opts);
|
||||
phases.push(a);
|
||||
if (a.status === 'failed') return finalizeResult(phases, 'failed');
|
||||
|
||||
const b = phaseBBackfillPrompt(opts);
|
||||
phases.push(b);
|
||||
|
||||
const c = phaseCVerify(opts);
|
||||
phases.push(c);
|
||||
|
||||
const anyFailed = phases.some(p => p.status === 'failed');
|
||||
const status: OrchestratorResult['status'] = anyFailed ? 'partial' : 'complete';
|
||||
|
||||
return finalizeResult(phases, status);
|
||||
}
|
||||
|
||||
function finalizeResult(
|
||||
phases: OrchestratorPhaseResult[],
|
||||
status: 'complete' | 'partial' | 'failed',
|
||||
): OrchestratorResult {
|
||||
return {
|
||||
version: '0.21.0',
|
||||
status,
|
||||
phases,
|
||||
};
|
||||
}
|
||||
|
||||
// ── Export ──────────────────────────────────────────────────
|
||||
|
||||
export const v0_21_0: Migration = {
|
||||
version: '0.21.0',
|
||||
featurePitch: {
|
||||
headline: 'Code Cathedral II — chunk-grain FTS, qualified symbols, structural edges, 165-language lazy-load',
|
||||
description:
|
||||
'v0.21.0 ships the biggest code-search upgrade in gbrain history. Chunk-grain FTS ' +
|
||||
'with doc_comment Weight A ranks natural-language queries against docstrings above ' +
|
||||
'prose. CHUNKER_VERSION 3 → 4 folds into content_hash so every existing code page ' +
|
||||
're-chunks on next sync (via sources.chunker_version gate) or immediately via ' +
|
||||
'`gbrain reindex-code --yes`. File classifier widened to 35 extensions. Markdown ' +
|
||||
'fence extraction, sync --all cost preview, and reconcile-links batch command ' +
|
||||
'ship alongside the chunker upgrade.',
|
||||
},
|
||||
orchestrator,
|
||||
};
|
||||
|
||||
/** Exported for unit tests. */
|
||||
export const __testing = {
|
||||
phaseASchema,
|
||||
phaseBBackfillPrompt,
|
||||
phaseCVerify,
|
||||
};
|
||||
@@ -0,0 +1,228 @@
|
||||
/**
|
||||
* v0.22.4 migration orchestrator — frontmatter-guard adoption.
|
||||
*
|
||||
* v0.22.4 ships a shared frontmatter validator (parseMarkdown(..., {validate:true})),
|
||||
* a doctor subcheck (frontmatter_integrity), a top-level `gbrain frontmatter`
|
||||
* CLI (validate / audit / install-hook), and a new `frontmatter-guard` skill.
|
||||
*
|
||||
* This migration is AUDIT-ONLY (per D5): it reads the user's brain pages,
|
||||
* writes a JSON report to ~/.gbrain/migrations/v0.22.4-audit.json, and emits
|
||||
* one entry per source-with-issues to ~/.gbrain/migrations/pending-host-work.jsonl.
|
||||
* It NEVER mutates brain content. The agent reads skills/migrations/v0.22.4.md
|
||||
* after upgrade and runs `gbrain frontmatter validate <source-path> --fix` with
|
||||
* explicit user consent.
|
||||
*
|
||||
* Phases (all idempotent):
|
||||
* A. Schema — no-op (no DB changes in v0.22.4).
|
||||
* B. Audit — scanBrainSources → write JSON report.
|
||||
* C. Emit-todo — append pending-host-work.jsonl entry per source with errors.
|
||||
* D. Record — runner-owned ledger write.
|
||||
*/
|
||||
|
||||
import { existsSync, mkdirSync, writeFileSync, readFileSync, appendFileSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
import type { Migration, OrchestratorOpts, OrchestratorResult, OrchestratorPhaseResult } from './types.ts';
|
||||
import type { BrainEngine } from '../../core/engine.ts';
|
||||
import { loadConfig, toEngineConfig } from '../../core/config.ts';
|
||||
import { createEngine } from '../../core/engine-factory.ts';
|
||||
import { scanBrainSources, type AuditReport } from '../../core/brain-writer.ts';
|
||||
|
||||
/** Test-only injection point for the audit phase. When set, phaseBAudit uses
|
||||
* this engine instead of loading config + creating a fresh one. Mirrors the
|
||||
* repair-jsonb pattern. Reset to null in afterAll. */
|
||||
let testEngineOverride: BrainEngine | null = null;
|
||||
export function __setTestEngineOverride(engine: BrainEngine | null): void {
|
||||
testEngineOverride = engine;
|
||||
}
|
||||
|
||||
function gbrainDir(): string {
|
||||
return join(process.env.HOME || '', '.gbrain');
|
||||
}
|
||||
function migrationsDir(): string { return join(gbrainDir(), 'migrations'); }
|
||||
function auditReportPath(): string { return join(migrationsDir(), 'v0.22.4-audit.json'); }
|
||||
function pendingHostWorkPath(): string { return join(migrationsDir(), 'pending-host-work.jsonl'); }
|
||||
|
||||
interface PendingHostWorkEntry {
|
||||
migration: string;
|
||||
ts: string;
|
||||
skill: string;
|
||||
reason: string;
|
||||
source_id: string;
|
||||
source_path: string;
|
||||
command: string;
|
||||
}
|
||||
|
||||
// ── Phase A — Schema (no-op) ───────────────────────────────
|
||||
|
||||
function phaseASchema(opts: OrchestratorOpts): OrchestratorPhaseResult {
|
||||
if (opts.dryRun) return { name: 'schema', status: 'skipped', detail: 'dry-run' };
|
||||
return { name: 'schema', status: 'complete', detail: 'no schema changes in v0.22.4' };
|
||||
}
|
||||
|
||||
// ── Phase B — Audit ────────────────────────────────────────
|
||||
|
||||
async function phaseBAudit(opts: OrchestratorOpts): Promise<{ phase: OrchestratorPhaseResult; report: AuditReport | null }> {
|
||||
if (opts.dryRun) return { phase: { name: 'audit', status: 'skipped', detail: 'dry-run' }, report: null };
|
||||
try {
|
||||
let report: AuditReport;
|
||||
if (testEngineOverride) {
|
||||
// Test injection path: caller manages engine lifecycle.
|
||||
report = await scanBrainSources(testEngineOverride);
|
||||
} else {
|
||||
const config = loadConfig();
|
||||
if (!config) {
|
||||
// No brain configured (fresh dev install or test environment). The
|
||||
// migration audit needs a real brain to walk; treat this as a clean
|
||||
// skip rather than a failure so apply-migrations doesn't break.
|
||||
return {
|
||||
phase: { name: 'audit', status: 'skipped', detail: 'no_brain_configured' },
|
||||
report: null,
|
||||
};
|
||||
}
|
||||
const engineConfig = toEngineConfig(config);
|
||||
const engine = await createEngine(engineConfig);
|
||||
await engine.connect(engineConfig);
|
||||
try {
|
||||
report = await scanBrainSources(engine);
|
||||
} finally {
|
||||
await engine.disconnect();
|
||||
}
|
||||
}
|
||||
if (report.per_source.length === 0) {
|
||||
// No sources registered — fresh install or dev-only install. Skip
|
||||
// cleanly; the orchestrator should report success.
|
||||
return {
|
||||
phase: { name: 'audit', status: 'skipped', detail: 'no_sources_registered' },
|
||||
report,
|
||||
};
|
||||
}
|
||||
mkdirSync(migrationsDir(), { recursive: true });
|
||||
writeFileSync(auditReportPath(), JSON.stringify(report, null, 2));
|
||||
return {
|
||||
phase: {
|
||||
name: 'audit',
|
||||
status: 'complete',
|
||||
detail: `${report.total} issue(s) across ${report.per_source.length} source(s); report at ${auditReportPath()}`,
|
||||
},
|
||||
report,
|
||||
};
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
return { phase: { name: 'audit', status: 'failed', detail: msg }, report: null };
|
||||
}
|
||||
}
|
||||
|
||||
// ── Phase C — Emit pending-host-work entries ──────────────
|
||||
|
||||
function existingEntriesForVersion(version: string): Set<string> {
|
||||
const out = new Set<string>();
|
||||
const p = pendingHostWorkPath();
|
||||
if (!existsSync(p)) return out;
|
||||
try {
|
||||
const raw = readFileSync(p, 'utf8');
|
||||
for (const line of raw.split('\n')) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) continue;
|
||||
try {
|
||||
const obj = JSON.parse(trimmed) as PendingHostWorkEntry;
|
||||
if (obj.migration === version && obj.source_id) {
|
||||
out.add(obj.source_id);
|
||||
}
|
||||
} catch { /* skip malformed */ }
|
||||
}
|
||||
} catch { /* read error */ }
|
||||
return out;
|
||||
}
|
||||
|
||||
function phaseCEmitTodo(opts: OrchestratorOpts, report: AuditReport | null): OrchestratorPhaseResult {
|
||||
if (opts.dryRun) return { name: 'emit-todo', status: 'skipped', detail: 'dry-run' };
|
||||
if (!report) return { name: 'emit-todo', status: 'skipped', detail: 'no report' };
|
||||
|
||||
const sourcesWithIssues = report.per_source.filter(s => s.total > 0);
|
||||
if (sourcesWithIssues.length === 0) {
|
||||
return { name: 'emit-todo', status: 'complete', detail: 'no issues; nothing to queue' };
|
||||
}
|
||||
|
||||
try {
|
||||
mkdirSync(migrationsDir(), { recursive: true });
|
||||
const already = existingEntriesForVersion('0.22.4');
|
||||
let added = 0;
|
||||
for (const src of sourcesWithIssues) {
|
||||
if (already.has(src.source_id)) continue;
|
||||
const entry: PendingHostWorkEntry = {
|
||||
migration: '0.22.4',
|
||||
ts: new Date().toISOString(),
|
||||
skill: 'skills/migrations/v0.22.4.md',
|
||||
reason: `${src.total} frontmatter issue(s) in source ${src.source_id}`,
|
||||
source_id: src.source_id,
|
||||
source_path: src.source_path,
|
||||
command: `gbrain frontmatter validate ${src.source_path} --fix`,
|
||||
};
|
||||
appendFileSync(pendingHostWorkPath(), JSON.stringify(entry) + '\n');
|
||||
added++;
|
||||
}
|
||||
return {
|
||||
name: 'emit-todo',
|
||||
status: 'complete',
|
||||
detail: `appended ${added} entr${added === 1 ? 'y' : 'ies'} to ${pendingHostWorkPath()}`,
|
||||
};
|
||||
} catch (e) {
|
||||
return { name: 'emit-todo', status: 'failed', detail: e instanceof Error ? e.message : String(e) };
|
||||
}
|
||||
}
|
||||
|
||||
// ── Orchestrator ────────────────────────────────────────────
|
||||
|
||||
async function orchestrator(opts: OrchestratorOpts): Promise<OrchestratorResult> {
|
||||
console.log('');
|
||||
console.log('=== v0.22.4 — frontmatter-guard adoption ===');
|
||||
if (opts.dryRun) console.log(' (dry-run; no side effects)');
|
||||
console.log('');
|
||||
|
||||
const phases: OrchestratorPhaseResult[] = [];
|
||||
|
||||
phases.push(phaseASchema(opts));
|
||||
|
||||
const { phase: bPhase, report } = await phaseBAudit(opts);
|
||||
phases.push(bPhase);
|
||||
if (bPhase.status === 'failed') {
|
||||
return { version: '0.22.4', status: 'partial', phases };
|
||||
}
|
||||
|
||||
phases.push(phaseCEmitTodo(opts, report));
|
||||
|
||||
const overallStatus: 'complete' | 'partial' | 'failed' =
|
||||
phases.some(p => p.status === 'failed') ? 'partial' : 'complete';
|
||||
|
||||
return {
|
||||
version: '0.22.4',
|
||||
status: overallStatus,
|
||||
phases,
|
||||
pending_host_work: report?.per_source.filter(s => s.total > 0).length ?? 0,
|
||||
};
|
||||
}
|
||||
|
||||
export const v0_22_4: Migration = {
|
||||
version: '0.22.4',
|
||||
featurePitch: {
|
||||
headline: 'Frontmatter-guard ships — broken brain pages can\'t hide',
|
||||
description:
|
||||
'gbrain v0.22.4 adds end-to-end frontmatter validation: a `gbrain frontmatter` CLI ' +
|
||||
'(validate / audit / install-hook), a `frontmatter_integrity` doctor subcheck, a ' +
|
||||
'pre-commit hook helper, and a new frontmatter-guard skill. The migration is audit-only ' +
|
||||
'(it never mutates your brain) — it scans every registered source, writes a per-source ' +
|
||||
'report to ~/.gbrain/migrations/v0.22.4-audit.json, and queues a TODO with the exact fix ' +
|
||||
'command. Run `gbrain frontmatter validate <source-path> --fix` to repair (creates .bak ' +
|
||||
'backups). Resolves all 7 check-resolvable warnings on master; ships frontmatter-guard.',
|
||||
},
|
||||
orchestrator,
|
||||
};
|
||||
|
||||
/** Exported for unit tests. */
|
||||
export const __testing = {
|
||||
phaseASchema,
|
||||
phaseBAudit,
|
||||
phaseCEmitTodo,
|
||||
auditReportPath,
|
||||
pendingHostWorkPath,
|
||||
};
|
||||
@@ -0,0 +1,177 @@
|
||||
/**
|
||||
* v0.20.0 Cathedral II Layer 8 D3 — reconcile-links batch command.
|
||||
*
|
||||
* Closes the v0.19.0 Layer 6 doc↔impl order-dependency bug. When a
|
||||
* markdown guide cites `src/core/sync.ts:42` but the code source
|
||||
* hasn't been synced yet, the forward-scan at import time inserts
|
||||
* nothing because `addLink`'s inner SELECT drops edges to missing
|
||||
* pages. The guide and the code eventually both exist, but the edge
|
||||
* never materialized.
|
||||
*
|
||||
* D3 fixes this batch-style: walk every markdown page, re-run
|
||||
* `extractCodeRefs`, and call `addLink(md, code, ..., 'documents')` +
|
||||
* reverse for each hit. ON CONFLICT DO NOTHING on the `links` table
|
||||
* makes the operation idempotent — edges that already exist stay,
|
||||
* new edges land.
|
||||
*
|
||||
* Why batch over per-import-reverse-scan: codex 2-phase review
|
||||
* flagged the per-import approach as O(N) ILIKE/JOIN queries per
|
||||
* code file imported. On a 47K-page brain first-syncing 5K code
|
||||
* files, that's 5K ILIKE scans. A user-triggered batch pass on an
|
||||
* already-synced brain is one walk, fully indexed via the existing
|
||||
* slug lookup in addLink.
|
||||
*/
|
||||
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { extractCodeRefs } from '../core/link-extraction.ts';
|
||||
import { slugifyCodePath } from '../core/sync.ts';
|
||||
import { createProgress } from '../core/progress.ts';
|
||||
import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts';
|
||||
|
||||
export interface ReconcileLinksResult {
|
||||
status: 'ok' | 'auto_link_disabled';
|
||||
markdownPagesScanned: number;
|
||||
codeRefsFound: number;
|
||||
edgesAttempted: number;
|
||||
edgesTargetsMissing: number;
|
||||
}
|
||||
|
||||
export interface ReconcileLinksOpts {
|
||||
sourceId?: string;
|
||||
dryRun?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Scan every markdown page for code-path references (e.g.
|
||||
* `src/core/sync.ts`, `lib/foo.py:42`) and create bidirectional
|
||||
* doc↔impl edges (`documents` + `documented_by`) for each hit
|
||||
* that resolves to a code page. Idempotent via ON CONFLICT DO
|
||||
* NOTHING in the underlying addLink path.
|
||||
*
|
||||
* Called by `gbrain reconcile-links` CLI surface. Respects the
|
||||
* `auto_link` config: if the user has disabled auto-linking on
|
||||
* put_page, reconcile-links doesn't silently re-populate those
|
||||
* edges either.
|
||||
*/
|
||||
export async function runReconcileLinks(
|
||||
engine: BrainEngine,
|
||||
opts: ReconcileLinksOpts = {},
|
||||
): Promise<ReconcileLinksResult> {
|
||||
// Respect auto_link config (same gate put_page uses). A user that
|
||||
// explicitly turned off auto-link doesn't want reconcile-links
|
||||
// writing edges back either.
|
||||
const autoLinkCfg = await engine.getConfig('auto_link');
|
||||
if (autoLinkCfg === 'false') {
|
||||
return {
|
||||
status: 'auto_link_disabled',
|
||||
markdownPagesScanned: 0,
|
||||
codeRefsFound: 0,
|
||||
edgesAttempted: 0,
|
||||
edgesTargetsMissing: 0,
|
||||
};
|
||||
}
|
||||
|
||||
const progress = createProgress(cliOptsToProgressOptions(getCliOptions()));
|
||||
|
||||
// Walk all markdown slugs. listPages(markdown-only filter) isn't exposed,
|
||||
// so filter at call time via page_kind. Not using getAllSlugs because we
|
||||
// also need compiled_truth + timeline for extractCodeRefs.
|
||||
const mdSlugs = (await engine.executeRaw<{ slug: string }>(
|
||||
`SELECT slug FROM pages WHERE page_kind = 'markdown' ORDER BY slug`,
|
||||
)).map(r => r.slug);
|
||||
|
||||
progress.start('reconcile_links.scan', mdSlugs.length);
|
||||
|
||||
let codeRefsFound = 0;
|
||||
let edgesAttempted = 0;
|
||||
let edgesTargetsMissing = 0;
|
||||
|
||||
// Fetch pages one at a time via getPage (no bulk read helper exists yet).
|
||||
// On a 47K-page brain this is the slow path; a v0.20.x follow-up can add
|
||||
// getPagesBatch. For the typical 2K–5K markdown count it's fine.
|
||||
for (const mdSlug of mdSlugs) {
|
||||
const page = await engine.getPage(mdSlug);
|
||||
if (!page) {
|
||||
progress.tick(1, mdSlug);
|
||||
continue;
|
||||
}
|
||||
const haystack = (page.compiled_truth || '') + '\n' + (page.timeline || '');
|
||||
const refs = extractCodeRefs(haystack);
|
||||
if (refs.length === 0) {
|
||||
progress.tick(1, mdSlug);
|
||||
continue;
|
||||
}
|
||||
codeRefsFound += refs.length;
|
||||
|
||||
if (opts.dryRun) {
|
||||
progress.tick(1, `${mdSlug} (+${refs.length} refs)`);
|
||||
continue;
|
||||
}
|
||||
|
||||
for (const ref of refs) {
|
||||
const codeSlug = slugifyCodePath(ref.path);
|
||||
const ctx = ref.line ? `cited at ${ref.path}:${ref.line}` : ref.path;
|
||||
edgesAttempted++;
|
||||
try {
|
||||
// Forward: guide documents code. addLink's inner SELECT drops
|
||||
// silently if codeSlug isn't a page yet (benign — counted below).
|
||||
await engine.addLink(mdSlug, codeSlug, ctx, 'documents', 'markdown', mdSlug, 'compiled_truth');
|
||||
await engine.addLink(codeSlug, mdSlug, ref.path, 'documented_by', 'markdown', mdSlug, 'compiled_truth');
|
||||
} catch (e: unknown) {
|
||||
// Per-link errors don't abort the batch. Track them for the summary.
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
if (/not found|does not exist/i.test(msg)) {
|
||||
edgesTargetsMissing++;
|
||||
} else {
|
||||
// Real error — log but keep going. Agents can inspect progress events.
|
||||
console.warn(`[reconcile-links] ${mdSlug} → ${codeSlug}: ${msg}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
progress.tick(1, `${mdSlug} (+${refs.length} refs)`);
|
||||
}
|
||||
|
||||
progress.finish();
|
||||
|
||||
return {
|
||||
status: 'ok',
|
||||
markdownPagesScanned: mdSlugs.length,
|
||||
codeRefsFound,
|
||||
edgesAttempted,
|
||||
edgesTargetsMissing,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* CLI entry. Parses argv, runs runReconcileLinks, prints a summary.
|
||||
* --dry-run reports counts without writing. --json emits machine output.
|
||||
*/
|
||||
export async function runReconcileLinksCli(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
const dryRun = args.includes('--dry-run');
|
||||
const jsonOut = args.includes('--json');
|
||||
|
||||
const result = await runReconcileLinks(engine, { dryRun });
|
||||
|
||||
if (jsonOut) {
|
||||
console.log(JSON.stringify(result));
|
||||
return;
|
||||
}
|
||||
|
||||
if (result.status === 'auto_link_disabled') {
|
||||
console.log(
|
||||
'[reconcile-links] auto_link is disabled in config; skipping. ' +
|
||||
'Set `gbrain config set auto_link true` to re-enable.',
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
const header = dryRun ? 'reconcile-links (dry run)' : 'reconcile-links';
|
||||
console.log(
|
||||
`${header}: scanned ${result.markdownPagesScanned} markdown pages, ` +
|
||||
`found ${result.codeRefsFound} code refs, ` +
|
||||
`attempted ${result.edgesAttempted} edges` +
|
||||
(result.edgesTargetsMissing > 0
|
||||
? ` (${result.edgesTargetsMissing} targets missing code page)`
|
||||
: ''),
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,324 @@
|
||||
/**
|
||||
* v0.21.0 Cathedral II Layer 13 (E2) — `gbrain reindex-code`.
|
||||
*
|
||||
* Explicit backfill for v0.19.0 → v0.21.0 brains. Layer 12's
|
||||
* `sources.chunker_version` gate forces a re-walk next sync on any source
|
||||
* whose working tree hasn't drifted, but users who want the benefits NOW
|
||||
* (before the next sync) get this: walk every page where type='code', read
|
||||
* compiled_truth + frontmatter.file, re-import via importCodeFile. Pages
|
||||
* flow through the same code path as normal sync (chunker + embeddings +
|
||||
* content_hash folding), so a reindex is bit-identical to a fresh sync.
|
||||
*
|
||||
* Flags:
|
||||
* --source <id> Scope to one sources row. Omit = all code pages.
|
||||
* --dry-run Preview cost + page count, exit 0.
|
||||
* --yes Skip interactive [y/N]. Required for non-TTY + non-JSON.
|
||||
* --json Machine-readable ConfirmationRequired / result envelope.
|
||||
* --force Bypass importCodeFile's content_hash early-return. Use
|
||||
* this for paranoid full reindex when content_hash equals
|
||||
* but you still want a re-chunk + re-embed pass.
|
||||
*
|
||||
* Batched in chunks of 100 pages to avoid OOM on 47K-page brains (codex
|
||||
* review Finding 4.4). Idempotent: re-running on already-reindexed pages
|
||||
* is a no-op unless --force is passed.
|
||||
*/
|
||||
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { importCodeFile } from '../core/import-file.ts';
|
||||
import { estimateTokens } from '../core/chunkers/code.ts';
|
||||
import { EMBEDDING_MODEL, estimateEmbeddingCostUsd } from '../core/embedding.ts';
|
||||
import { errorFor, serializeError } from '../core/errors.ts';
|
||||
import { createInterface } from 'readline';
|
||||
import { createProgress } from '../core/progress.ts';
|
||||
import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts';
|
||||
|
||||
export interface ReindexCodeOpts {
|
||||
sourceId?: string;
|
||||
dryRun?: boolean;
|
||||
yes?: boolean;
|
||||
json?: boolean;
|
||||
force?: boolean;
|
||||
noEmbed?: boolean;
|
||||
/** Page batch size. Default 100 (codex Finding 4.4 OOM protection). */
|
||||
batchSize?: number;
|
||||
}
|
||||
|
||||
export interface ReindexCodeResult {
|
||||
status: 'ok' | 'dry_run' | 'cancelled' | 'source_id_required';
|
||||
codePages: number;
|
||||
reindexed: number;
|
||||
skipped: number;
|
||||
failed: number;
|
||||
totalTokens: number;
|
||||
costUsd: number;
|
||||
model: string;
|
||||
failures?: Array<{ slug: string; error: string }>;
|
||||
}
|
||||
|
||||
interface CodePageRow {
|
||||
slug: string;
|
||||
compiled_truth: string;
|
||||
frontmatter: Record<string, unknown> | null;
|
||||
}
|
||||
|
||||
async function fetchCodePages(
|
||||
engine: BrainEngine,
|
||||
sourceId: string | undefined,
|
||||
batchSize: number,
|
||||
offset: number,
|
||||
): Promise<CodePageRow[]> {
|
||||
// Direct SQL: listPages doesn't expose source_id filtering, and we need
|
||||
// compiled_truth + frontmatter anyway (not just the Page shape).
|
||||
const sourceClause = sourceId ? `AND p.source_id = '${sourceId.replace(/'/g, "''")}'` : '';
|
||||
const rows = await engine.executeRaw<CodePageRow>(
|
||||
`SELECT p.slug, p.compiled_truth, p.frontmatter
|
||||
FROM pages p
|
||||
WHERE p.type = 'code' ${sourceClause}
|
||||
ORDER BY p.slug
|
||||
LIMIT ${batchSize} OFFSET ${offset}`,
|
||||
);
|
||||
return rows;
|
||||
}
|
||||
|
||||
async function countCodePages(engine: BrainEngine, sourceId: string | undefined): Promise<number> {
|
||||
const sourceClause = sourceId ? `AND p.source_id = '${sourceId.replace(/'/g, "''")}'` : '';
|
||||
const rows = await engine.executeRaw<{ n: string | number }>(
|
||||
`SELECT COUNT(*)::text AS n FROM pages p WHERE p.type = 'code' ${sourceClause}`,
|
||||
);
|
||||
if (rows.length === 0) return 0;
|
||||
const raw = rows[0]!.n;
|
||||
return typeof raw === 'string' ? parseInt(raw, 10) : raw;
|
||||
}
|
||||
|
||||
/**
|
||||
* Estimate total embedding cost for a reindex. Walks every code page's
|
||||
* compiled_truth and sums tokens. Conservative: does not try to detect
|
||||
* unchanged chunks (the incremental embedding cache in importCodeFile does
|
||||
* that; this estimate is the ceiling, not the floor).
|
||||
*/
|
||||
async function estimateReindexCost(
|
||||
engine: BrainEngine,
|
||||
sourceId: string | undefined,
|
||||
batchSize: number,
|
||||
): Promise<{ totalTokens: number; totalPages: number }> {
|
||||
let totalTokens = 0;
|
||||
let totalPages = 0;
|
||||
let offset = 0;
|
||||
while (true) {
|
||||
const batch = await fetchCodePages(engine, sourceId, batchSize, offset);
|
||||
if (batch.length === 0) break;
|
||||
for (const row of batch) {
|
||||
if (row.compiled_truth) totalTokens += estimateTokens(row.compiled_truth);
|
||||
totalPages++;
|
||||
}
|
||||
offset += batch.length;
|
||||
if (batch.length < batchSize) break;
|
||||
}
|
||||
return { totalTokens, totalPages };
|
||||
}
|
||||
|
||||
async function promptYesNo(question: string): Promise<boolean> {
|
||||
return new Promise((resolve) => {
|
||||
const rl = createInterface({ input: process.stdin, output: process.stdout });
|
||||
rl.question(question, (answer) => {
|
||||
rl.close();
|
||||
const a = answer.trim().toLowerCase();
|
||||
resolve(a === 'y' || a === 'yes');
|
||||
});
|
||||
rl.on('close', () => resolve(false));
|
||||
});
|
||||
}
|
||||
|
||||
export async function runReindexCode(
|
||||
engine: BrainEngine,
|
||||
opts: ReindexCodeOpts = {},
|
||||
): Promise<ReindexCodeResult> {
|
||||
const batchSize = opts.batchSize ?? 100;
|
||||
|
||||
const { totalTokens, totalPages } = await estimateReindexCost(engine, opts.sourceId, batchSize);
|
||||
const costUsd = estimateEmbeddingCostUsd(totalTokens);
|
||||
|
||||
if (opts.dryRun) {
|
||||
return {
|
||||
status: 'dry_run',
|
||||
codePages: totalPages,
|
||||
reindexed: 0,
|
||||
skipped: 0,
|
||||
failed: 0,
|
||||
totalTokens,
|
||||
costUsd,
|
||||
model: EMBEDDING_MODEL,
|
||||
};
|
||||
}
|
||||
|
||||
if (totalPages === 0) {
|
||||
return {
|
||||
status: 'ok',
|
||||
codePages: 0,
|
||||
reindexed: 0,
|
||||
skipped: 0,
|
||||
failed: 0,
|
||||
totalTokens: 0,
|
||||
costUsd: 0,
|
||||
model: EMBEDDING_MODEL,
|
||||
};
|
||||
}
|
||||
|
||||
// Walk every code page, re-run importCodeFile with compiled_truth as
|
||||
// the content source. relativePath comes from frontmatter.file (set by
|
||||
// the original importCodeFile call). Progress via stderr reporter.
|
||||
const reporter = createProgress(cliOptsToProgressOptions(getCliOptions()));
|
||||
reporter.start('reindex_code.pages', totalPages);
|
||||
|
||||
let reindexed = 0;
|
||||
let skipped = 0;
|
||||
let failed = 0;
|
||||
const failures: Array<{ slug: string; error: string }> = [];
|
||||
let offset = 0;
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
const batch = await fetchCodePages(engine, opts.sourceId, batchSize, offset);
|
||||
if (batch.length === 0) break;
|
||||
|
||||
for (const row of batch) {
|
||||
const fm = row.frontmatter ?? {};
|
||||
const relPath = typeof fm.file === 'string' ? fm.file : null;
|
||||
if (!relPath) {
|
||||
failed++;
|
||||
failures.push({ slug: row.slug, error: 'missing frontmatter.file' });
|
||||
reporter.tick();
|
||||
continue;
|
||||
}
|
||||
if (!row.compiled_truth) {
|
||||
failed++;
|
||||
failures.push({ slug: row.slug, error: 'missing compiled_truth' });
|
||||
reporter.tick();
|
||||
continue;
|
||||
}
|
||||
try {
|
||||
const result = await importCodeFile(engine, relPath, row.compiled_truth, {
|
||||
noEmbed: opts.noEmbed,
|
||||
force: opts.force,
|
||||
});
|
||||
if (result.status === 'imported') reindexed++;
|
||||
else if (result.status === 'skipped') skipped++;
|
||||
else {
|
||||
failed++;
|
||||
failures.push({ slug: row.slug, error: result.error ?? result.status });
|
||||
}
|
||||
} catch (e: unknown) {
|
||||
failed++;
|
||||
failures.push({ slug: row.slug, error: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
reporter.tick();
|
||||
}
|
||||
|
||||
offset += batch.length;
|
||||
if (batch.length < batchSize) break;
|
||||
}
|
||||
} finally {
|
||||
reporter.finish();
|
||||
}
|
||||
|
||||
return {
|
||||
status: 'ok',
|
||||
codePages: totalPages,
|
||||
reindexed,
|
||||
skipped,
|
||||
failed,
|
||||
totalTokens,
|
||||
costUsd,
|
||||
model: EMBEDDING_MODEL,
|
||||
failures: failures.length > 0 ? failures : undefined,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* CLI entrypoint. Parses argv, wires cost-preview gate + JSON/TTY branching,
|
||||
* delegates to runReindexCode. Exit codes: 0 on success/dry-run, 2 on
|
||||
* ConfirmationRequired (matches sync --all), 1 on runtime error.
|
||||
*/
|
||||
export async function runReindexCodeCli(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
const sourceIdx = args.indexOf('--source');
|
||||
const sourceId = sourceIdx >= 0 ? args[sourceIdx + 1] : undefined;
|
||||
const dryRun = args.includes('--dry-run');
|
||||
const yes = args.includes('--yes') || args.includes('-y');
|
||||
const json = args.includes('--json');
|
||||
const force = args.includes('--force');
|
||||
const noEmbed = args.includes('--no-embed');
|
||||
|
||||
if (dryRun) {
|
||||
const result = await runReindexCode(engine, { sourceId, dryRun: true, yes, json, force, noEmbed });
|
||||
if (json) {
|
||||
console.log(JSON.stringify(result));
|
||||
} else {
|
||||
console.log(
|
||||
`reindex-code preview: ${result.codePages} code page(s), ` +
|
||||
`~${result.totalTokens.toLocaleString()} tokens, ` +
|
||||
`est. $${result.costUsd.toFixed(2)} on ${result.model}.`,
|
||||
);
|
||||
console.log('--dry-run: exit without reindexing.');
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Cost preview + gate, before touching the DB.
|
||||
if (!noEmbed) {
|
||||
const preview = await estimateReindexCost(engine, sourceId, 100);
|
||||
const costUsd = estimateEmbeddingCostUsd(preview.totalTokens);
|
||||
const previewMsg =
|
||||
`reindex-code: ${preview.totalPages} code page(s), ` +
|
||||
`~${preview.totalTokens.toLocaleString()} tokens, ` +
|
||||
`est. $${costUsd.toFixed(2)} on ${EMBEDDING_MODEL}.`;
|
||||
|
||||
if (preview.totalPages === 0) {
|
||||
if (json) {
|
||||
console.log(JSON.stringify({ status: 'ok', codePages: 0, reindexed: 0, skipped: 0, failed: 0, totalTokens: 0, costUsd: 0, model: EMBEDDING_MODEL }));
|
||||
} else {
|
||||
console.log('No code pages to reindex.');
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if (!yes) {
|
||||
const isTTY = Boolean(process.stdout.isTTY) && Boolean(process.stdin.isTTY);
|
||||
if (!isTTY || json) {
|
||||
const envelope = serializeError(errorFor({
|
||||
class: 'ConfirmationRequired',
|
||||
code: 'cost_preview_requires_yes',
|
||||
message: previewMsg,
|
||||
hint: 'Pass --yes to proceed, or --dry-run to see the preview and exit 0.',
|
||||
}));
|
||||
console.log(JSON.stringify({ error: envelope, preview, costUsd, model: EMBEDDING_MODEL }));
|
||||
process.exit(2);
|
||||
}
|
||||
console.log(previewMsg);
|
||||
const answer = await promptYesNo('Proceed? [y/N] ');
|
||||
if (!answer) {
|
||||
console.log('Cancelled.');
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const result = await runReindexCode(engine, { sourceId, yes, json, force, noEmbed });
|
||||
if (json) {
|
||||
console.log(JSON.stringify(result));
|
||||
} else {
|
||||
console.log(
|
||||
`reindex-code: ${result.reindexed} reindexed, ${result.skipped} skipped, ${result.failed} failed ` +
|
||||
`(${result.codePages} total code pages, ~${result.totalTokens.toLocaleString()} tokens, ` +
|
||||
`est. $${result.costUsd.toFixed(2)}).`,
|
||||
);
|
||||
if (result.failures && result.failures.length > 0) {
|
||||
console.log(`\n${result.failures.length} failure(s):`);
|
||||
for (const f of result.failures.slice(0, 10)) {
|
||||
console.log(` ${f.slug}: ${f.error}`);
|
||||
}
|
||||
if (result.failures.length > 10) {
|
||||
console.log(` ... and ${result.failures.length - 10} more`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+15
-3
@@ -1,7 +1,19 @@
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { startMcpServer } from '../mcp/server.ts';
|
||||
import { startHttpTransport } from '../mcp/http-transport.ts';
|
||||
|
||||
export async function runServe(engine: BrainEngine) {
|
||||
console.error('Starting GBrain MCP server (stdio)...');
|
||||
await startMcpServer(engine);
|
||||
export async function runServe(engine: BrainEngine, args: string[] = []) {
|
||||
const useHttp = args.includes('--http');
|
||||
const portIdx = args.indexOf('--port');
|
||||
const port = portIdx >= 0 ? parseInt(args[portIdx + 1]) || 8787 : 8787;
|
||||
|
||||
if (useHttp) {
|
||||
console.error(`Starting GBrain MCP server (HTTP on port ${port})...`);
|
||||
await startHttpTransport({ port, engine });
|
||||
// Keep alive
|
||||
await new Promise(() => {});
|
||||
} else {
|
||||
console.error('Starting GBrain MCP server (stdio)...');
|
||||
await startMcpServer(engine);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,245 @@
|
||||
import { join } from 'path';
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { loadStorageConfig, validateStorageConfig, getStorageTier } from '../core/storage-config.ts';
|
||||
import type { StorageConfig, StorageTier } from '../core/storage-config.ts';
|
||||
import { walkBrainRepo, type DiskFileEntry } from '../core/disk-walk.ts';
|
||||
import { getDefaultSourcePath } from '../core/source-resolver.ts';
|
||||
|
||||
/**
|
||||
* Distinct nominal types for the two tier-keyed numeric maps. Both shapes
|
||||
* are `Record<StorageTier, number>` structurally — but they carry
|
||||
* semantically different units (page COUNT vs disk BYTES). Distinct types
|
||||
* make accidental swaps a compile-time error rather than a silent display
|
||||
* bug. Issue #11 of the eng review.
|
||||
*/
|
||||
export type PageCountsByTier = Record<StorageTier, number> & { __brand?: 'page-counts' };
|
||||
export type DiskUsageByTier = Record<StorageTier, number> & { __brand?: 'disk-bytes' };
|
||||
|
||||
/**
|
||||
* Pure-data result of a storage-status query. No side effects, no I/O
|
||||
* beyond the engine call and one filesystem walk. Consumed by both the
|
||||
* JSON formatter and the human formatter; kept narrow so it's a stable
|
||||
* MCP/scripting contract (D14: storage_status is read-only MCP-exposed).
|
||||
*/
|
||||
export interface StorageStatusResult {
|
||||
config: StorageConfig | null;
|
||||
repoPath: string | null;
|
||||
totalPages: number;
|
||||
pagesByTier: PageCountsByTier;
|
||||
missingFiles: Array<{ slug: string; expectedPath: string }>;
|
||||
diskUsageByTier: DiskUsageByTier;
|
||||
warnings: string[];
|
||||
}
|
||||
|
||||
// ── Dispatcher ────────────────────────────────────────────
|
||||
|
||||
export async function runStorage(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
const subcommand = args[0];
|
||||
if (!subcommand || subcommand === 'status') {
|
||||
await runStorageStatus(engine, args.slice(1));
|
||||
return;
|
||||
}
|
||||
console.error(`Unknown storage subcommand: ${subcommand}`);
|
||||
console.error('Available subcommands: status');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
async function runStorageStatus(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
warnIfPGLite(engine);
|
||||
|
||||
// Resolution chain (D5, Issue #3): explicit --repo → typed accessor → null.
|
||||
// No cwd fallback. The original silent footgun is dead.
|
||||
let repoPath: string | null = null;
|
||||
const repoIdx = args.indexOf('--repo');
|
||||
if (repoIdx !== -1 && args[repoIdx + 1]) {
|
||||
repoPath = args[repoIdx + 1];
|
||||
} else {
|
||||
repoPath = await getDefaultSourcePath(engine);
|
||||
}
|
||||
|
||||
const result = await getStorageStatus(engine, repoPath);
|
||||
|
||||
if (args.includes('--json')) {
|
||||
console.log(formatStorageStatusJson(result));
|
||||
return;
|
||||
}
|
||||
console.log(formatStorageStatusHuman(result));
|
||||
}
|
||||
|
||||
/**
|
||||
* D4: storage tiering on PGLite is a partial feature. The "DB" the pages
|
||||
* live in IS the local file gbrain uses for everything else, so "db_only"
|
||||
* has no real offload effect. The .gitignore management still helps
|
||||
* (keeps bulk content out of git history), so we warn but proceed.
|
||||
*
|
||||
* Once-per-process via a module-local flag — sub-commands invoked from a
|
||||
* single CLI run share the same warning.
|
||||
*/
|
||||
let _pgliteWarned = false;
|
||||
function warnIfPGLite(engine: BrainEngine): void {
|
||||
if (_pgliteWarned) return;
|
||||
if (engine.kind !== 'pglite') return;
|
||||
_pgliteWarned = true;
|
||||
console.warn(
|
||||
`Note: storage tiering has limited effect on PGLite — pages live in your ` +
|
||||
`local database file regardless of tier. The .gitignore management still ` +
|
||||
`keeps bulk content out of git history. To get full tiering, migrate to ` +
|
||||
`Postgres with \`gbrain migrate --to supabase\`.`,
|
||||
);
|
||||
}
|
||||
|
||||
/** Reset for tests. */
|
||||
export function __resetPGLiteWarn(): void {
|
||||
_pgliteWarned = false;
|
||||
}
|
||||
|
||||
// ── Pure data ─────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Compute the storage status against the given engine + brain repo path.
|
||||
*
|
||||
* Side-effect-free apart from the engine.listPages call and one recursive
|
||||
* filesystem walk. Pure for testability — formatters are tested separately.
|
||||
*
|
||||
* Returns null `config` when no gbrain.yml is present at repoPath. In that
|
||||
* case pagesByTier is all zeros for db_tracked/db_only and totals roll up
|
||||
* into unspecified.
|
||||
*/
|
||||
export async function getStorageStatus(
|
||||
engine: BrainEngine,
|
||||
repoPath: string | null,
|
||||
): Promise<StorageStatusResult> {
|
||||
const config = repoPath ? loadStorageConfig(repoPath) : null;
|
||||
const warnings = config ? validateStorageConfig(config) : [];
|
||||
|
||||
const pagesByTier: PageCountsByTier = { db_tracked: 0, db_only: 0, unspecified: 0 };
|
||||
const diskUsageByTier: DiskUsageByTier = { db_tracked: 0, db_only: 0, unspecified: 0 };
|
||||
const missingFiles: Array<{ slug: string; expectedPath: string }> = [];
|
||||
|
||||
// Single recursive walk of the brain repo (Issue #14). Replaces per-page
|
||||
// existsSync+statSync — was ~400K syscalls on 200K-page brains, now ~one
|
||||
// per directory + one stat per .md file, plus O(1) lookups below.
|
||||
const fileMap: Map<string, DiskFileEntry> = repoPath ? walkBrainRepo(repoPath) : new Map();
|
||||
|
||||
const pages = await engine.listPages({ limit: 1_000_000 });
|
||||
|
||||
for (const page of pages) {
|
||||
const tier = config ? getStorageTier(page.slug, config) : 'unspecified';
|
||||
pagesByTier[tier]++;
|
||||
if (!repoPath) continue;
|
||||
const entry = fileMap.get(page.slug);
|
||||
if (entry) {
|
||||
diskUsageByTier[tier] += entry.size;
|
||||
} else if (config && tier === 'db_only') {
|
||||
missingFiles.push({ slug: page.slug, expectedPath: join(repoPath, page.slug + '.md') });
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
config,
|
||||
repoPath,
|
||||
totalPages: pages.length,
|
||||
pagesByTier,
|
||||
missingFiles,
|
||||
diskUsageByTier,
|
||||
warnings,
|
||||
};
|
||||
}
|
||||
|
||||
// ── JSON formatter ────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Serialize StorageStatusResult to a stable JSON contract. Indented for
|
||||
* human readability; agents/orchestrators can parse with a standard
|
||||
* JSON.parse. Schema is the StorageStatusResult interface above.
|
||||
*/
|
||||
export function formatStorageStatusJson(result: StorageStatusResult): string {
|
||||
return JSON.stringify(result, null, 2);
|
||||
}
|
||||
|
||||
// ── Human formatter ───────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Render StorageStatusResult to ASCII text suitable for terminal output.
|
||||
* D10 lock: ASCII separators only — universally portable. No unicode
|
||||
* box-drawing.
|
||||
*/
|
||||
export function formatStorageStatusHuman(result: StorageStatusResult): string {
|
||||
const lines: string[] = [];
|
||||
lines.push('Storage Status');
|
||||
lines.push('==============');
|
||||
lines.push('');
|
||||
|
||||
if (!result.config) {
|
||||
lines.push('No gbrain.yml configuration found.');
|
||||
if (result.repoPath) lines.push(`Checked: ${result.repoPath}/gbrain.yml`);
|
||||
lines.push('');
|
||||
lines.push('All pages are stored in git by default.');
|
||||
lines.push(`Total pages: ${result.totalPages}`);
|
||||
return lines.join('\n');
|
||||
}
|
||||
|
||||
lines.push(`Repository: ${result.repoPath}`);
|
||||
lines.push(`Total pages: ${result.totalPages}`);
|
||||
lines.push('');
|
||||
lines.push('Storage Tiers:');
|
||||
lines.push('-------------');
|
||||
lines.push(`DB tracked: ${result.pagesByTier.db_tracked.toLocaleString()} pages`);
|
||||
lines.push(`DB only: ${result.pagesByTier.db_only.toLocaleString()} pages`);
|
||||
lines.push(`Unspecified: ${result.pagesByTier.unspecified.toLocaleString()} pages`);
|
||||
|
||||
if (result.diskUsageByTier.db_tracked > 0 || result.diskUsageByTier.db_only > 0) {
|
||||
lines.push('');
|
||||
lines.push('Disk Usage:');
|
||||
lines.push('-----------');
|
||||
if (result.diskUsageByTier.db_tracked > 0) {
|
||||
lines.push(`DB tracked: ${formatBytes(result.diskUsageByTier.db_tracked)}`);
|
||||
}
|
||||
if (result.diskUsageByTier.db_only > 0) {
|
||||
lines.push(`DB only: ${formatBytes(result.diskUsageByTier.db_only)}`);
|
||||
}
|
||||
if (result.diskUsageByTier.unspecified > 0) {
|
||||
lines.push(`Unspecified: ${formatBytes(result.diskUsageByTier.unspecified)}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (result.missingFiles.length > 0) {
|
||||
lines.push('');
|
||||
lines.push('Missing Files (need restore):');
|
||||
lines.push('-----------------------------');
|
||||
for (const missing of result.missingFiles.slice(0, 10)) {
|
||||
lines.push(` ${missing.slug}`);
|
||||
}
|
||||
if (result.missingFiles.length > 10) {
|
||||
lines.push(` ... and ${result.missingFiles.length - 10} more`);
|
||||
}
|
||||
lines.push('');
|
||||
lines.push(`Use: gbrain export --restore-only --repo "${result.repoPath}"`);
|
||||
}
|
||||
|
||||
if (result.warnings.length > 0) {
|
||||
lines.push('');
|
||||
lines.push('Warnings:');
|
||||
lines.push('---------');
|
||||
for (const warning of result.warnings) lines.push(` ! ${warning}`);
|
||||
}
|
||||
|
||||
lines.push('');
|
||||
lines.push('Configuration:');
|
||||
lines.push('--------------');
|
||||
lines.push('DB tracked directories:');
|
||||
for (const dir of result.config.db_tracked) lines.push(` - ${dir}`);
|
||||
lines.push('');
|
||||
lines.push('DB-only directories:');
|
||||
for (const dir of result.config.db_only) lines.push(` - ${dir}`);
|
||||
|
||||
return lines.join('\n');
|
||||
}
|
||||
|
||||
function formatBytes(bytes: number): string {
|
||||
if (bytes === 0) return '0 B';
|
||||
const k = 1024;
|
||||
const sizes = ['B', 'KB', 'MB', 'GB', 'TB'];
|
||||
const i = Math.floor(Math.log(bytes) / Math.log(k));
|
||||
return parseFloat((bytes / Math.pow(k, i)).toFixed(1)) + ' ' + sizes[i];
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user