Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 edf7fc6b5a fix(skills): green CI for vault skills — resolver health, fixture lint, privacy literal, llms bundle
- RESOLVER.md rows now round-trip against SKILL.md frontmatter triggers
  ('put this setup change in the vault'; replace 'safe/free gbrain index'
  with 'import my vault to gbrain').
- Rewrite both routing-eval.jsonl fixture sets: positives embed a trigger
  phrase in natural context (fixture linter rejects verbatim tautologies),
  declare ambiguous_with for legitimate co-fires (capture, idea-ingest),
  and negatives no longer collide with skillpack-harvest/setup triggers.
  check:resolver --strict is clean (0 errors, 0 warnings).
- Privacy: the vault-capture regression test sources the banned fork-name
  pattern from harvest-lint's DEFAULT_PRIVATE_PATTERNS instead of the
  literal, so scripts/check-privacy.sh passes.
- Regenerate llms-full.txt after the RESOLVER.md edit (build-llms
  freshness gate).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 11:03:38 -07:00
b7f42f099a feat(skills): skillify vault skills — obsidian-gbrain-safe-index + skill-vault-capture-policy (takeover of #2685)
Salvages the intended 8-file change from PR #2685 (two vault skills with
routing-eval fixtures + unit/E2E tests, registered in RESOLVER.md,
manifest.json, and openclaw.plugin.json) without the ~670 unrelated
third-party skill dumps that contaminated that branch.

Also scrubs private user/agent names from skill-vault-capture-policy per
the repo privacy rule, with a regression test.

Takeover of #2685.

Co-authored-by: MortdAdam <MortdAdam@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:27:59 -07:00
30 changed files with 554 additions and 732 deletions
+3 -1
View File
@@ -1411,7 +1411,8 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
| "get more out of gbrain", "is my brain set up right", "weekly brain checkup", "advise me on my brain", "gbrain advisor" | `skills/gbrain-advisor/SKILL.md` |
| Save or load reports | `skills/reports/SKILL.md` |
| "Create a skill", "improve this skill" | `skills/skill-creator/SKILL.md` |
| "Skillify this", "is this a skill?", "make this proper" | `skills/skillify/SKILL.md` |
| "save this learning to the vault", "capture this skill in Obsidian", "record this workflow in my notes", "put this setup change in the vault" | `skills/skill-vault-capture-policy/SKILL.md` |
| "Skillify this", "is this a skill?", "make this proper", "add tests and evals for this" | `skills/skillify/SKILL.md` |
| "Compress my resolver", "AGENTS.md too large", "RESOLVER.md too big", "functional area dispatcher", "shrink routing table" | `skills/functional-area-resolver/SKILL.md` |
| "Is gbrain healthy?", morning health check, skillpack-check | `skills/skillpack-check/SKILL.md` |
| "harvest this skill into gbrain", "publish this skill to gbrain", "lift this skill upstream", "share this skill with other gbrain clients", "promote my skill to gbrain" | `skills/skillpack-harvest/SKILL.md` |
@@ -1429,6 +1430,7 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
| "Set up GBrain", first boot | `skills/setup/SKILL.md` |
| "Now what?", "fill my brain", "cold start", "bootstrap", "import my data", "what should I import first" | `skills/cold-start/SKILL.md` |
| "Migrate from Obsidian/Notion/Logseq" | `skills/migrate/SKILL.md` |
| "Connect Obsidian to gbrain", "import my vault to gbrain", "sync vault and gbrain", "embed gbrain after vault update", "is gbrain synced with my vault" | `skills/obsidian-gbrain-safe-index/SKILL.md` |
| Brain health check, maintenance run | `skills/maintain/SKILL.md` |
| "Extract links", "build link graph", "populate timeline" | `skills/maintain/SKILL.md` (extraction sections) |
| "Run dream", "process today's session", "synthesize my conversations", "consolidate yesterday's conversations", "what patterns did you see", "did the dream cycle run" | `skills/maintain/SKILL.md` (dream cycle section) |
+3 -2
View File
@@ -39,8 +39,8 @@
"skills/briefing",
"skills/citation-fixer",
"skills/concept-synthesis",
"skills/cross-modal-review",
"skills/cron-scheduler",
"skills/cross-modal-review",
"skills/daily-task-manager",
"skills/daily-task-prep",
"skills/data-research",
@@ -54,10 +54,11 @@
"skills/media-ingest",
"skills/meeting-ingestion",
"skills/minion-orchestrator",
"skills/obsidian-gbrain-safe-index",
"skills/perplexity-research",
"skills/query",
"skills/reports",
"skills/repo-architecture",
"skills/reports",
"skills/signal-detector",
"skills/skill-creator",
"skills/skillify",
+3 -1
View File
@@ -60,7 +60,8 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
| "get more out of gbrain", "is my brain set up right", "weekly brain checkup", "advise me on my brain", "gbrain advisor" | `skills/gbrain-advisor/SKILL.md` |
| Save or load reports | `skills/reports/SKILL.md` |
| "Create a skill", "improve this skill" | `skills/skill-creator/SKILL.md` |
| "Skillify this", "is this a skill?", "make this proper" | `skills/skillify/SKILL.md` |
| "save this learning to the vault", "capture this skill in Obsidian", "record this workflow in my notes", "put this setup change in the vault" | `skills/skill-vault-capture-policy/SKILL.md` |
| "Skillify this", "is this a skill?", "make this proper", "add tests and evals for this" | `skills/skillify/SKILL.md` |
| "Compress my resolver", "AGENTS.md too large", "RESOLVER.md too big", "functional area dispatcher", "shrink routing table" | `skills/functional-area-resolver/SKILL.md` |
| "Is gbrain healthy?", morning health check, skillpack-check | `skills/skillpack-check/SKILL.md` |
| "harvest this skill into gbrain", "publish this skill to gbrain", "lift this skill upstream", "share this skill with other gbrain clients", "promote my skill to gbrain" | `skills/skillpack-harvest/SKILL.md` |
@@ -78,6 +79,7 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
| "Set up GBrain", first boot | `skills/setup/SKILL.md` |
| "Now what?", "fill my brain", "cold start", "bootstrap", "import my data", "what should I import first" | `skills/cold-start/SKILL.md` |
| "Migrate from Obsidian/Notion/Logseq" | `skills/migrate/SKILL.md` |
| "Connect Obsidian to gbrain", "import my vault to gbrain", "sync vault and gbrain", "embed gbrain after vault update", "is gbrain synced with my vault" | `skills/obsidian-gbrain-safe-index/SKILL.md` |
| Brain health check, maintenance run | `skills/maintain/SKILL.md` |
| "Extract links", "build link graph", "populate timeline" | `skills/maintain/SKILL.md` (extraction sections) |
| "Run dream", "process today's session", "synthesize my conversations", "consolidate yesterday's conversations", "what patterns did you see", "did the dream cycle run" | `skills/maintain/SKILL.md` (dream cycle section) |
+11 -1
View File
@@ -2,7 +2,7 @@
"name": "gbrain",
"version": "0.32.3.0",
"conformance_version": "1.0.0",
"description": "Personal knowledge brain with hybrid RAG search \u2014 GStack mod for agent platforms",
"description": "Personal knowledge brain with hybrid RAG search GStack mod for agent platforms",
"skills": [
{
"name": "ingest",
@@ -34,6 +34,11 @@
"path": "migrate/SKILL.md",
"description": "Universal migration from Obsidian, Notion, Logseq, markdown, CSV, JSON, Roam"
},
{
"name": "obsidian-gbrain-safe-index",
"path": "obsidian-gbrain-safe-index/SKILL.md",
"description": "Connect and sync an Obsidian-style Markdown vault with gbrain using a cost-controlled import-first workflow, explicit embedding gates, and durable skill/workflow capture."
},
{
"name": "setup",
"path": "setup/SKILL.md",
@@ -263,6 +268,11 @@
"name": "skill-optimizer",
"path": "skill-optimizer/SKILL.md",
"description": "Self-evolving skill optimization via gbrain skillopt — SkillOpt-paper-grounded text-space optimizer with validation gating (median-of-3 + epsilon=0.05), bundled-skill safety, bootstrap review sentinel, per-skill DB lock, and atomic versioned writes."
},
{
"name": "skill-vault-capture-policy",
"path": "skill-vault-capture-policy/SKILL.md",
"description": "Capture durable operational learnings, new agent skills, and environment/setup changes into the Obsidian vault instead of transient chat memory."
}
],
"dependencies": {
+137
View File
@@ -0,0 +1,137 @@
---
name: obsidian-gbrain-safe-index
version: 1.0.0
description: |
Connect, maintain, and sync an Obsidian-style Markdown vault with gbrain while preserving a cost-controlled workflow: the vault remains the source of truth, gbrain is the searchable/embedded index, durable skills/workflows are captured into the vault, and paid embedding runs only after explicit approval.
triggers:
- "connect Obsidian to gbrain"
- "import my vault to gbrain"
- "sync vault and gbrain"
- "capture this skill in my vault"
- "embed gbrain after vault update"
- "is gbrain synced with my vault"
tools:
- terminal
- read_file
- search_files
- write_file
- patch
mutating: true
---
# Obsidian → gbrain Safe Index and Capture
## Contract
This skill guarantees:
- Treats the user's Obsidian-style Markdown vault as the source of truth before gbrain indexing.
- Keeps gbrain in conservative mode unless the user explicitly approves a more expensive mode.
- Imports vault changes with `--no-embed` first, then embeds only after explicit approval for the specific paid action.
- Captures durable new skills, workflows, and environment learnings into the vault instead of leaving them only in chat memory.
- Verifies every sync with concrete `gbrain stats`, search mode, and, when embeddings run, exact embedded chunk counts.
## Phases
1. **Resolve the vault path.**
- Prefer an existing environment variable such as `OBSIDIAN_VAULT_PATH` or `WIKI_PATH`.
- If no path is configured, search likely note directories and ask the user before writing.
- Verify the directory exists and contains markdown files or an `.obsidian` directory.
2. **Read vault operating rules before writing.**
- If the vault has `SCHEMA.md`, `index.md`, `log.md`, `AGENTS.md`, or similar operating files, read them before ingest/query/major edit.
- Respect immutable source folders such as `raw/` when the vault declares them.
- Use the vault's native link convention, usually Obsidian `[[wikilinks]]`, for durable relationships.
3. **MECE/capture decision.**
- If new knowledge belongs on an existing page, update that page.
- If it is a distinct recurring workflow or operational policy, create a small meta or concept page following the vault schema.
- Update the vault index/catalog for every new page when the vault maintains one.
- Append a log entry for meaningful vault updates when the vault maintains a log.
4. **Safe gbrain import path.**
- Pre-check source directory; do not import a nonexistent path.
- Run `gbrain config set search.mode conservative` before/after risky reinit steps.
- Run `gbrain import "$OBSIDIAN_VAULT_PATH" --no-embed`.
- Run `gbrain extract links --source fs --dir "$OBSIDIAN_VAULT_PATH"` when wikilinks changed materially.
5. **Paid embedding gate.**
- Do not run `gbrain embed --stale` unless the user explicitly asks or a prior instruction clearly approved this exact paid action.
- Before embedding, verify provider readiness with `gbrain providers test --model <provider:model>`.
- Confirm the configured embedding dimensions match the local schema.
- After embedding, verify `gbrain stats` and record exact `Pages`, `Chunks`, `Embedded`, and `Links` counts.
6. **Final verification and vault echo.**
- Run `gbrain stats` and `gbrain search modes`.
- If vault files changed, re-import with `--no-embed`; if embedding was approved, embed stale chunks afterward.
- Report what changed, what was free/local, what used API billing, and what remains pending.
## Output Format
Use a compact status table:
| Item | Status |
|---|---|
| Vault path | `/path` |
| Vault updated | yes/no + files |
| gbrain mode | conservative/balanced/tokenmax |
| Import | `--no-embed` completed / skipped / failed |
| Pages/chunks | exact counts from `gbrain stats` |
| Embeddings | exact count; note whether this run used API billing |
| Links | exact count |
| Background jobs | none / list exact jobs |
Then include:
- **Safe next step:** free/local action.
- **Paid next step:** embedding/LLM action, if any, with explicit approval requirement.
## Anti-Patterns
- Creating a duplicate skill/page when an existing Obsidian, gbrain, or vault-ingest skill already covers the workflow.
- Running `gbrain embed --stale`, `gbrain dream`, `gbrain autopilot --install`, `gbrain onboard --auto`, or `tokenmax` without explicit cost approval.
- Importing a nonexistent or wrong directory and treating a zero-page import as success.
- Forgetting to update the vault index/catalog and log after creating or materially updating vault pages.
- Recording API keys, tokens, or raw secrets in the vault or final response.
## Tools Used
- `read_file` — read vault schema/index/log and target notes.
- `search_files` — find existing vault pages and avoid duplicates.
- `write_file` / `patch` — create or update vault pages.
- `terminal` — run `gbrain`, `git`, and environment checks with secret values redacted.
## Safe Commands
```bash
export PATH="$HOME/.bun/bin:$PATH"
gbrain config set search.mode conservative
gbrain import "$OBSIDIAN_VAULT_PATH" --no-embed
gbrain extract links --source fs --dir "$OBSIDIAN_VAULT_PATH"
gbrain stats
gbrain search modes
```
## Paid / Approval-Gated Commands
```bash
gbrain providers test --model <provider:model>
gbrain embed --stale
gbrain dream
gbrain autopilot --install
gbrain onboard --auto --max-usd 5
gbrain config set search.mode tokenmax
```
## Verification Checklist
- [ ] Vault path exists and is the intended source.
- [ ] Vault operating files were read before edits when present.
- [ ] Existing pages/skills were searched to avoid duplicates.
- [ ] New/updated vault pages follow the vault schema and link convention.
- [ ] Vault index/catalog updated for new pages when present.
- [ ] Vault log appended for meaningful actions when present.
- [ ] `gbrain import ... --no-embed` completed.
- [ ] `gbrain stats` recorded pages/chunks/embeddings/links.
- [ ] `gbrain search modes` confirms conservative mode unless a different mode was explicitly approved.
- [ ] No paid/background commands ran without approval.
@@ -0,0 +1,11 @@
// Routing eval fixtures for skills/obsidian-gbrain-safe-index.
// Positive cases: intents embed a trigger phrase in natural surrounding context
// (never verbatim-identical to a trigger — the fixture linter rejects tautologies).
{"intent": "help me connect Obsidian to gbrain for my notes", "expected_skill": "obsidian-gbrain-safe-index"}
{"intent": "import my vault to gbrain but skip embeddings for now", "expected_skill": "obsidian-gbrain-safe-index"}
{"intent": "please sync vault and gbrain after I edit notes", "expected_skill": "obsidian-gbrain-safe-index"}
{"intent": "run embed gbrain after vault update tonight", "expected_skill": "obsidian-gbrain-safe-index"}
{"intent": "hey is gbrain synced with my vault right now", "expected_skill": "obsidian-gbrain-safe-index"}
// Negative cases: related but owned by other skills. Assert NO route to this skill.
{"intent": "migrate my notes from Notion to gbrain", "expected_skill": null}
{"intent": "what is on my calendar tomorrow", "expected_skill": null}
+100
View File
@@ -0,0 +1,100 @@
---
name: skill-vault-capture-policy
version: 1.0.0
description: |
Use when the user wants a durable operational learning, new agent skill, or
important environment/setup change to be captured into the Obsidian vault
instead of left only in transient chat memory. Covers the capture rule,
preferred page patterns, and index/log update obligations.
triggers:
- "save this learning to the vault"
- "capture this skill in Obsidian"
- "record this workflow in my notes"
- "put this setup change in the vault"
- "should we add this to the knowledge base"
tools:
- read_file
- search_files
- write_file
- patch
mutating: true
---
# Skill Vault Capture Policy
## Contract
This skill guarantees:
- Durable operational learnings, newly adopted skills, and important environment/setup changes are captured into the Obsidian vault rather than left only in chat memory.
- An existing page is updated when the knowledge clearly belongs there; a new page is created only when the topic is distinct and likely to recur.
- Every new vault page is added to `index.md`.
- Every meaningful create/update appends a dated entry to `log.md`.
- Small linked pages are preferred over one giant running note.
## Phases
1. **Classify the learning.**
- New gbrain operating rule, cost control, or embedding/provider change.
- New agent-fork / harness / coding-tool integration fact.
- New Obsidian vault workflow or structure decision.
- New recurring agent skill that changes how the agent should operate here.
2. **Avoid duplicates.**
- Search the vault for an existing page that already owns the topic.
- If found, update it with a new section or dated note rather than creating a near-duplicate.
3. **Create when distinct.**
- Place new pages under the vault schema: `_meta/` for operating notes, `concepts/` for workflows, `entities/` for tools/people.
- Use YAML frontmatter and at least two `[[wikilinks]]` unless it is a short seed page.
4. **Update navigation.**
- Add the page to `index.md` under the correct type heading.
- Append a `## [YYYY-MM-DD] create|update | subject` entry to `log.md`.
5. **Report the capture.**
- State which files changed and whether `index.md` / `log.md` were updated.
## Output Format
Use a short status block:
| Item | Status |
|---|---|
| Learning classified | type |
| Page created/updated | path |
| index.md updated | yes/no |
| log.md updated | yes/no |
## Anti-Patterns
- Leaving durable learnings only in chat memory.
- Creating a near-duplicate page instead of updating the existing one.
- Forgetting to update `index.md` and `log.md`.
- Writing one giant running note instead of small linked pages.
- Recording secrets, API keys, or raw credentials in the vault.
## Tools Used
- `read_file` — read `SCHEMA.md`, `index.md`, `log.md`, and target pages.
- `search_files` — find existing pages to avoid duplicates.
- `write_file` / `patch` — create or update vault pages and navigation.
## Safe Commands
```bash
# inspect vault navigation before writing
read SCHEMA.md index.md log.md
# create or update a page, then refresh catalog/log
# index.md: add [[page-slug]] under the matching type heading
# log.md: append ## [YYYY-MM-DD] create|update | subject
```
## Verification Checklist
- [ ] Vault schema/read files were checked before writing.
- [ ] Existing pages were searched to avoid duplicates.
- [ ] New/updated page has frontmatter and wikilinks.
- [ ] `index.md` updated for new pages.
- [ ] `log.md` appended for meaningful actions.
- [ ] No secrets or raw credentials were written.
@@ -0,0 +1,10 @@
// Routing eval fixtures for skills/skill-vault-capture-policy.
// Positive cases: intents embed a trigger phrase in natural surrounding context
// (never verbatim-identical to a trigger — the fixture linter rejects tautologies).
{"intent": "please save this learning to the vault so we keep it", "expected_skill": "skill-vault-capture-policy", "ambiguous_with": ["idea-ingest"]}
{"intent": "we should capture this skill in Obsidian for reuse", "expected_skill": "skill-vault-capture-policy", "ambiguous_with": ["capture"]}
{"intent": "can you record this workflow in my notes for next time", "expected_skill": "skill-vault-capture-policy"}
{"intent": "put this setup change in the vault before we forget", "expected_skill": "skill-vault-capture-policy"}
// Negative cases: related but owned by other skills or out of scope.
{"intent": "connect my Obsidian vault to gbrain", "expected_skill": null}
{"intent": "what is on my calendar tomorrow", "expected_skill": null}
+4 -82
View File
@@ -527,9 +527,6 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
process.on('SIGINT', () => { void shutdown('SIGINT'); });
let consecutiveErrors = 0;
// Parser-probe fixture warning is once-per-process, not once-per-cycle
// (compiled-binary installs have no source tree; don't spam the log).
let parserProbeFixtureWarned = false;
// v0.37.7.0 #1162 — counter for consecutive reconnect failures.
// Reset on every successful health probe or reconnect. Threshold
// controlled by GBRAIN_AUTOPILOT_MAX_RECONNECT_FAILS env (default 30).
@@ -1076,36 +1073,17 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
// loop. Probe runs even when cycleOk=false (probe may surface signal
// explaining why the cycle is failing).
try {
const { resolveProbeEnabled, resolveProbeMaxUsd, runNightlyQualityProbe } = await import('../core/cycle/nightly-quality-probe.ts');
// Dual-plane read: `gbrain config set` (what the doctor enable hint
// prints) writes the DB plane; ~/.gbrain/config.json is the fallback.
let dbEnabled: string | null = null;
let dbMaxUsd: string | null = null;
try {
dbEnabled = await engine.getConfig('autopilot.nightly_quality_probe.enabled');
dbMaxUsd = await engine.getConfig('autopilot.nightly_quality_probe.max_usd');
} catch { /* DB unavailable → file plane only */ }
const probeEnabled = resolveProbeEnabled(dbEnabled, cfg?.autopilot?.nightly_quality_probe?.enabled);
const probeEnabled = cfg?.autopilot?.nightly_quality_probe?.enabled === true;
if (probeEnabled) {
const { runNightlyQualityProbe } = await import('../core/cycle/nightly-quality-probe.ts');
const { runLongMemEvalForProbe, runCrossModalBatchForProbe } = await import('../core/cycle/nightly-probe-adapters.ts');
const { isAvailable } = await import('../core/ai/gateway.ts');
const { existsSync } = await import('node:fs');
const { fileURLToPath } = await import('node:url');
const { join } = await import('node:path');
const maxUsd = resolveProbeMaxUsd(dbMaxUsd, cfg?.autopilot?.nightly_quality_probe?.max_usd);
// The committed fixture (test/fixtures/longmemeval-nightly.jsonl)
// lives in the gbrain PACKAGE, not the brain repo — repoPath is
// sync.repo_path (the user's brain), where the fixture never
// exists, so the probe error'd on every real install. Resolve the
// package root from the module location; keep repoPath as the
// fallback for setups that vendor the fixture into the brain repo.
const pkgRoot = fileURLToPath(new URL('../..', import.meta.url));
const fixtureAtPkgRoot = existsSync(join(pkgRoot, 'test', 'fixtures', 'longmemeval-nightly.jsonl'));
const maxUsd = Number(cfg?.autopilot?.nightly_quality_probe?.max_usd ?? 5);
await runNightlyQualityProbe({
isEnabled: () => true, // already gated above; phase re-checks for defense-in-depth
hasEmbeddingProvider: () => isAvailable('embedding'),
resolveMaxUsd: () => maxUsd,
resolveRepoRoot: () => (fixtureAtPkgRoot ? pkgRoot : repoPath ?? gbrainHomePath('.')),
resolveRepoRoot: () => repoPath ?? gbrainHomePath('.'),
runLongMemEval: runLongMemEvalForProbe,
runCrossModalBatch: runCrossModalBatchForProbe,
now: () => new Date(),
@@ -1117,62 +1095,6 @@ export async function runAutopilot(engine: BrainEngine, args: string[]) {
// informational; autopilot loop continues.
}
// 4.6 — Nightly conversation-parser probe (v0.41.16.0 phase module;
// the scheduler wire-up was deferred at ship and is added here). Same
// posture as 4.5: the phase owns its gates (enabled/mode-gate, LLM
// key), the wiring owns invocation + the audit row, and a probe
// failure NEVER crashes the autopilot loop. Per D10 the probe is
// default-ON for search.mode=tokenmax, opt-in otherwise.
try {
const { runConversationParserNightlyProbe } = await import('../core/conversation-parser/nightly-probe.ts');
const { logParserProbeEvent, parserProbeRanWithin } = await import('../core/audit-parser-probe.ts');
const { isAvailable } = await import('../core/ai/gateway.ts');
const { existsSync } = await import('node:fs');
const { fileURLToPath } = await import('node:url');
const { join } = await import('node:path');
// Flag reads dual-plane: the DB row (`gbrain config set …`) wins,
// ~/.gbrain/config.json is the fallback. search.mode lives on the
// DB plane only (mode.ts owns it).
let parserDbEnabled: string | null = null;
let dbSearchMode: string | null = null;
try {
parserDbEnabled = await engine.getConfig('autopilot.conversation_parser_probe.enabled');
dbSearchMode = await engine.getConfig('search.mode');
} catch { /* DB unavailable → file plane only */ }
const parserEnabled = parserDbEnabled != null
? parserDbEnabled === 'true'
: cfg?.autopilot?.conversation_parser_probe?.enabled === true;
const searchMode = dbSearchMode ?? '';
// Fixtures are committed in the gbrain package (test/fixtures/…),
// NOT the brain repo — resolve from the module location. Compiled
// binaries carry no source tree: skip quietly instead of writing
// failure rows that would flip doctor to WARN on every binary install.
const pkgRoot = fileURLToPath(new URL('../..', import.meta.url));
const fixturePath = join(pkgRoot, 'test', 'fixtures', 'conversation-formats', 'all.jsonl');
const adversarialPath = join(pkgRoot, 'test', 'fixtures', 'conversation-formats', 'adversarial.jsonl');
const shouldInvoke = parserEnabled || searchMode === 'tokenmax';
if (shouldInvoke && existsSync(fixturePath) && existsSync(adversarialPath)) {
const result = await runConversationParserNightlyProbe({
isEnabled: () => parserEnabled,
searchMode: () => searchMode,
hasLlmKey: () => isAvailable('chat'),
resolveFixturePath: () => fixturePath,
resolveAdversarialPath: () => adversarialPath,
now: () => new Date(),
shouldSkipForRateLimit: () => parserProbeRanWithin(24 * 60 * 60 * 1000),
});
// rate_limited is a non-run: the loop ticks every few minutes, so
// logging every skip would flood the audit file with no-signal rows.
if (result.outcome !== 'rate_limited') logParserProbeEvent(result);
} else if (shouldInvoke && !parserProbeFixtureWarned) {
parserProbeFixtureWarned = true;
console.error(`[parser-probe] fixtures not found under ${pkgRoot}; skipping (probe needs a source-checkout install)`);
}
} catch (e) {
logError('autopilot.parser_probe', e);
// Informational, like 4.5: do NOT bump consecutiveErrors.
}
// Wait for next cycle
await new Promise(r => setTimeout(r, interval * 1000));
}
+14 -79
View File
@@ -2960,54 +2960,6 @@ function _resolveSyncFreshnessHours(varName: string, fallback: number): number {
* branch (disabled / enabled-no-events / enabled-all-pass / enabled-with-failures)
* without spinning up the audit JSONL or a real config file.
*/
/**
* Pure function form of the conversation_parser_probe_health check.
* Mirrors computeNightlyQualityProbeHealthCheck: skip-with-hint when the
* probe is off and silent, surface the last 7 days of audit events when
* it has run, WARN on any non-pass outcome.
*
* `effectiveEnabled` folds the D10 mode-gate in: explicitly enabled OR
* search.mode=tokenmax (where the probe is default-on).
*/
export function computeConversationParserProbeHealthCheck(
effectiveEnabled: boolean,
events: ReadonlyArray<{ outcome: string; ts: string; reason?: string }>,
): Check {
const name = 'conversation_parser_probe_health';
if (!effectiveEnabled && events.length === 0) {
return {
name,
status: 'ok',
message:
'disabled (opt-in; default-on only for search.mode=tokenmax). Enable with: ' +
'`gbrain config set autopilot.conversation_parser_probe.enabled true`',
};
}
if (events.length === 0) {
return {
name,
status: 'ok',
message: 'enabled but no probe events in the last 7 days (next run by autopilot; fixtures require a source-checkout install).',
};
}
const bad = events.filter(e => e.outcome !== 'pass');
const latest = events[events.length - 1]!;
if (bad.length > 0) {
return {
name,
status: 'warn',
message:
`${bad.length}/${events.length} probe run(s) in the last 7 days did not pass; ` +
`latest: ${latest.outcome}${latest.reason ? ` (${latest.reason})` : ''}`,
};
}
return {
name,
status: 'ok',
message: `${events.length} probe run(s) in the last 7 days, all pass (latest ${latest.ts}).`,
};
}
export function computeNightlyQualityProbeHealthCheck(
probeEnabled: boolean,
events: ReadonlyArray<{ outcome: string; ts: string; detail?: string }>,
@@ -4891,17 +4843,10 @@ export async function buildChecks(
try {
const { readRecentQualityProbeEvents } = await import('../core/audit-quality-probe.ts');
const { loadConfig } = await import('../core/config.ts');
const { resolveProbeEnabled } = await import('../core/cycle/nightly-quality-probe.ts');
let probeEnabled = false;
try {
// Dual-plane read, matching the autopilot gate: the DB row (what the
// enable hint's `gbrain config set` writes) wins; file plane fallback.
let dbVal: string | null = null;
try {
dbVal = engine ? await engine.getConfig('autopilot.nightly_quality_probe.enabled') : null;
} catch { /* DB unavailable → file plane only */ }
const cfg = loadConfig();
probeEnabled = resolveProbeEnabled(dbVal, (cfg as any)?.autopilot?.nightly_quality_probe?.enabled);
probeEnabled = Boolean((cfg as any)?.autopilot?.nightly_quality_probe?.enabled);
} catch { /* config unavailable → treat as disabled */ }
const events = readRecentQualityProbeEvents(7);
const check = computeNightlyQualityProbeHealthCheck(probeEnabled, events);
@@ -5085,29 +5030,19 @@ export async function buildChecks(
// 3d.5 v0.41.13.0 — conversation_parser_probe_health. Mode-gated
// per D10: ON when search.mode=tokenmax, opt-in for other modes.
// Surfaces the last 7 days of nightly-probe audit events; warn on any
// non-pass outcome (fail / budget_exceeded / adversarial_false_positive).
// (Until the autopilot wire-up this was a hardcoded "Skipped" stub.)
try {
const { readRecentParserProbeEvents } = await import('../core/audit-parser-probe.ts');
let parserProbeEnabled = false;
try {
let dbVal: string | null = null;
let dbMode: string | null = null;
try {
dbVal = engine ? await engine.getConfig('autopilot.conversation_parser_probe.enabled') : null;
dbMode = engine ? await engine.getConfig('search.mode') : null;
} catch { /* DB unavailable → file plane only */ }
const { loadConfig } = await import('../core/config.ts');
const fileVal = (loadConfig() as any)?.autopilot?.conversation_parser_probe?.enabled;
const flagOn = dbVal != null ? dbVal === 'true' : fileVal === true;
parserProbeEnabled = flagOn || dbMode === 'tokenmax';
} catch { /* config unavailable → treat as disabled */ }
const parserEvents = readRecentParserProbeEvents(7);
checks.push(computeConversationParserProbeHealthCheck(parserProbeEnabled, parserEvents));
} catch {
// Best-effort; audit-log read failure shouldn't stop doctor.
}
// Surface the last 7 days of nightly-probe events; warn on FAIL /
// BUDGET_EXCEEDED / adversarial_false_positive.
//
// v0.41.13.0 ships the probe as opt-in (autopilot wiring deferred
// to T7 in the cathedral plan); this check skips with an enable
// hint until the probe has at least one audit event written.
checks.push({
name: 'conversation_parser_probe_health',
status: 'ok',
message:
'Skipped (nightly probe is opt-in; enable with ' +
'`gbrain config set autopilot.conversation_parser_probe.enabled true`)',
});
// 3e. home_dir_in_worktree (v0.35.8.0). Walks up from `gbrainPath()`
// looking for a `.git` directory OR file. If found, warns: `~/.gbrain/`
+2 -15
View File
@@ -76,7 +76,7 @@ FLAGS:
dimensions (goal, depth, sourcing, specificity, useful).
--cycles N 1-3. Default: 3 in TTY, 1 in non-TTY (T11). Each
cycle is 3 model calls; verdict aggregates over them.
--slot-a-model <id> Override default 'openai:gpt-5.2'.
--slot-a-model <id> Override default 'openai:gpt-4o'.
--slot-b-model <id> Override default 'anthropic:claude-opus-4-7'.
--slot-c-model <id> Override default 'google:gemini-1.5-pro'.
--receipt-dir <path> Default: gbrainPath('eval-receipts').
@@ -468,14 +468,6 @@ interface BatchRow {
question_id: string;
question: string;
hypothesis: string;
/**
* Gold answer from the benchmark dataset, when the upstream eval emits
* it (eval-longmemeval does). Folded into the judge task so CORRECTNESS
* is verifiable — without it a judge panel that sees only
* {question, hypothesis} cannot validate a terse factual answer against
* a haystack it never saw.
*/
answer?: string;
}
/**
@@ -589,7 +581,6 @@ function readBatchRows(path: string): BatchReadResult {
question_id: typeof obj.question_id === 'string' ? obj.question_id : `line-${lineNo}`,
question: obj.question,
hypothesis: obj.hypothesis,
...(typeof obj.answer === 'string' && obj.answer.length > 0 ? { answer: obj.answer } : {}),
});
}
if (summarySkipped > 0) {
@@ -706,11 +697,7 @@ async function runBatchMode(parsed: ParsedArgs, opts: RunCrossModalOpts): Promis
fn: async (row, idx) => {
process.stderr.write(`[eval cross-modal batch] ${idx + 1}/${rows.length} ${row.question_id} starting...\n`);
return await runEvalFn({
// With a gold answer the judges can actually verify correctness;
// without one they see only {question, hypothesis} and cannot.
task: row.answer
? `${row.question}\n\nExpected answer (gold label from the benchmark dataset): ${row.answer}`
: row.question,
task: row.question,
output: row.hypothesis,
slug: row.question_id,
dimensions,
+3 -17
View File
@@ -33,7 +33,6 @@ import {
type AliasMap,
} from '../eval/longmemeval/extract.ts';
import { extractCandidateEntities } from '../core/think/entity-extract.ts';
import { splitProviderModelId } from '../core/model-id.ts';
import { resolveEntitySlugWithSource, type ResolutionSource } from '../core/entities/resolve.ts';
import { formatTrajectoryBlock } from '../core/trajectory-format.ts';
@@ -470,22 +469,14 @@ export async function runEvalLongMemEval(args: string[], runOpts: RunOpts = {}):
});
// Wrap Anthropic SDK so its `.messages.create` shape matches ThinkLLMClient.
// Same pattern as src/core/think/index.ts:247-249 — EXCEPT think's default
// client routes through the gateway, which parses `provider:model` recipe
// ids. This eval's client is a raw SDK by design (hermetic, no gateway
// dependency), and resolveModel returns RECIPE ids (`anthropic:claude-…`);
// passing one through unstripped 404s every answer/extractor call, which
// surfaces downstream as all-upstream_error batches in the nightly probe.
const toSdkModel = (m: string): string => splitProviderModelId(m).model || m;
// Same pattern as src/core/think/index.ts:247-249.
const realClient = new Anthropic();
const client: ThinkLLMClient = runOpts.client ?? {
create: (params, callOpts) =>
realClient.messages.create({ ...params, model: toSdkModel(params.model) }, callOpts),
create: (params, callOpts) => realClient.messages.create(params, callOpts),
};
// v0.40.2.0 — separate extractor client (defaults to same SDK).
const extractorClient: ThinkLLMClient = runOpts.extractorClient ?? {
create: (params, callOpts) =>
realClient.messages.create({ ...params, model: toSdkModel(params.model) }, callOpts),
create: (params, callOpts) => realClient.messages.create(params, callOpts),
};
const trajectoryEnabled = !opts.noTrajectory;
const extractorModel = trajectoryEnabled
@@ -760,11 +751,6 @@ async function runOneQuestion(
// v0.40.1.0 (Track D / T2) — copy question_type into the row so the
// by_type_summary can be rebuilt from the file on resume runs.
question_type: q.question_type,
// Gold answer for downstream consumers that verify correctness (the
// cross-modal --batch judge folds it into the task; evaluate_qa.py
// ignores unknown fields). Without it a judge can't validate a terse
// factual hypothesis against a haystack it never saw.
...(q.answer !== undefined ? { answer: q.answer } : {}),
hypothesis,
retrieved_session_ids: retrievedSessionIds,
...(recallHit !== undefined ? { recall_hit: recallHit } : {}),
-63
View File
@@ -1,63 +0,0 @@
/**
* Nightly conversation-parser probe audit trail.
*
* One event per REAL probe run lands in
* `~/.gbrain/audit/parser-probe-YYYY-Www.jsonl` (ISO-week rotation via the
* shared audit-writer primitive; honors `GBRAIN_AUDIT_DIR`).
* Scheduler-cadence skips (`rate_limited`) are NOT logged — the autopilot
* loop ticks every few minutes, so logging every skip would flood the
* audit file with rows that carry no signal.
*
* Read by `gbrain doctor`'s `conversation_parser_probe_health` check and
* by the autopilot wiring's 24h rate-limit gate (`parserProbeRanWithin`).
*/
import { createAuditWriter } from './audit/audit-writer.ts';
import type { NightlyProbeResult } from './conversation-parser/nightly-probe.ts';
export type ParserProbeAuditEvent = NightlyProbeResult;
const writer = createAuditWriter<ParserProbeAuditEvent>({
featureName: 'parser-probe',
errorLabel: 'gbrain',
errorMessagePrefix: 'parser-probe audit ',
errorTrailer: '; probe continues',
});
/** Append one parser-probe event. Best-effort; never throws. */
export function logParserProbeEvent(event: ParserProbeAuditEvent): void {
writer.log(event);
}
/**
* Read recent parser-probe events (current + previous ISO week, filtered
* to the window). Missing files and corrupt rows are skipped silently.
*/
export function readRecentParserProbeEvents(
days = 7,
now: Date = new Date(),
): ParserProbeAuditEvent[] {
return writer.readRecent(days, now);
}
/** Exposed for tests pinning the rotation edge cases. */
export function computeParserProbeAuditFilename(now: Date = new Date()): string {
return writer.computeFilename(now);
}
/**
* 24h rate-limit gate for the autopilot wiring: true when any audited run
* happened within `windowMs` of `now`. Only REAL outcomes are audited (see
* module header), so a pass/fail today blocks re-runs until tomorrow while
* scheduler-cadence skips never extend the window.
*/
export function parserProbeRanWithin(
windowMs: number,
now: Date = new Date(),
): boolean {
const cutoff = now.getTime() - windowMs;
return readRecentParserProbeEvents(2, now).some((ev) => {
const ts = Date.parse(ev.ts);
return Number.isFinite(ts) && ts >= cutoff;
});
}
-10
View File
@@ -105,16 +105,6 @@ export interface GBrainConfig {
*/
max_usd?: number;
};
/**
* v0.41.16.0 — nightly conversation-parser probe. Per D10: default ON
* for `search.mode=tokenmax` brains, opt-in for conservative/balanced.
* ~$0.05/night with the committed fixtures × Haiku polish. Gated
* INSIDE the autopilot tick body, like nightly_quality_probe.
*/
conversation_parser_probe?: {
/** Enable for non-tokenmax modes. Defaults to false. */
enabled?: boolean;
};
/**
* v0.42.x (#1685 GAP D) — extract_atoms backlog auto-drain. Default ON so a
* pack-gated silent backlog never piles up unseen; daily-spend-capped so the
@@ -17,11 +17,11 @@
* Cost: ~$0.05/night with default fixtures × Haiku polish. Bounded
* by the active BudgetTracker the autopilot loop creates per-tick.
*
* Wired into the autopilot loop (step 4.6 in autopilot.ts), following
* the same shape as `src/core/cycle/nightly-quality-probe.ts`
* (v0.40.1.0 Track D / T6): the wiring resolves fixtures from the
* gbrain package root, writes real outcomes to the parser-probe audit
* trail (`audit-parser-probe.ts`), and never crashes the loop.
* **Wiring into the autopilot loop is deferred to a follow-up**
* (filed in TODOS.md). v0.41.16.0 ships the phase as a callable
* module so doctor + future cron drivers can invoke it; the
* scheduler wire-up follows the same shape as
* `src/core/cycle/nightly-quality-probe.ts` (v0.40.1.0 Track D / T6).
*
* Test seam: all dependencies are injected via NightlyProbeDeps so
* unit tests don't touch real LLMs or real fixtures.
+1 -6
View File
@@ -44,12 +44,7 @@ export const DEFAULT_DIMENSIONS: string[] = [
* `--slot-a-model`, `--slot-b-model`, `--slot-c-model` on the CLI.
*/
export const DEFAULT_SLOTS: SlotConfig[] = [
// Every default MUST be listed in its recipe's chat touchpoint (pinned by
// test/cross-modal-default-slots.test.ts) — `openai:gpt-4o` sat here after
// the OpenAI recipe dropped it, so slot A errored "not listed for OpenAI
// chat" on every install and the 3-slot panel could never reach its
// 2-model quorum without a Google key (verdict: permanently inconclusive).
{ id: 'A', model: 'openai:gpt-5.2' },
{ id: 'A', model: 'openai:gpt-4o' },
{ id: 'B', model: 'anthropic:claude-opus-4-7' },
{ id: 'C', model: 'google:gemini-1.5-pro' },
];
-24
View File
@@ -70,28 +70,6 @@ export async function runLongMemEvalForProbe(args: LongMemEvalProbeArgs): Promis
* the batch input) or unparseable (cross-modal wrote garbage). Both
* cases are paste-ready in the error message.
*/
/**
* QA-shaped judge dimensions for the nightly probe. The batch judge's
* DEFAULT_DIMENSIONS rubric (DEPTH / SOURCING / SPECIFICITY / …) is built
* for rich agent responses; LongMemEval hypotheses are deliberately terse
* factual answers ("in widget-co") that can never score ≥7 on DEPTH or
* SOURCING — so with the default rubric the probe FAILs every night even
* when retrieval + answering are perfectly healthy. The probe owns its
* invocation of the eval tool and passes dimensions matching the
* fixture's QA shape instead.
*
* NOTE: the `--dimensions` CLI flag splits on commas, so these dimension
* descriptions must stay comma-free.
*/
export const PROBE_QA_DIMENSIONS: string[] = [
// No faithfulness/grounding dimension on purpose: the judge never sees
// the haystack, so any accurate detail beyond the terse gold label reads
// as "invented" and correct answers fail (verified empirically — a
// correct "before + dates" answer scored 4/10 on such a dimension).
'CORRECTNESS — Does the hypothesis state the same fact as the expected answer? A terse direct answer is ideal.',
'DIRECTNESS — Does it answer THIS question without hedging or padding or answering something else?',
];
export async function runCrossModalBatchForProbe(
args: CrossModalProbeArgs,
): Promise<{ exitCode: number; summary: CrossModalBatchSummary }> {
@@ -103,8 +81,6 @@ export async function runCrossModalBatchForProbe(
args.summaryPath,
'--max-usd',
String(args.maxUsd),
'--dimensions',
PROBE_QA_DIMENSIONS.join(','),
'--yes',
'--json',
]);
+11 -43
View File
@@ -62,42 +62,6 @@ export interface NightlyProbeDeps {
now: () => Date;
}
/**
* Dual-plane flag resolution (same precedent as `mcp.publish_skills` in
* serve-http.ts): the DB config row — what `gbrain config set` writes —
* wins when present; the file plane (~/.gbrain/config.json) is the
* fallback. Doctor's paste-ready enable hint says `gbrain config set
* autopilot.nightly_quality_probe.enabled true`, so the gate MUST read
* the DB plane — a file-only read turns that hint into a silent no-op.
*/
export function resolveProbeEnabled(
dbVal: string | null | undefined,
fileVal: unknown,
): boolean {
if (dbVal != null) return dbVal === 'true';
return fileVal === true;
}
/**
* Same dual-plane rule for the per-run cost cap. Malformed or negative
* values on either plane fall through to the next plane / the default.
*/
export function resolveProbeMaxUsd(
dbVal: string | null | undefined,
fileVal: unknown,
fallback: number = DEFAULT_MAX_USD,
): number {
if (dbVal != null) {
const n = Number(dbVal);
if (Number.isFinite(n) && n >= 0) return n;
}
if (fileVal != null) {
const n = Number(fileVal);
if (Number.isFinite(n) && n >= 0) return n;
}
return fallback;
}
/**
* Pure function: decide whether the probe should run given the audit
* history. Returns reason when skipping.
@@ -137,17 +101,21 @@ export async function runNightlyQualityProbe(deps: NightlyProbeDeps): Promise<Ni
return { outcome: 'disabled', exit_code: 0, detail: 'feature flag off' };
}
// 24h rate limit — skip WITHOUT an audit row. The autopilot loop invokes
// the probe every cycle (~5-10 min), so all but one invocation per day
// lands here; logging each skip floods the audit file (~hundreds of
// rows/day) and — because doctor treats any non-pass outcome as bad
// signal — flips nightly_quality_probe_health to a permanent WARN the
// moment the probe is enabled. A skip is a non-event: the real runs are
// the signal, and their rows are what gates the next 24h window.
// 24h rate limit — skip + audit "rate_limited".
const now = deps.now();
const recent = readRecentQualityProbeEvents(2, now); // 2-day window is enough for 24h check
const decision = shouldRunNightly(now, recent);
if (!decision.run) {
logQualityProbeEvent({
outcome: 'rate_limited',
exit_code: 0,
pass_count: 0,
fail_count: 0,
inconclusive_count: 0,
error_count: 0,
est_cost_usd: 0,
detail: 'already ran within 24h window',
});
return { outcome: 'rate_limited', exit_code: 0, detail: 'already ran within 24h' };
}
-5
View File
@@ -75,11 +75,6 @@ export const CANONICAL_PRICING: Record<string, ModelPricing> = {
'openai:gpt-4o': { input: 2.50, output: 10.00 },
'openai:gpt-4o-mini': { input: 0.15, output: 0.60 },
'openai:gpt-5': { input: 5.00, output: 20.00 },
// gpt-5.2: rates from the OpenAI recipe chat touchpoint (verified
// 2026-04-20). Needed here because it's the cross-modal DEFAULT_SLOTS
// slot-A model — without a canonical entry estimateCost silently drops
// slot A from the --max-usd pre-flight and est_cost_usd audit rows.
'openai:gpt-5.2': { input: 1.25, output: 10.00 },
'openai:gpt-5.5': { input: 4.00, output: 16.00 },
// ── Google ─────────────────────────────────────────────────────────────
-102
View File
@@ -1,102 +0,0 @@
/**
* Tests for the parser-probe audit trail + the 24h rate-limit gate.
*
* Uses GBRAIN_AUDIT_DIR override pointed at a tmpdir for hermeticity
* (same pattern as audit-slug-fallback.serial.test.ts). Serial because
* the env override is process-global.
*/
import { afterEach, beforeEach, describe, expect, test } from 'bun:test';
import { mkdtempSync, rmSync, readdirSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import {
computeParserProbeAuditFilename,
logParserProbeEvent,
parserProbeRanWithin,
readRecentParserProbeEvents,
type ParserProbeAuditEvent,
} from '../src/core/audit-parser-probe.ts';
let auditDir: string;
let savedEnv: string | undefined;
beforeEach(() => {
auditDir = mkdtempSync(join(tmpdir(), 'parser-probe-audit-'));
savedEnv = process.env.GBRAIN_AUDIT_DIR;
process.env.GBRAIN_AUDIT_DIR = auditDir;
});
afterEach(() => {
if (savedEnv === undefined) delete process.env.GBRAIN_AUDIT_DIR;
else process.env.GBRAIN_AUDIT_DIR = savedEnv;
rmSync(auditDir, { recursive: true, force: true });
});
function makeEvent(overrides: Partial<ParserProbeAuditEvent> = {}): ParserProbeAuditEvent {
return {
schema_version: 1,
ts: new Date().toISOString(),
outcome: 'pass',
fixtures_total: 12,
fixtures_passed: 12,
recall_mean: 0.98,
participants_recall_mean: 0.97,
adversarial_false_positives: 0,
failed_fixture_ids: [],
...overrides,
};
}
describe('parser-probe audit trail', () => {
test('log + readRecent round-trip', () => {
logParserProbeEvent(makeEvent({ outcome: 'fail', reason: '2 fixture(s) failed' }));
const events = readRecentParserProbeEvents(7);
expect(events.length).toBe(1);
expect(events[0]!.outcome).toBe('fail');
expect(events[0]!.reason).toBe('2 fixture(s) failed');
const files = readdirSync(auditDir);
expect(files.length).toBe(1);
expect(files[0]).toMatch(/^parser-probe-\d{4}-W\d{2}\.jsonl$/);
});
test('filename uses ISO-week rotation with the parser-probe prefix', () => {
// Year-boundary edge pinned by the shared writer's own tests; here we
// pin the prefix wiring.
expect(computeParserProbeAuditFilename(new Date('2026-07-06T12:00:00Z'))).toBe(
'parser-probe-2026-W28.jsonl',
);
});
test('readRecent filters by window', () => {
const old = new Date(Date.now() - 10 * 86400000).toISOString();
logParserProbeEvent(makeEvent({ ts: old }));
expect(readRecentParserProbeEvents(7).length).toBe(0);
});
});
describe('parserProbeRanWithin — 24h rate-limit gate', () => {
const DAY_MS = 24 * 60 * 60 * 1000;
test('false when no runs are audited', () => {
expect(parserProbeRanWithin(DAY_MS)).toBe(false);
});
test('true when a run landed within the window', () => {
logParserProbeEvent(makeEvent({ ts: new Date(Date.now() - 60_000).toISOString() }));
expect(parserProbeRanWithin(DAY_MS)).toBe(true);
});
test('false when the last run is older than the window', () => {
logParserProbeEvent(makeEvent({ ts: new Date(Date.now() - 25 * 3600_000).toISOString() }));
expect(parserProbeRanWithin(DAY_MS)).toBe(false);
});
test('non-pass outcomes also hold the window (mirrors quality-probe semantics)', () => {
logParserProbeEvent(makeEvent({
outcome: 'no_embedding_key',
ts: new Date(Date.now() - 3600_000).toISOString(),
}));
expect(parserProbeRanWithin(DAY_MS)).toBe(true);
});
});
+4 -20
View File
@@ -31,15 +31,10 @@ describe('autopilot wiring: nightly quality probe', () => {
expect(SOURCE).toContain(`runCrossModalBatchForProbe`);
});
test('feature flag gate present: dual-plane read (DB row wins, file plane fallback)', () => {
test('feature flag gate present: cfg.autopilot.nightly_quality_probe.enabled', () => {
// Per D10: the scheduler ONLY checks the feature flag. The 24h rate-limit
// lives inside runNightlyQualityProbe itself (no scheduler-side precheck).
// The flag resolves through resolveProbeEnabled so `gbrain config set
// autopilot.nightly_quality_probe.enabled true` (the doctor hint, DB
// plane) and ~/.gbrain/config.json (file plane) BOTH work — a file-only
// read made the printed hint a silent no-op.
expect(SOURCE).toContain(`getConfig('autopilot.nightly_quality_probe.enabled')`);
expect(SOURCE).toMatch(/resolveProbeEnabled\(dbEnabled,\s*cfg\?\.autopilot\?\.nightly_quality_probe\?\.enabled\)/);
expect(SOURCE).toContain(`nightly_quality_probe?.enabled === true`);
});
test('NO scheduler-side rate-limit check (D10 simplification)', () => {
@@ -69,23 +64,12 @@ describe('autopilot wiring: nightly quality probe', () => {
expect(SOURCE).toContain(`now:`);
});
test('resolveRepoRoot prefers the gbrain package root (committed fixture home), not the brain repoPath', () => {
// The DI harness in nightly-quality-probe.test.ts passes process.cwd()
// (= the gbrain repo in CI), which papered over the wiring passing
// repoPath (= sync.repo_path, the user's BRAIN repo, where the fixture
// never exists). Pin the package-root resolution + existence check.
expect(SOURCE).toMatch(/fileURLToPath\(new URL\('\.\.\/\.\.', import\.meta\.url\)\)/);
expect(SOURCE).toContain(`'longmemeval-nightly.jsonl'`);
expect(SOURCE).toMatch(/fixtureAtPkgRoot \? pkgRoot : repoPath/);
});
test('hasEmbeddingProvider reads from gateway.isAvailable("embedding") (codex round-2 #12 — in-process, not subprocess)', () => {
expect(SOURCE).toContain(`isAvailable('embedding')`);
expect(SOURCE).toContain(`gateway`);
});
test('max_usd resolves dual-plane (default = 5 pinned by resolveProbeMaxUsd unit tests)', () => {
expect(SOURCE).toContain(`getConfig('autopilot.nightly_quality_probe.max_usd')`);
expect(SOURCE).toMatch(/resolveProbeMaxUsd\(dbMaxUsd,\s*cfg\?\.autopilot\?\.nightly_quality_probe\?\.max_usd\)/);
test('max_usd default = 5 when config unset (matches plan default per D10)', () => {
expect(SOURCE).toMatch(/max_usd\s*\?\?\s*5/);
});
});
@@ -1,77 +0,0 @@
/**
* Source-shape regression tests for the autopilot wiring of
* `runConversationParserNightlyProbe` (step 4.6).
*
* Same rationale as autopilot-nightly-probe-wiring.test.ts: the loop is
* hard to drive end-to-end, so these pin the structural protections —
* the dual-plane flag read, the D10 tokenmax mode-gate, the package-root
* fixture resolution, the audit-flood guard, and the try/catch posture.
*
* The probe's own gate/scoring logic is pinned by the module's unit
* tests; the audit trail by audit-parser-probe.serial.test.ts.
*/
import { describe, test, expect } from 'bun:test';
import { readFileSync } from 'node:fs';
import { resolve } from 'node:path';
const AUTOPILOT_SRC = resolve('src/commands/autopilot.ts');
const SOURCE = readFileSync(AUTOPILOT_SRC, 'utf-8');
describe('autopilot wiring: conversation-parser probe', () => {
test('invokes the phase module and the audit trail', () => {
expect(SOURCE).toContain(`runConversationParserNightlyProbe`);
expect(SOURCE).toContain(`conversation-parser/nightly-probe`);
expect(SOURCE).toContain(`logParserProbeEvent`);
expect(SOURCE).toContain(`audit-parser-probe`);
});
test('flag reads dual-plane: DB row (gbrain config set) wins, file plane fallback', () => {
expect(SOURCE).toContain(`getConfig('autopilot.conversation_parser_probe.enabled')`);
expect(SOURCE).toContain(`cfg?.autopilot?.conversation_parser_probe?.enabled === true`);
});
test('D10 mode-gate present: tokenmax brains run the probe by default', () => {
expect(SOURCE).toMatch(/parserEnabled \|\| searchMode === 'tokenmax'/);
});
test('fixtures resolve from the gbrain package root, NOT the brain repoPath', () => {
// The committed fixtures live in the gbrain source tree; resolving
// them against sync.repo_path would point into the user's brain repo.
expect(SOURCE).toMatch(/fileURLToPath\(new URL\('\.\.\/\.\.', import\.meta\.url\)\)/);
expect(SOURCE).toContain(`'conversation-formats', 'all.jsonl'`);
expect(SOURCE).toContain(`'conversation-formats', 'adversarial.jsonl'`);
});
test('missing fixtures skip quietly (no audit row, once-per-process stderr note)', () => {
// Compiled-binary installs carry no source tree; writing failure rows
// would flip doctor to WARN on every binary install.
expect(SOURCE).toContain(`parserProbeFixtureWarned`);
});
test('rate_limited outcomes are NOT audit-logged (flood guard)', () => {
expect(SOURCE).toMatch(/outcome !== 'rate_limited'\) logParserProbeEvent\(result\)/);
});
test('rate-limit gate delegates to the audit module, not inline event reads', () => {
expect(SOURCE).toContain(`parserProbeRanWithin(24 * 60 * 60 * 1000)`);
});
test('LLM-key gate reads gateway.isAvailable("chat") in-process', () => {
expect(SOURCE).toContain(`isAvailable('chat')`);
});
test('probe call wrapped in try/catch that does NOT bump consecutiveErrors', () => {
expect(SOURCE).toMatch(/catch[\s\S]*?autopilot\.parser_probe[\s\S]*?do NOT bump consecutiveErrors/);
});
test('DI shape: the exact 7 fields of the parser probe NightlyProbeDeps', () => {
expect(SOURCE).toContain(`isEnabled:`);
expect(SOURCE).toContain(`searchMode:`);
expect(SOURCE).toContain(`hasLlmKey:`);
expect(SOURCE).toContain(`resolveFixturePath:`);
expect(SOURCE).toContain(`resolveAdversarialPath:`);
expect(SOURCE).toContain(`shouldSkipForRateLimit:`);
expect(SOURCE).toContain(`now:`);
});
});
-47
View File
@@ -1,47 +0,0 @@
/**
* Consistency guard: every cross-modal DEFAULT_SLOTS model must be listed
* in its recipe's chat touchpoint. `openai:gpt-4o` drifted out of the
* OpenAI recipe while remaining the slot-A default — the gateway then
* rejected slot A ("not listed for OpenAI chat") on every install, and the
* 3-slot judge panel could never reach its 2-model quorum without a Google
* key, pinning every batch verdict at inconclusive (which the nightly
* quality probe surfaces as a doctor WARN).
*/
import { describe, expect, test } from 'bun:test';
import { DEFAULT_SLOTS } from '../src/core/cross-modal-eval/runner.ts';
import { getRecipe } from '../src/core/ai/recipes/index.ts';
import { splitProviderModelId } from '../src/core/model-id.ts';
import { canonicalLookup } from '../src/core/model-pricing.ts';
describe('cross-modal DEFAULT_SLOTS ↔ recipe consistency', () => {
test('every default slot model is listed in its recipe chat touchpoint', () => {
for (const slot of DEFAULT_SLOTS) {
const { provider, model } = splitProviderModelId(slot.model);
expect(provider).not.toBeNull();
const recipe = getRecipe(provider!);
expect(recipe, `slot ${slot.id}: unknown recipe "${provider}"`).toBeDefined();
const chatModels = recipe!.touchpoints.chat?.models ?? [];
expect(
chatModels,
`slot ${slot.id}: "${model}" not listed for ${provider} chat — the judge slot can never run`,
).toContain(model);
}
});
test('every default slot model has a canonical pricing entry', () => {
// Without one, estimateCost silently drops the slot from the
// --max-usd pre-flight and est_cost_usd audit rows (~1/3 under-count).
for (const slot of DEFAULT_SLOTS) {
expect(
canonicalLookup(slot.model),
`slot ${slot.id}: "${slot.model}" missing from CANONICAL_PRICING`,
).toBeDefined();
}
});
test('slots span three distinct providers (uncorrelated blind spots)', () => {
const providers = new Set(DEFAULT_SLOTS.map(s => splitProviderModelId(s.model).provider));
expect(providers.size).toBe(3);
});
});
-51
View File
@@ -1,51 +0,0 @@
/**
* Tests for computeConversationParserProbeHealthCheck — the pure function
* behind doctor's conversation_parser_probe_health check, which replaced
* the v0.41.13.0 hardcoded "Skipped" stub when the autopilot wiring
* landed. Mirrors the branch coverage style of the quality-probe check.
*/
import { describe, expect, test } from 'bun:test';
import { computeConversationParserProbeHealthCheck } from '../src/commands/doctor.ts';
const ev = (outcome: string, reason?: string, ts = new Date().toISOString()) => ({
outcome,
ts,
...(reason !== undefined ? { reason } : {}),
});
describe('computeConversationParserProbeHealthCheck', () => {
test('disabled + no events → ok with paste-ready enable hint', () => {
const check = computeConversationParserProbeHealthCheck(false, []);
expect(check.status).toBe('ok');
expect(check.message).toContain('gbrain config set autopilot.conversation_parser_probe.enabled true');
});
test('enabled + no events yet → ok, next run by autopilot', () => {
const check = computeConversationParserProbeHealthCheck(true, []);
expect(check.status).toBe('ok');
expect(check.message).toContain('no probe events');
});
test('disabled flag but events exist (tokenmax mode-gate ran it) → events win over the hint', () => {
const check = computeConversationParserProbeHealthCheck(false, [ev('pass')]);
expect(check.status).toBe('ok');
expect(check.message).toContain('all pass');
});
test('any non-pass outcome in the window → warn, latest surfaced with reason', () => {
const check = computeConversationParserProbeHealthCheck(true, [
ev('pass'),
ev('adversarial_false_positive', '1 adversarial fixture(s) parsed to non-empty'),
]);
expect(check.status).toBe('warn');
expect(check.message).toContain('adversarial_false_positive');
expect(check.message).toContain('parsed to non-empty');
});
test('all pass → ok with run count', () => {
const check = computeConversationParserProbeHealthCheck(true, [ev('pass'), ev('pass')]);
expect(check.status).toBe('ok');
expect(check.message).toContain('2 probe run(s)');
});
});
@@ -0,0 +1,61 @@
/**
* E2E smoke for skills/obsidian-gbrain-safe-index.
*
* Verifies the from-trigger-to-side-effect path that skillify requires:
* a real user trigger phrase routes to the skill, the resolver/check
* pipeline treats it as reachable, and the skill file exposes the
* gbrain commands the workflow actually runs.
*
* This stays local-only (no paid embedding, no external API): it asserts
* the documented safe/import path is present and parseable, not that it
* mutates a live brain.
*/
import { describe, expect, it } from 'bun:test';
import { existsSync, readFileSync } from 'fs';
import { join } from 'path';
const SKILLS = join(import.meta.dir, '..', '..', 'skills');
const SKILL_MD = join(SKILLS, 'obsidian-gbrain-safe-index', 'SKILL.md');
const RESOLVER = join(SKILLS, 'RESOLVER.md');
const TRIGGER_PHRASES = [
'connect my Obsidian vault to gbrain',
'import my vault to gbrain',
'sync vault and gbrain',
'capture this skill in my vault',
'embed gbrain after vault update',
'is gbrain synced with my vault',
];
describe('obsidian-gbrain-safe-index E2E', () => {
it('resolver maps real trigger phrasings to the skill', () => {
const resolver = readFileSync(RESOLVER, 'utf-8');
expect(resolver).toContain('obsidian-gbrain-safe-index/SKILL.md');
// Each representative phrase shares a token substring with a resolver row.
const rows = resolver
.split('\n')
.filter((l) => l.includes('obsidian-gbrain-safe-index/SKILL.md'))
.join('\n');
for (const phrase of TRIGGER_PHRASES) {
const hit = phrase
.toLowerCase()
.split(/\s+/)
.some((tok) => tok.length > 3 && rows.toLowerCase().includes(tok));
expect(hit, `no resolver token for: ${phrase}`).toBe(true);
}
});
it('skill documents the gbrain import-first safe path', () => {
const body = readFileSync(SKILL_MD, 'utf-8');
expect(body).toContain('gbrain import');
expect(body).toContain('--no-embed');
expect(body).toContain('gbrain config set search.mode conservative');
// Paid gate must be explicit, not a silent default.
expect(body).toContain('gbrain embed --stale');
});
it('skill is reachable from the skill tree', () => {
expect(existsSync(SKILL_MD)).toBe(true);
});
});
@@ -0,0 +1,52 @@
/**
* E2E smoke for skills/skill-vault-capture-policy.
*
* Verifies the from-trigger-to-side-effect path: a real capture request
* routes to the skill, and the skill documents the vault navigation
* obligations (index.md / log.md) that make a capture durable.
*
* Local-only: it asserts documented behavior, not live vault writes.
*/
import { describe, expect, it } from 'bun:test';
import { existsSync, readFileSync } from 'fs';
import { join } from 'path';
const SKILLS = join(import.meta.dir, '..', '..', 'skills');
const SKILL_MD = join(SKILLS, 'skill-vault-capture-policy', 'SKILL.md');
const RESOLVER = join(SKILLS, 'RESOLVER.md');
const TRIGGER_PHRASES = [
'save this learning to the vault',
'capture this skill in Obsidian',
'record this workflow in my notes',
'put this setup change in the knowledge base',
];
describe('skill-vault-capture-policy E2E', () => {
it('resolver maps real capture phrasings to the skill', () => {
const resolver = readFileSync(RESOLVER, 'utf-8');
expect(resolver).toContain('skill-vault-capture-policy/SKILL.md');
const rows = resolver
.split('\n')
.filter((l) => l.includes('skill-vault-capture-policy/SKILL.md'))
.join('\n');
for (const phrase of TRIGGER_PHRASES) {
const hit = phrase
.toLowerCase()
.split(/\s+/)
.some((tok) => tok.length > 3 && rows.toLowerCase().includes(tok));
expect(hit, `no resolver token for: ${phrase}`).toBe(true);
}
});
it('skill documents index.md and log.md update obligations', () => {
const body = readFileSync(SKILL_MD, 'utf-8');
expect(body).toContain('index.md');
expect(body).toContain('log.md');
});
it('skill is reachable from the skill tree', () => {
expect(existsSync(SKILL_MD)).toBe(true);
});
});
-74
View File
@@ -1,74 +0,0 @@
// Regression test for the nightly-quality-probe config-plane split-brain.
//
// The doctor check prints a paste-ready enable hint — `gbrain config set
// autopilot.nightly_quality_probe.enabled true` — which writes the DB config
// plane. But both the autopilot gate and the doctor check used to read ONLY
// the file plane (~/.gbrain/config.json via loadConfig), so following the
// hint was a silent no-op: the probe never ran and doctor kept reporting
// "disabled (opt-in)".
//
// resolveProbeEnabled / resolveProbeMaxUsd pin the dual-plane rule (same
// precedent as `mcp.publish_skills` in serve-http.ts): DB row wins when
// present, file plane is the fallback.
import { describe, expect, test } from 'bun:test';
import {
resolveProbeEnabled,
resolveProbeMaxUsd,
} from '../src/core/cycle/nightly-quality-probe.ts';
describe('resolveProbeEnabled — dual-plane flag resolution', () => {
test('DB plane "true" enables regardless of file plane (the doctor hint path)', () => {
expect(resolveProbeEnabled('true', undefined)).toBe(true);
expect(resolveProbeEnabled('true', false)).toBe(true);
});
test('explicit DB "false" wins over file-plane true (config set off sticks)', () => {
expect(resolveProbeEnabled('false', true)).toBe(false);
});
test('file plane is the fallback when no DB row exists', () => {
expect(resolveProbeEnabled(null, true)).toBe(true);
expect(resolveProbeEnabled(undefined, true)).toBe(true);
expect(resolveProbeEnabled(null, undefined)).toBe(false);
expect(resolveProbeEnabled(null, false)).toBe(false);
});
test('file plane stays strict boolean — string "true" in config.json does not enable', () => {
// Matches the pre-fix autopilot gate (`=== true`); the doctor check used
// Boolean(...) and could disagree with autopilot on a string value.
// Both call sites now share this helper, so they can no longer diverge.
expect(resolveProbeEnabled(null, 'true')).toBe(false);
expect(resolveProbeEnabled(null, 1)).toBe(false);
});
test('non-"true" DB strings are off (mcp.publish_skills semantics)', () => {
expect(resolveProbeEnabled('1', true)).toBe(false);
expect(resolveProbeEnabled('yes', true)).toBe(false);
expect(resolveProbeEnabled('', true)).toBe(false);
});
});
describe('resolveProbeMaxUsd — dual-plane cost cap resolution', () => {
test('DB plane wins when parseable', () => {
expect(resolveProbeMaxUsd('2.5', 10)).toBe(2.5);
expect(resolveProbeMaxUsd('0', 10)).toBe(0);
});
test('malformed or negative DB value falls through to file plane', () => {
expect(resolveProbeMaxUsd('banana', 3)).toBe(3);
expect(resolveProbeMaxUsd('-1', 3)).toBe(3);
});
test('file plane used when no DB row; default when both absent/invalid', () => {
expect(resolveProbeMaxUsd(null, 7)).toBe(7);
expect(resolveProbeMaxUsd(null, '4')).toBe(4);
expect(resolveProbeMaxUsd(null, undefined)).toBe(5);
expect(resolveProbeMaxUsd(null, 'banana')).toBe(5);
expect(resolveProbeMaxUsd(undefined, -2)).toBe(5);
});
test('explicit fallback override is honored', () => {
expect(resolveProbeMaxUsd(null, undefined, 12)).toBe(12);
});
});
+4 -7
View File
@@ -132,20 +132,17 @@ describe('runNightlyQualityProbe (DI stub harness)', () => {
});
});
test('enabled + recent run within 24h → outcome: rate_limited, NO audit row', async () => {
test('enabled + recent run within 24h → outcome: rate_limited', async () => {
// Pre-seed a recent audit event by running the probe once first.
await withEnv({ GBRAIN_AUDIT_DIR: auditTmp }, async () => {
// First run succeeds.
await runNightlyQualityProbe(makeDeps());
// Second run, same hour → rate_limited. A skip is a non-event: the
// autopilot loop invokes the probe every cycle (~5-10 min), so
// logging each skip would flood the audit file and flip doctor's
// any-non-pass-is-bad filter to a permanent WARN.
// Second run, same hour → rate_limited.
const r2 = await runNightlyQualityProbe(makeDeps());
expect(r2.outcome).toBe('rate_limited');
const events = await readEvents();
expect(events.length).toBe(1);
expect(events[0].outcome).toBe('pass');
expect(events.length).toBe(2);
expect(events[1].outcome).toBe('rate_limited');
});
});
+51
View File
@@ -0,0 +1,51 @@
import { describe, expect, it } from 'bun:test';
import { readFileSync, existsSync } from 'fs';
import { join } from 'path';
const SKILL_DIR = join(import.meta.dir, '..', 'skills', 'obsidian-gbrain-safe-index');
const SKILL_MD = join(SKILL_DIR, 'SKILL.md');
const RESOLVER = join(import.meta.dir, '..', 'skills', 'RESOLVER.md');
function parseFrontmatter(raw: string): Record<string, unknown> {
const m = raw.match(/^---\n([\s\S]*?)\n---/);
if (!m) throw new Error('no frontmatter');
const out: Record<string, unknown> = {};
for (const line of m[1].split('\n')) {
const mm = line.match(/^([a-zA-Z_]+):\s*(.*)$/);
if (mm) out[mm[1]] = mm[2].trim();
}
return out;
}
describe('obsidian-gbrain-safe-index skill', () => {
it('has a SKILL.md with required frontmatter', () => {
expect(existsSync(SKILL_MD)).toBe(true);
const fm = parseFrontmatter(readFileSync(SKILL_MD, 'utf-8'));
expect(fm['name']).toBe('obsidian-gbrain-safe-index');
expect(fm['description']).toBeTruthy();
});
it('has the required conformance sections', () => {
const body = readFileSync(SKILL_MD, 'utf-8');
for (const section of ['## Contract', '## Phases', '## Output Format', '## Anti-Patterns']) {
expect(body.includes(section), `missing ${section}`).toBe(true);
}
});
it('is registered in RESOLVER.md', () => {
expect(existsSync(RESOLVER)).toBe(true);
const resolver = readFileSync(RESOLVER, 'utf-8');
expect(resolver.includes('obsidian-gbrain-safe-index/SKILL.md')).toBe(true);
});
it('has routing-eval fixtures that exercise real trigger phrasings', () => {
const evalPath = join(SKILL_DIR, 'routing-eval.jsonl');
expect(existsSync(evalPath)).toBe(true);
const lines = readFileSync(evalPath, 'utf-8')
.split('\n')
.filter((l) => l.trim() && !l.trim().startsWith('//'))
.map((l) => JSON.parse(l));
const positives = lines.filter((l) => l.expected_skill === 'obsidian-gbrain-safe-index');
expect(positives.length).toBeGreaterThanOrEqual(5);
});
});
+64
View File
@@ -0,0 +1,64 @@
import { describe, expect, it } from 'bun:test';
import { readFileSync, existsSync } from 'fs';
import { join } from 'path';
import { DEFAULT_PRIVATE_PATTERNS } from '../src/core/skillpack/harvest-lint.ts';
const SKILL_DIR = join(import.meta.dir, '..', 'skills', 'skill-vault-capture-policy');
const SKILL_MD = join(SKILL_DIR, 'SKILL.md');
const RESOLVER = join(import.meta.dir, '..', 'skills', 'RESOLVER.md');
function parseFrontmatter(raw: string): Record<string, unknown> {
const m = raw.match(/^---\n([\s\S]*?)\n---/);
if (!m) throw new Error('no frontmatter');
const out: Record<string, unknown> = {};
for (const line of m[1].split('\n')) {
const mm = line.match(/^([a-zA-Z_]+):\s*(.*)$/);
if (mm) out[mm[1]] = mm[2].trim();
}
return out;
}
describe('skill-vault-capture-policy skill', () => {
it('has a SKILL.md with required frontmatter', () => {
expect(existsSync(SKILL_MD)).toBe(true);
const fm = parseFrontmatter(readFileSync(SKILL_MD, 'utf-8'));
expect(fm['name']).toBe('skill-vault-capture-policy');
expect(fm['description']).toBeTruthy();
});
it('has the required conformance sections', () => {
const body = readFileSync(SKILL_MD, 'utf-8');
for (const section of ['## Contract', '## Phases', '## Output Format', '## Anti-Patterns']) {
expect(body.includes(section), `missing ${section}`).toBe(true);
}
});
it('is registered in RESOLVER.md', () => {
expect(existsSync(RESOLVER)).toBe(true);
expect(readFileSync(RESOLVER, 'utf-8').includes('skill-vault-capture-policy/SKILL.md')).toBe(true);
});
it('contains no private user or agent-fork names (privacy rule)', () => {
for (const file of [SKILL_MD, join(SKILL_DIR, 'routing-eval.jsonl')]) {
const body = readFileSync(file, 'utf-8');
// DEFAULT_PRIVATE_PATTERNS[0] is the banned fork-name pattern; sourced
// from harvest-lint so this file never contains the literal itself
// (scripts/check-privacy.sh would reject it).
const forkName = new RegExp(DEFAULT_PRIVATE_PATTERNS[0], 'i');
for (const name of [/\bAdam\b/, /\bHermes\b/, /\bHerdr\b/, /\bArk\b/, forkName]) {
expect(name.test(body), `private name ${name} in ${file}`).toBe(false);
}
}
});
it('has routing-eval fixtures', () => {
const evalPath = join(SKILL_DIR, 'routing-eval.jsonl');
expect(existsSync(evalPath)).toBe(true);
const positives = readFileSync(evalPath, 'utf-8')
.split('\n')
.filter((l) => l.trim() && !l.trim().startsWith('//'))
.map((l) => JSON.parse(l))
.filter((l) => l.expected_skill === 'skill-vault-capture-policy');
expect(positives.length).toBeGreaterThanOrEqual(4);
});
});