mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-16 18:02:30 +00:00
Compare commits
9
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7e1c4f312f | ||
|
|
d9eb027bdd | ||
|
|
62e009d192 | ||
|
|
314fefa560 | ||
|
|
7f841fae7f | ||
|
|
1fabbb9849 | ||
|
|
64920f83c9 | ||
|
|
e861b92da7 | ||
|
|
e5b3e7aba7 |
@@ -71,8 +71,8 @@ GBrain is designed to be installed and operated by an AI agent. The fastest path
|
||||
|
||||
If you don't already have an AI agent platform running, start with one of these. Both are designed to read GBrain's install protocol and execute it:
|
||||
|
||||
- **[OpenClaw](https://github.com/openclawagents/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM)
|
||||
- **[Hermes](https://github.com/openclawagents/hermes)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click)
|
||||
- **[OpenClaw](https://github.com/openclaw/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM)
|
||||
- **[Hermes](https://github.com/NousResearch/hermes-agent)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click)
|
||||
|
||||
Then paste this into your agent:
|
||||
|
||||
|
||||
@@ -131,7 +131,9 @@ into gbrain so other clients can scaffold it. Default behavior:
|
||||
`~/.gbrain/harvest-private-patterns.txt` plus built-in defaults
|
||||
(canonical private fork name, common email regex, Slack channel pattern). Any
|
||||
match → rollback (delete the harvested files) and exit non-zero.
|
||||
- `openclaw.plugin.json` updated with the new slug, sorted.
|
||||
- `openclaw.plugin.json` updated with the new slug, sorted. Harvest must preserve
|
||||
the top-level OpenClaw-native plugin fields (`id`, `configSchema`, `contracts`)
|
||||
because OpenClaw validates those before it can install the package.
|
||||
- `--no-lint` bypasses the linter (after a manual editorial scrub).
|
||||
|
||||
Use the `skillpack-harvest` skill (its companion editorial workflow)
|
||||
|
||||
@@ -233,13 +233,14 @@ keep it or `git checkout` to throw it away. Nothing is committed for you.
|
||||
|
||||
**For a skill that ships with gbrain** (anything under the gbrain repo's own
|
||||
`skills/`): SkillOpt refuses to overwrite it by default and writes the winner to
|
||||
`skills/<name>/skillopt/best.md` instead, so an optimization pass can never
|
||||
silently mutate a skill other people depend on. Two ways to handle that:
|
||||
`skills/<name>/skillopt/proposed.md` instead (while keeping `best.md` as the
|
||||
optimizer's current-best pointer), so an optimization pass can never silently
|
||||
mutate a skill other people depend on. Two ways to handle that:
|
||||
|
||||
```bash
|
||||
# See the proposed improvement without touching SKILL.md (works for ANY skill):
|
||||
gbrain skillopt meeting-prep --split 1:1:1 --no-mutate
|
||||
# → writes skills/meeting-prep/skillopt/best.md (the proposed rewrite), prints its path. Copy what you want.
|
||||
# → writes skills/meeting-prep/skillopt/proposed.md, updates best.md, and prints the proposal path.
|
||||
|
||||
# Actually rewrite a bundled skill (explicit opt-in + an independent held-out set):
|
||||
gbrain skillopt brain-ops --split 1:1:1 --allow-mutate-bundled \
|
||||
|
||||
+2
-2
@@ -1565,8 +1565,8 @@ GBrain is designed to be installed and operated by an AI agent. The fastest path
|
||||
|
||||
If you don't already have an AI agent platform running, start with one of these. Both are designed to read GBrain's install protocol and execute it:
|
||||
|
||||
- **[OpenClaw](https://github.com/openclawagents/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM)
|
||||
- **[Hermes](https://github.com/openclawagents/hermes)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click)
|
||||
- **[OpenClaw](https://github.com/openclaw/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM)
|
||||
- **[Hermes](https://github.com/NousResearch/hermes-agent)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click)
|
||||
|
||||
Then paste this into your agent:
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
{
|
||||
"id": "gbrain-context-engine",
|
||||
"name": "gbrain",
|
||||
"version": "0.32.3.0",
|
||||
"description": "Personal knowledge brain with Postgres + pgvector hybrid search",
|
||||
|
||||
@@ -23,7 +23,7 @@ health_checks:
|
||||
label: "Auth provider"
|
||||
checks:
|
||||
- type: http
|
||||
url: "$CLAWVISOR_URL/health"
|
||||
url: "$CLAWVISOR_URL/ready"
|
||||
label: "ClawVisor"
|
||||
- type: env_exists
|
||||
name: GOOGLE_CLIENT_ID
|
||||
@@ -135,7 +135,7 @@ Tell the user:
|
||||
|
||||
Validate:
|
||||
```bash
|
||||
curl -sf "$CLAWVISOR_URL/health" && echo "PASS: ClawVisor reachable" || echo "FAIL"
|
||||
curl -sf "$CLAWVISOR_URL/ready" && echo "PASS: ClawVisor reachable" || echo "FAIL"
|
||||
```
|
||||
|
||||
**STOP until ClawVisor validates.**
|
||||
|
||||
@@ -0,0 +1,308 @@
|
||||
---
|
||||
id: contacts-to-brain
|
||||
name: Contacts-to-Brain
|
||||
version: 0.1.0
|
||||
description: Google Contacts become canonical people/ pages, enriching brain entities with ground-truth name/email/phone/org data.
|
||||
category: sense
|
||||
requires: [credential-gateway]
|
||||
secrets:
|
||||
- name: CLAWVISOR_URL
|
||||
description: ClawVisor gateway URL (Option A — recommended, handles OAuth for you)
|
||||
where: https://clawvisor.com — create an agent, activate Google Contacts service
|
||||
- name: CLAWVISOR_AGENT_TOKEN
|
||||
description: ClawVisor agent token (Option A)
|
||||
where: https://clawvisor.com — agent settings, copy the agent token
|
||||
- name: GOOGLE_CLIENT_ID
|
||||
description: Google OAuth2 client ID (Option B — direct API access, you manage tokens)
|
||||
where: https://console.cloud.google.com/apis/credentials — create OAuth 2.0 Client ID
|
||||
- name: GOOGLE_CLIENT_SECRET
|
||||
description: Google OAuth2 client secret (Option B)
|
||||
where: https://console.cloud.google.com/apis/credentials — same page as client ID
|
||||
health_checks:
|
||||
- type: any_of
|
||||
label: "Auth provider"
|
||||
checks:
|
||||
- type: http
|
||||
url: "$CLAWVISOR_URL/ready"
|
||||
label: "ClawVisor"
|
||||
- type: env_exists
|
||||
name: GOOGLE_CLIENT_ID
|
||||
label: "Google OAuth"
|
||||
setup_time: 15 min
|
||||
cost_estimate: "$0 (both options are free)"
|
||||
---
|
||||
|
||||
# Contacts-to-Brain: Your Address Book Becomes Ground Truth
|
||||
|
||||
Calendar attendees and email senders are the bulk of `people/<slug>` brain pages.
|
||||
Your Google Contacts is the ground-truth directory — canonical name, email, phone,
|
||||
organization — for those same entities. Syncing it closes the "who is this person"
|
||||
loop automatically and feeds enrichment.
|
||||
|
||||
## IMPORTANT: Instructions for the Agent
|
||||
|
||||
**You are the installer.** Follow these steps precisely.
|
||||
|
||||
**Why this matters:** email-to-brain and calendar-to-brain create people pages
|
||||
from whatever name string the API returned — sometimes an email prefix, sometimes
|
||||
a nickname. Contacts carries the authoritative record. After this recipe runs,
|
||||
"j.smith@acme-example.com" and "Jon S" resolve to the same person page with the
|
||||
right display name, phone, and company.
|
||||
|
||||
**The output is staging files, not direct writes:** the deterministic collector
|
||||
dumps contact records to `brain/contacts/.staging/`; YOU (the agent) merge them
|
||||
into `people/<slug>` pages using judgment — the Notability Gate in
|
||||
`skills/_brain-filing-rules.md` applies. Not every contact deserves a page.
|
||||
|
||||
**Do not skip steps. Verify after each step.**
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Google Contacts (People API, paginated)
|
||||
↓ (ClawVisor credential gateway: list_contacts / get_contact / search_contacts)
|
||||
Contacts Sync Script (deterministic Node.js)
|
||||
↓ Outputs:
|
||||
├── brain/contacts/.raw/contacts-{date}.json (raw API responses, provenance)
|
||||
└── brain/contacts/.staging/{slug}.md (one markdown record per contact)
|
||||
↓
|
||||
Agent reads staging files
|
||||
↓ Judgment calls (Notability Gate):
|
||||
├── Merge into existing people/<slug> pages (name/email/phone/org enrichment)
|
||||
├── Create new pages ONLY for notable contacts not already in brain
|
||||
└── Skip the rest (staging is not the brain)
|
||||
```
|
||||
|
||||
## Opinionated Defaults
|
||||
|
||||
**Staging record format** (one file per contact, deterministic):
|
||||
```markdown
|
||||
# Alice Example
|
||||
|
||||
- **Emails:** alice@acme-example.com, alice@gmail.com
|
||||
- **Phone:** +1 555 0100
|
||||
- **Organization:** Acme Example — VP Engineering
|
||||
- **Source:** Google Contacts (resourceName people/c123, synced 2026-07-21)
|
||||
```
|
||||
|
||||
**Enrichment, not duplication:** if `people/alice-example.md` already exists,
|
||||
append missing fields to it with a `[Source: Google Contacts]` citation. Do NOT
|
||||
create a second page. Slug-match by normalized name, then by email against
|
||||
existing page content.
|
||||
|
||||
**Notability Gate (from `skills/_brain-filing-rules.md`):** a contact with no
|
||||
brain presence gets a new page only if they appear elsewhere in the brain
|
||||
(calendar attendee, email correspondent) or the user confirms they matter.
|
||||
When in doubt, DON'T create — a junk page degrades search quality.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
1. **GBrain installed and configured** (`gbrain doctor` passes)
|
||||
2. **Node.js 18+** (for the sync script)
|
||||
3. **Google Contacts access** via ONE of:
|
||||
- **Option A: ClawVisor** (recommended, handles OAuth for you, no token management)
|
||||
- **Option B: Google OAuth2 directly** (you manage tokens, no extra service needed)
|
||||
|
||||
## Setup Flow
|
||||
|
||||
### Step 1: Choose and Configure Contacts Access
|
||||
|
||||
Ask the user: "How do you want to connect to Google Contacts?
|
||||
|
||||
**Option A: ClawVisor (recommended)**
|
||||
ClawVisor handles OAuth, token refresh, and encryption. If you already use
|
||||
ClawVisor for email-to-brain or calendar-to-brain, this uses the same setup —
|
||||
just activate the Google Contacts service on your existing agent.
|
||||
|
||||
**Option B: Google OAuth2 directly**
|
||||
Connect to the Google People API directly. No extra service needed, but you
|
||||
manage OAuth tokens yourself."
|
||||
|
||||
#### Option A: ClawVisor Setup
|
||||
|
||||
Tell the user:
|
||||
"I need your ClawVisor URL and agent token.
|
||||
1. Go to https://clawvisor.com
|
||||
2. Create an agent (or use existing)
|
||||
3. Activate the **Google Contacts** service
|
||||
4. Create a standing task with purpose: 'Full contacts access for people
|
||||
enrichment: list contacts, read contact details, search contacts across
|
||||
all connected Google accounts.'
|
||||
IMPORTANT: Be EXPANSIVE in the task purpose. Narrow purposes block requests.
|
||||
5. Copy the gateway URL and agent token"
|
||||
|
||||
Validate:
|
||||
```bash
|
||||
curl -sf "$CLAWVISOR_URL/ready" && echo "PASS: ClawVisor reachable" || echo "FAIL"
|
||||
```
|
||||
|
||||
**STOP until ClawVisor validates.**
|
||||
|
||||
#### Option B: Google OAuth2 Setup
|
||||
|
||||
Same flow as `recipes/credential-gateway.md` Option B, with the contacts scope:
|
||||
|
||||
1. https://console.cloud.google.com/apis/credentials — create an OAuth client ID
|
||||
(Desktop app), consent screen scope: `https://www.googleapis.com/auth/contacts.readonly`
|
||||
2. Enable the People API: https://console.cloud.google.com/apis/library/people.googleapis.com
|
||||
3. Run the OAuth flow; store tokens in `~/.gbrain/google-tokens.json` (auto-refresh on expiry)
|
||||
|
||||
Validate:
|
||||
```bash
|
||||
[ -n "$GOOGLE_CLIENT_ID" ] && [ -n "$GOOGLE_CLIENT_SECRET" ] \
|
||||
&& echo "PASS: Google OAuth credentials set" \
|
||||
|| echo "FAIL: Missing GOOGLE_CLIENT_ID or GOOGLE_CLIENT_SECRET"
|
||||
```
|
||||
|
||||
**STOP until OAuth flow completes and tokens are stored.**
|
||||
|
||||
### Step 2: Set Up the Contacts Sync Script
|
||||
|
||||
```bash
|
||||
mkdir -p contacts-sync
|
||||
cd contacts-sync
|
||||
npm init -y
|
||||
```
|
||||
|
||||
The sync script needs these capabilities:
|
||||
|
||||
1. **Paginated retrieval** — `list_contacts` (People API `people.connections.list`)
|
||||
returns pages of up to 100; follow `nextPageToken` until exhausted. Request
|
||||
fields: names, emailAddresses, phoneNumbers, organizations, metadata.
|
||||
2. **Deterministic staging output** — one markdown file per contact at
|
||||
`brain/contacts/.staging/{slug}.md`, slug from normalized display name
|
||||
(fall back to email prefix). Same contact = same file on every run (idempotent).
|
||||
3. **Raw JSON preservation** — save raw API responses to
|
||||
`brain/contacts/.raw/contacts-{date}.json` for provenance.
|
||||
4. **Skip empty records** — contacts with no name AND no email are noise; drop them.
|
||||
|
||||
### Step 3: Run the Full Sync
|
||||
|
||||
```bash
|
||||
node contacts-sync.mjs
|
||||
```
|
||||
|
||||
Verify:
|
||||
```bash
|
||||
ls brain/contacts/.staging/ | head -10
|
||||
```
|
||||
|
||||
Should show one file per contact, e.g. `alice-example.md`, `charlie-example.md`.
|
||||
|
||||
### Step 4: Enrich People Pages (Agent Judgment)
|
||||
|
||||
This is YOUR job (the agent). For each staging record:
|
||||
|
||||
1. **Check brain**: `gbrain search "contact name"` — do they have a
|
||||
`people/<slug>` page? Also search by email address.
|
||||
2. **Existing page** → merge the ground-truth fields (canonical name, emails,
|
||||
phone, organization) into the page, each with a
|
||||
`[Source: Google Contacts]` citation. Fix a wrong/partial display name.
|
||||
3. **No page** → apply the Notability Gate: create a page only if the contact
|
||||
already appears in the brain (calendar, email) or is clearly relevant.
|
||||
Otherwise skip.
|
||||
4. **Back-link** per the Iron Law in `skills/_brain-filing-rules.md`: an
|
||||
organization with a brain page gets a link from the person's page and back.
|
||||
|
||||
After enrichment, import and embed:
|
||||
```bash
|
||||
gbrain sync --no-pull --no-embed && gbrain embed --stale
|
||||
```
|
||||
|
||||
Verify:
|
||||
```bash
|
||||
gbrain search "alice-example" --limit 3
|
||||
```
|
||||
|
||||
Should return the enriched people page with contact details.
|
||||
|
||||
### Step 5: Set Up Weekly Sync
|
||||
|
||||
Contacts change slowly; once a week is plenty:
|
||||
```bash
|
||||
# Cron: every Sunday at 9 AM
|
||||
0 9 * * 0 cd /path/to/contacts-sync && node contacts-sync.mjs
|
||||
```
|
||||
|
||||
After each sync, re-run the Step 4 enrichment pass over CHANGED staging files
|
||||
only (compare mtime or diff against git), then:
|
||||
```bash
|
||||
gbrain sync --no-pull --no-embed && gbrain embed --stale
|
||||
```
|
||||
|
||||
### Step 6: Log Setup Completion
|
||||
|
||||
```bash
|
||||
mkdir -p ~/.gbrain/integrations/contacts-to-brain
|
||||
echo '{"ts":"'$(date -u +%Y-%m-%dT%H:%M:%SZ)'","event":"setup_complete","source_version":"0.1.0","status":"ok","details":{"contacts":"CONTACT_COUNT"}}' >> ~/.gbrain/integrations/contacts-to-brain/heartbeat.jsonl
|
||||
```
|
||||
|
||||
Tell the user: "Contacts-to-brain is set up. [N] contacts staged; [M] people
|
||||
pages enriched with ground-truth contact data. Weekly sync keeps it current."
|
||||
|
||||
## Implementation Guide
|
||||
|
||||
### Pagination
|
||||
|
||||
```
|
||||
list_all_contacts():
|
||||
contacts = []
|
||||
token = null
|
||||
do:
|
||||
page = list_contacts({ pageSize: 100, pageToken: token,
|
||||
personFields: 'names,emailAddresses,phoneNumbers,organizations,metadata' })
|
||||
contacts += page.connections
|
||||
token = page.nextPageToken
|
||||
while token
|
||||
return contacts
|
||||
```
|
||||
|
||||
### Slug Normalization
|
||||
|
||||
```
|
||||
slugify(contact):
|
||||
name = contact.names?[0]?.displayName
|
||||
if not name:
|
||||
name = contact.emailAddresses?[0]?.value.split('@')[0]
|
||||
return name.toLowerCase().normalize('NFD')
|
||||
.replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, '')
|
||||
```
|
||||
|
||||
Collision (two contacts, same slug): suffix with the email domain
|
||||
(`alice-example-acme-example`) rather than overwrite.
|
||||
|
||||
### What the Agent Should Test After Setup
|
||||
|
||||
1. **Idempotency:** run the sync twice. `git status` on `brain/contacts/.staging/`
|
||||
shows no changes on the second run.
|
||||
2. **Pagination:** with 250+ contacts, verify the staging count matches the
|
||||
Google Contacts count (not capped at 100).
|
||||
3. **Notability Gate:** verify a one-off contact with no brain presence did NOT
|
||||
get a `people/` page.
|
||||
4. **Enrichment merge:** verify an existing people page gained contact fields
|
||||
without losing its prior content, each with a `[Source: Google Contacts]`
|
||||
citation.
|
||||
|
||||
## Cost Estimate
|
||||
|
||||
| Component | Monthly Cost |
|
||||
|-----------|-------------|
|
||||
| ClawVisor (free tier) | $0 |
|
||||
| Google People API | $0 (within free quota) |
|
||||
| **Total** | **$0** |
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
**No contacts returned:**
|
||||
- Check ClawVisor has the Google Contacts service activated
|
||||
- Check the standing task purpose is expansive enough
|
||||
- Option B: verify the People API is enabled and the token carries
|
||||
`contacts.readonly`
|
||||
|
||||
**Duplicate people pages after enrichment:**
|
||||
- The agent matched by name but the brain page slug differs — search by email
|
||||
address too before creating, then merge and delete the duplicate
|
||||
|
||||
**Contacts with no name:**
|
||||
- The sync script falls back to the email prefix; records with neither name
|
||||
nor email are dropped as noise
|
||||
@@ -23,7 +23,7 @@ health_checks:
|
||||
label: "Auth provider"
|
||||
checks:
|
||||
- type: http
|
||||
url: "$CLAWVISOR_URL/health"
|
||||
url: "$CLAWVISOR_URL/ready"
|
||||
label: "ClawVisor"
|
||||
- type: env_exists
|
||||
name: GOOGLE_CLIENT_ID
|
||||
@@ -75,7 +75,7 @@ Tell the user:
|
||||
3. Activate the services you need:
|
||||
- **Gmail** (for email-to-brain)
|
||||
- **Google Calendar** (for calendar-to-brain)
|
||||
- **Google Contacts** (for enrichment)
|
||||
- **Google Contacts** (for contacts-to-brain)
|
||||
4. Create a standing task with a broad purpose. CRITICAL: be EXPANSIVE.
|
||||
|
||||
Good purpose: 'Full executive assistant access to Gmail, Calendar, and
|
||||
@@ -88,7 +88,7 @@ Tell the user:
|
||||
|
||||
Validate:
|
||||
```bash
|
||||
curl -sf "$CLAWVISOR_URL/health" \
|
||||
curl -sf "$CLAWVISOR_URL/ready" \
|
||||
&& echo "PASS: ClawVisor reachable" \
|
||||
|| echo "FAIL: ClawVisor not reachable — check the URL"
|
||||
```
|
||||
@@ -171,7 +171,7 @@ can now access your Google services."
|
||||
|
||||
## How to Verify
|
||||
|
||||
1. **ClawVisor:** `curl $CLAWVISOR_URL/health` returns OK.
|
||||
1. **ClawVisor:** `curl $CLAWVISOR_URL/ready` returns OK.
|
||||
2. **Google OAuth:** Tokens exist at `~/.gbrain/google-tokens.json`.
|
||||
3. **Gmail access:** Run the email collector — it should pull recent messages.
|
||||
4. **Calendar access:** Run the calendar sync — it should pull today's events.
|
||||
|
||||
@@ -23,7 +23,7 @@ health_checks:
|
||||
label: "Auth provider"
|
||||
checks:
|
||||
- type: http
|
||||
url: "$CLAWVISOR_URL/health"
|
||||
url: "$CLAWVISOR_URL/ready"
|
||||
label: "ClawVisor"
|
||||
- type: env_exists
|
||||
name: GOOGLE_CLIENT_ID
|
||||
@@ -130,7 +130,7 @@ Tell the user:
|
||||
|
||||
Validate:
|
||||
```bash
|
||||
curl -sf "$CLAWVISOR_URL/health" && echo "PASS: ClawVisor reachable" || echo "FAIL"
|
||||
curl -sf "$CLAWVISOR_URL/ready" && echo "PASS: ClawVisor reachable" || echo "FAIL"
|
||||
```
|
||||
|
||||
**STOP until ClawVisor validates.**
|
||||
@@ -328,7 +328,7 @@ threads you already replied to. Sent mail acts as a negative filter.
|
||||
## Troubleshooting
|
||||
|
||||
**No emails collected:**
|
||||
- Check ClawVisor health: `curl $CLAWVISOR_URL/health`
|
||||
- Check ClawVisor health: `curl $CLAWVISOR_URL/ready`
|
||||
- Check standing task is active and has Gmail service enabled
|
||||
- Check task purpose is expansive enough (narrow purposes block requests)
|
||||
|
||||
|
||||
@@ -266,4 +266,5 @@ editorial pass.
|
||||
(e.g. `src/commands/<slug>.ts` if the host SKILL.md declares it
|
||||
in frontmatter)
|
||||
- gbrain's `openclaw.plugin.json` — adds the slug to `skills:`
|
||||
array, sorted alphabetically
|
||||
array, sorted alphabetically, without removing OpenClaw-native plugin fields
|
||||
like `id`, `configSchema`, or `contracts`
|
||||
|
||||
@@ -57,6 +57,8 @@ This mode guarantees:
|
||||
- `skills/manifest.json` lists every skill directory
|
||||
- `skills/RESOLVER.md` references every skill in the manifest
|
||||
- `openclaw.plugin.json` `skills[]` round-trips with both
|
||||
- `openclaw.plugin.json` keeps OpenClaw install-required native plugin fields
|
||||
(`id`, object `configSchema`, and `contracts.contextEngines` when applicable)
|
||||
- No MECE violations (duplicate triggers across skills)
|
||||
|
||||
### Phases
|
||||
@@ -72,7 +74,7 @@ This mode guarantees:
|
||||
### Automation
|
||||
|
||||
```bash
|
||||
bun test test/skills-conformance.test.ts test/resolver.test.ts
|
||||
bun test test/skills-conformance.test.ts test/resolver.test.ts test/openclaw-plugin-manifest.test.ts
|
||||
```
|
||||
|
||||
The CI-gated check is the package.json `test` script.
|
||||
|
||||
+8
-12
@@ -54,7 +54,7 @@ export function bigintToStringReplacer(_key: string, value: unknown): unknown {
|
||||
}
|
||||
|
||||
// CLI-only commands that bypass the operation layer
|
||||
export const CLI_ONLY = new Set(['init', 'reinit-pglite', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'sync', 'extract', 'extract-conversation-facts', 'enrich', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'sources', 'mounts', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test', 'providers', 'storage', 'repos', 'code-def', 'code-refs', 'reindex', 'reindex-code', 'reindex-frontmatter', 'code-callers', 'code-callees', 'reconcile-links', 'frontmatter', 'auth', 'friction', 'claw-test', 'book-mirror', 'takes', 'think', 'salience', 'anomalies', 'calibration', 'transcripts', 'models', 'remote', 'recall', 'forget', 'edges-backfill', 'facts', 'cache', 'ze-switch', 'founder', 'brainstorm', 'lsd', 'schema', 'capture', 'onboard', 'conversation-parser', 'status', 'connect', 'skillopt', 'quarantine', 'self-upgrade', 'advisor', 'watch', 'reindex-search-vector']);
|
||||
export const CLI_ONLY = new Set(['init', 'reinit-pglite', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'sync', 'extract', 'extract-conversation-facts', 'enrich', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'maintain', 'sources', 'mounts', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test', 'providers', 'storage', 'repos', 'code-def', 'code-refs', 'reindex', 'reindex-code', 'reindex-frontmatter', 'code-callers', 'code-callees', 'reconcile-links', 'frontmatter', 'auth', 'friction', 'claw-test', 'book-mirror', 'takes', 'think', 'salience', 'anomalies', 'calibration', 'transcripts', 'models', 'remote', 'recall', 'forget', 'edges-backfill', 'cache', 'ze-switch', 'founder', 'brainstorm', 'lsd', 'schema', 'capture', 'onboard', 'conversation-parser', 'status', 'connect', 'skillopt', 'quarantine', 'self-upgrade', 'advisor', 'watch', 'reindex-search-vector']);
|
||||
// CLI-only commands whose handlers print their own --help text. These are
|
||||
// excluded from the generic short-circuit so detailed per-command and
|
||||
// per-subcommand usage stays reachable.
|
||||
@@ -78,6 +78,8 @@ const CLI_ONLY_SELF_HELP = new Set([
|
||||
'capture',
|
||||
// v0.42 self-upgrade ships its own usage (flags + the agent-skill story).
|
||||
'self-upgrade',
|
||||
// maintain (#3015) prints its own usage block (modes + not-auto-applied list).
|
||||
'maintain',
|
||||
// v0.43 (#2095): watch ships WATCH_HELP (flags + the stdin-turn protocol).
|
||||
'watch',
|
||||
// v0.37 fix wave (Lane D.4 + CDX2-12): sync's --no-embed flag was
|
||||
@@ -991,8 +993,6 @@ const THIN_CLIENT_REFUSED_COMMANDS = new Set([
|
||||
// hint pointing at the routable MCP tools; per-subcommand splits are
|
||||
// a v0.31.x follow-up TODO.
|
||||
'takes', 'sources',
|
||||
// #1867: fence-backfill edits local .md fences + stamps the local DB.
|
||||
'facts',
|
||||
// v0.32 thin-client routing audit (Codex round 2 findings #2, #4):
|
||||
// - `pages` purge-deleted is admin+localOnly (operations.ts:856-864)
|
||||
// - `files` list / file_url MCP ops are localOnly (operations.ts:1769-1879)
|
||||
@@ -1028,7 +1028,6 @@ const THIN_CLIENT_REFUSE_HINTS: Record<string, string> = {
|
||||
migrate: "migrate runs on the host's local engine. Run on the host machine.",
|
||||
'apply-migrations': 'schema migrations run on the host. SSH and run there.',
|
||||
'repair-jsonb': 'repair-jsonb operates on the local DB only.',
|
||||
facts: 'facts fence-backfill edits local entity-page fences. Run on the host machine.',
|
||||
integrity: 'integrity scans local files. Run on the host machine.',
|
||||
serve: 'serve starts a server. Run on the host, not the thin client.',
|
||||
dream: 'dream runs the autopilot cycle on the host. `gbrain remote ping` queues one. (Native `gbrain dream` thin-client routing planned for v0.31.2.)',
|
||||
@@ -1760,6 +1759,11 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
await runOrphans(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'maintain': {
|
||||
const { runMaintain } = await import('./commands/maintain.ts');
|
||||
await runMaintain(engine, args);
|
||||
break;
|
||||
}
|
||||
// v0.32.7 CJK wave — post-upgrade markdown re-chunk sweep.
|
||||
// v0.36 Phase 3 wave — `gbrain reindex --multimodal` re-embeds content_chunks
|
||||
// into the unified Voyage multimodal-3 column.
|
||||
@@ -1858,13 +1862,6 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
await runEdgesBackfill(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'facts': {
|
||||
// #1867 — re-runnable fence-backfill for row_num-NULL legacy fact
|
||||
// rows (idempotent v0_32_2 phase B, exposed as an operator command).
|
||||
const { runFactsCommand } = await import('./commands/facts.ts');
|
||||
await runFactsCommand(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'whoknows': {
|
||||
// v0.33 (Issue #?): expertise + relationship-proximity routing.
|
||||
// MCP op `find_experts` (read-scoped) backs the same code path; CLI
|
||||
@@ -2345,7 +2342,6 @@ TOOLS
|
||||
check-backlinks <check|fix> [dir] Find/fix missing back-links across brain
|
||||
lint <dir|file> [--fix] Catch LLM artifacts, placeholder dates, bad frontmatter
|
||||
orphans [--json] [--count] Find pages with no inbound wikilinks
|
||||
facts fence-backfill [--dry-run] Fence legacy fact rows (row_num NULL) onto entity pages
|
||||
salience [--days N] [--kind P] v0.29: pages ranked by emotional + activity salience
|
||||
anomalies [--since D] [--sigma N] v0.29: cohort-based statistical anomalies (tag, type)
|
||||
transcripts recent [--days N] v0.29: recent raw .txt transcripts (local-only)
|
||||
|
||||
+34
-4
@@ -581,7 +581,7 @@ async function embedPage(
|
||||
for (let j = 0; j < toEmbed.length; j++) {
|
||||
embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]);
|
||||
}
|
||||
const updated: ChunkInput[] = chunks.map(c => ({
|
||||
const updated: ChunkInput[] = chunks.map(c => preserveCodeMetadata(c, {
|
||||
chunk_index: c.chunk_index,
|
||||
chunk_text: c.chunk_text,
|
||||
chunk_source: c.chunk_source,
|
||||
@@ -605,6 +605,31 @@ async function embedPage(
|
||||
slog(`${slug}: embedded ${toEmbed.length} chunks`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Carry code-chunk metadata (language, symbol_name, symbol_type, line range,
|
||||
* parent scope, doc comment, qualified name) from a loaded Chunk back into a
|
||||
* ChunkInput destined for upsertChunks.
|
||||
*
|
||||
* Issue #769: every re-embed used to strip these fields, and upsertChunks
|
||||
* overwrites (does not COALESCE) the metadata columns from EXCLUDED, so
|
||||
* each pass clobbered code-def's primary index to NULL. Pulling the
|
||||
* preservation into one helper keeps the three re-embed call sites
|
||||
* (embedPage, embedAll non-stale, embedAllStale) in lock-step.
|
||||
*/
|
||||
function preserveCodeMetadata(loaded: any, base: ChunkInput): ChunkInput {
|
||||
return {
|
||||
...base,
|
||||
language: loaded.language ?? undefined,
|
||||
symbol_name: loaded.symbol_name ?? undefined,
|
||||
symbol_type: loaded.symbol_type ?? undefined,
|
||||
start_line: loaded.start_line ?? undefined,
|
||||
end_line: loaded.end_line ?? undefined,
|
||||
parent_symbol_path: loaded.parent_symbol_path ?? undefined,
|
||||
doc_comment: loaded.doc_comment ?? undefined,
|
||||
symbol_name_qualified: loaded.symbol_name_qualified ?? undefined,
|
||||
};
|
||||
}
|
||||
|
||||
async function embedAll(
|
||||
engine: BrainEngine,
|
||||
staleOnly: boolean,
|
||||
@@ -717,8 +742,10 @@ async function embedAll(
|
||||
for (let j = 0; j < toEmbed.length; j++) {
|
||||
embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]);
|
||||
}
|
||||
// Preserve ALL chunks, only update embeddings for stale ones
|
||||
const updated: ChunkInput[] = chunks.map(c => ({
|
||||
// Preserve ALL chunks, only update embeddings for stale ones.
|
||||
// preserveCodeMetadata threads code-chunk metadata (#769) so re-embed
|
||||
// doesn't clobber language/symbol_name/symbol_type to NULL.
|
||||
const updated: ChunkInput[] = chunks.map(c => preserveCodeMetadata(c, {
|
||||
chunk_index: c.chunk_index,
|
||||
chunk_text: c.chunk_text,
|
||||
chunk_source: c.chunk_source,
|
||||
@@ -1012,7 +1039,10 @@ async function embedAllStale(
|
||||
for (let j = 0; j < stale.length; j++) {
|
||||
staleIdxToEmbedding.set(stale[j].chunk_index, embeddings[j]);
|
||||
}
|
||||
const merged: ChunkInput[] = existing.map(c => ({
|
||||
// preserveCodeMetadata threads code-chunk metadata (#769) so the
|
||||
// autopilot --stale path doesn't clobber language/symbol_name/etc
|
||||
// to NULL on every cycle.
|
||||
const merged: ChunkInput[] = existing.map(c => preserveCodeMetadata(c, {
|
||||
chunk_index: c.chunk_index,
|
||||
chunk_text: c.chunk_text,
|
||||
chunk_source: c.chunk_source,
|
||||
|
||||
@@ -1651,7 +1651,7 @@ async function extractTimelineFromDB(
|
||||
* make re-extraction idempotent). EVERY processed page is stamped, including
|
||||
* zero-link pages — they WERE processed.
|
||||
*/
|
||||
async function extractStaleFromDB(
|
||||
export async function extractStaleFromDB(
|
||||
engine: BrainEngine,
|
||||
opts: {
|
||||
dryRun: boolean;
|
||||
|
||||
@@ -1,48 +0,0 @@
|
||||
/**
|
||||
* gbrain facts — fact-store maintenance surface (#1867).
|
||||
*
|
||||
* `fence-backfill` re-runs the v0_32_2 fence-backfill phase on demand.
|
||||
* Remote `extract_facts` deposits that predate the fence-write backstop
|
||||
* (and any legacy DB-only insert) leave `row_num IS NULL` rows that the
|
||||
* cycle extract_facts guard refuses to reconcile past — previously the
|
||||
* only remedy was the one-shot v0_32_2 migration, which the ledger marks
|
||||
* complete and never re-runs. The phase is idempotent (only touches
|
||||
* `row_num IS NULL` rows), so exposing it as a command is safe to re-run
|
||||
* any time the backlog reappears.
|
||||
*/
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { setCliExitVerdict } from '../core/cli-force-exit.ts';
|
||||
import { phaseBFenceFacts } from './migrations/v0_32_2.ts';
|
||||
|
||||
function printHelp(): void {
|
||||
process.stderr.write(
|
||||
`Usage: gbrain facts fence-backfill [--dry-run]\n\n` +
|
||||
`Fence-backfill: appends every legacy fact row (row_num IS NULL) to its\n` +
|
||||
`entity page's \`## Facts\` fence and stamps row_num + source_markdown_slug\n` +
|
||||
`back onto the DB row. Idempotent — re-runs only pick up rows still\n` +
|
||||
`missing a fence assignment. Clears the backlog that makes the cycle's\n` +
|
||||
`extract_facts phase skip fence→DB reconciliation.\n\n` +
|
||||
` --dry-run report what would be fenced; no FS or DB writes\n`,
|
||||
);
|
||||
}
|
||||
|
||||
export async function runFactsCommand(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
const sub = args[0];
|
||||
if (!sub || sub === '--help' || sub === '-h') {
|
||||
printHelp();
|
||||
return;
|
||||
}
|
||||
if (sub !== 'fence-backfill') {
|
||||
process.stderr.write(`Unknown facts subcommand: ${sub}\n`);
|
||||
printHelp();
|
||||
setCliExitVerdict(1);
|
||||
return;
|
||||
}
|
||||
|
||||
const dryRun = args.includes('--dry-run');
|
||||
const result = await phaseBFenceFacts(engine, { dryRun });
|
||||
process.stderr.write(
|
||||
`fence-backfill: ${result.status}${result.detail ? ` — ${result.detail}` : ''}\n`,
|
||||
);
|
||||
if (result.status === 'failed') setCliExitVerdict(1);
|
||||
}
|
||||
@@ -334,6 +334,10 @@ export async function scanIntegrity(
|
||||
if (!page) continue;
|
||||
// Skip grandfathered pages (opted out of brain-integrity enforcement)
|
||||
if ((page.frontmatter as Record<string, unknown> | undefined)?.validate === false) continue;
|
||||
// Skip code pages: indexed source files aren't prose. 'tweet' in an
|
||||
// identifier or comment is not a bare-tweet citation gap, and auto-repair
|
||||
// would inject wikilink brackets into source code.
|
||||
if (page.type === 'code') continue;
|
||||
pagesScanned++;
|
||||
bareHits.push(...findBareTweetHits(page.compiled_truth, slug));
|
||||
externalHits.push(...findExternalLinks(page.compiled_truth, slug));
|
||||
@@ -363,6 +367,9 @@ async function scanIntegrityBatch(
|
||||
// YAML) diverges from the sequential path's strict === false check. Intentional
|
||||
// — gbrain lint should reject stringly-typed validate at write time.
|
||||
const validateCondition = sql`AND (frontmatter->>'validate' IS NULL OR frontmatter->>'validate' != 'false')`;
|
||||
// Mirror of the sequential path's `page.type === 'code'` skip: code pages
|
||||
// (indexed source files) are never prose-integrity candidates.
|
||||
const codeCondition = sql`AND type IS DISTINCT FROM 'code'`;
|
||||
|
||||
// v0.32.8: scan ONE row per (source_id, slug) pair, not one per slug.
|
||||
// Pre-fix used DISTINCT ON (slug) which collapsed multi-source rows into
|
||||
@@ -372,7 +379,7 @@ async function scanIntegrityBatch(
|
||||
const rows = await sql`
|
||||
SELECT slug, compiled_truth, frontmatter
|
||||
FROM pages
|
||||
WHERE 1=1 ${typeCondition} ${validateCondition}
|
||||
WHERE 1=1 ${typeCondition} ${validateCondition} ${codeCondition}
|
||||
ORDER BY source_id, slug
|
||||
LIMIT ${limit}
|
||||
`;
|
||||
@@ -461,6 +468,9 @@ async function cmdAuto(args: string[]): Promise<void> {
|
||||
|
||||
const page = await engine.getPage(slug, { sourceId: source_id });
|
||||
if (!page) continue;
|
||||
// Never auto-repair code pages — injecting tweet citations into
|
||||
// indexed source files corrupts them. Same gate as scanIntegrity.
|
||||
if (page.type === 'code') continue;
|
||||
|
||||
pagesProcessed++;
|
||||
progress.tick(1, slug);
|
||||
|
||||
@@ -0,0 +1,224 @@
|
||||
/**
|
||||
* gbrain maintain — conservative self-healing maintenance.
|
||||
*
|
||||
* This command automates the safe parts of the operator runbook:
|
||||
* - stale link/timeline extraction
|
||||
* - stale per-source dream cycles when doctor reports cycle_freshness
|
||||
*
|
||||
* It deliberately does NOT mutate source files, apply schema-pack upgrades, or
|
||||
* invent semantic hub links. Those need review or a separate command with an
|
||||
* auditable proposal surface.
|
||||
*/
|
||||
|
||||
import { existsSync } from 'fs';
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import type { BrainHealth } from '../core/types.ts';
|
||||
import { buildChecks, computeDoctorReport, type DoctorReport, type Check } from './doctor.ts';
|
||||
import { extractStaleFromDB } from './extract.ts';
|
||||
import { runCycle, type CycleReport } from '../core/cycle.ts';
|
||||
|
||||
type ActionStatus = 'ok' | 'would_apply' | 'applied' | 'blocked' | 'skipped';
|
||||
|
||||
export interface MaintenanceAction {
|
||||
name: string;
|
||||
status: ActionStatus;
|
||||
message: string;
|
||||
details?: Record<string, unknown>;
|
||||
}
|
||||
|
||||
export interface MaintainOptions {
|
||||
json: boolean;
|
||||
safe: boolean;
|
||||
dryRun: boolean;
|
||||
help: boolean;
|
||||
}
|
||||
|
||||
export interface MaintainReport {
|
||||
mode: 'dry-run' | 'safe';
|
||||
before: {
|
||||
health: BrainHealth;
|
||||
doctor: DoctorReport;
|
||||
};
|
||||
actions: MaintenanceAction[];
|
||||
after: {
|
||||
health: BrainHealth;
|
||||
doctor: DoctorReport;
|
||||
};
|
||||
}
|
||||
|
||||
export function parseMaintainArgs(args: string[]): MaintainOptions {
|
||||
const safe = args.includes('--safe');
|
||||
return {
|
||||
json: args.includes('--json'),
|
||||
safe,
|
||||
dryRun: args.includes('--dry-run') || !safe,
|
||||
help: args.includes('--help') || args.includes('-h'),
|
||||
};
|
||||
}
|
||||
|
||||
export function extractCycleFreshnessSourceIds(checks: Check[]): string[] {
|
||||
const ids = new Set<string>();
|
||||
for (const check of checks) {
|
||||
if (check.name !== 'cycle_freshness' || check.status === 'ok') continue;
|
||||
const re = /Source '([^']+)' last cycled/g;
|
||||
for (const match of check.message.matchAll(re)) {
|
||||
const id = match[1]?.trim();
|
||||
if (id) ids.add(id);
|
||||
}
|
||||
}
|
||||
return [...ids].sort();
|
||||
}
|
||||
|
||||
async function buildDoctorReport(engine: BrainEngine): Promise<DoctorReport> {
|
||||
const checks = await buildChecks(engine, ['--json', '--scope=brain']);
|
||||
return computeDoctorReport(checks);
|
||||
}
|
||||
|
||||
async function runStaleExtraction(
|
||||
engine: BrainEngine,
|
||||
beforeHealth: BrainHealth,
|
||||
dryRun: boolean,
|
||||
): Promise<MaintenanceAction> {
|
||||
if (beforeHealth.stale_pages <= 0) {
|
||||
return { name: 'extract_stale', status: 'ok', message: 'No stale pages.' };
|
||||
}
|
||||
|
||||
if (dryRun) {
|
||||
return {
|
||||
name: 'extract_stale',
|
||||
status: 'would_apply',
|
||||
message: `Would run DB-backed stale extraction for ${beforeHealth.stale_pages} page(s).`,
|
||||
details: { stale_pages: beforeHealth.stale_pages },
|
||||
};
|
||||
}
|
||||
|
||||
const result = await extractStaleFromDB(engine, {
|
||||
dryRun: false,
|
||||
jsonMode: false,
|
||||
includeFrontmatter: false,
|
||||
catchUp: false,
|
||||
});
|
||||
|
||||
return {
|
||||
name: 'extract_stale',
|
||||
status: 'applied',
|
||||
message: `Processed ${result.pagesProcessed} stale page(s); ${result.staleRemaining} remain.`,
|
||||
details: {
|
||||
links_created: result.linksCreated,
|
||||
timeline_created: result.timelineCreated,
|
||||
pages_processed: result.pagesProcessed,
|
||||
stale_remaining: result.staleRemaining,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
async function runCycleFreshnessMaintenance(
|
||||
engine: BrainEngine,
|
||||
beforeDoctor: DoctorReport,
|
||||
dryRun: boolean,
|
||||
): Promise<MaintenanceAction[]> {
|
||||
const sourceIds = extractCycleFreshnessSourceIds(beforeDoctor.checks);
|
||||
if (sourceIds.length === 0) {
|
||||
return [{ name: 'cycle_freshness', status: 'ok', message: 'All sources cycled recently.' }];
|
||||
}
|
||||
|
||||
if (dryRun) {
|
||||
return sourceIds.map((sourceId) => ({
|
||||
name: 'cycle_freshness',
|
||||
status: 'would_apply',
|
||||
message: `Would run source-scoped dream cycle for ${sourceId}.`,
|
||||
details: { source_id: sourceId },
|
||||
}));
|
||||
}
|
||||
|
||||
const sources = await engine.listAllSources();
|
||||
const actions: MaintenanceAction[] = [];
|
||||
|
||||
for (const sourceId of sourceIds) {
|
||||
const source = sources.find((s) => s.id === sourceId);
|
||||
const localPath = source?.local_path ?? null;
|
||||
const brainDir = localPath && existsSync(localPath) ? localPath : null;
|
||||
const report: CycleReport = await runCycle(engine, {
|
||||
brainDir,
|
||||
dryRun: false,
|
||||
pull: false,
|
||||
sourceId,
|
||||
});
|
||||
actions.push({
|
||||
name: 'cycle_freshness',
|
||||
status: report.status === 'failed' ? 'blocked' : 'applied',
|
||||
message: `Ran source-scoped dream cycle for ${sourceId}: ${report.status}.`,
|
||||
details: {
|
||||
source_id: sourceId,
|
||||
brain_dir: brainDir,
|
||||
cycle_status: report.status,
|
||||
phases: report.phases.map((p) => ({ phase: p.phase, status: p.status })),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
return actions;
|
||||
}
|
||||
|
||||
export async function runMaintain(engine: BrainEngine, args: string[]): Promise<MaintainReport | void> {
|
||||
const opts = parseMaintainArgs(args);
|
||||
if (opts.help) {
|
||||
console.log(`Usage: gbrain maintain [--safe] [--dry-run] [--json]
|
||||
|
||||
Conservative self-healing maintenance.
|
||||
|
||||
Modes:
|
||||
--dry-run Preview safe actions without writes. Default when --safe is absent.
|
||||
--safe Apply safe actions: stale extraction and source cycle freshness.
|
||||
--json Emit a structured before/action/after report.
|
||||
|
||||
Not auto-applied:
|
||||
source-file frontmatter fixes, schema-pack upgrades, atom-pack changes,
|
||||
semantic hub-link guesses, and destructive cleanup.
|
||||
`);
|
||||
return;
|
||||
}
|
||||
|
||||
const beforeHealth = await engine.getHealth();
|
||||
const beforeDoctor = await buildDoctorReport(engine);
|
||||
const actions: MaintenanceAction[] = [];
|
||||
|
||||
actions.push(await runStaleExtraction(engine, beforeHealth, opts.dryRun));
|
||||
actions.push(...await runCycleFreshnessMaintenance(engine, beforeDoctor, opts.dryRun));
|
||||
|
||||
const afterHealth = await engine.getHealth();
|
||||
const afterDoctor = await buildDoctorReport(engine);
|
||||
const report: MaintainReport = {
|
||||
mode: opts.dryRun ? 'dry-run' : 'safe',
|
||||
before: { health: beforeHealth, doctor: beforeDoctor },
|
||||
actions,
|
||||
after: { health: afterHealth, doctor: afterDoctor },
|
||||
};
|
||||
|
||||
if (opts.json) {
|
||||
console.log(JSON.stringify(report, null, 2));
|
||||
} else {
|
||||
printMaintainReport(report);
|
||||
}
|
||||
return report;
|
||||
}
|
||||
|
||||
function printMaintainReport(report: MaintainReport): void {
|
||||
console.log(`GBrain maintain (${report.mode})`);
|
||||
console.log(
|
||||
`Before: brain_score=${Math.round(report.before.health.brain_score)}/100 ` +
|
||||
`stale=${report.before.health.stale_pages} islands=${report.before.health.orphan_pages} ` +
|
||||
`doctor=${report.before.doctor.status}`,
|
||||
);
|
||||
for (const action of report.actions) {
|
||||
console.log(` ${action.status}: ${action.name} — ${action.message}`);
|
||||
}
|
||||
console.log(
|
||||
`After: brain_score=${Math.round(report.after.health.brain_score)}/100 ` +
|
||||
`stale=${report.after.health.stale_pages} islands=${report.after.health.orphan_pages} ` +
|
||||
`doctor=${report.after.doctor.status}`,
|
||||
);
|
||||
if (report.mode === 'dry-run') {
|
||||
console.log('Run `gbrain maintain --safe` to apply safe actions.');
|
||||
}
|
||||
}
|
||||
@@ -39,7 +39,6 @@ import type { BrainEngine } from '../../core/engine.ts';
|
||||
import { loadConfig, toEngineConfig } from '../../core/config.ts';
|
||||
import { createEngine } from '../../core/engine-factory.ts';
|
||||
import { upsertFactRow, parseFactsFence } from '../../core/facts-fence.ts';
|
||||
import { resolvePageFilePath } from '../../core/markdown.ts';
|
||||
|
||||
let testEngineOverride: BrainEngine | null = null;
|
||||
export function __setTestEngineOverride(engine: BrainEngine | null): void {
|
||||
@@ -149,16 +148,9 @@ function isLocalPathDirty(localPath: string): boolean {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Exported (not just via `__testing`) because `gbrain facts fence-backfill`
|
||||
* (#1867) re-runs this phase on demand: remote `extract_facts` deposits that
|
||||
* predate the fence-write backstop leave row_num-NULL rows the cycle guard
|
||||
* refuses to reconcile past. The phase is idempotent (only touches
|
||||
* `row_num IS NULL` rows), so re-running is always safe.
|
||||
*/
|
||||
export async function phaseBFenceFacts(
|
||||
async function phaseBFenceFacts(
|
||||
engine: BrainEngine | null,
|
||||
opts: Pick<OrchestratorOpts, 'dryRun'>,
|
||||
opts: OrchestratorOpts,
|
||||
): Promise<OrchestratorPhaseResult> {
|
||||
if (opts.dryRun) {
|
||||
// Dry-run: report what WOULD happen without touching FS or DB.
|
||||
@@ -246,11 +238,7 @@ export async function phaseBFenceFacts(
|
||||
for (const [key, group] of groups) {
|
||||
const [sourceId, entitySlug] = key.split('\0');
|
||||
const localPath = localPathById.get(sourceId)!;
|
||||
// resolvePageFilePath, NOT a bare join — non-default sources fence
|
||||
// into `<local_path>/.sources/<id>/<slug>.md`, the same path the
|
||||
// fence-write backstop and put_page write-through compute. A bare
|
||||
// join here diverges fence and DB for non-default sources (#2044).
|
||||
const filePath = resolvePageFilePath(localPath, entitySlug, sourceId);
|
||||
const filePath = join(localPath, `${entitySlug}.md`);
|
||||
const tmpPath = `${filePath}.tmp`;
|
||||
|
||||
try {
|
||||
@@ -281,21 +269,6 @@ export async function phaseBFenceFacts(
|
||||
const existingFence = parseFactsFence(body);
|
||||
const existingKeySet = new Set(existingFence.facts.map(f => `${f.claim}\0${f.source ?? ''}`));
|
||||
|
||||
// Seed appended row_nums from MAX(fence max, DB max) — same #2044
|
||||
// divergence guard as writeFactsToFence. When the fence at the
|
||||
// resolved path is missing/behind but the DB already holds stamped
|
||||
// rows for this slug (legacy wrong-path fence writes), fence-max+1
|
||||
// collides with idx_facts_fence_key on the post-rename UPDATE,
|
||||
// failing the page and leaving fence and DB disagreeing.
|
||||
const dbMaxRows = await engine.executeRaw<{ max: number | string | null }>(
|
||||
`SELECT MAX(row_num) AS max FROM facts
|
||||
WHERE source_id = $1 AND source_markdown_slug = $2`,
|
||||
[sourceId, entitySlug],
|
||||
);
|
||||
const dbMaxRowNum = Number(dbMaxRows[0]?.max ?? 0) || 0;
|
||||
const fenceMaxRowNum = existingFence.facts.reduce((m, f) => Math.max(m, f.rowNum), 0);
|
||||
let nextRowNum = Math.max(fenceMaxRowNum, dbMaxRowNum) + 1;
|
||||
|
||||
const assignments: Array<{ id: string; row_num: number }> = [];
|
||||
for (const row of group) {
|
||||
const key = `${row.fact}\0${row.source ?? ''}`;
|
||||
@@ -318,7 +291,6 @@ export async function phaseBFenceFacts(
|
||||
.toISOString().slice(0, 10)
|
||||
: undefined;
|
||||
const { body: updated, rowNum } = upsertFactRow(body, {
|
||||
rowNum: nextRowNum++,
|
||||
claim: row.fact,
|
||||
kind: row.kind,
|
||||
confidence: row.confidence,
|
||||
@@ -409,7 +381,7 @@ async function phaseCVerify(
|
||||
for (const g of groups) {
|
||||
const localPath = localPathById.get(g.source_id);
|
||||
if (!localPath) continue;
|
||||
const filePath = resolvePageFilePath(localPath, g.source_markdown_slug, g.source_id);
|
||||
const filePath = join(localPath, `${g.source_markdown_slug}.md`);
|
||||
if (!existsSync(filePath)) {
|
||||
mismatches.push(`${g.source_markdown_slug} (file missing)`);
|
||||
continue;
|
||||
|
||||
+10
-55
@@ -15,6 +15,11 @@
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { createProgress, startHeartbeat } from '../core/progress.ts';
|
||||
import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts';
|
||||
import {
|
||||
shouldExcludeFromOrphanReporting,
|
||||
loadOrphanPolicyOverrides,
|
||||
type OrphanPolicyOverrides,
|
||||
} from '../core/orphan-policy.ts';
|
||||
|
||||
// --- Types ---
|
||||
|
||||
@@ -32,65 +37,14 @@ export interface OrphanResult {
|
||||
excluded: number;
|
||||
}
|
||||
|
||||
// --- Filter constants ---
|
||||
|
||||
/** Slug suffixes that are always auto-generated root files */
|
||||
const AUTO_SUFFIX_PATTERNS = ['/_index', '/log'];
|
||||
|
||||
/** Page slugs that are pseudo-pages by convention */
|
||||
const PSEUDO_SLUGS = new Set(['_atlas', '_index', '_stats', '_orphans', '_scratch', 'claude']);
|
||||
|
||||
/** Slug segment that marks raw sources */
|
||||
const RAW_SEGMENT = '/raw/';
|
||||
|
||||
/** Slug prefixes where no inbound links is expected */
|
||||
const DENY_PREFIXES = [
|
||||
'output/',
|
||||
'dashboards/',
|
||||
'scripts/',
|
||||
'templates/',
|
||||
'openclaw/config/',
|
||||
];
|
||||
|
||||
/** First slug segments where no inbound links is expected */
|
||||
const FIRST_SEGMENT_EXCLUSIONS = new Set([
|
||||
'scratch',
|
||||
'thoughts',
|
||||
'catalog',
|
||||
'entities',
|
||||
'raw',
|
||||
'atoms',
|
||||
'skills',
|
||||
]);
|
||||
|
||||
// --- Filter logic ---
|
||||
|
||||
/**
|
||||
* Returns true if a slug should be excluded from orphan reporting by default.
|
||||
* These are pages where having no inbound links is expected / not a content problem.
|
||||
*/
|
||||
export function shouldExclude(slug: string): boolean {
|
||||
// Pseudo-pages (exact match)
|
||||
if (PSEUDO_SLUGS.has(slug)) return true;
|
||||
|
||||
// Auto-generated suffix patterns
|
||||
for (const suffix of AUTO_SUFFIX_PATTERNS) {
|
||||
if (slug.endsWith(suffix)) return true;
|
||||
}
|
||||
|
||||
// Raw source slugs
|
||||
if (slug.includes(RAW_SEGMENT)) return true;
|
||||
|
||||
// Deny-prefix slugs
|
||||
for (const prefix of DENY_PREFIXES) {
|
||||
if (slug.startsWith(prefix)) return true;
|
||||
}
|
||||
|
||||
// First-segment exclusions
|
||||
const firstSegment = slug.split('/')[0];
|
||||
if (FIRST_SEGMENT_EXCLUSIONS.has(firstSegment)) return true;
|
||||
|
||||
return false;
|
||||
export function shouldExclude(slug: string, overrides?: OrphanPolicyOverrides): boolean {
|
||||
return shouldExcludeFromOrphanReporting(slug, overrides);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -156,6 +110,7 @@ export async function findOrphans(
|
||||
let allOrphans: { slug: string; title: string; domain: string | null }[];
|
||||
let total: number;
|
||||
let excludedAll: number;
|
||||
const overrides = includePseudo ? undefined : await loadOrphanPolicyOverrides(engine);
|
||||
try {
|
||||
allOrphans = await engine.findOrphanPages(
|
||||
sourceIds ? { sourceIds } : sourceId ? { sourceId } : undefined,
|
||||
@@ -184,7 +139,7 @@ export async function findOrphans(
|
||||
total = liveRows.length;
|
||||
excludedAll = includePseudo
|
||||
? 0
|
||||
: liveRows.reduce((n, r) => n + (shouldExclude(r.slug) ? 1 : 0), 0);
|
||||
: liveRows.reduce((n, r) => n + (shouldExclude(r.slug, overrides) ? 1 : 0), 0);
|
||||
} finally {
|
||||
stopHb();
|
||||
progress.finish();
|
||||
@@ -192,7 +147,7 @@ export async function findOrphans(
|
||||
|
||||
const filtered = includePseudo
|
||||
? allOrphans
|
||||
: allOrphans.filter(row => !shouldExclude(row.slug));
|
||||
: allOrphans.filter(row => !shouldExclude(row.slug, overrides));
|
||||
|
||||
const orphans: OrphanPage[] = filtered.map(row => ({
|
||||
slug: row.slug,
|
||||
|
||||
@@ -54,6 +54,7 @@ import {
|
||||
import {
|
||||
generatePerChunkSynopsis,
|
||||
SYNOPSIS_PROMPT_VERSION,
|
||||
SYNOPSIS_DOC_MAX_CHARS,
|
||||
type GeneratePerChunkSynopsisResult,
|
||||
} from './page-summary.ts';
|
||||
import {
|
||||
@@ -103,8 +104,17 @@ function getEmbeddingModelTag(): string {
|
||||
export function computeCorpusGeneration(args: {
|
||||
crMode: CRMode;
|
||||
haikuModel: string;
|
||||
/**
|
||||
* Resolved `SYNOPSIS_DOC_MAX_CHARS` for per_chunk_synopsis runs. When
|
||||
* present, folded into the hash so changes to
|
||||
* `GBRAIN_SYNOPSIS_DOC_MAX_CHARS` invalidate the prior cache cleanly.
|
||||
* Omit for `crMode !== 'per_chunk_synopsis'` — title / none modes
|
||||
* don't consult the cap and the field stays out of the hash for
|
||||
* back-compat with pre-cap embeddings.
|
||||
*/
|
||||
synopsisDocMaxChars?: number;
|
||||
}): string {
|
||||
return createHash('sha256')
|
||||
const h = createHash('sha256')
|
||||
.update(args.crMode)
|
||||
.update('|')
|
||||
.update(String(SYNOPSIS_PROMPT_VERSION))
|
||||
@@ -113,9 +123,11 @@ export function computeCorpusGeneration(args: {
|
||||
.update('|')
|
||||
.update(String(TITLE_WRAPPER_VERSION))
|
||||
.update('|')
|
||||
.update(getEmbeddingModelTag())
|
||||
.digest('hex')
|
||||
.slice(0, 16);
|
||||
.update(getEmbeddingModelTag());
|
||||
if (args.synopsisDocMaxChars !== undefined) {
|
||||
h.update('|doc_cap=').update(String(args.synopsisDocMaxChars));
|
||||
}
|
||||
return h.digest('hex').slice(0, 16);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -253,7 +265,11 @@ export async function reembedPageWithContextualRetrieval(
|
||||
args.pageSlug,
|
||||
args.sourceId,
|
||||
resolution.mode,
|
||||
computeCorpusGeneration({ crMode: resolution.mode, haikuModel: args.haikuModel ?? DEFAULT_HAIKU_MODEL }),
|
||||
computeCorpusGeneration({
|
||||
crMode: resolution.mode,
|
||||
haikuModel: args.haikuModel ?? DEFAULT_HAIKU_MODEL,
|
||||
synopsisDocMaxChars: resolution.mode === 'per_chunk_synopsis' ? SYNOPSIS_DOC_MAX_CHARS : undefined,
|
||||
}),
|
||||
);
|
||||
return { kind: 'skipped', reason: 'no_chunks' };
|
||||
}
|
||||
@@ -282,6 +298,7 @@ export async function reembedPageWithContextualRetrieval(
|
||||
const corpus_generation = computeCorpusGeneration({
|
||||
crMode: attemptMode,
|
||||
haikuModel,
|
||||
synopsisDocMaxChars: attemptMode === 'per_chunk_synopsis' ? SYNOPSIS_DOC_MAX_CHARS : undefined,
|
||||
});
|
||||
|
||||
// ── PHASE 2: single DB transaction ───────────────────────────
|
||||
|
||||
@@ -176,11 +176,9 @@ export async function runExtractFacts(
|
||||
if (legacyCount > 0) {
|
||||
result.guardTriggered = true;
|
||||
result.warnings.push(
|
||||
`extract_facts: ${legacyCount} legacy fact rows pending fence backfill ` +
|
||||
`(row_num IS NULL — v0.31 rows or remote extract_facts deposits that ` +
|
||||
`predate the fence backstop). Run \`gbrain facts fence-backfill\` ` +
|
||||
`(idempotent, re-runnable) before this phase can safely reconcile ` +
|
||||
`fence → DB.`,
|
||||
`extract_facts: ${legacyCount} legacy v0.31 fact rows pending fence backfill. ` +
|
||||
`Run \`gbrain apply-migrations --yes\` to complete v0_32_2 before this phase ` +
|
||||
`can safely reconcile fence → DB.`,
|
||||
);
|
||||
return result;
|
||||
}
|
||||
|
||||
+6
-7
@@ -1739,14 +1739,13 @@ export interface BrainEngine {
|
||||
* single-row supersede flow because fence reconciliation is the canonical
|
||||
* source-of-truth direction, not the consolidator path.
|
||||
*
|
||||
* Insertion runs in a single transaction. A collision on the v51
|
||||
* partial UNIQUE index `(source_id, source_markdown_slug, row_num)`
|
||||
* skips ONLY that row (ON CONFLICT DO NOTHING, #2044) — the rest of
|
||||
* the batch still commits, so a redundant deposit against an
|
||||
* already-indexed fence row is idempotent instead of a hard failure.
|
||||
* Insertion is atomic per call: all rows commit in a single transaction
|
||||
* or none commit (the transaction rolls back on any constraint
|
||||
* violation, e.g. the v51 partial UNIQUE index on
|
||||
* `(source_id, source_markdown_slug, row_num)`).
|
||||
*
|
||||
* Returns the inserted ids in input-order (colliding rows omitted) so
|
||||
* callers can correlate fence-row → DB-id without a separate lookup.
|
||||
* Returns the inserted ids in input-order so callers can correlate
|
||||
* fence-row → DB-id without a separate lookup.
|
||||
*/
|
||||
insertFacts(
|
||||
rows: Array<NewFact & { row_num: number; source_markdown_slug: string }>,
|
||||
|
||||
@@ -218,27 +218,11 @@ export async function writeFactsToFence(
|
||||
}
|
||||
|
||||
// 2. Upsert each fact onto the fence in input order. row_num
|
||||
// monotonically increases, append-only, seeded from the MAX of
|
||||
// the fence and the DB index (#2044): when fence and DB have
|
||||
// diverged (e.g. legacy writes that stamped DB rows against a
|
||||
// fence at a path this code no longer reads), fence-max+1 can
|
||||
// collide with an existing DB row_num, tripping
|
||||
// idx_facts_fence_key and rolling back the whole batch.
|
||||
const dbMaxRows = await engine.executeRaw<{ max: number | string | null }>(
|
||||
`SELECT MAX(row_num) AS max FROM facts
|
||||
WHERE source_id = $1 AND source_markdown_slug = $2`,
|
||||
[target.sourceId, target.slug],
|
||||
);
|
||||
const dbMaxRowNum = Number(dbMaxRows[0]?.max ?? 0) || 0;
|
||||
const fenceMaxRowNum = parseFactsFence(body).facts
|
||||
.reduce((m, f) => Math.max(m, f.rowNum), 0);
|
||||
let nextRowNum = Math.max(fenceMaxRowNum, dbMaxRowNum) + 1;
|
||||
|
||||
// monotonically increases (max-existing + 1 per call, append-only).
|
||||
const assignedRowNums: number[] = [];
|
||||
for (const f of facts) {
|
||||
const validFromStr = (f.validFrom ?? new Date()).toISOString().slice(0, 10);
|
||||
const { body: updated, rowNum } = upsertFactRow(body, {
|
||||
rowNum: nextRowNum++,
|
||||
claim: f.fact,
|
||||
kind: (f.kind ?? 'fact') as 'fact' | 'event' | 'preference' | 'commitment' | 'belief',
|
||||
confidence: f.confidence ?? 1.0,
|
||||
|
||||
@@ -733,6 +733,11 @@ export async function importFromContent(
|
||||
: computeCorpusGeneration({
|
||||
crMode: effectiveCRMode,
|
||||
haikuModel: 'anthropic:claude-haiku-4-5-20251001',
|
||||
// Inline import-file path never uses per_chunk_synopsis (refuses
|
||||
// upstream); pass undefined so the doc-cap field stays out of
|
||||
// the hash here. Per_chunk_synopsis runs through the Minion
|
||||
// backfill handler which threads SYNOPSIS_DOC_MAX_CHARS through
|
||||
// the service layer.
|
||||
});
|
||||
|
||||
// Transaction wraps all DB writes. Every per-page tx call carries the
|
||||
|
||||
@@ -489,7 +489,22 @@ export async function extractPageLinks(
|
||||
// text inside `[[...]]` before any `|`), NOT the display alias
|
||||
// (ref.name = match[2]). `[[struktura|the project]]` must resolve
|
||||
// `struktura`, not "the project". The display text is for context only.
|
||||
const matches = await resolver.resolveBasenameMatches(ref.slug);
|
||||
//
|
||||
// The literal may be path-qualified (`[[notes/struktura]]`). The FS
|
||||
// path (resolveSlugAll) strips the dirname before its basename lookup,
|
||||
// but this path passed the raw literal to an index keyed by final
|
||||
// segments only — so every slash-containing wikilink outside
|
||||
// DIR_PATTERN silently resolved to nothing. Query by the final
|
||||
// segment, then use the written path as a disambiguation filter
|
||||
// (the analogue of the FS ancestor walk honoring the written path):
|
||||
// a match must end with the literal, so `[[notes/struktura]]` can
|
||||
// resolve to `vault/notes/struktura` but never to `wiki/struktura`.
|
||||
const slashIdx = ref.slug.lastIndexOf('/');
|
||||
const basename = slashIdx === -1 ? ref.slug : ref.slug.slice(slashIdx + 1);
|
||||
let matches = await resolver.resolveBasenameMatches(basename);
|
||||
if (slashIdx !== -1) {
|
||||
matches = matches.filter(m => m === ref.slug || m.endsWith(`/${ref.slug}`));
|
||||
}
|
||||
if (matches.length === 0) continue;
|
||||
const idx = content.indexOf(ref.slug);
|
||||
const context = idx >= 0 ? excerpt(content, idx, 240) : ref.name;
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
/**
|
||||
* Shared orphan-reporting exclusion policy.
|
||||
*
|
||||
* These are pages where "no inbound links" is expected and should not count
|
||||
* against health. Keep this in core so the CLI orphan report and engine health
|
||||
* dashboard cannot drift.
|
||||
*
|
||||
* Defaults are GBrain-wide conventions only. Brain-specific exclusions
|
||||
* (private folder names, one-off fixture slugs) belong in the brain's own
|
||||
* config, not here:
|
||||
*
|
||||
* gbrain config set orphans.exclude_prefixes "my-private-folder/,archive/"
|
||||
* gbrain config set orphans.exclude_slugs "some-one-off-page"
|
||||
*/
|
||||
|
||||
const AUTO_SUFFIX_PATTERNS = ['/_index', '/log'];
|
||||
|
||||
const PSEUDO_SLUGS = new Set(['_atlas', '_index', '_stats', '_orphans', '_scratch', 'claude']);
|
||||
|
||||
const RAW_SEGMENT = '/raw/';
|
||||
|
||||
const DENY_PREFIXES = [
|
||||
'output/',
|
||||
'dashboards/',
|
||||
'scripts/',
|
||||
'templates/',
|
||||
'_templates/',
|
||||
'openclaw/config/',
|
||||
'extracts/',
|
||||
];
|
||||
|
||||
const FIRST_SEGMENT_EXCLUSIONS = new Set([
|
||||
'scratch',
|
||||
'thoughts',
|
||||
'catalog',
|
||||
'entities',
|
||||
'raw',
|
||||
'atoms',
|
||||
'skills',
|
||||
'dreaming',
|
||||
'daily',
|
||||
]);
|
||||
|
||||
const ROOT_DATE_SLUG = /^\d{4}-\d{2}-\d{2}(?:-.+)?$/;
|
||||
|
||||
function isAgentWorkspaceConvention(slug: string): boolean {
|
||||
if (!slug.startsWith('agents/')) return false;
|
||||
if (slug.includes('/memory/dreaming/')) return true;
|
||||
return /^agents\/[^/]+\/(?:agents|identity|soul|tools|user|heartbeat|dreams|dormant)$/.test(slug);
|
||||
}
|
||||
|
||||
/** Per-brain additions to the convention defaults (from config). */
|
||||
export interface OrphanPolicyOverrides {
|
||||
excludePrefixes?: string[];
|
||||
excludeSlugs?: string[];
|
||||
}
|
||||
|
||||
/** Config keys for per-brain orphan exclusions (comma-separated values). */
|
||||
export const ORPHAN_EXCLUDE_PREFIXES_KEY = 'orphans.exclude_prefixes';
|
||||
export const ORPHAN_EXCLUDE_SLUGS_KEY = 'orphans.exclude_slugs';
|
||||
|
||||
function parseList(value: string | null): string[] {
|
||||
if (!value) return [];
|
||||
return value.split(',').map(s => s.trim()).filter(Boolean);
|
||||
}
|
||||
|
||||
/**
|
||||
* Load per-brain orphan exclusions from the brain config table. Callers with
|
||||
* an engine in hand (getHealth, `gbrain orphans`) pass the result as the
|
||||
* second argument to shouldExcludeFromOrphanReporting.
|
||||
*/
|
||||
export async function loadOrphanPolicyOverrides(
|
||||
engine: { getConfig(key: string): Promise<string | null> },
|
||||
): Promise<OrphanPolicyOverrides> {
|
||||
const [prefixes, slugs] = await Promise.all([
|
||||
engine.getConfig(ORPHAN_EXCLUDE_PREFIXES_KEY),
|
||||
engine.getConfig(ORPHAN_EXCLUDE_SLUGS_KEY),
|
||||
]);
|
||||
return { excludePrefixes: parseList(prefixes), excludeSlugs: parseList(slugs) };
|
||||
}
|
||||
|
||||
export function shouldExcludeFromOrphanReporting(
|
||||
slug: string,
|
||||
overrides?: OrphanPolicyOverrides,
|
||||
): boolean {
|
||||
if (PSEUDO_SLUGS.has(slug)) return true;
|
||||
|
||||
for (const suffix of AUTO_SUFFIX_PATTERNS) {
|
||||
if (slug.endsWith(suffix)) return true;
|
||||
}
|
||||
|
||||
if (slug.includes(RAW_SEGMENT)) return true;
|
||||
if (slug.includes('/daily/')) return true;
|
||||
|
||||
for (const prefix of DENY_PREFIXES) {
|
||||
if (slug.startsWith(prefix)) return true;
|
||||
}
|
||||
|
||||
const firstSegment = slug.split('/')[0];
|
||||
if (FIRST_SEGMENT_EXCLUSIONS.has(firstSegment)) return true;
|
||||
|
||||
if (ROOT_DATE_SLUG.test(slug)) return true;
|
||||
|
||||
if (slug.startsWith('_brain-')) return true;
|
||||
|
||||
if (isAgentWorkspaceConvention(slug)) return true;
|
||||
|
||||
if (overrides) {
|
||||
if (overrides.excludeSlugs?.includes(slug)) return true;
|
||||
for (const prefix of overrides.excludePrefixes ?? []) {
|
||||
if (slug.startsWith(prefix)) return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
@@ -44,6 +44,33 @@ const HAIKU_MAX_TOKENS = 200;
|
||||
/** Default model when caller doesn't override. Resolves through the gateway. */
|
||||
const DEFAULT_SYNOPSIS_MODEL = 'anthropic:claude-haiku-4-5-20251001';
|
||||
|
||||
/**
|
||||
* Hard cap on `documentText` length (chars) before send.
|
||||
*
|
||||
* 2026-05-25 fix wave: small local chat models (Gemma 4 E2B, Qwen3 4B) get
|
||||
* dramatically slower on long contexts even with 131K-token windows declared.
|
||||
* A 73K-char page synopsis on Gemma 4 E2B takes 60-120s, exceeding the
|
||||
* worker's default 30s `lockDuration` and tripping `lock-lost` errors.
|
||||
*
|
||||
* Truncate to a budget that fits a small model's effective throughput while
|
||||
* preserving enough document context for the synopsis to be useful. Truncates
|
||||
* the TAIL because the head (title, frontmatter, intro) carries the
|
||||
* document-level anchor the synopsis needs.
|
||||
*
|
||||
* Override per workload via `GBRAIN_SYNOPSIS_DOC_MAX_CHARS`. Default 32768
|
||||
* (~8K tokens at 4 chars/tok) keeps small-model synopsis under ~30s.
|
||||
* Anthropic Haiku is unaffected at this cap; bump higher when running
|
||||
* frontier models if you want richer document anchoring.
|
||||
*/
|
||||
export const SYNOPSIS_DOC_MAX_CHARS = (() => {
|
||||
const env = process.env.GBRAIN_SYNOPSIS_DOC_MAX_CHARS;
|
||||
if (env && /^\d+$/.test(env)) {
|
||||
const n = parseInt(env, 10);
|
||||
if (n >= 512 && n <= 1_048_576) return n;
|
||||
}
|
||||
return 32768;
|
||||
})();
|
||||
|
||||
/**
|
||||
* Synopsis prompt version. Folded into corpus_generation so prompt edits
|
||||
* invalidate prior embeddings via the v0.40.3.0 query_cache.page_generations
|
||||
@@ -188,11 +215,19 @@ function buildUserPrompt(
|
||||
documentText: string,
|
||||
chunkText: string,
|
||||
): string {
|
||||
// Tail-truncate `documentText` to `SYNOPSIS_DOC_MAX_CHARS` so small local
|
||||
// chat models don't stall on >100KB pages. Head preserved (title block,
|
||||
// frontmatter, intro paragraphs carry the document-level anchor).
|
||||
let trimmedDoc = documentText;
|
||||
if (documentText.length > SYNOPSIS_DOC_MAX_CHARS) {
|
||||
trimmedDoc = documentText.slice(0, SYNOPSIS_DOC_MAX_CHARS) +
|
||||
`\n\n[... ${documentText.length - SYNOPSIS_DOC_MAX_CHARS} chars truncated for synopsis budget ...]`;
|
||||
}
|
||||
return [
|
||||
`<page_title>${pageTitle}</page_title>`,
|
||||
'',
|
||||
'<full_document>',
|
||||
documentText,
|
||||
trimmedDoc,
|
||||
'</full_document>',
|
||||
'',
|
||||
'<chunk>',
|
||||
|
||||
+59
-41
@@ -57,6 +57,8 @@ import { finalizeLastSeen } from './chronicle/last-seen.ts';
|
||||
import { computeAnomaliesFromBuckets } from './cycle/anomaly.ts';
|
||||
import { resolveBoostMap, resolveHardExcludes } from './search/source-boost.ts';
|
||||
import { buildSourceFactorCase, buildHardExcludeClause, buildVisibilityClause, buildRecencyComponentSql, buildBestPerPagePoolCte, buildOrFallbackWebsearchQuery } from './search/sql-ranking.ts';
|
||||
import { shouldExcludeFromOrphanReporting, loadOrphanPolicyOverrides } from './orphan-policy.ts';
|
||||
import { LINK_EXTRACTOR_VERSION_TS } from './link-extraction.ts';
|
||||
import {
|
||||
normalizeEngineColumn,
|
||||
buildVectorCastFragment,
|
||||
@@ -2322,6 +2324,10 @@ export class PGLiteEngine implements BrainEngine {
|
||||
// v0.40.3.0 D24 NULL→non-NULL race fix mirrors postgres-engine.ts. Two writers
|
||||
// racing on the same chunk previously raced last-write-wins; the fix lets the
|
||||
// fresher `embedded_at` win in the text-unchanged branch.
|
||||
//
|
||||
// Code-chunk metadata columns follow the same chunk_text-gated CASE pattern as `embedding`
|
||||
// (#769). Re-chunk trusts EXCLUDED outright; pure re-embed COALESCEs so a caller carrying
|
||||
// only embedding-shaped fields doesn't clobber metadata to NULL.
|
||||
await this.db.query(
|
||||
`INSERT INTO content_chunks ${cols} VALUES ${rowParts.join(', ')}
|
||||
ON CONFLICT (page_id, chunk_index) DO UPDATE SET
|
||||
@@ -2345,14 +2351,14 @@ export class PGLiteEngine implements BrainEngine {
|
||||
THEN EXCLUDED.embedded_at
|
||||
ELSE content_chunks.embedded_at
|
||||
END,
|
||||
language = EXCLUDED.language,
|
||||
symbol_name = EXCLUDED.symbol_name,
|
||||
symbol_type = EXCLUDED.symbol_type,
|
||||
start_line = EXCLUDED.start_line,
|
||||
end_line = EXCLUDED.end_line,
|
||||
parent_symbol_path = EXCLUDED.parent_symbol_path,
|
||||
doc_comment = EXCLUDED.doc_comment,
|
||||
symbol_name_qualified = EXCLUDED.symbol_name_qualified,
|
||||
language = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.language ELSE COALESCE(EXCLUDED.language, content_chunks.language) END,
|
||||
symbol_name = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name ELSE COALESCE(EXCLUDED.symbol_name, content_chunks.symbol_name) END,
|
||||
symbol_type = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_type ELSE COALESCE(EXCLUDED.symbol_type, content_chunks.symbol_type) END,
|
||||
start_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.start_line ELSE COALESCE(EXCLUDED.start_line, content_chunks.start_line) END,
|
||||
end_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.end_line ELSE COALESCE(EXCLUDED.end_line, content_chunks.end_line) END,
|
||||
parent_symbol_path = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.parent_symbol_path ELSE COALESCE(EXCLUDED.parent_symbol_path, content_chunks.parent_symbol_path) END,
|
||||
doc_comment = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.doc_comment ELSE COALESCE(EXCLUDED.doc_comment, content_chunks.doc_comment) END,
|
||||
symbol_name_qualified = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name_qualified ELSE COALESCE(EXCLUDED.symbol_name_qualified, content_chunks.symbol_name_qualified) END,
|
||||
modality = EXCLUDED.modality,
|
||||
embedding_image = COALESCE(EXCLUDED.embedding_image, content_chunks.embedding_image)`,
|
||||
params
|
||||
@@ -4102,13 +4108,11 @@ export class PGLiteEngine implements BrainEngine {
|
||||
): Promise<{ inserted: number; ids: number[] }> {
|
||||
if (rows.length === 0) return { inserted: 0, ids: [] };
|
||||
|
||||
// Single transaction; per-row INSERTs (not multi-row VALUES) keep the
|
||||
// embedding-vs-no-embedding branching readable; batch sizes are small
|
||||
// (5-30 rows per page in practice) so the loop overhead is negligible
|
||||
// vs the embedding compute cost. #2044: ON CONFLICT DO NOTHING on the
|
||||
// v51 partial UNIQUE index makes a residual fence/DB row_num collision
|
||||
// skip that row instead of rolling back the whole batch (parity with
|
||||
// postgres-engine.ts).
|
||||
// Single transaction so the v51 partial UNIQUE index can roll back the
|
||||
// whole batch on constraint violation. Per-row INSERTs (not multi-row
|
||||
// VALUES) keep the embedding-vs-no-embedding branching readable; batch
|
||||
// sizes are small (5-30 rows per page in practice) so the loop overhead
|
||||
// is negligible vs the embedding compute cost.
|
||||
const ids = await this.db.transaction(async (tx) => {
|
||||
const out: number[] = [];
|
||||
for (const input of rows) {
|
||||
@@ -4151,11 +4155,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
$14, $15,
|
||||
$16, $17, $18, $19,
|
||||
$20
|
||||
)
|
||||
ON CONFLICT (source_id, source_markdown_slug, row_num)
|
||||
WHERE row_num IS NOT NULL
|
||||
DO NOTHING
|
||||
RETURNING id`
|
||||
) RETURNING id`
|
||||
: `INSERT INTO facts (
|
||||
source_id, entity_slug, fact, kind, visibility, notability, context,
|
||||
valid_from, valid_until, source, source_session, confidence,
|
||||
@@ -4169,16 +4169,12 @@ export class PGLiteEngine implements BrainEngine {
|
||||
$15, $16,
|
||||
$17, $18, $19, $20,
|
||||
$21
|
||||
)
|
||||
ON CONFLICT (source_id, source_markdown_slug, row_num)
|
||||
WHERE row_num IS NOT NULL
|
||||
DO NOTHING
|
||||
RETURNING id`,
|
||||
) RETURNING id`,
|
||||
embedStr === null
|
||||
? [ctx.source_id, entitySlug, input.fact, kind, visibility, notability, context, validFrom, validUntil, input.source, sourceSession, confidence, embeddedAt, input.row_num, input.source_markdown_slug, claimMetric, claimValue, claimUnit, claimPeriod, eventType]
|
||||
: [ctx.source_id, entitySlug, input.fact, kind, visibility, notability, context, validFrom, validUntil, input.source, sourceSession, confidence, embedStr, embeddedAt, input.row_num, input.source_markdown_slug, claimMetric, claimValue, claimUnit, claimPeriod, eventType],
|
||||
);
|
||||
if (ins.rows[0]) out.push(ins.rows[0].id);
|
||||
out.push(ins.rows[0].id);
|
||||
}
|
||||
return out;
|
||||
});
|
||||
@@ -5212,26 +5208,31 @@ export class PGLiteEngine implements BrainEngine {
|
||||
const { rows: [h] } = await this.db.query(`
|
||||
WITH entity_pages AS (
|
||||
SELECT id, slug FROM pages WHERE type IN ('person', 'company')
|
||||
),
|
||||
narrative_pages AS (
|
||||
-- Composition-aware score: code source files and calendar daily
|
||||
-- files are orphans-by-design (no inbound wikilinks, no Timeline
|
||||
-- fence). Excluding them from the orphan/link-density/timeline
|
||||
-- denominators keeps a bulk code/calendar import from cratering
|
||||
-- brain_score.
|
||||
SELECT id FROM pages WHERE type IS NULL OR type NOT IN ('code', 'calendar-index')
|
||||
)
|
||||
SELECT
|
||||
(SELECT count(*) FROM pages) as page_count,
|
||||
(SELECT count(*) FROM narrative_pages) as narrative_page_count,
|
||||
(SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float /
|
||||
GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage,
|
||||
(SELECT count(*) FROM pages p
|
||||
WHERE p.updated_at < (SELECT MAX(te.created_at) FROM timeline_entries te WHERE te.page_id = p.id)
|
||||
) as stale_pages,
|
||||
-- Bug 11 — orphan = islanded (no inbound AND no outbound).
|
||||
-- See BrainHealth.orphan_pages docstring; docs updated to match this.
|
||||
(SELECT count(*) FROM pages p
|
||||
WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id)
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)
|
||||
) as orphan_pages,
|
||||
0 as stale_pages,
|
||||
-- Bug 11 — orphan = islanded (no inbound AND no outbound). The raw
|
||||
-- list is filtered in TS using the shared orphan-reporting policy.
|
||||
0 as orphan_pages,
|
||||
(SELECT count(*) FROM links l
|
||||
WHERE NOT EXISTS (SELECT 1 FROM pages p WHERE p.id = l.to_page_id)
|
||||
) as dead_links,
|
||||
(SELECT count(*) FROM content_chunks WHERE embedded_at IS NULL) as missing_embeddings,
|
||||
(SELECT count(*) FROM links) as link_count,
|
||||
(SELECT count(DISTINCT page_id) FROM timeline_entries) as pages_with_timeline,
|
||||
(SELECT count(DISTINCT te.page_id) FROM timeline_entries te
|
||||
WHERE te.page_id IN (SELECT id FROM narrative_pages)) as pages_with_timeline,
|
||||
(SELECT count(*) FROM entity_pages e
|
||||
WHERE EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = e.id))::float /
|
||||
GREATEST((SELECT count(*) FROM entity_pages), 1)::float as link_coverage,
|
||||
@@ -5250,17 +5251,34 @@ export class PGLiteEngine implements BrainEngine {
|
||||
LIMIT 5
|
||||
`);
|
||||
|
||||
const { rows: islandedRows } = await this.db.query(`
|
||||
SELECT p.slug
|
||||
FROM pages p
|
||||
-- Narrative pages only (same type filter as the narrative_pages CTE):
|
||||
-- code/calendar-index pages are orphans-by-design and must not count
|
||||
-- against the noOrphans component (#1144).
|
||||
WHERE (p.type IS NULL OR p.type NOT IN ('code', 'calendar-index'))
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id)
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)
|
||||
`);
|
||||
|
||||
const r = h as Record<string, unknown>;
|
||||
const pageCount = Number(r.page_count);
|
||||
// Composition-aware denominators (excludes code/calendar-index pages);
|
||||
// a code-only brain has nothing narrative to penalize → full marks.
|
||||
const narrativePageCount = Number(r.narrative_page_count);
|
||||
const embedCoverage = Number(r.embed_coverage);
|
||||
const orphanPages = Number(r.orphan_pages);
|
||||
const stalePages = await this.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS });
|
||||
const orphanOverrides = await loadOrphanPolicyOverrides(this);
|
||||
const orphanPages = (islandedRows as { slug: string }[])
|
||||
.filter(row => !shouldExcludeFromOrphanReporting(row.slug, orphanOverrides)).length;
|
||||
const deadLinks = Number(r.dead_links);
|
||||
const linkCount = Number(r.link_count);
|
||||
const pagesWithTimeline = Number(r.pages_with_timeline);
|
||||
|
||||
const linkDensity = pageCount > 0 ? Math.min(linkCount / pageCount, 1) : 0;
|
||||
const timelineCoverageDensity = pageCount > 0 ? Math.min(pagesWithTimeline / pageCount, 1) : 0;
|
||||
const noOrphans = pageCount > 0 ? 1 - (orphanPages / pageCount) : 1;
|
||||
const linkDensity = narrativePageCount > 0 ? Math.min(linkCount / narrativePageCount, 1) : 1;
|
||||
const timelineCoverageDensity = narrativePageCount > 0 ? Math.min(pagesWithTimeline / narrativePageCount, 1) : 1;
|
||||
const noOrphans = narrativePageCount > 0 ? 1 - (orphanPages / narrativePageCount) : 1;
|
||||
const noDeadLinks = pageCount > 0 ? 1 - Math.min(deadLinks / pageCount, 1) : 1;
|
||||
// Bug 11 — per-component points. Sum equals brainScore by construction
|
||||
// so `doctor` can render a breakdown that adds up to the total.
|
||||
@@ -5281,7 +5299,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
return {
|
||||
page_count: pageCount,
|
||||
embed_coverage: embedCoverage,
|
||||
stale_pages: Number(r.stale_pages),
|
||||
stale_pages: stalePages,
|
||||
orphan_pages: orphanPages,
|
||||
missing_embeddings: Number(r.missing_embeddings),
|
||||
brain_score: brainScore,
|
||||
|
||||
+60
-38
@@ -67,6 +67,8 @@ import { resolveBoostMap, resolveHardExcludes } from './search/source-boost.ts';
|
||||
import { buildSourceFactorCase, buildHardExcludeClause, buildVisibilityClause, buildRecencyComponentSql, buildBestPerPagePoolCte, buildOrFallbackWebsearchQuery } from './search/sql-ranking.ts';
|
||||
import { DEFAULT_EMBEDDING_MODEL, DEFAULT_EMBEDDING_DIMENSIONS } from './ai/defaults.ts';
|
||||
import { DELETE_BATCH_SIZE } from './engine-constants.ts';
|
||||
import { shouldExcludeFromOrphanReporting, loadOrphanPolicyOverrides } from './orphan-policy.ts';
|
||||
import { LINK_EXTRACTOR_VERSION_TS } from './link-extraction.ts';
|
||||
|
||||
function escapeSqlStringLiteral(value: string): string {
|
||||
return value.replace(/'/g, "''");
|
||||
@@ -2473,6 +2475,13 @@ export class PostgresEngine implements BrainEngine {
|
||||
// - new is fresher (embedded_at > existing.embedded_at) → take new
|
||||
// - otherwise → keep existing (slower writer with stale embedding loses)
|
||||
// Mirrored in pglite-engine.ts; pinned by test/e2e/concurrent-embed-race.test.ts.
|
||||
//
|
||||
// Code-chunk metadata columns (language / symbol_name / symbol_type / line range /
|
||||
// parent_symbol_path / doc_comment / symbol_name_qualified) follow the SAME chunk_text-gated
|
||||
// CASE pattern as `embedding` (#769). Re-chunk (chunk_text changed) trusts EXCLUDED outright;
|
||||
// pure re-embed (chunk_text unchanged) COALESCEs so a caller that only carries embedding
|
||||
// doesn't clobber metadata to NULL. Without this, every embed --stale pass nuked code-def's
|
||||
// primary index for thousands of chunks at once.
|
||||
await sql.unsafe(
|
||||
`INSERT INTO content_chunks ${cols} VALUES ${rows.join(', ')}
|
||||
ON CONFLICT (page_id, chunk_index) DO UPDATE SET
|
||||
@@ -2496,14 +2505,14 @@ export class PostgresEngine implements BrainEngine {
|
||||
THEN EXCLUDED.embedded_at
|
||||
ELSE content_chunks.embedded_at
|
||||
END,
|
||||
language = EXCLUDED.language,
|
||||
symbol_name = EXCLUDED.symbol_name,
|
||||
symbol_type = EXCLUDED.symbol_type,
|
||||
start_line = EXCLUDED.start_line,
|
||||
end_line = EXCLUDED.end_line,
|
||||
parent_symbol_path = EXCLUDED.parent_symbol_path,
|
||||
doc_comment = EXCLUDED.doc_comment,
|
||||
symbol_name_qualified = EXCLUDED.symbol_name_qualified,
|
||||
language = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.language ELSE COALESCE(EXCLUDED.language, content_chunks.language) END,
|
||||
symbol_name = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name ELSE COALESCE(EXCLUDED.symbol_name, content_chunks.symbol_name) END,
|
||||
symbol_type = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_type ELSE COALESCE(EXCLUDED.symbol_type, content_chunks.symbol_type) END,
|
||||
start_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.start_line ELSE COALESCE(EXCLUDED.start_line, content_chunks.start_line) END,
|
||||
end_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.end_line ELSE COALESCE(EXCLUDED.end_line, content_chunks.end_line) END,
|
||||
parent_symbol_path = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.parent_symbol_path ELSE COALESCE(EXCLUDED.parent_symbol_path, content_chunks.parent_symbol_path) END,
|
||||
doc_comment = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.doc_comment ELSE COALESCE(EXCLUDED.doc_comment, content_chunks.doc_comment) END,
|
||||
symbol_name_qualified = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name_qualified ELSE COALESCE(EXCLUDED.symbol_name_qualified, content_chunks.symbol_name_qualified) END,
|
||||
modality = EXCLUDED.modality,
|
||||
embedding_image = COALESCE(EXCLUDED.embedding_image, content_chunks.embedding_image)`,
|
||||
params as Parameters<typeof sql.unsafe>[1],
|
||||
@@ -4302,12 +4311,10 @@ export class PostgresEngine implements BrainEngine {
|
||||
// ONCE per process so the cast matches the actual column type
|
||||
// (halfvec vs vector). The probe is cached after first call.
|
||||
const castSuffix = await this.resolveFactsEmbeddingCast();
|
||||
// Single transaction; per-row INSERTs (not multi-row VALUES) keep the
|
||||
// embedding-vs-no-embedding branching readable; batch sizes are small
|
||||
// (5-30 rows per page in practice). #2044: ON CONFLICT DO NOTHING on
|
||||
// the v51 partial UNIQUE index makes a residual fence/DB row_num
|
||||
// collision skip that row instead of rolling back the whole batch —
|
||||
// the fence stays system-of-record and reconciliation catches up.
|
||||
// Single transaction so the v51 partial UNIQUE index can roll back
|
||||
// the whole batch on constraint violation. Per-row INSERTs (not
|
||||
// multi-row VALUES) keep the embedding-vs-no-embedding branching
|
||||
// readable; batch sizes are small (5-30 rows per page in practice).
|
||||
// No supersede flow in this path — fence reconciliation is the
|
||||
// canonical source-of-truth direction, not the consolidator path.
|
||||
const ids = await sql.begin(async (tx) => {
|
||||
@@ -4348,13 +4355,9 @@ export class PostgresEngine implements BrainEngine {
|
||||
${input.row_num}, ${input.source_markdown_slug},
|
||||
${claimMetric}, ${claimValue}, ${claimUnit}, ${claimPeriod},
|
||||
${eventType}
|
||||
)
|
||||
ON CONFLICT (source_id, source_markdown_slug, row_num)
|
||||
WHERE row_num IS NOT NULL
|
||||
DO NOTHING
|
||||
RETURNING id
|
||||
) RETURNING id
|
||||
`;
|
||||
if (ins[0]) out.push(Number(ins[0].id));
|
||||
out.push(Number(ins[0].id));
|
||||
}
|
||||
return out;
|
||||
});
|
||||
@@ -5319,32 +5322,35 @@ export class PostgresEngine implements BrainEngine {
|
||||
async getHealth(): Promise<BrainHealth> {
|
||||
const sql = this.sql;
|
||||
// Bug 11 doc-drift fix — orphan_pages means "islanded" (no inbound AND
|
||||
// no outbound links), aligning both engines with the user-facing
|
||||
// definition. The type comment previously said "no inbound" but the
|
||||
// SQL required both — docs now match code so users can trust the
|
||||
// number. A hub page that links out to many but has no back-references
|
||||
// is working as intended, not an orphan.
|
||||
// no outbound links). The raw islanded list is filtered through the same
|
||||
// policy as `gbrain orphans` so convention pages do not count against
|
||||
// dashboard health.
|
||||
const [h] = await sql`
|
||||
WITH entity_pages AS (
|
||||
SELECT id, slug FROM pages WHERE type IN ('person', 'company')
|
||||
),
|
||||
narrative_pages AS (
|
||||
-- Composition-aware score: code source files and calendar daily
|
||||
-- files are orphans-by-design (no inbound wikilinks, no Timeline
|
||||
-- fence). Excluding them from the orphan/link-density/timeline
|
||||
-- denominators keeps a bulk code/calendar import from cratering
|
||||
-- brain_score.
|
||||
SELECT id FROM pages WHERE type IS NULL OR type NOT IN ('code', 'calendar-index')
|
||||
)
|
||||
SELECT
|
||||
(SELECT count(*) FROM pages) as page_count,
|
||||
(SELECT count(*) FROM narrative_pages) as narrative_page_count,
|
||||
(SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float /
|
||||
GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage,
|
||||
(SELECT count(*) FROM pages p
|
||||
WHERE p.updated_at < (SELECT MAX(te.created_at) FROM timeline_entries te WHERE te.page_id = p.id)
|
||||
) as stale_pages,
|
||||
(SELECT count(*) FROM pages p
|
||||
WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id)
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)
|
||||
) as orphan_pages,
|
||||
0 as stale_pages,
|
||||
0 as orphan_pages,
|
||||
(SELECT count(*) FROM links l
|
||||
WHERE NOT EXISTS (SELECT 1 FROM pages p WHERE p.id = l.to_page_id)
|
||||
) as dead_links,
|
||||
(SELECT count(*) FROM content_chunks WHERE embedded_at IS NULL) as missing_embeddings,
|
||||
(SELECT count(*) FROM links) as link_count,
|
||||
(SELECT count(DISTINCT page_id) FROM timeline_entries) as pages_with_timeline,
|
||||
(SELECT count(DISTINCT te.page_id) FROM timeline_entries te
|
||||
WHERE te.page_id IN (SELECT id FROM narrative_pages)) as pages_with_timeline,
|
||||
(SELECT count(*) FROM entity_pages e
|
||||
WHERE EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = e.id))::float /
|
||||
GREATEST((SELECT count(*) FROM entity_pages), 1)::float as link_coverage,
|
||||
@@ -5362,17 +5368,33 @@ export class PostgresEngine implements BrainEngine {
|
||||
LIMIT 5
|
||||
`;
|
||||
|
||||
const islandedRows = await sql<{ slug: string }[]>`
|
||||
SELECT p.slug
|
||||
FROM pages p
|
||||
-- Narrative pages only (same type filter as the narrative_pages CTE):
|
||||
-- code/calendar-index pages are orphans-by-design and must not count
|
||||
-- against the noOrphans component (#1144).
|
||||
WHERE (p.type IS NULL OR p.type NOT IN ('code', 'calendar-index'))
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id)
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)
|
||||
`;
|
||||
|
||||
const pageCount = Number(h.page_count);
|
||||
// Composition-aware denominators (excludes code/calendar-index pages);
|
||||
// a code-only brain has nothing narrative to penalize → full marks.
|
||||
const narrativePageCount = Number(h.narrative_page_count);
|
||||
const embedCoverage = Number(h.embed_coverage);
|
||||
const orphanPages = Number(h.orphan_pages);
|
||||
const stalePages = await this.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS });
|
||||
const orphanOverrides = await loadOrphanPolicyOverrides(this);
|
||||
const orphanPages = islandedRows.filter(row => !shouldExcludeFromOrphanReporting(row.slug, orphanOverrides)).length;
|
||||
const deadLinks = Number(h.dead_links);
|
||||
const linkCount = Number(h.link_count);
|
||||
const pagesWithTimeline = Number(h.pages_with_timeline);
|
||||
|
||||
// brain_score: 0-100 weighted average
|
||||
const linkDensity = pageCount > 0 ? Math.min(linkCount / pageCount, 1) : 0;
|
||||
const timelineCoverageWhole = pageCount > 0 ? Math.min(pagesWithTimeline / pageCount, 1) : 0;
|
||||
const noOrphans = pageCount > 0 ? 1 - (orphanPages / pageCount) : 1;
|
||||
const linkDensity = narrativePageCount > 0 ? Math.min(linkCount / narrativePageCount, 1) : 1;
|
||||
const timelineCoverageWhole = narrativePageCount > 0 ? Math.min(pagesWithTimeline / narrativePageCount, 1) : 1;
|
||||
const noOrphans = narrativePageCount > 0 ? 1 - (orphanPages / narrativePageCount) : 1;
|
||||
const noDeadLinks = pageCount > 0 ? 1 - Math.min(deadLinks / pageCount, 1) : 1;
|
||||
// Per-component points. Sum equals brainScore by construction.
|
||||
//
|
||||
@@ -5392,7 +5414,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
return {
|
||||
page_count: pageCount,
|
||||
embed_coverage: embedCoverage,
|
||||
stale_pages: Number(h.stale_pages),
|
||||
stale_pages: stalePages,
|
||||
orphan_pages: orphanPages,
|
||||
missing_embeddings: Number(h.missing_embeddings),
|
||||
brain_score: brainScore,
|
||||
|
||||
@@ -93,7 +93,13 @@ import { resolveLrSchedule } from './lr-schedule.ts';
|
||||
import { preflight, formatPreflightReport } from './preflight.ts';
|
||||
import { isRejected, loadRejectedBuffer, makeRejectedEntry, saveRejectedBuffer } from './rejected-buffer.ts';
|
||||
import { runReflect, runOneShotRewrite, describeJudges } from './reflect.ts';
|
||||
import { acceptCandidate, bestPath, revertAllPending, skillPath, writeProposed } from './version-store.ts';
|
||||
import {
|
||||
acceptCandidate,
|
||||
proposedPath as proposedFilePath,
|
||||
revertAllPending,
|
||||
skillPath,
|
||||
writeProposed,
|
||||
} from './version-store.ts';
|
||||
import { runValidationGate, scoreSkillOnTasks } from './validate-gate.ts';
|
||||
import { ROLLOUT_SUCCESS_THRESHOLD } from './types.ts';
|
||||
import type { SkillOptOpts, EditOp, RunReceipt, BenchmarkTask } from './types.ts';
|
||||
@@ -702,9 +708,9 @@ async function runOptimizationLoop(
|
||||
// to the catch's assignment values only (it can't prove the async callback ran).
|
||||
const finalOutcome = outcome as 'accepted' | 'no_improvement' | 'aborted' | 'errored';
|
||||
if (!mutateDecision.mutate && finalOutcome === 'accepted') {
|
||||
// best.md was written by writeProposed() in the accept branch (no-mutate
|
||||
// path); it doubles as proposed.md for human review. SKILL.md untouched.
|
||||
proposedPath = bestPath(skillsDir, skillName);
|
||||
// writeProposed() emitted both the best pointer and the stable review
|
||||
// artifact in the accept branch. SKILL.md remains untouched.
|
||||
proposedPath = proposedFilePath(skillsDir, skillName);
|
||||
} else if (mutateDecision.mutate) {
|
||||
mutatedSkillFile = finalOutcome === 'accepted';
|
||||
}
|
||||
|
||||
@@ -23,6 +23,7 @@
|
||||
*
|
||||
* history.json
|
||||
* best.md
|
||||
* proposed.md
|
||||
* versions/
|
||||
* v0001_e1_s1.md
|
||||
* v0002_e1_s2.md
|
||||
@@ -52,6 +53,10 @@ export function bestPath(skillsDir: string, skillName: string): string {
|
||||
return path.join(skilloptDir(skillsDir, skillName), 'best.md');
|
||||
}
|
||||
|
||||
export function proposedPath(skillsDir: string, skillName: string): string {
|
||||
return path.join(skilloptDir(skillsDir, skillName), 'proposed.md');
|
||||
}
|
||||
|
||||
export function skillPath(skillsDir: string, skillName: string): string {
|
||||
return path.join(skillsDir, skillName, 'SKILL.md');
|
||||
}
|
||||
@@ -171,17 +176,18 @@ export function acceptCandidate(input: AcceptInput): AcceptResult {
|
||||
}
|
||||
|
||||
/**
|
||||
* Write the candidate to `best.md` (which doubles as `proposed.md`) WITHOUT
|
||||
* touching SKILL.md or the history ledger. Used by the `--no-mutate` /
|
||||
* bundled-without-allow paths: the optimizer found a better candidate but the
|
||||
* caller opted out of in-place mutation, so we surface it for human review.
|
||||
* Returns the path written. Atomic (.tmp + rename).
|
||||
* Write the candidate to both `best.md` and `proposed.md` WITHOUT touching
|
||||
* SKILL.md or the history ledger. `best.md` remains the optimizer's current
|
||||
* best pointer; `proposed.md` is the stable human-review artifact promised by
|
||||
* `--no-mutate`. Returns the proposal path. Each write is atomic (.tmp + rename).
|
||||
*/
|
||||
export function writeProposed(skillsDir: string, skillName: string, candidateText: string): string {
|
||||
const p = bestPath(skillsDir, skillName);
|
||||
fs.mkdirSync(path.dirname(p), { recursive: true });
|
||||
atomicWrite(p, candidateText);
|
||||
return p;
|
||||
const best = bestPath(skillsDir, skillName);
|
||||
const proposed = proposedPath(skillsDir, skillName);
|
||||
fs.mkdirSync(path.dirname(best), { recursive: true });
|
||||
atomicWrite(best, candidateText);
|
||||
atomicWrite(proposed, candidateText);
|
||||
return proposed;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -63,25 +63,26 @@ interface PluginCtx {
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
export function register(api: PluginApi) {
|
||||
api.registerContextEngine(ENGINE_ID, (ctx: PluginCtx) => {
|
||||
const hostResolver =
|
||||
typeof ctx.resolveEntities === 'function'
|
||||
? ctx.resolveEntities
|
||||
: typeof ctx.brainQuery === 'function'
|
||||
? ctx.brainQuery
|
||||
: undefined;
|
||||
return createGBrainContextEngine({
|
||||
workspaceDir: ctx.workspaceDir,
|
||||
resolveEntities: hostResolver,
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
const entry: PluginEntry = {
|
||||
id: 'gbrain-context-engine',
|
||||
name: 'GBrain Context Engine',
|
||||
description: 'Deterministic temporal/spatial context injection on every turn',
|
||||
|
||||
register(api: PluginApi) {
|
||||
api.registerContextEngine(ENGINE_ID, (ctx: PluginCtx) => {
|
||||
const hostResolver =
|
||||
typeof ctx.resolveEntities === 'function'
|
||||
? ctx.resolveEntities
|
||||
: typeof ctx.brainQuery === 'function'
|
||||
? ctx.brainQuery
|
||||
: undefined;
|
||||
return createGBrainContextEngine({
|
||||
workspaceDir: ctx.workspaceDir,
|
||||
resolveEntities: hostResolver,
|
||||
});
|
||||
});
|
||||
},
|
||||
register,
|
||||
};
|
||||
|
||||
export default entry;
|
||||
|
||||
@@ -119,6 +119,40 @@ describe('Bug 11 — orphan_pages is "no inbound links"', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('#1144 — brain_score is composition-aware', () => {
|
||||
test('code and calendar-index pages do not count as orphans or dilute density metrics', async () => {
|
||||
// Two linked narrative pages + a flood of orphan-by-design pages.
|
||||
await engine.putPage('people/alice', { type: 'person', title: 'Alice', compiled_truth: 'a', frontmatter: {} });
|
||||
await engine.putPage('people/bob', { type: 'person', title: 'Bob', compiled_truth: 'b', frontmatter: {} });
|
||||
const aId = (await (engine as any).db.query(`SELECT id FROM pages WHERE slug='people/alice'`)).rows[0].id;
|
||||
const bId = (await (engine as any).db.query(`SELECT id FROM pages WHERE slug='people/bob'`)).rows[0].id;
|
||||
await (engine as any).db.query(
|
||||
`INSERT INTO links (from_page_id, to_page_id, link_type) VALUES ($1, $2, 'mentions')`,
|
||||
[aId, bId],
|
||||
);
|
||||
for (let i = 0; i < 10; i++) {
|
||||
await engine.putPage(`code/src/f${i}.py`, { type: 'code', title: `f${i}.py`, compiled_truth: 'def f(): pass', frontmatter: {} });
|
||||
}
|
||||
await engine.putPage('daily/calendar/2026-01-01', { type: 'calendar-index', title: '2026-01-01', compiled_truth: 'events', frontmatter: {} });
|
||||
|
||||
const h = await engine.getHealth();
|
||||
// 11 unlinked code/calendar pages exist, but both narrative pages are
|
||||
// linked → 0 orphans, full no-orphans marks.
|
||||
expect(h.orphan_pages).toBe(0);
|
||||
expect(h.no_orphans_score).toBe(15);
|
||||
// Link density: 1 link / 2 narrative pages, not 1 / 13 total pages.
|
||||
expect(h.link_density_score).toBe(Math.round(0.5 * 25));
|
||||
});
|
||||
|
||||
test('a brain with ONLY non-narrative pages gets full composition marks', async () => {
|
||||
await engine.putPage('code/src/only.py', { type: 'code', title: 'only.py', compiled_truth: 'x = 1', frontmatter: {} });
|
||||
const h = await engine.getHealth();
|
||||
expect(h.no_orphans_score).toBe(15);
|
||||
expect(h.link_density_score).toBe(25);
|
||||
expect(h.timeline_coverage_score).toBe(15);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Bug 11 — doctor renders brain_score breakdown', () => {
|
||||
test('doctor source contains brain_score breakdown rendering', async () => {
|
||||
const source = await Bun.file(new URL('../src/commands/doctor.ts', import.meta.url)).text();
|
||||
|
||||
@@ -763,3 +763,48 @@ describeBoth('Engine parity — federated sourceIds[] secondary reads (#2200)',
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
// #1144 — brain_score is composition-aware: code / calendar-index pages are
|
||||
// orphans-by-design and must not feed the orphan/link-density/timeline
|
||||
// denominators. Both engines must agree.
|
||||
async function seedComposition(eng: BrainEngine) {
|
||||
await eng.putPage('people/comp-alice', { type: 'person', title: 'Alice', compiled_truth: 'a', timeline: '' });
|
||||
await eng.putPage('people/comp-bob', { type: 'person', title: 'Bob', compiled_truth: 'b', timeline: '' });
|
||||
await eng.addLink('people/comp-alice', 'people/comp-bob', 'knows', 'mentions', 'markdown');
|
||||
for (let i = 0; i < 5; i++) {
|
||||
await eng.putPage(`code/src/comp-f${i}.py`, { type: 'code', title: `f${i}.py`, compiled_truth: 'def f(): pass', timeline: '' });
|
||||
}
|
||||
await eng.putPage('daily/calendar/2026-01-01', { type: 'calendar-index', title: '2026-01-01', compiled_truth: 'events', timeline: '' });
|
||||
}
|
||||
|
||||
describeBoth('Engine parity — getHealth composition-aware brain_score (#1144)', () => {
|
||||
let pgEngine: BrainEngine;
|
||||
let pgliteEngine: PGLiteEngine;
|
||||
|
||||
beforeAll(async () => {
|
||||
pgEngine = await setupDB();
|
||||
await seedComposition(pgEngine);
|
||||
pgliteEngine = new PGLiteEngine();
|
||||
await pgliteEngine.connect({});
|
||||
await pgliteEngine.initSchema();
|
||||
await seedComposition(pgliteEngine);
|
||||
}, 90_000);
|
||||
|
||||
afterAll(async () => {
|
||||
await pgliteEngine.disconnect();
|
||||
await teardownDB();
|
||||
}, 30_000);
|
||||
|
||||
test('orphan/link-density/timeline components identical and exclude code + calendar-index pages', async () => {
|
||||
const pg = await pgEngine.getHealth();
|
||||
const pglite = await pgliteEngine.getHealth();
|
||||
for (const k of ['orphan_pages', 'no_orphans_score', 'link_density_score', 'timeline_coverage_score', 'brain_score'] as const) {
|
||||
expect(pg[k]).toBe(pglite[k]);
|
||||
}
|
||||
// 6 non-narrative pages exist but both narrative pages are linked → 0 orphans.
|
||||
expect(pg.orphan_pages).toBe(0);
|
||||
expect(pg.no_orphans_score).toBe(15);
|
||||
// 1 link / 2 narrative pages, not / 8 total pages.
|
||||
expect(pg.link_density_score).toBe(Math.round(0.5 * 25));
|
||||
});
|
||||
});
|
||||
|
||||
@@ -172,6 +172,59 @@ describe('issue #972 — DB-source (gbrain extract links --source db)', () => {
|
||||
expect(strk!.link_type).toBe('wikilink_basename');
|
||||
});
|
||||
|
||||
test('flag ON → path-qualified wikilink outside DIR_PATTERN resolves via DB path', async () => {
|
||||
// `[[notes/struktura]]` — `notes` is not in DIR_PATTERN, so the ref
|
||||
// reaches the generic pass with its dirname intact. Regression: the DB
|
||||
// path queried the basename index with the raw literal (which is keyed
|
||||
// by final segments only), so path-qualified wikilinks outside
|
||||
// DIR_PATTERN silently produced zero edges while the FS path resolved
|
||||
// the identical content.
|
||||
await engine.putPage('notes/struktura', {
|
||||
type: 'concept' as any, title: 'Struktura Notes',
|
||||
compiled_truth: '', timeline: '',
|
||||
});
|
||||
await engine.putPage('concepts/knowledge-graph', {
|
||||
type: 'concept', title: 'Knowledge Graph',
|
||||
compiled_truth: 'Background in [[notes/struktura]].', timeline: '',
|
||||
});
|
||||
await engine.setConfig('link_resolution.global_basename', 'true');
|
||||
|
||||
await runExtract(engine, ['links', '--source', 'db']);
|
||||
|
||||
const outLinks = await engine.getLinks('concepts/knowledge-graph');
|
||||
const strk = outLinks.find(l => l.to_slug === 'notes/struktura');
|
||||
expect(strk).toBeDefined();
|
||||
expect(strk!.link_type).toBe('wikilink_basename');
|
||||
expect(strk!.link_source).toBe('wikilink-resolved');
|
||||
});
|
||||
|
||||
test('path-qualified wikilink never attaches to a basename-only sibling', async () => {
|
||||
// Both notes/struktura and wiki/struktura exist. The author wrote
|
||||
// `[[notes/struktura]]` — the written path must exclude wiki/struktura
|
||||
// (a bare `[[struktura]]` would legitimately match both).
|
||||
await engine.putPage('notes/struktura', {
|
||||
type: 'concept' as any, title: 'Struktura Notes',
|
||||
compiled_truth: '', timeline: '',
|
||||
});
|
||||
await engine.putPage('wiki/struktura', {
|
||||
type: 'concept' as any, title: 'Struktura Wiki',
|
||||
compiled_truth: '', timeline: '',
|
||||
});
|
||||
await engine.putPage('concepts/x', {
|
||||
type: 'concept', title: 'X',
|
||||
compiled_truth: 'See [[notes/struktura]].', timeline: '',
|
||||
});
|
||||
await engine.setConfig('link_resolution.global_basename', 'true');
|
||||
|
||||
await runExtract(engine, ['links', '--source', 'db']);
|
||||
|
||||
const outLinks = await engine.getLinks('concepts/x');
|
||||
const basenameLinks = outLinks
|
||||
.filter(l => l.link_type === 'wikilink_basename')
|
||||
.map(l => l.to_slug);
|
||||
expect(basenameLinks).toEqual(['notes/struktura']);
|
||||
});
|
||||
|
||||
test('flag OFF → no basename edges via DB path (back-compat)', async () => {
|
||||
await engine.putPage('projects/struktura', {
|
||||
type: 'project', title: 'Struktura',
|
||||
|
||||
@@ -140,6 +140,35 @@ describeE2E('scanIntegrity batch parity (E2E, Postgres-only)', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('code pages', () => {
|
||||
test('type:code page is skipped on both paths (#1143)', async () => {
|
||||
const engine = getEngine();
|
||||
|
||||
await engine.putPage('people/alice', {
|
||||
type: 'person',
|
||||
title: 'Alice',
|
||||
compiled_truth: 'Alice tweeted about something.',
|
||||
timeline: '',
|
||||
frontmatter: {},
|
||||
});
|
||||
await engine.putPage('code/src/bot.py', {
|
||||
type: 'code',
|
||||
title: 'src/bot.py (python)',
|
||||
compiled_truth: '# the user tweeted about this feature\ndef send_tweet():\n pass',
|
||||
timeline: '',
|
||||
frontmatter: { language: 'python', file: 'src/bot.py' },
|
||||
});
|
||||
|
||||
const batchResult = await scanIntegrity(engine, { limit: 100, batchLoad: true });
|
||||
const seqResult = await scanIntegrity(engine, { limit: 100, batchLoad: false });
|
||||
|
||||
expect(batchResult.pagesScanned).toBe(seqResult.pagesScanned);
|
||||
expect(batchResult.pagesScanned).toBe(1);
|
||||
expect(batchResult.bareHits.map(h => h.slug)).not.toContain('code/src/bot.py');
|
||||
expect(seqResult.bareHits.map(h => h.slug)).not.toContain('code/src/bot.py');
|
||||
});
|
||||
});
|
||||
|
||||
describe('topPages', () => {
|
||||
test('topPages ordering matches between paths', async () => {
|
||||
const engine = getEngine();
|
||||
|
||||
@@ -39,6 +39,7 @@ import { runSkillOpt } from '../../src/core/skillopt/orchestrator.ts';
|
||||
import {
|
||||
bestPath,
|
||||
loadHistory,
|
||||
proposedPath,
|
||||
skillPath,
|
||||
} from '../../src/core/skillopt/version-store.ts';
|
||||
import { loadRejectedBuffer } from '../../src/core/skillopt/rejected-buffer.ts';
|
||||
@@ -741,7 +742,7 @@ describe('skillopt T3 — F11 held-out gate, ablation opts, no-DB-pollution', ()
|
||||
} finally { fixture.cleanup(); }
|
||||
});
|
||||
|
||||
test('--no-mutate writes proposed.md (best.md), leaves SKILL.md untouched', async () => {
|
||||
test('--no-mutate writes proposed.md and best.md, leaves SKILL.md untouched', async () => {
|
||||
const fixture = setupFixture(SKILL_PEOPLE_ONLY, CITATIONS_BENCHMARK);
|
||||
try {
|
||||
installStub({
|
||||
@@ -753,10 +754,9 @@ describe('skillopt T3 — F11 held-out gate, ablation opts, no-DB-pollution', ()
|
||||
const result = await runOnce(fixture, { noMutate: true });
|
||||
expect(result.outcome).toBe('accepted');
|
||||
expect(result.mutatedSkillFile).toBe(false);
|
||||
expect(result.proposedPath).toBeDefined();
|
||||
// proposed.md (best.md) exists and carries the improvement.
|
||||
expect(fs.existsSync(result.proposedPath!)).toBe(true);
|
||||
expect(result.proposedPath).toBe(proposedPath(fixture.skillsDir, SKILL));
|
||||
expect(fs.readFileSync(result.proposedPath!, 'utf8')).toContain('## Citations');
|
||||
expect(fs.readFileSync(bestPath(fixture.skillsDir, SKILL), 'utf8')).toContain('## Citations');
|
||||
// SKILL.md on disk is UNCHANGED (still People-only).
|
||||
const skill = fs.readFileSync(skillPath(fixture.skillsDir, SKILL), 'utf8');
|
||||
expect(skill).not.toContain('## Citations');
|
||||
|
||||
@@ -803,3 +803,107 @@ describe('embedAllStale --source threading (D7)', () => {
|
||||
expect((firstCallOpts as { sourceId?: string }).sourceId).toBe('media-corpus');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────
|
||||
// Code metadata preservation across re-embed (regression for #769)
|
||||
// ────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// gbrain v0.30.1 and earlier silently clobbered code-chunk metadata
|
||||
// (language, symbol_name, symbol_type, start_line, end_line,
|
||||
// parent_symbol_path, doc_comment, symbol_name_qualified) on every
|
||||
// re-embed pass. The chunker populated those columns at import time,
|
||||
// but embed.ts loaded chunks via getChunks then mapped them to a
|
||||
// stripped ChunkInput carrying only 5 fields. upsertChunks then
|
||||
// OVERWROTE (not COALESCEd) the metadata columns from EXCLUDED, so
|
||||
// re-embed wiped them to NULL. End result on a real brain: 4875 code
|
||||
// pages, 47866 chunks, all with NULL language/symbol_name/symbol_type;
|
||||
// code-def returned 0 hits across every indexed repo.
|
||||
//
|
||||
// All three runEmbed paths (--stale autopilot, --all, --slugs) must
|
||||
// thread metadata through the re-upsert. Tests below assert that the
|
||||
// engine.upsertChunks call carries the same metadata it loaded.
|
||||
|
||||
describe('runEmbed preserves code-chunk metadata across re-embed (regression for #769)', () => {
|
||||
const fullCodeChunk = {
|
||||
chunk_index: 0,
|
||||
chunk_text: '[Java] foo/Bar.java:10-20 method baz',
|
||||
chunk_source: 'compiled_truth' as const,
|
||||
embedded_at: null,
|
||||
token_count: 12,
|
||||
language: 'java',
|
||||
symbol_name: 'baz',
|
||||
symbol_type: 'function',
|
||||
start_line: 10,
|
||||
end_line: 20,
|
||||
parent_symbol_path: ['Bar'],
|
||||
doc_comment: 'does the thing',
|
||||
symbol_name_qualified: 'Bar.baz',
|
||||
};
|
||||
|
||||
function metadataOf(chunk: any) {
|
||||
return {
|
||||
language: chunk.language,
|
||||
symbol_name: chunk.symbol_name,
|
||||
symbol_type: chunk.symbol_type,
|
||||
start_line: chunk.start_line,
|
||||
end_line: chunk.end_line,
|
||||
parent_symbol_path: chunk.parent_symbol_path,
|
||||
doc_comment: chunk.doc_comment,
|
||||
symbol_name_qualified: chunk.symbol_name_qualified,
|
||||
};
|
||||
}
|
||||
|
||||
test('--stale (autopilot path) carries code metadata into upsertChunks', async () => {
|
||||
const stale = [{
|
||||
slug: 'code-page',
|
||||
chunk_index: 0,
|
||||
chunk_text: fullCodeChunk.chunk_text,
|
||||
chunk_source: 'compiled_truth',
|
||||
model: null,
|
||||
token_count: 12,
|
||||
}];
|
||||
let upsertChunkArgs: any[] | null = null;
|
||||
const engine = mockEngine({
|
||||
countStaleChunks: async () => 1,
|
||||
listStaleChunks: async () => stale,
|
||||
getChunks: async () => [fullCodeChunk],
|
||||
upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; },
|
||||
});
|
||||
|
||||
await runEmbed(engine, ['--stale']);
|
||||
|
||||
expect(upsertChunkArgs).not.toBeNull();
|
||||
expect(upsertChunkArgs!).toHaveLength(1);
|
||||
expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk));
|
||||
});
|
||||
|
||||
test('--all (full re-embed) carries code metadata into upsertChunks', async () => {
|
||||
let upsertChunkArgs: any[] | null = null;
|
||||
const engine = mockEngine({
|
||||
listPages: async () => [{ slug: 'code-page' }],
|
||||
getChunks: async () => [fullCodeChunk],
|
||||
upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; },
|
||||
});
|
||||
|
||||
await runEmbed(engine, ['--all']);
|
||||
|
||||
expect(upsertChunkArgs).not.toBeNull();
|
||||
expect(upsertChunkArgs!).toHaveLength(1);
|
||||
expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk));
|
||||
});
|
||||
|
||||
test('--slugs (per-page embed) carries code metadata into upsertChunks', async () => {
|
||||
let upsertChunkArgs: any[] | null = null;
|
||||
const engine = mockEngine({
|
||||
getPage: async () => ({ slug: 'code-page', compiled_truth: 'x', timeline: '' }),
|
||||
getChunks: async () => [fullCodeChunk],
|
||||
upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; },
|
||||
});
|
||||
|
||||
await runEmbed(engine, ['--slugs', 'code-page']);
|
||||
|
||||
expect(upsertChunkArgs).not.toBeNull();
|
||||
expect(upsertChunkArgs!).toHaveLength(1);
|
||||
expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk));
|
||||
});
|
||||
});
|
||||
|
||||
@@ -304,9 +304,7 @@ describe('runExtractFacts — empty-fence guard (Codex R2-#7)', () => {
|
||||
expect(r.legacyRowsPending).toBe(1);
|
||||
expect(r.factsInserted).toBe(0);
|
||||
expect(r.factsDeleted).toBe(0);
|
||||
// #1867: the remedy hint points at the re-runnable backfill command,
|
||||
// not the one-shot v0_32_2 migration (which the ledger never re-runs).
|
||||
expect(r.warnings.some(w => w.includes('gbrain facts fence-backfill'))).toBe(true);
|
||||
expect(r.warnings.some(w => w.includes('apply-migrations'))).toBe(true);
|
||||
|
||||
// Legacy row was NOT touched.
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
|
||||
@@ -1,133 +0,0 @@
|
||||
/**
|
||||
* #1867 — `gbrain facts fence-backfill` command tests.
|
||||
*
|
||||
* The command re-runs the (idempotent) v0_32_2 phase B on demand so
|
||||
* row_num-NULL backlogs — remote extract_facts deposits that predate
|
||||
* the fence-write backstop — can be cleared without re-running the
|
||||
* one-shot migration. Real PGLite + real tempdir filesystem.
|
||||
*/
|
||||
|
||||
import { describe, test, expect, beforeAll, afterAll, beforeEach } from 'bun:test';
|
||||
import { mkdtempSync, rmSync, existsSync, readFileSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
|
||||
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
import { runFactsCommand } from '../src/commands/facts.ts';
|
||||
import { phaseBFenceFacts } from '../src/commands/migrations/v0_32_2.ts';
|
||||
|
||||
let engine: PGLiteEngine;
|
||||
let brainDir: string;
|
||||
|
||||
beforeAll(async () => {
|
||||
engine = new PGLiteEngine();
|
||||
await engine.connect({});
|
||||
await engine.initSchema();
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
await engine.disconnect();
|
||||
try {
|
||||
if (brainDir) rmSync(brainDir, { recursive: true, force: true });
|
||||
} catch { /* best-effort */ }
|
||||
});
|
||||
|
||||
beforeEach(async () => {
|
||||
brainDir = mkdtempSync(join(tmpdir(), 'facts-backfill-cmd-test-'));
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
await (engine as any).db.query('DELETE FROM facts');
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
await (engine as any).db.query(
|
||||
`UPDATE sources SET local_path = $1 WHERE id = 'default'`,
|
||||
[brainDir],
|
||||
);
|
||||
});
|
||||
|
||||
async function seedLegacyFact(fact: string): Promise<void> {
|
||||
// The row_num-NULL shape a remote extract_facts deposit leaves behind
|
||||
// when it lands via the legacy DB-only path.
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
await (engine as any).db.query(
|
||||
`INSERT INTO facts (source_id, entity_slug, fact, kind, visibility, notability,
|
||||
valid_from, source, confidence)
|
||||
VALUES ('default', 'people/alice', $1, 'fact', 'private', 'medium',
|
||||
now(), 'mcp:extract_facts', 1.0)`,
|
||||
[fact],
|
||||
);
|
||||
}
|
||||
|
||||
describe('gbrain facts fence-backfill', () => {
|
||||
test('fences row_num-NULL rows and stamps the DB', async () => {
|
||||
await seedLegacyFact('Deposited remotely');
|
||||
|
||||
await runFactsCommand(engine, ['fence-backfill']);
|
||||
|
||||
// The fence exists on disk with the claim.
|
||||
const filePath = join(brainDir, 'people/alice.md');
|
||||
expect(existsSync(filePath)).toBe(true);
|
||||
expect(readFileSync(filePath, 'utf-8')).toContain('Deposited remotely');
|
||||
|
||||
// The backlog is cleared: no row_num-NULL rows remain.
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const rows = await (engine as any).db.query(
|
||||
'SELECT row_num, source_markdown_slug FROM facts',
|
||||
);
|
||||
expect(rows.rows).toHaveLength(1);
|
||||
expect(rows.rows[0].row_num).toBe(1);
|
||||
expect(rows.rows[0].source_markdown_slug).toBe('people/alice');
|
||||
});
|
||||
|
||||
test('re-run is a no-op (idempotent)', async () => {
|
||||
await seedLegacyFact('Deposited remotely');
|
||||
await runFactsCommand(engine, ['fence-backfill']);
|
||||
await runFactsCommand(engine, ['fence-backfill']);
|
||||
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const rows = await (engine as any).db.query('SELECT id FROM facts');
|
||||
expect(rows.rows).toHaveLength(1);
|
||||
const body = readFileSync(join(brainDir, 'people/alice.md'), 'utf-8');
|
||||
expect(body.match(/Deposited remotely/g)).toHaveLength(1);
|
||||
});
|
||||
|
||||
test('--dry-run reports without writing', async () => {
|
||||
await seedLegacyFact('Deposited remotely');
|
||||
await runFactsCommand(engine, ['fence-backfill', '--dry-run']);
|
||||
|
||||
expect(existsSync(join(brainDir, 'people/alice.md'))).toBe(false);
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const rows = await (engine as any).db.query(
|
||||
'SELECT row_num FROM facts',
|
||||
);
|
||||
expect(rows.rows[0].row_num).toBeNull();
|
||||
});
|
||||
|
||||
test('diverged page: appends past the DB row_num max instead of colliding (#2044 class)', async () => {
|
||||
// The #2044 divergence shape the backfill must survive: the DB already
|
||||
// holds stamped rows 1..3 for the slug (legacy wrong-path fence write),
|
||||
// but the fence at the resolved path is missing. Fence-max+1 (= 1) would
|
||||
// collide with the stamped rows on the post-rename UPDATE, failing the
|
||||
// page and leaving the renamed fence disagreeing with the DB.
|
||||
for (let n = 1; n <= 3; n++) {
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
await (engine as any).db.query(
|
||||
`INSERT INTO facts (source_id, entity_slug, fact, kind, visibility, notability,
|
||||
valid_from, source, confidence, row_num, source_markdown_slug)
|
||||
VALUES ('default', 'people/alice', $1, 'fact', 'private', 'medium',
|
||||
now(), 'mcp:extract_facts', 1.0, $2, 'people/alice')`,
|
||||
[`stamped ${n}`, n],
|
||||
);
|
||||
}
|
||||
await seedLegacyFact('Deposited remotely');
|
||||
|
||||
const result = await phaseBFenceFacts(engine, { dryRun: false });
|
||||
expect(result.status).toBe('complete');
|
||||
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const rows = await (engine as any).db.query(
|
||||
`SELECT row_num FROM facts WHERE fact = 'Deposited remotely'`,
|
||||
);
|
||||
expect(rows.rows[0].row_num).toBe(4);
|
||||
const body = readFileSync(join(brainDir, 'people/alice.md'), 'utf-8');
|
||||
expect(body).toContain('| 4 | Deposited remotely |');
|
||||
});
|
||||
});
|
||||
@@ -292,49 +292,6 @@ describe('lookupSourceLocalPath', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('writeFactsToFence — fence/DB divergence (#2044)', () => {
|
||||
test('seeds row_num past the DB max when the fence lags the DB', async () => {
|
||||
// Simulate the divergence class from #2044: DB rows were stamped with
|
||||
// row_nums against a fence written at a path this code no longer reads
|
||||
// (e.g. the pre-"Local patch 2026-06-11" wrong-path writes). The page
|
||||
// on disk has NO fence, but the DB already holds row_num 1..3 for the
|
||||
// slug. Pre-fix, the next deposit re-assigned row_num=1 from fence
|
||||
// text alone and the whole insertFacts batch failed on
|
||||
// idx_facts_fence_key.
|
||||
for (let n = 1; n <= 3; n++) {
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
await (engine as any).db.query(
|
||||
`INSERT INTO facts (source_id, entity_slug, fact, kind, visibility, notability,
|
||||
valid_from, source, confidence, row_num, source_markdown_slug)
|
||||
VALUES ('default', 'people/dana', $1, 'fact', 'private', 'medium',
|
||||
now(), 'mcp:extract_facts', 1.0, $2, 'people/dana')`,
|
||||
[`old claim ${n}`, n],
|
||||
);
|
||||
}
|
||||
|
||||
const result = await writeFactsToFence(
|
||||
engine,
|
||||
{ sourceId: 'default', localPath: brainDir, slug: 'people/dana' },
|
||||
[baseInput({ fact: 'second deposit' })],
|
||||
);
|
||||
|
||||
expect(result.fenceWriteFailed).toBeUndefined();
|
||||
expect(result.inserted).toBe(1);
|
||||
|
||||
// The new row landed PAST the DB max, not at fence-max+1 (= 1).
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const rows = await (engine as any).db.query(
|
||||
'SELECT row_num FROM facts WHERE id = $1',
|
||||
[result.ids[0]],
|
||||
);
|
||||
expect(rows.rows[0].row_num).toBe(4);
|
||||
|
||||
// And the on-disk fence carries the same row_num — fence and DB agree.
|
||||
const body = readFileSync(join(brainDir, 'people/dana.md'), 'utf-8');
|
||||
expect(body).toContain('| 4 | second deposit |');
|
||||
});
|
||||
});
|
||||
|
||||
// Cleanup any leftover tempdirs after the whole suite.
|
||||
afterAll(() => {
|
||||
// No-op: each test cleaned up via the beforeEach; this is a safety net.
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
* - Batch insert N rows persists row_num + source_markdown_slug
|
||||
* - Empty batch is a no-op
|
||||
* - Returns ids in input-order
|
||||
* - v51 partial UNIQUE collision skips only the colliding row (#2044)
|
||||
* - v51 partial UNIQUE index rolls back the whole batch on a collision
|
||||
* - deleteFactsForPage scopes by (source_id, source_markdown_slug);
|
||||
* never touches other pages or pre-v51 NULL-source_markdown_slug rows
|
||||
* - deleteFactsForPage on an empty page returns deleted:0 (idempotent)
|
||||
@@ -135,29 +135,30 @@ describe('engine.insertFacts — batch insert', () => {
|
||||
});
|
||||
});
|
||||
|
||||
test('v51 partial UNIQUE collision skips ONLY the colliding row (#2044)', async () => {
|
||||
test('v51 partial UNIQUE index rolls back the whole batch on collision', async () => {
|
||||
// Seed row #1 first.
|
||||
await engine.insertFacts([fixtureFact(1, { fact: 'seeded' })], { source_id: 'default' });
|
||||
|
||||
// Batch-insert rows that include a colliding row_num=1. Pre-#2044 this
|
||||
// threw and rolled back the whole batch, making a second remote
|
||||
// extract_facts deposit to an already-fenced page a hard failure. Now
|
||||
// ON CONFLICT DO NOTHING skips the colliding row and keeps the rest.
|
||||
const result = await engine.insertFacts(
|
||||
[
|
||||
fixtureFact(2, { fact: 'second' }),
|
||||
fixtureFact(1, { fact: 'collides' }), // row_num=1 on same (source_id, source_markdown_slug)
|
||||
fixtureFact(3, { fact: 'third' }),
|
||||
],
|
||||
{ source_id: 'default' },
|
||||
);
|
||||
expect(result.inserted).toBe(2);
|
||||
expect(result.ids).toHaveLength(2);
|
||||
// Now try to batch-insert rows that include a colliding row_num=1.
|
||||
let threw = false;
|
||||
try {
|
||||
await engine.insertFacts(
|
||||
[
|
||||
fixtureFact(2, { fact: 'second' }),
|
||||
fixtureFact(1, { fact: 'collides' }), // row_num=1 on same (source_id, source_markdown_slug)
|
||||
fixtureFact(3, { fact: 'third' }),
|
||||
],
|
||||
{ source_id: 'default' },
|
||||
);
|
||||
} catch {
|
||||
threw = true;
|
||||
}
|
||||
expect(threw).toBe(true);
|
||||
|
||||
// The seeded row survives untouched; the colliding claim is skipped.
|
||||
// Verify the transaction rolled back — only the seeded row should remain.
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const rows = await (engine as any).db.query('SELECT fact FROM facts ORDER BY id');
|
||||
expect(rows.rows.map((r: { fact: string }) => r.fact)).toEqual(['seeded', 'second', 'third']);
|
||||
expect(rows.rows.map((r: { fact: string }) => r.fact)).toEqual(['seeded']);
|
||||
});
|
||||
|
||||
test('different source_markdown_slug values DO NOT collide on the same row_num', async () => {
|
||||
|
||||
@@ -201,6 +201,15 @@ describe('scanIntegrity', () => {
|
||||
timeline: '',
|
||||
frontmatter: { validate: false },
|
||||
});
|
||||
// Indexed source file — 'tweeted about' in a comment must NOT be flagged
|
||||
// as a bare-tweet citation gap (#1143: auto-repair would corrupt source).
|
||||
await engine.putPage('code/src/bot.py', {
|
||||
type: 'code',
|
||||
title: 'src/bot.py (python)',
|
||||
compiled_truth: '# the user tweeted about this feature\ndef send_tweet():\n pass',
|
||||
timeline: '',
|
||||
frontmatter: { language: 'python', file: 'src/bot.py' },
|
||||
});
|
||||
}, 60_000);
|
||||
|
||||
afterAll(async () => {
|
||||
@@ -223,6 +232,13 @@ describe('scanIntegrity', () => {
|
||||
expect(slugs).not.toContain('people/legacy');
|
||||
});
|
||||
|
||||
test('skips type:code pages (#1143 — bare-tweet false positives on source files)', async () => {
|
||||
const res = await scanIntegrity(engine);
|
||||
const slugs = res.bareHits.map(h => h.slug);
|
||||
expect(slugs).not.toContain('code/src/bot.py');
|
||||
expect(res.pagesScanned).toBe(2);
|
||||
});
|
||||
|
||||
test('honors limit', async () => {
|
||||
const res = await scanIntegrity(engine, { limit: 1 });
|
||||
expect(res.pagesScanned).toBe(1);
|
||||
|
||||
@@ -403,6 +403,77 @@ describe('extractPageLinks', () => {
|
||||
expect(candidates).toEqual([]);
|
||||
});
|
||||
|
||||
test('path-qualified wikilink outside DIR_PATTERN queries by final segment', async () => {
|
||||
// `[[notes/struktura]]` (dir not in DIR_PATTERN) falls to the generic
|
||||
// pass. The resolver's basename index is keyed by final path segments,
|
||||
// so the lookup must strip the dirname — mirroring the FS path
|
||||
// (resolveSlugAll). Regression: the raw literal was passed through,
|
||||
// which never matched, so these links silently dropped.
|
||||
const seen: string[] = [];
|
||||
const resolver: SlugResolver = {
|
||||
resolve: async () => null,
|
||||
resolveBasenameMatches: async (name) => {
|
||||
seen.push(name);
|
||||
return name === 'struktura' ? ['notes/struktura'] : [];
|
||||
},
|
||||
};
|
||||
const { candidates } = await extractPageLinks(
|
||||
'concepts/x', 'See [[notes/struktura]].',
|
||||
{}, 'concept', resolver, { globalBasename: true },
|
||||
);
|
||||
expect(seen).toContain('struktura');
|
||||
expect(seen).not.toContain('notes/struktura');
|
||||
expect(candidates.map(c => c.targetSlug)).toEqual(['notes/struktura']);
|
||||
expect(candidates[0].linkType).toBe('wikilink_basename');
|
||||
expect(candidates[0].linkSource).toBe('wikilink-resolved');
|
||||
});
|
||||
|
||||
test('path-qualified wikilink keeps only matches ending with the written path', async () => {
|
||||
// The written path disambiguates: `[[notes/struktura]]` must never
|
||||
// attach to `wiki/struktura` even though both share the basename.
|
||||
const resolver: SlugResolver = {
|
||||
resolve: async () => null,
|
||||
resolveBasenameMatches: async (name) =>
|
||||
name === 'struktura' ? ['notes/struktura', 'wiki/struktura'] : [],
|
||||
};
|
||||
const { candidates } = await extractPageLinks(
|
||||
'concepts/x', 'See [[notes/struktura]].',
|
||||
{}, 'concept', resolver, { globalBasename: true },
|
||||
);
|
||||
expect(candidates.map(c => c.targetSlug)).toEqual(['notes/struktura']);
|
||||
});
|
||||
|
||||
test('path-qualified wikilink matches a deeper real slug by path suffix', async () => {
|
||||
// The page lives at vault/notes/struktura; the author wrote the shorter
|
||||
// tail `[[notes/struktura]]`. Suffix matching connects them, while the
|
||||
// basename-only sibling `wiki/struktura` stays excluded.
|
||||
const resolver: SlugResolver = {
|
||||
resolve: async () => null,
|
||||
resolveBasenameMatches: async (name) =>
|
||||
name === 'struktura' ? ['vault/notes/struktura', 'wiki/struktura'] : [],
|
||||
};
|
||||
const { candidates } = await extractPageLinks(
|
||||
'concepts/x', 'See [[notes/struktura]].',
|
||||
{}, 'concept', resolver, { globalBasename: true },
|
||||
);
|
||||
expect(candidates.map(c => c.targetSlug)).toEqual(['vault/notes/struktura']);
|
||||
});
|
||||
|
||||
test('path-qualified self-link is dropped like the bare form', async () => {
|
||||
// `[[notes/struktura]]` written on notes/struktura itself must not
|
||||
// produce a self-loop (same guard as the bare `[[own-tail]]` case).
|
||||
const resolver: SlugResolver = {
|
||||
resolve: async () => null,
|
||||
resolveBasenameMatches: async (name) =>
|
||||
name === 'struktura' ? ['notes/struktura'] : [],
|
||||
};
|
||||
const { candidates } = await extractPageLinks(
|
||||
'notes/struktura', 'See [[notes/struktura]].',
|
||||
{}, 'concept', resolver, { globalBasename: true },
|
||||
);
|
||||
expect(candidates).toEqual([]);
|
||||
});
|
||||
|
||||
test('bare wikilink resolution does not interfere with DIR_PATTERN wikilinks', async () => {
|
||||
// 2b refs (people/alice) take the verb-inferred type;
|
||||
// 2c refs (struktura) take wikilink_basename. Same call.
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import {
|
||||
extractCycleFreshnessSourceIds,
|
||||
parseMaintainArgs,
|
||||
} from '../src/commands/maintain.ts';
|
||||
import type { Check } from '../src/commands/doctor.ts';
|
||||
|
||||
describe('maintain args', () => {
|
||||
test('defaults to dry-run unless --safe is explicit', () => {
|
||||
expect(parseMaintainArgs([])).toMatchObject({
|
||||
safe: false,
|
||||
dryRun: true,
|
||||
json: false,
|
||||
});
|
||||
});
|
||||
|
||||
test('--safe enables mutating safe mode', () => {
|
||||
expect(parseMaintainArgs(['--safe', '--json'])).toMatchObject({
|
||||
safe: true,
|
||||
dryRun: false,
|
||||
json: true,
|
||||
});
|
||||
});
|
||||
|
||||
test('--dry-run wins over --safe', () => {
|
||||
expect(parseMaintainArgs(['--safe', '--dry-run'])).toMatchObject({
|
||||
safe: true,
|
||||
dryRun: true,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('cycle freshness source extraction', () => {
|
||||
test('extracts stale source ids from doctor messages', () => {
|
||||
const checks: Check[] = [
|
||||
{
|
||||
name: 'cycle_freshness',
|
||||
status: 'fail',
|
||||
message: "Source 'brain-sync-remote-teffur' last cycled 40h ago. Run `gbrain dream --source <id>`.",
|
||||
},
|
||||
{
|
||||
name: 'cycle_freshness',
|
||||
status: 'fail',
|
||||
message: "Source 'wiki' last cycled 25h ago. Source 'wiki' last cycled 25h ago.",
|
||||
},
|
||||
];
|
||||
|
||||
expect(extractCycleFreshnessSourceIds(checks)).toEqual([
|
||||
'brain-sync-remote-teffur',
|
||||
'wiki',
|
||||
]);
|
||||
});
|
||||
|
||||
test('ignores ok and unrelated checks', () => {
|
||||
const checks: Check[] = [
|
||||
{ name: 'cycle_freshness', status: 'ok', message: "Source 'fresh' last cycled recently." },
|
||||
{ name: 'frontmatter_integrity', status: 'warn', message: "Source 'wiki' has frontmatter issues." },
|
||||
];
|
||||
|
||||
expect(extractCycleFreshnessSourceIds(checks)).toEqual([]);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,17 @@
|
||||
import { describe, expect, it } from 'bun:test';
|
||||
import { readFileSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
|
||||
describe('root OpenClaw plugin manifest', () => {
|
||||
it('declares the id required by OpenClaw plugin installs', () => {
|
||||
const manifest = JSON.parse(readFileSync(join(import.meta.dir, '..', 'openclaw.plugin.json'), 'utf8'));
|
||||
const entrySource = readFileSync(join(import.meta.dir, '..', 'src', 'openclaw-context-engine.ts'), 'utf8');
|
||||
const entryId = entrySource.match(/id:\s*'([^']+)'/)?.[1];
|
||||
|
||||
expect(manifest.id).toBe(entryId);
|
||||
expect(manifest.configSchema).toBeDefined();
|
||||
expect(typeof manifest.configSchema).toBe('object');
|
||||
expect(manifest.contracts?.contextEngines).toContain('gbrain-context');
|
||||
expect(entrySource).toContain('export function register');
|
||||
});
|
||||
});
|
||||
@@ -186,11 +186,67 @@ describe('shouldExclude — orphan filter regression (preserve curation)', () =>
|
||||
expect(shouldExclude('entities/anonymous')).toBe(true);
|
||||
expect(shouldExclude('atoms/fact-123')).toBe(true);
|
||||
expect(shouldExclude('skills/gbrain-operations')).toBe(true);
|
||||
expect(shouldExclude('dreaming/light/2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('daily/2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('agent-openclaw/daily/2026-07-20')).toBe(true);
|
||||
});
|
||||
|
||||
test('workspace convention slugs are excluded', () => {
|
||||
expect(shouldExclude('_brain-conventions')).toBe(true);
|
||||
expect(shouldExclude('_templates/decision')).toBe(true);
|
||||
expect(shouldExclude('extracts/2026-06-30/takes.proposed/round-single')).toBe(true);
|
||||
expect(shouldExclude('2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('2026-07-20-qa-sweep')).toBe(true);
|
||||
expect(shouldExclude('agents/arya/identity')).toBe(true);
|
||||
expect(shouldExclude('agents/arya/memory/dreaming/deep/2026-07-20')).toBe(true);
|
||||
});
|
||||
|
||||
test('regular slugs are NOT excluded', () => {
|
||||
expect(shouldExclude('people/alice')).toBe(false);
|
||||
expect(shouldExclude('companies/acme')).toBe(false);
|
||||
expect(shouldExclude('writing/post-1')).toBe(false);
|
||||
expect(shouldExclude('agents/arya/qa-reports/launch-review')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('getHealth orphan_pages uses shared exclusion policy', () => {
|
||||
test('excluded convention islands do not count against health', async () => {
|
||||
await engine.putPage('_templates/decision', {
|
||||
type: 'template', title: 'Decision', compiled_truth: 'template', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('skills/arya/source-check', {
|
||||
type: 'concept', title: 'Skill', compiled_truth: 'skill', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('agents/arya/identity', {
|
||||
type: 'note', title: 'Identity', compiled_truth: 'identity', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('people/alice', {
|
||||
type: 'person', title: 'Alice', compiled_truth: 'real island', timeline: '', frontmatter: {},
|
||||
});
|
||||
|
||||
const health = await engine.getHealth();
|
||||
|
||||
expect(health.orphan_pages).toBe(1);
|
||||
});
|
||||
|
||||
test('per-brain config overrides (orphans.exclude_*) also apply to health', async () => {
|
||||
await engine.putPage('my-private-folder/secret-ref', {
|
||||
type: 'note', title: 'Ref', compiled_truth: 'ref', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('one-off-fixture-page', {
|
||||
type: 'note', title: 'Fixture', compiled_truth: 'fixture', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('people/alice', {
|
||||
type: 'person', title: 'Alice', compiled_truth: 'real island', timeline: '', frontmatter: {},
|
||||
});
|
||||
|
||||
expect((await engine.getHealth()).orphan_pages).toBe(3);
|
||||
|
||||
await engine.setConfig('orphans.exclude_prefixes', 'my-private-folder/');
|
||||
await engine.setConfig('orphans.exclude_slugs', 'one-off-fixture-page');
|
||||
expect((await engine.getHealth()).orphan_pages).toBe(1);
|
||||
|
||||
await engine.unsetConfig('orphans.exclude_prefixes');
|
||||
await engine.unsetConfig('orphans.exclude_slugs');
|
||||
});
|
||||
});
|
||||
|
||||
@@ -66,6 +66,10 @@ describe('shouldExclude', () => {
|
||||
expect(shouldExclude('templates/meeting-note')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes deny-prefix: _templates/', () => {
|
||||
expect(shouldExclude('_templates/meeting-note')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes deny-prefix: openclaw/config/', () => {
|
||||
expect(shouldExclude('openclaw/config/agent')).toBe(true);
|
||||
});
|
||||
@@ -86,10 +90,44 @@ describe('shouldExclude', () => {
|
||||
expect(shouldExclude('entities/product-hunt')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes first-segment: skills, dreaming, and daily', () => {
|
||||
expect(shouldExclude('skills/arya/source-check')).toBe(true);
|
||||
expect(shouldExclude('dreaming/light/2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('daily/2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('agent-openclaw/daily/2026-07-20')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes root date logs and agent workspace conventions', () => {
|
||||
expect(shouldExclude('_brain-conventions')).toBe(true);
|
||||
expect(shouldExclude('2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('2026-07-20-qa-sweep')).toBe(true);
|
||||
expect(shouldExclude('agents/arya/identity')).toBe(true);
|
||||
expect(shouldExclude('agents/arya/memory/dreaming/deep/2026-07-20')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes generated extracts', () => {
|
||||
expect(shouldExclude('extracts/2026-06-30/takes.proposed/round-single')).toBe(true);
|
||||
});
|
||||
|
||||
test('brain-specific exclusions come from config overrides, not global defaults', () => {
|
||||
// No baked-in defaults for these:
|
||||
expect(shouldExclude('my-private-folder/some-secret-ref.md')).toBe(false);
|
||||
expect(shouldExclude('one-off-fixture-page')).toBe(false);
|
||||
// The per-brain config plane (orphans.exclude_prefixes / exclude_slugs):
|
||||
const overrides = {
|
||||
excludePrefixes: ['my-private-folder/'],
|
||||
excludeSlugs: ['one-off-fixture-page'],
|
||||
};
|
||||
expect(shouldExclude('my-private-folder/some-secret-ref.md', overrides)).toBe(true);
|
||||
expect(shouldExclude('one-off-fixture-page', overrides)).toBe(true);
|
||||
expect(shouldExclude('people/jane-doe', overrides)).toBe(false);
|
||||
});
|
||||
|
||||
test('does NOT exclude a normal content page', () => {
|
||||
expect(shouldExclude('companies/acme')).toBe(false);
|
||||
expect(shouldExclude('people/jane-doe')).toBe(false);
|
||||
expect(shouldExclude('projects/gbrain')).toBe(false);
|
||||
expect(shouldExclude('agents/arya/qa-reports/launch-review')).toBe(false);
|
||||
});
|
||||
|
||||
test('does NOT exclude a page ending with log-like text that is not /log', () => {
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
/**
|
||||
* #1098 — ClawVisor health checks must hit /ready (surfaces db/vault
|
||||
* degradation), never /health (bare liveness, masks those failure modes).
|
||||
* #1099 — the credential-gateway recipe tells users to activate Google
|
||||
* Contacts; the canonical consumer recipe must actually ship.
|
||||
*/
|
||||
|
||||
import { describe, test, expect } from 'bun:test';
|
||||
import { readFileSync, existsSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
|
||||
const RECIPES = join(import.meta.dir, '../recipes');
|
||||
|
||||
describe('#1098 — ClawVisor health checks use /ready', () => {
|
||||
for (const f of ['email-to-brain.md', 'calendar-to-brain.md', 'credential-gateway.md', 'contacts-to-brain.md']) {
|
||||
test(`${f} references $CLAWVISOR_URL/ready, never /health`, () => {
|
||||
const text = readFileSync(join(RECIPES, f), 'utf-8');
|
||||
expect(text).not.toContain('$CLAWVISOR_URL/health');
|
||||
expect(text).toContain('$CLAWVISOR_URL/ready');
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
describe('#1099 — canonical contacts-to-brain recipe ships', () => {
|
||||
test('recipes/contacts-to-brain.md exists with the sibling frontmatter shape', () => {
|
||||
const p = join(RECIPES, 'contacts-to-brain.md');
|
||||
expect(existsSync(p)).toBe(true);
|
||||
const text = readFileSync(p, 'utf-8');
|
||||
expect(text).toContain('id: contacts-to-brain');
|
||||
expect(text).toContain('requires: [credential-gateway]');
|
||||
expect(text).toContain('health_checks:');
|
||||
expect(text).toContain('contacts.readonly');
|
||||
});
|
||||
});
|
||||
@@ -12,9 +12,11 @@ import {
|
||||
bestPath,
|
||||
historyPath,
|
||||
loadHistory,
|
||||
proposedPath,
|
||||
revertAllPending,
|
||||
skillPath,
|
||||
versionsDir,
|
||||
writeProposed,
|
||||
} from '../../src/core/skillopt/version-store.ts';
|
||||
|
||||
let tmpDir: string;
|
||||
@@ -79,6 +81,19 @@ describe('acceptCandidate (D8 two-phase commit)', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('writeProposed', () => {
|
||||
test('writes distinct best and proposed artifacts without mutating SKILL.md (#2635)', () => {
|
||||
const candidate = '---\nname: test\n---\nproposed body\n';
|
||||
|
||||
const written = writeProposed(tmpDir, SKILL, candidate);
|
||||
|
||||
expect(written).toBe(proposedPath(tmpDir, SKILL));
|
||||
expect(fs.readFileSync(bestPath(tmpDir, SKILL), 'utf8')).toBe(candidate);
|
||||
expect(fs.readFileSync(proposedPath(tmpDir, SKILL), 'utf8')).toBe(candidate);
|
||||
expect(fs.readFileSync(skillPath(tmpDir, SKILL), 'utf8')).toContain('baseline body');
|
||||
});
|
||||
});
|
||||
|
||||
describe('revertAllPending (D8 crash recovery)', () => {
|
||||
test('no-op when no pending rows', () => {
|
||||
const reverted = revertAllPending(tmpDir, SKILL);
|
||||
|
||||
Reference in New Issue
Block a user