mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-15 01:12:20 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7bee2e48f3 | ||
|
|
53bb974eaf | ||
|
|
bdd23cdede | ||
|
|
45689dd1bd |
@@ -61,10 +61,7 @@ jobs:
|
||||
- name: Run JSONB double-encode parity tests on real Postgres
|
||||
env:
|
||||
DATABASE_URL: postgresql://postgres:postgres@localhost:5432/gbrain_test
|
||||
# --timeout also raises bun's 5s default hook budget (beforeAll/afterAll
|
||||
# do NOT inherit a test's third-arg timeout; verified on bun 1.3.x).
|
||||
# Every runner script in scripts/ passes it; bare invocations must too.
|
||||
run: bun test --timeout=60000 test/e2e/op-checkpoint-jsonb-parity.test.ts test/e2e/jsonb-roundtrip.test.ts
|
||||
run: bun test test/e2e/op-checkpoint-jsonb-parity.test.ts test/e2e/jsonb-roundtrip.test.ts
|
||||
|
||||
tier1:
|
||||
name: Tier 1 (Mechanical)
|
||||
@@ -91,7 +88,7 @@ jobs:
|
||||
bun-version: 1.3.13
|
||||
- run: bun install
|
||||
- name: Run Tier 1 E2E tests
|
||||
run: bun test --timeout=60000 test/e2e/mechanical.test.ts test/e2e/mcp.test.ts
|
||||
run: bun test test/e2e/mechanical.test.ts test/e2e/mcp.test.ts
|
||||
env:
|
||||
DATABASE_URL: postgresql://postgres:postgres@localhost:5432/gbrain_test
|
||||
|
||||
@@ -158,7 +155,7 @@ jobs:
|
||||
}
|
||||
EOF
|
||||
- name: Run Tier 2 skill tests
|
||||
run: bun test --timeout=60000 test/e2e/skills.test.ts test/e2e/zeroentropy-live.test.ts
|
||||
run: bun test test/e2e/skills.test.ts test/e2e/zeroentropy-live.test.ts
|
||||
env:
|
||||
DATABASE_URL: postgresql://postgres:postgres@localhost:5432/gbrain_test
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
|
||||
@@ -29,9 +29,7 @@ jobs:
|
||||
with:
|
||||
bun-version: 1.3.13
|
||||
- run: bun install
|
||||
# --timeout matches every scripts/ runner and covers hook budgets too
|
||||
# (bunfig.toml's timeout key is ignored by bun; hooks default to 5s).
|
||||
- run: bun test --timeout=60000
|
||||
- run: bun test
|
||||
- run: bun run verify
|
||||
- run: bun build --compile --target=${{ matrix.target }} --outfile bin/${{ matrix.artifact }} src/cli.ts
|
||||
- name: Attest build provenance
|
||||
|
||||
@@ -113,11 +113,6 @@ jobs:
|
||||
key: bun-cache-${{ runner.os }}-${{ hashFiles('bun.lock') }}
|
||||
- run: bun install
|
||||
- run: bun run verify
|
||||
# Guard: no bare `bun test` in workflows/scripts — bun ignores
|
||||
# bunfig.toml's timeout, and hooks (beforeAll/afterAll) get the 5s
|
||||
# default regardless of per-test third-arg timeouts. Runs directly
|
||||
# (not via verify's CHECKS array) to avoid a package.json edit.
|
||||
- run: bash scripts/check-bun-test-timeout.sh
|
||||
|
||||
serial-tests:
|
||||
# *.serial.test.ts at --max-concurrency=1. Lives in its own runner so
|
||||
|
||||
@@ -67,19 +67,6 @@ Per-file detail is in `docs/architecture/KEY_FILES.md`.
|
||||
text, the cast parses it). Guarded by `scripts/check-jsonb-pattern.sh` (template grep) +
|
||||
`scripts/check-jsonb-params.mjs` (positional AST scanner); the real backstop is the DATABASE_URL-gated
|
||||
e2e parity tests, since PGLite can't surface the bug. Full rule in `docs/ENGINES.md`.
|
||||
- **Engine-live paths avoid runtime dynamic `import()` for helper dependencies.** In
|
||||
`src/core/pglite-engine.ts`, `src/core/postgres-engine.ts`, and
|
||||
`src/core/migrate.ts`, dependencies previously reached through runtime dynamic
|
||||
imports use static top-level imports. The only current dynamic-`import()` exceptions
|
||||
are the four `ai/gateway.ts` lookups in both engines'
|
||||
`initSchema()` and `_upsertChunksOnce()` methods; each remains lazy inside a
|
||||
local `try/catch` because the gateway has a large provider/config closure and,
|
||||
more importantly, eager evaluation would occur before the catch and could
|
||||
turn a recoverable default/config-row fallback into a module-load failure.
|
||||
Every exception carries `engine-dynamic-import-ok` on the import line.
|
||||
`scripts/check-engine-dynamic-import.sh` enforces the rule. For history, use
|
||||
`git log -G'await[[:space:]]+import\\('`, not `git log -S`: a dynamic-to-static
|
||||
rewrite can preserve the searched token while changing its context.
|
||||
- **Engine parity.** `src/core/postgres-engine.ts` and `src/core/pglite-engine.ts` move in
|
||||
lockstep — a new method/SQL shape lands in BOTH, pinned by `test/e2e/engine-parity.test.ts`.
|
||||
Forward-referenced columns/indexes go in the bootstrap probe set (guarded by
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -229,7 +229,7 @@ add `GBRAIN_AUDIT_FULL=1` (v0.43+ TODO; not yet wired).
|
||||
- Per-source pack-upgrade (the handler accepts `sourceId` but
|
||||
`findPackSuccessors` doesn't yet pass it through)
|
||||
- Cross-brain federated mounts that disagree on canonical packs
|
||||
- Automatic rollback (today: manual SQL or `gbrain pages restore`)
|
||||
- Automatic rollback (today: manual SQL or `gbrain restore`)
|
||||
- LLM-assisted mapping_rules codegen from production data (`gbrain
|
||||
schema detect-mappings`; deferred to v0.43+)
|
||||
|
||||
|
||||
@@ -214,7 +214,7 @@ gbrain schema downgrade
|
||||
|
||||
1. `git revert <merge-commit>` — restores the code.
|
||||
2. `gbrain schema downgrade --to gbrain-base` — restores config.
|
||||
3. (Optional) `gbrain pages purge-deleted --older-than 0h` — drops
|
||||
3. (Optional) `gbrain purge-deleted --older-than 0h` — drops
|
||||
v0.39-typed pages that no longer have a matching type in the active
|
||||
pack.
|
||||
|
||||
|
||||
@@ -19,11 +19,13 @@ entire DB from scratch.
|
||||
|
||||
This means:
|
||||
|
||||
- **Disaster recovery is one command.** If your DB volume corrupts, if
|
||||
Postgres eats itself, if PGLite's WASM lock wedges — you don't need
|
||||
a backup. You wipe the DB, re-import from your brain repo, and the
|
||||
derived state regenerates. v0.32.3 ships `gbrain rebuild
|
||||
--confirm-destructive` as the documented one-liner.
|
||||
- **Disaster recovery is a short, boring sequence.** If your DB volume
|
||||
corrupts, if Postgres eats itself, if PGLite's WASM lock wedges — you
|
||||
don't need a backup. You wipe the derived tables (on PGLite,
|
||||
`gbrain reinit-pglite` wipes the whole embedded DB), re-import from
|
||||
your brain repo with `gbrain sync`, and `gbrain extract all`
|
||||
regenerates the derived state. See "Disaster recovery" below for the
|
||||
exact commands.
|
||||
- **Multi-machine sync is git.** Your brain is a repo. Push from one
|
||||
machine, pull from another, and the second machine's DB rebuilds on
|
||||
its next sync. No "back up the database" step.
|
||||
@@ -146,11 +148,9 @@ The promise the rule makes:
|
||||
# Snapshot what's there
|
||||
gbrain stats > /tmp/before.txt
|
||||
|
||||
# Wipe and rebuild
|
||||
gbrain rebuild --confirm-destructive # v0.32.3 — deletes derived tables
|
||||
# (pages + content_chunks survive
|
||||
# the CASCADE-safe design)
|
||||
# OR manually for v0.32.2:
|
||||
# Wipe and rebuild — delete the derived tables (pages + content_chunks
|
||||
# survive the CASCADE-safe design), then re-derive from the repo.
|
||||
# On PGLite, `gbrain reinit-pglite` wipes the whole embedded DB instead.
|
||||
psql -c 'DELETE FROM facts; DELETE FROM takes; DELETE FROM links; DELETE FROM timeline_entries;'
|
||||
gbrain sync
|
||||
gbrain extract all
|
||||
|
||||
@@ -108,8 +108,8 @@ Every primitive ships with a documented rollback:
|
||||
| Operation | Rollback |
|
||||
|-----------|----------|
|
||||
| Retype | `frontmatter.legacy_type = <original>` preserved on every page (D8). One SQL UPDATE restores types: `UPDATE pages SET type = frontmatter->>'legacy_type' WHERE frontmatter ? 'legacy_type'`. |
|
||||
| Page-to-link | Source page soft-deleted with 72h TTL. `gbrain pages restore <slug>` within 72h. Link row stays harmless if source restored. |
|
||||
| Page-to-alias | Source page soft-deleted with 72h TTL. `gbrain pages restore <slug>` within 72h. Alias row stays harmless (or `DELETE FROM slug_aliases WHERE alias_slug = <slug>` to clean up). |
|
||||
| Page-to-link | Source page soft-deleted with 72h TTL. `gbrain restore <slug>` within 72h. Link row stays harmless if source restored. |
|
||||
| Page-to-alias | Source page soft-deleted with 72h TTL. `gbrain restore <slug>` within 72h. Alias row stays harmless (or `DELETE FROM slug_aliases WHERE alias_slug = <slug>` to clean up). |
|
||||
| Active-pack flip | `gbrain schema use gbrain-base` reverses the flip. |
|
||||
|
||||
## What if my brain doesn't fit?
|
||||
|
||||
@@ -183,6 +183,6 @@ This also means the best AI agent setups will be open source by default. Closed,
|
||||
|
||||
Software distribution reimagined: the package is a markdown file, the runtime is a sufficiently smart model, the package manager is your AI agent, and the app store is a git repo.
|
||||
|
||||
`gbrain install voice-agent`
|
||||
`gbrain skillpack scaffold voice-agent`
|
||||
|
||||
That's it.
|
||||
|
||||
@@ -69,7 +69,7 @@ update_brain_page(slug, new_info, source):
|
||||
page = gbrain get {slug}
|
||||
|
||||
// TIMELINE: always APPEND (never edit existing entries)
|
||||
gbrain add_timeline_entry {slug} {
|
||||
gbrain timeline-add {slug} {
|
||||
date: today,
|
||||
summary: new_info.summary,
|
||||
detail: new_info.detail,
|
||||
|
||||
@@ -46,10 +46,10 @@ on user_shares_media(url_or_file):
|
||||
|
||||
# Step 4: Extract and cross-reference entities
|
||||
for person in transcript.mentioned_people:
|
||||
gbrain add_link <slug> <person_slug>
|
||||
gbrain add_link <person_slug> <slug>
|
||||
gbrain add_timeline_entry <person_slug> \
|
||||
--entry "Discussed in {video_title}: {what_was_said}" \
|
||||
gbrain link <slug> <person_slug>
|
||||
gbrain link <person_slug> <slug>
|
||||
gbrain timeline-add <person_slug> {date} \
|
||||
"Discussed in {video_title}: {what_was_said}" \
|
||||
--source "YouTube: {url}"
|
||||
|
||||
# PATTERN 2: Social Media Bundles
|
||||
@@ -80,8 +80,8 @@ on user_shares_media(url_or_file):
|
||||
|
||||
# Extract entities and cross-reference
|
||||
for entity in bundle.mentioned_entities:
|
||||
gbrain add_link <slug> <entity_slug>
|
||||
gbrain add_link <entity_slug> <slug>
|
||||
gbrain link <slug> <entity_slug>
|
||||
gbrain link <entity_slug> <slug>
|
||||
|
||||
# PATTERN 3: PDFs and Documents
|
||||
elif media.type == "pdf" or media.type == "document":
|
||||
@@ -109,8 +109,8 @@ on user_shares_media(url_or_file):
|
||||
"""
|
||||
|
||||
for entity in document.mentioned_entities:
|
||||
gbrain add_link <slug> <entity_slug>
|
||||
gbrain add_link <entity_slug> <slug>
|
||||
gbrain link <slug> <entity_slug>
|
||||
gbrain link <entity_slug> <slug>
|
||||
|
||||
# Always sync after ingestion
|
||||
gbrain sync
|
||||
@@ -127,7 +127,7 @@ on user_shares_media(url_or_file):
|
||||
## How to Verify
|
||||
|
||||
1. Ingest a YouTube video. Run `gbrain get media/youtube/{slug}`. Confirm the page has: the agent's analysis (not just a summary), key quotes with speaker attribution, and the full diarized transcript.
|
||||
2. Run `gbrain get_links media/youtube/{slug}`. Confirm back-links exist to brain pages for every person and company mentioned in the video.
|
||||
2. Run `gbrain call get_links '{"slug": "media/youtube/{slug}"}'`. Confirm back-links exist to brain pages for every person and company mentioned in the video.
|
||||
3. Pick a person mentioned in the video. Run `gbrain get <person_slug>`. Confirm their timeline has a new entry referencing the video with specific context.
|
||||
4. Ingest a tweet. Confirm the brain page includes the thread context, linked article summaries, and entity cross-references -- not just the tweet text.
|
||||
5. Run `gbrain search "{topic_from_video}"`. Confirm the media page appears in search results (verifies the content is indexed and searchable).
|
||||
|
||||
@@ -49,23 +49,23 @@ on enrich(entity, trigger):
|
||||
data["contacts"] = google_contacts(entity.email) # Contact data
|
||||
|
||||
# Step 5: Store raw data (auditable, re-processable)
|
||||
gbrain put_raw_data <entity_slug> \
|
||||
--data '{"sources": {"crustdata": {"fetched_at": "...", "data": {...}}, ...}}'
|
||||
gbrain call put_raw_data \
|
||||
'{"slug": "<entity_slug>", "data": {"sources": {"crustdata": {"fetched_at": "...", "data": {...}}, ...}}}'
|
||||
# Overwrite on re-enrichment, don't append
|
||||
|
||||
# Step 6: Write to brain page
|
||||
if path == "CREATE":
|
||||
gbrain put <entity_slug> --content "<compiled_truth_from_all_sources>"
|
||||
gbrain add_timeline_entry <entity_slug> --entry "Page created via enrichment"
|
||||
gbrain timeline-add <entity_slug> {date} "Page created via enrichment"
|
||||
elif path == "UPDATE":
|
||||
# Append timeline, update compiled truth ONLY if materially new
|
||||
gbrain add_timeline_entry <entity_slug> --entry "Enriched: {new_signal}"
|
||||
gbrain timeline-add <entity_slug> {date} "Enriched: {new_signal}"
|
||||
# Flag contradictions -- don't silently resolve them
|
||||
|
||||
# Step 7: Cross-reference the graph
|
||||
gbrain add_link <person_slug> <company_slug> # person -> company
|
||||
gbrain add_link <company_slug> <person_slug> # company -> person
|
||||
gbrain add_link <person_slug> <deal_slug> # person -> deal
|
||||
gbrain link <person_slug> <company_slug> # person -> company
|
||||
gbrain link <company_slug> <person_slug> # company -> person
|
||||
gbrain link <person_slug> <deal_slug> # person -> deal
|
||||
# Every entity page links to every other entity page that references it
|
||||
|
||||
# People page sections (not a LinkedIn profile -- a living portrait):
|
||||
@@ -94,8 +94,8 @@ on enrich(entity, trigger):
|
||||
## How to Verify
|
||||
|
||||
1. Enrich a Tier 1 person. Run `gbrain get <slug>` and confirm the page has Executive Summary, State, What They Believe, Contact, and Timeline sections populated from multiple sources.
|
||||
2. Run `gbrain get_raw_data <slug>`. Confirm raw API responses are stored with `sources.{provider}.fetched_at` timestamps.
|
||||
3. Run `gbrain get_links <slug>`. Confirm cross-reference links exist to the person's company page, deal pages, and related entities.
|
||||
2. Run `gbrain call get_raw_data '{"slug": "<slug>"}'`. Confirm raw API responses are stored with `sources.{provider}.fetched_at` timestamps.
|
||||
3. Run `gbrain call get_links '{"slug": "<slug>"}'`. Confirm cross-reference links exist to the person's company page, deal pages, and related entities.
|
||||
4. Check a page that was enriched AND has a user-written Assessment. Confirm the Assessment section was preserved, not overwritten by API data.
|
||||
5. Try to re-enrich the same person. Confirm the system checks the `fetched_at` timestamp and skips if less than a week old.
|
||||
|
||||
|
||||
@@ -53,7 +53,7 @@ on upcoming_meeting(meeting):
|
||||
"last_interaction": page.timeline[0], # most recent
|
||||
"open_threads": page.open_threads,
|
||||
"relationship_temperature": page.relationship,
|
||||
"relevant_deals": gbrain get_links <attendee_slug>,
|
||||
"relevant_deals": gbrain call get_links '{"slug": "<attendee_slug>"}',
|
||||
}
|
||||
else:
|
||||
briefing[attendee] = "No brain page -- consider enriching"
|
||||
@@ -67,14 +67,14 @@ on inbox_cleared():
|
||||
for email in processed_emails:
|
||||
if email.contained_new_information:
|
||||
# Update the sender's brain page with new signal
|
||||
gbrain add_timeline_entry <sender_slug> \
|
||||
--entry "Email re: {subject}. Key info: {extracted_signal}" \
|
||||
gbrain timeline-add <sender_slug> {date} \
|
||||
"Email re: {subject}. Key info: {extracted_signal}" \
|
||||
--source "email from {sender} re {subject}, {date}"
|
||||
|
||||
# Update any mentioned entity pages too
|
||||
for entity in email.mentioned_entities:
|
||||
gbrain add_timeline_entry <entity_slug> \
|
||||
--entry "{what_was_said_about_them}" \
|
||||
gbrain timeline-add <entity_slug> {date} \
|
||||
"{what_was_said_about_them}" \
|
||||
--source "email from {sender}, {date}"
|
||||
|
||||
# WORKFLOW 4: Scheduling Nudges
|
||||
|
||||
@@ -32,15 +32,15 @@ on new_meeting_transcript(meeting):
|
||||
|
||||
# Step 3: Propagate to ALL entity pages (MANDATORY -- most agents skip this)
|
||||
for person in meeting.attendees + meeting.mentioned_people:
|
||||
gbrain add_timeline_entry <person_slug> \
|
||||
--entry "Met in '{meeting.title}' on {date}. Key points: ..." \
|
||||
gbrain timeline-add <person_slug> {date} \
|
||||
"Met in '{meeting.title}' on {date}. Key points: ..." \
|
||||
--source "Meeting notes '{meeting.title}', {date}"
|
||||
# Update their State section if new information surfaced
|
||||
# Update company pages for each person's company if relevant
|
||||
|
||||
for company in meeting.mentioned_companies:
|
||||
gbrain add_timeline_entry <company_slug> \
|
||||
--entry "Discussed in '{meeting.title}': {what_was_said}" \
|
||||
gbrain timeline-add <company_slug> {date} \
|
||||
"Discussed in '{meeting.title}': {what_was_said}" \
|
||||
--source "Meeting notes '{meeting.title}', {date}"
|
||||
|
||||
# Step 4: Extract action items
|
||||
@@ -49,8 +49,8 @@ on new_meeting_transcript(meeting):
|
||||
|
||||
# Step 5: Back-link everything (bidirectional graph)
|
||||
for entity in all_entities_mentioned:
|
||||
gbrain add_link <slug> <entity_slug> # meeting -> entity
|
||||
gbrain add_link <entity_slug> <slug> # entity -> meeting
|
||||
gbrain link <slug> <entity_slug> # meeting -> entity
|
||||
gbrain link <entity_slug> <slug> # entity -> meeting
|
||||
|
||||
# Step 6: Sync so new pages are immediately searchable
|
||||
gbrain sync
|
||||
@@ -73,7 +73,7 @@ on new_meeting_transcript(meeting):
|
||||
1. After ingesting a meeting, run `gbrain get meetings/{date}-{slug}`. Confirm the page has the agent's analysis above the bar and the full diarized transcript below it.
|
||||
2. For each attendee, run `gbrain get <attendee_slug>`. Check that their timeline has a new entry referencing the meeting with specific insights (not just "attended meeting").
|
||||
3. Pick a company mentioned in the meeting. Run `gbrain get <company_slug>`. Confirm a timeline entry exists referencing what was discussed about the company.
|
||||
4. Run `gbrain get_links meetings/{date}-{slug}`. Verify back-links exist to all attendee and entity pages.
|
||||
4. Run `gbrain call get_links '{"slug": "meetings/{date}-{slug}"}'`. Verify back-links exist to all attendee and entity pages.
|
||||
5. Run `gbrain search "{meeting_topic}"`. Confirm the meeting page appears in search results (verifies sync ran).
|
||||
|
||||
---
|
||||
|
||||
@@ -91,7 +91,7 @@ first):
|
||||
6. The seeded `default` source.
|
||||
|
||||
So inside `~/.gstack/plans/` on a brain that pinned `gstack` to
|
||||
`~/.gstack` via `.gbrain-source`, `gbrain put-page` implicitly writes to
|
||||
`~/.gstack` via `.gbrain-source`, `gbrain put` implicitly writes to
|
||||
the `gstack` source. Outside any registered directory with no env/dotfile
|
||||
set, it writes to the default.
|
||||
|
||||
@@ -188,10 +188,10 @@ citations keep working.
|
||||
|
||||
```bash
|
||||
# Pass --source explicitly
|
||||
gbrain put-page topics/ai ... --source wiki
|
||||
gbrain put topics/ai ... --source wiki
|
||||
|
||||
# Or rely on the dotfile / env / CWD match
|
||||
cd ~/.gstack && gbrain put-page plans/multi-repo ...
|
||||
cd ~/.gstack && gbrain put plans/multi-repo ...
|
||||
# → source auto-resolves to gstack
|
||||
```
|
||||
|
||||
|
||||
@@ -20,8 +20,8 @@ on every_inbound_message(message):
|
||||
for entity in entities:
|
||||
existing = gbrain search "{entity.name}"
|
||||
if existing:
|
||||
gbrain add_timeline_entry <entity_slug> \
|
||||
--entry "{what_was_said}" \
|
||||
gbrain timeline-add <entity_slug> {date} \
|
||||
"{what_was_said}" \
|
||||
--source "User, direct message, {timestamp}"
|
||||
# else: flag for enrichment if important enough
|
||||
|
||||
@@ -64,13 +64,13 @@ on nightly_schedule("02:00"):
|
||||
# The brain COMPOUNDS overnight.
|
||||
|
||||
# 5a: Entity sweep -- find unlinked mentions
|
||||
pages = gbrain list_pages
|
||||
pages = gbrain list
|
||||
for page in pages:
|
||||
mentions = extract_entity_mentions(page.content)
|
||||
existing_links = gbrain get_links <page.slug>
|
||||
existing_links = gbrain call get_links '{"slug": "<page.slug>"}'
|
||||
for mention in mentions:
|
||||
if mention not in existing_links:
|
||||
gbrain add_link <page.slug> <mention_slug> # fix broken graph
|
||||
gbrain link <page.slug> <mention_slug> # fix broken graph
|
||||
|
||||
# 5b: Citation audit -- find facts without sources
|
||||
for page in pages:
|
||||
@@ -80,7 +80,7 @@ on nightly_schedule("02:00"):
|
||||
|
||||
# 5c: Memory consolidation -- update compiled truth from timeline
|
||||
for page in stale_pages(older_than="7d"):
|
||||
timeline = gbrain get_timeline <page.slug>
|
||||
timeline = gbrain timeline <page.slug>
|
||||
if timeline.has_new_entries_since_last_consolidation:
|
||||
# Re-synthesize compiled truth from accumulated timeline
|
||||
updated_truth = consolidate(page.compiled_truth, timeline.new_entries)
|
||||
@@ -110,11 +110,11 @@ on nightly_schedule("02:00"):
|
||||
|
||||
## How to Verify
|
||||
|
||||
1. Send a message mentioning a person with a brain page. Confirm the agent detects the entity and adds a timeline entry to their page (`gbrain get_timeline <slug>`).
|
||||
1. Send a message mentioning a person with a brain page. Confirm the agent detects the entity and adds a timeline entry to their page (`gbrain timeline <slug>`).
|
||||
2. Ask the agent about someone in the brain. Confirm it runs `gbrain search` or `gbrain get` BEFORE reaching for external APIs (check the tool call order).
|
||||
3. Write a new page with `gbrain put`, then immediately run `gbrain search` for it. Confirm it appears in results (verifies sync ran).
|
||||
4. Run `gbrain doctor`. Confirm it returns a health report with database status, page count, and any flagged issues.
|
||||
5. After a dream cycle runs, check a page that had unlinked entity mentions. Confirm new links were added (`gbrain get_links <slug>`).
|
||||
5. After a dream cycle runs, check a page that had unlinked entity mentions. Confirm new links were added (`gbrain call get_links '{"slug": "<slug>"}'`).
|
||||
|
||||
---
|
||||
*Part of the [GBrain Skillpack](../GBRAIN_SKILLPACK.md).*
|
||||
|
||||
@@ -47,8 +47,8 @@ on user_message(message):
|
||||
|
||||
# Step 3: Cross-link to everything that shaped the thinking
|
||||
for entity in idea.influences:
|
||||
gbrain add_link originals/{slug} <entity_slug>
|
||||
gbrain add_link <entity_slug> originals/{slug}
|
||||
gbrain link originals/{slug} <entity_slug>
|
||||
gbrain link <entity_slug> originals/{slug}
|
||||
|
||||
# Step 4: Sync
|
||||
gbrain sync
|
||||
@@ -79,7 +79,7 @@ on user_message(message):
|
||||
|
||||
1. Generate an original idea in conversation (e.g., "I call this the 'ambition debt' problem -- every year you delay going big, the compound interest works against you"). Confirm a new page appears at `brain/originals/ambition-debt` with `gbrain get originals/ambition-debt`.
|
||||
2. Check that the page uses the user's exact phrasing for the title and slug -- not a sanitized version.
|
||||
3. Run `gbrain get_links originals/ambition-debt`. Confirm cross-links exist to related people, meetings, or other originals.
|
||||
3. Run `gbrain call get_links '{"slug": "originals/ambition-debt"}'`. Confirm cross-links exist to related people, meetings, or other originals.
|
||||
4. Express a take on someone else's idea (e.g., "I think Thiel's contrarian question is wrong because..."). Confirm it goes to `originals/` (synthesis is original), not `concepts/`.
|
||||
5. Run `gbrain search "ambition debt"`. Confirm the originals page appears in search results and is discoverable.
|
||||
|
||||
|
||||
@@ -87,7 +87,7 @@ expect it.
|
||||
| `version` | string | yes | Your plugin's semver. Informational. |
|
||||
| `plugin_version` | string | yes | Contract lock. Must equal `"gbrain-plugin-v1"` for v0.15. |
|
||||
| `subagents` | string | no | Subdir name (default `subagents`). Escape-attempts are rejected. |
|
||||
| `description` | string | no | Shown in future `gbrain plugin list`. |
|
||||
| `description` | string | no | Shown in a future plugin-listing command. |
|
||||
|
||||
## Subagent definition files
|
||||
|
||||
|
||||
+1
-1
@@ -250,7 +250,7 @@ All 30 GBrain operations are available remotely, including `sync_brain` and
|
||||
directory where `gbrain serve` was launched. Symlinks, `..` traversal, and absolute
|
||||
paths outside cwd are rejected. Page slugs and filenames are allowlist-validated
|
||||
(alphanumeric + hyphens; no control chars, RTL overrides, or backslashes). Local
|
||||
CLI callers (`gbrain file upload ...`) keep unrestricted filesystem access since
|
||||
CLI callers (`gbrain files upload ...`) keep unrestricted filesystem access since
|
||||
the user owns the machine.
|
||||
|
||||
## Deployment Options
|
||||
|
||||
@@ -1,690 +0,0 @@
|
||||
# Engine Dynamic-Import Reconciliation Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Reconstruct the missing engine-path static-import hardening, preserve the four load-bearing lazy gateway fallbacks, and prevent unreviewed dynamic imports from returning.
|
||||
|
||||
**Architecture:** Make the 13 safe engine/migration import statements static and leave only four line-marked `ai/gateway.ts` imports inside their existing soft-failure `try/catch` boundaries. Enforce that current state with a repository-anchored Bash wrapper delegating to a fail-closed TypeScript AST scanner, a hermetic Bun regression test, package/verify wiring, and current-state architecture documentation.
|
||||
|
||||
**Tech Stack:** TypeScript compiler API, Bun test runner, Bash, Git, generated llms documentation bundles.
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- Reconstruct directly on branch `claude/kind-meitner-330c90`, based on investigated `origin/master` commit `6136e139972a5449630b4f47f5ed7b4cbe5b811b` plus design commit `d7f52d8c`.
|
||||
- Do not merge or cherry-pick `48ada48f`, `248bfe55`, `ef4cf7a8`, or either historical branch wholesale.
|
||||
- Do not modify `VERSION`, `CHANGELOG.md`, `TODOS.md`, or release metadata; this is a no-version-bump reconciliation.
|
||||
- Keep all four `await import('./ai/gateway.ts')` calls lazy: PGLite and Postgres `initSchema`, plus both `_upsertChunksOnce` methods.
|
||||
- Every allowed lazy gateway line must carry `engine-dynamic-import-ok`; there is no file-level exemption.
|
||||
- Preserve the stronger gateway rationale: the static closure is large, and eager module evaluation would occur outside the local `try/catch`, potentially converting a recoverable configuration/import failure into a module-load-time hard failure.
|
||||
- Describe the hoists as engine-path hardening. Do not claim every dynamic import deterministically causes a Windows crash; system-wide commit exhaustion confounded prior measurements.
|
||||
- Keep shared PGLite/Postgres behavior in parity.
|
||||
- Invoke repository shell scripts through `bash` in `package.json`.
|
||||
- Capture complete test/check output to workspace-local `.context/*.txt` files before inspecting it; never pipe a test command directly through `head` or `tail`.
|
||||
- Use `git log -G`, not `git log -S`, for any additional dynamic-to-static import history work.
|
||||
- Keep every implementation and verification commit local. Do not push, create a PR, comment upstream, or otherwise publish without explicit user approval after local completion.
|
||||
- Before editing any affected function, run GBrain `code_blast` and `code_callers` for that symbol and inspect any disambiguation candidates.
|
||||
|
||||
---
|
||||
|
||||
## File Map
|
||||
|
||||
- Create `scripts/check-engine-dynamic-import.sh` — repository-anchored Bash wrapper for default and explicit input routing.
|
||||
- Create `scripts/check-engine-dynamic-import.ts` — TypeScript AST policy scanner for runtime `import()` expressions, parse/read failures, and exact-line comment-trivia opt-outs.
|
||||
- Create `test/scripts/check-engine-dynamic-import.test.ts` — 22 hermetic adversarial, CRLF, fail-closed, real-tree, and wiring tests.
|
||||
- Modify `src/core/pglite-engine.ts` — hoist three safe import statements and mark two deliberate gateway imports.
|
||||
- Modify `src/core/postgres-engine.ts` — hoist eight safe import statements and mark two deliberate gateway imports.
|
||||
- Modify `src/core/migrate.ts` — hoist two safe migration helper import statements.
|
||||
- Modify `package.json` — expose `check:engine-dynamic-import` and append it to `check:all` through `bash`.
|
||||
- Modify `scripts/run-verify-parallel.sh` — add the package check to the authoritative verify dispatcher.
|
||||
- Modify `CLAUDE.md` — add the cross-cutting current-state invariant.
|
||||
- Modify `docs/architecture/KEY_FILES.md` — update current-state entries for the three engine-path files.
|
||||
- Regenerate `llms.txt` and `llms-full.txt` — required derived bundles after CLAUDE/reference documentation changes.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Establish and enforce the source invariant
|
||||
|
||||
**Files:**
|
||||
- Create: `scripts/check-engine-dynamic-import.sh`
|
||||
- Create: `scripts/check-engine-dynamic-import.ts`
|
||||
- Create: `test/scripts/check-engine-dynamic-import.test.ts`
|
||||
- Modify: `src/core/pglite-engine.ts`
|
||||
- Modify: `src/core/postgres-engine.ts`
|
||||
- Modify: `src/core/migrate.ts`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: shell positional arguments `FILE...`; without arguments, the guard scans the three repository files.
|
||||
- Produces: `scripts/check-engine-dynamic-import.sh [FILE...]`, exit `0` when every runtime dynamic import is allowed and exit `1` after reporting every `file:line:text` violation plus every read/parse error on stderr.
|
||||
- Produces: one line-level opt-out token, `engine-dynamic-import-ok`, accepted only in real comment trivia on the same physical line as the deliberately lazy import.
|
||||
- Fails closed on missing/unreadable inputs, TypeScript parse diagnostics, and scanner/process failures; comments, strings, templates, regex literals, and type-position `import(...)` syntax are not runtime imports.
|
||||
|
||||
- [ ] **Step 1: Record call-graph blast radius before touching functions**
|
||||
|
||||
First call `sources_list` and select the source whose registered path is this gbrain checkout. Then run `code_blast` and `code_callers` for these qualified symbols with that exact `source_id`, following `did_you_mean`/`candidates` when a method name is ambiguous:
|
||||
|
||||
```text
|
||||
src/core/pglite-engine.ts::PGLiteEngine.initSchema
|
||||
src/core/pglite-engine.ts::PGLiteEngine.batchRetry
|
||||
src/core/pglite-engine.ts::PGLiteEngine._upsertChunksOnce
|
||||
src/core/pglite-engine.ts::PGLiteEngine.mergeOntologyFact
|
||||
src/core/pglite-engine.ts::PGLiteEngine.getRecentSalience
|
||||
src/core/postgres-engine.ts::PostgresEngine.disconnect
|
||||
src/core/postgres-engine.ts::PostgresEngine.initSchema
|
||||
src/core/postgres-engine.ts::PostgresEngine.batchRetry
|
||||
src/core/postgres-engine.ts::PostgresEngine._upsertChunksOnce
|
||||
src/core/postgres-engine.ts::PostgresEngine.mergeOntologyFact
|
||||
src/core/postgres-engine.ts::PostgresEngine.reconnect
|
||||
src/core/postgres-engine.ts::PostgresEngine.getRecentSalience
|
||||
src/core/migrate.ts::runMigrationSQLWithRetry
|
||||
src/core/migrate.ts::runMigrations
|
||||
```
|
||||
|
||||
Use `depth: 5`, `max_nodes: 200`, and `limit: 100`. Expected: no caller requires a signature or behavior change; the patch only changes module binding time and retains all local fallback/error handling.
|
||||
|
||||
- [ ] **Step 2: Write the failing guard regression test**
|
||||
|
||||
Create `test/scripts/check-engine-dynamic-import.test.ts` as a hermetic subprocess suite. The completed 22-test surface covers:
|
||||
|
||||
- unmarked runtime `import()` rejection, including bare and trivia-separated forms;
|
||||
- same-line markers in real line or multiline block-comment trivia;
|
||||
- rejection of markers on prior lines or inside strings, templates, and module paths;
|
||||
- comments and comment-like delimiters inside strings, templates, and regex literals;
|
||||
- live code after same-line or multiline block comments close;
|
||||
- CRLF input and complete multi-file violation aggregation;
|
||||
- missing/readable mixed inputs and TypeScript parse diagnostics;
|
||||
- default repository anchoring when invoked from a foreign Git repository;
|
||||
- the reconciled three-file source scan plus package/parallel-verifier wiring.
|
||||
|
||||
Use the TypeScript parser rather than a partial lexical reimplementation. On Windows, set the test default to 30 seconds because each case launches Git Bash and Bun, whose startup can exceed Bun's 5-second per-test default.
|
||||
|
||||
- [ ] **Step 3: Run the test to prove the pre-implementation red state**
|
||||
|
||||
```bash
|
||||
bun test test/scripts/check-engine-dynamic-import.test.ts > .context/engine-dynamic-import-red.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit 0
|
||||
```
|
||||
|
||||
Expected: non-zero Bun result captured inside the log. At minimum, the `exists` assertion fails because `scripts/check-engine-dynamic-import.sh` does not exist. Read `.context/engine-dynamic-import-red.txt`; do not infer the result from a truncated pipeline.
|
||||
|
||||
- [ ] **Step 4: Add the CRLF-safe, fail-closed guard**
|
||||
|
||||
Create `scripts/check-engine-dynamic-import.sh` as a thin LF-terminated wrapper. Resolve its own directory first; when no explicit files are passed, anchor the repository with `git -C "$SCRIPT_DIR/.."` and scan the two engines plus `migrate.ts`. Delegate with `exec bun "$SCRIPT_DIR/check-engine-dynamic-import.ts" "${FILES[@]}"` so scanner failures propagate.
|
||||
|
||||
Create `scripts/check-engine-dynamic-import.ts` using the TypeScript compiler API:
|
||||
|
||||
- read every requested file and aggregate read failures;
|
||||
- parse as TypeScript and aggregate parse diagnostics;
|
||||
- walk the AST for `CallExpression`s whose expression is `ImportKeyword`;
|
||||
- locate all marker occurrences in the full source and use `ts.getTokenAtPosition` to admit only occurrences outside AST tokens (real comment trivia), recording their physical source lines;
|
||||
- require each runtime import's line to have an admitted marker or report its original `file:line:text`;
|
||||
- print every read/parse error and every violation before exiting nonzero.
|
||||
|
||||
This preserves CRLF line accounting, ignores comment/literal/type-only false positives, catches every legal runtime `import()` shape the TypeScript parser recognizes, rejects marker spoofing, and fails closed.
|
||||
|
||||
- [ ] **Step 5: Run the guard test to prove the source-tree midpoint is still red**
|
||||
|
||||
```bash
|
||||
bun test test/scripts/check-engine-dynamic-import.test.ts > .context/engine-dynamic-import-midpoint.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit 0
|
||||
```
|
||||
|
||||
Expected: the synthetic violation, marker, comments, and CRLF cases pass. The default repository scan fails and reports all 17 current imports: 13 unmarked safe candidates plus the four not-yet-marked gateway calls.
|
||||
|
||||
- [ ] **Step 6: Hoist the three safe PGLite import statements**
|
||||
|
||||
Replace the existing `retry.ts` import and add the ontology/recency imports near the top of `src/core/pglite-engine.ts`:
|
||||
|
||||
```ts
|
||||
// Engine-path imports stay static unless a call site carries an explicit
|
||||
// engine-dynamic-import-ok justification. The gateway is the only current
|
||||
// exception because its local try/catch preserves a soft fallback.
|
||||
import {
|
||||
withRetry,
|
||||
BULK_RETRY_OPTS,
|
||||
resolveBulkRetryOpts,
|
||||
computeNextDelay,
|
||||
isRetryableConnError,
|
||||
type BatchAuditSite,
|
||||
} from './retry.ts';
|
||||
import {
|
||||
valueHash,
|
||||
normalizeDimension,
|
||||
isNovelDimension,
|
||||
} from './chronicle/ontology.ts';
|
||||
import {
|
||||
resolveRecencyDecayMap,
|
||||
DEFAULT_FALLBACK,
|
||||
} from './search/recency-decay.ts';
|
||||
```
|
||||
|
||||
Delete only these three in-method destructuring imports, leaving their uses unchanged:
|
||||
|
||||
```ts
|
||||
const { isRetryableConnError } = await import('./retry.ts');
|
||||
const { valueHash, normalizeDimension, isNovelDimension } = await import('./chronicle/ontology.ts');
|
||||
const { resolveRecencyDecayMap, DEFAULT_FALLBACK } = await import('./search/recency-decay.ts');
|
||||
```
|
||||
|
||||
- [ ] **Step 7: Mark both PGLite gateway soft-failure boundaries**
|
||||
|
||||
In `PGLiteEngine.initSchema`, preserve the `try/catch` and accessors, changing only the rationale and import line:
|
||||
|
||||
```ts
|
||||
try {
|
||||
// Keep the gateway lazy: its static closure is large, and evaluation inside
|
||||
// this try/catch preserves the unconfigured-gateway default fallback.
|
||||
const gw = await import('./ai/gateway.ts'); // engine-dynamic-import-ok
|
||||
// Both accessors THROW when the gateway is unconfigured (they never
|
||||
// return falsy), so the catch below is the only fallback path (#3461).
|
||||
dims = gw.getEmbeddingDimensions();
|
||||
model = gw.getEmbeddingModel();
|
||||
} catch { /* gateway not configured — use defaults */ }
|
||||
```
|
||||
|
||||
In `PGLiteEngine._upsertChunksOnce`, preserve the config-row and compile-time fallback chain:
|
||||
|
||||
```ts
|
||||
try {
|
||||
// Keep the gateway lazy so module-load failure remains inside this soft
|
||||
// fallback boundary; eager evaluation would bypass the config-row fallback.
|
||||
const gw = await import('./ai/gateway.ts'); // engine-dynamic-import-ok
|
||||
resolvedModel = gw.getEmbeddingModel();
|
||||
} catch {
|
||||
```
|
||||
|
||||
- [ ] **Step 8: Hoist the eight safe Postgres import statements**
|
||||
|
||||
Replace the existing `retry.ts` import and add these imports near the top of `src/core/postgres-engine.ts`:
|
||||
|
||||
```ts
|
||||
// Engine-path imports stay static unless a call site carries an explicit
|
||||
// engine-dynamic-import-ok justification. The gateway is the only current
|
||||
// exception because its local try/catch preserves a soft fallback.
|
||||
import {
|
||||
withRetry,
|
||||
BULK_RETRY_OPTS,
|
||||
resolveBulkRetryOpts,
|
||||
computeNextDelay,
|
||||
isRetryableConnError,
|
||||
type BatchAuditSite,
|
||||
} from './retry.ts';
|
||||
import { isConnectionEndedError } from './retry-matcher.ts';
|
||||
import {
|
||||
valueHash,
|
||||
normalizeDimension,
|
||||
isNovelDimension,
|
||||
} from './chronicle/ontology.ts';
|
||||
import {
|
||||
resolveRecencyDecayMap,
|
||||
DEFAULT_FALLBACK,
|
||||
} from './search/recency-decay.ts';
|
||||
import { logDbDisconnect } from './audit/db-disconnect-audit.ts';
|
||||
import { logPoolRecovery } from './audit/pool-recovery-audit.ts';
|
||||
```
|
||||
|
||||
Delete the eight safe dynamic-import statements while keeping their surrounding `try/catch` blocks and calls unchanged:
|
||||
|
||||
```ts
|
||||
const { logDbDisconnect } = await import('./audit/db-disconnect-audit.ts');
|
||||
const { isRetryableConnError } = await import('./retry.ts');
|
||||
const { valueHash, normalizeDimension, isNovelDimension } = await import('./chronicle/ontology.ts');
|
||||
const { isConnectionEndedError } = await import('./retry-matcher.ts');
|
||||
const { logPoolRecovery } = await import('./audit/pool-recovery-audit.ts');
|
||||
const { logPoolRecovery } = await import('./audit/pool-recovery-audit.ts');
|
||||
const { logPoolRecovery } = await import('./audit/pool-recovery-audit.ts');
|
||||
const { resolveRecencyDecayMap, DEFAULT_FALLBACK } = await import('./search/recency-decay.ts');
|
||||
```
|
||||
|
||||
Update the stale `batchRetry` comment from “Lazy-import to avoid a circular dep concern” to current truth:
|
||||
|
||||
```ts
|
||||
// retry.ts is already in this module's static graph through withRetry, so
|
||||
// classifying the exhausted error does not need a second runtime import.
|
||||
```
|
||||
|
||||
- [ ] **Step 9: Mark both Postgres gateway soft-failure boundaries**
|
||||
|
||||
In `PostgresEngine.initSchema`, mirror the PGLite rationale and preserve behavior:
|
||||
|
||||
```ts
|
||||
try {
|
||||
// Keep the gateway lazy: its static closure is large, and evaluation inside
|
||||
// this try/catch preserves the unconfigured-gateway default fallback.
|
||||
const gw = await import('./ai/gateway.ts'); // engine-dynamic-import-ok
|
||||
// Both accessors THROW when the gateway is unconfigured (they never
|
||||
// return falsy), so the catch below is the only fallback path (#3461).
|
||||
dims = gw.getEmbeddingDimensions();
|
||||
model = gw.getEmbeddingModel();
|
||||
} catch { /* gateway not yet configured — use defaults */ }
|
||||
```
|
||||
|
||||
In `PostgresEngine._upsertChunksOnce`, preserve the DB-config fallback:
|
||||
|
||||
```ts
|
||||
try {
|
||||
// Keep the gateway lazy so module-load failure remains inside this soft
|
||||
// fallback boundary; eager evaluation would bypass the config-row fallback.
|
||||
const gw = await import('./ai/gateway.ts'); // engine-dynamic-import-ok
|
||||
resolvedModel = gw.getEmbeddingModel();
|
||||
} catch {
|
||||
```
|
||||
|
||||
- [ ] **Step 10: Hoist the two migration helper import statements**
|
||||
|
||||
Add these static imports at the top of `src/core/migrate.ts`:
|
||||
|
||||
```ts
|
||||
// runMigrations executes while an initialized engine is live. Keep its helper
|
||||
// modules in the static graph rather than importing them from async handlers.
|
||||
import {
|
||||
isStatementTimeoutError,
|
||||
isRetryableConnError,
|
||||
} from './retry-matcher.ts';
|
||||
import { repairTimelineDedupIndex } from './timeline-dedup-repair.ts';
|
||||
```
|
||||
|
||||
Delete only these two local destructuring imports:
|
||||
|
||||
```ts
|
||||
const { isStatementTimeoutError, isRetryableConnError } = await import('./retry-matcher.ts');
|
||||
const { repairTimelineDedupIndex } = await import('./timeline-dedup-repair.ts');
|
||||
```
|
||||
|
||||
- [ ] **Step 11: Run the complete guard test and direct guard**
|
||||
|
||||
```bash
|
||||
bun test test/scripts/check-engine-dynamic-import.test.ts > .context/engine-dynamic-import-green.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`; the full guard regression suite passes.
|
||||
|
||||
```bash
|
||||
bash scripts/check-engine-dynamic-import.sh > .context/engine-dynamic-import-guard.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`; output contains `check-engine-dynamic-import: ok (3 file(s) scanned)`.
|
||||
|
||||
- [ ] **Step 12: Prove the guard leaves exactly four marked dynamic imports**
|
||||
|
||||
```bash
|
||||
git grep -n -F "import('./ai/gateway.ts'); // engine-dynamic-import-ok" -- src/core/pglite-engine.ts src/core/postgres-engine.ts src/core/migrate.ts > .context/engine-dynamic-import-sites.txt; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exactly four lines, all importing `./ai/gateway.ts` and all carrying `engine-dynamic-import-ok`; no match in `src/core/migrate.ts`.
|
||||
|
||||
- [ ] **Step 13: Run focused behavior tests**
|
||||
|
||||
```bash
|
||||
bun test test/chronicle-ontology.test.ts test/chronicle-ontology-ops.test.ts test/recency-decay.test.ts test/core/retry.test.ts test/retry-matcher.test.ts test/audit/pool-recovery-audit.test.ts test/migrate-retry.test.ts test/timeline-dedup-repair.test.ts > .context/engine-dynamic-import-focused.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`. If Windows resource pressure aborts the process, record the exact exit code and rerun the failing file alone; do not relabel an infrastructure abort as a source pass.
|
||||
|
||||
- [ ] **Step 14: Commit the source invariant locally**
|
||||
|
||||
```bash
|
||||
git add scripts/check-engine-dynamic-import.sh scripts/check-engine-dynamic-import.ts test/scripts/check-engine-dynamic-import.test.ts src/core/pglite-engine.ts src/core/postgres-engine.ts src/core/migrate.ts
|
||||
```
|
||||
|
||||
```bash
|
||||
git commit -m "fix(engine): reconcile dynamic import hardening"
|
||||
```
|
||||
|
||||
Expected: one local commit; no version or release files staged.
|
||||
|
||||
---
|
||||
|
||||
### Task 2: Wire the guard into repository checks
|
||||
|
||||
**Files:**
|
||||
- Modify: `test/scripts/check-engine-dynamic-import.test.ts`
|
||||
- Modify: `package.json`
|
||||
- Modify: `scripts/run-verify-parallel.sh`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `scripts/check-engine-dynamic-import.sh` from Task 1.
|
||||
- Produces: package script `check:engine-dynamic-import` and verify dry-list entry of the same name.
|
||||
|
||||
- [ ] **Step 1: Add failing wiring assertions**
|
||||
|
||||
Add these imports/constants to `test/scripts/check-engine-dynamic-import.test.ts`:
|
||||
|
||||
```ts
|
||||
const PACKAGE_JSON = resolve(REPO_ROOT, 'package.json');
|
||||
```
|
||||
|
||||
Append this test block:
|
||||
|
||||
```ts
|
||||
describe('engine dynamic-import guard wiring', () => {
|
||||
it('is invoked through bash by check:all', () => {
|
||||
const pkg = JSON.parse(readFileSync(PACKAGE_JSON, 'utf8')) as {
|
||||
scripts: Record<string, string>;
|
||||
};
|
||||
expect(pkg.scripts['check:engine-dynamic-import']).toBe(
|
||||
'bash scripts/check-engine-dynamic-import.sh',
|
||||
);
|
||||
expect(pkg.scripts['check:all']).toContain(
|
||||
'bash scripts/check-engine-dynamic-import.sh',
|
||||
);
|
||||
});
|
||||
|
||||
it('is listed by the authoritative verify dispatcher', () => {
|
||||
const result = spawnSync(BASH, [VERIFY_DISPATCHER, '--dry-list'], {
|
||||
cwd: REPO_ROOT,
|
||||
encoding: 'utf8',
|
||||
timeout: 30_000,
|
||||
});
|
||||
expect(result.status).toBe(0);
|
||||
expect(new Set((result.stdout ?? '').trim().split('\n'))).toContain(
|
||||
'check:engine-dynamic-import',
|
||||
);
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run the test and verify both wiring assertions fail**
|
||||
|
||||
```bash
|
||||
bun test test/scripts/check-engine-dynamic-import.test.ts > .context/engine-dynamic-import-wiring-red.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit 0
|
||||
```
|
||||
|
||||
Expected: non-zero Bun result. The source guard tests remain green; package-script and verify-list assertions fail because the wiring is absent.
|
||||
|
||||
- [ ] **Step 3: Add the package scripts**
|
||||
|
||||
In `package.json`, add this script alongside the other `check:*` entries:
|
||||
|
||||
```json
|
||||
"check:engine-dynamic-import": "bash scripts/check-engine-dynamic-import.sh"
|
||||
```
|
||||
|
||||
Append the guard to the existing `check:all` chain, preserving every existing check:
|
||||
|
||||
```text
|
||||
&& bash scripts/check-engine-dynamic-import.sh
|
||||
```
|
||||
|
||||
Do not rewrite any existing shell entry without its `bash` prefix.
|
||||
|
||||
- [ ] **Step 4: Add the authoritative verify entry**
|
||||
|
||||
In `scripts/run-verify-parallel.sh`, add this stable `CHECKS` entry near the other source-shape guards:
|
||||
|
||||
```bash
|
||||
"check:engine-dynamic-import"
|
||||
```
|
||||
|
||||
- [ ] **Step 5: Run the regression test and package check**
|
||||
|
||||
```bash
|
||||
bun test test/scripts/check-engine-dynamic-import.test.ts > .context/engine-dynamic-import-wiring-green.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`; the full guard regression suite passes.
|
||||
|
||||
```bash
|
||||
bun run check:engine-dynamic-import > .context/engine-dynamic-import-package-check.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0` and three files scanned.
|
||||
|
||||
- [ ] **Step 6: Commit the wiring locally**
|
||||
|
||||
```bash
|
||||
git add package.json scripts/run-verify-parallel.sh test/scripts/check-engine-dynamic-import.test.ts
|
||||
```
|
||||
|
||||
```bash
|
||||
git commit -m "test(engine): guard dynamic import policy"
|
||||
```
|
||||
|
||||
Expected: one local commit with the guard wiring and its regression assertions.
|
||||
|
||||
---
|
||||
|
||||
### Task 3: Document the current-state invariant
|
||||
|
||||
**Files:**
|
||||
- Modify: `CLAUDE.md`
|
||||
- Modify: `docs/architecture/KEY_FILES.md`
|
||||
- Regenerate: `llms.txt`
|
||||
- Regenerate: `llms-full.txt`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: the four-marked-import source state and the `check:engine-dynamic-import` package surface.
|
||||
- Produces: current-state contributor guidance and fresh generated documentation bundles.
|
||||
|
||||
- [ ] **Step 1: Add the cross-cutting invariant to `CLAUDE.md`**
|
||||
|
||||
Add this bullet under “Cross-cutting invariants” near the other language/filesystem guards:
|
||||
|
||||
```md
|
||||
- **Engine-live paths use static imports by default.** In
|
||||
`src/core/pglite-engine.ts`, `src/core/postgres-engine.ts`, and
|
||||
`src/core/migrate.ts`, helper modules are top-level imports. The only current
|
||||
exceptions are the four `ai/gateway.ts` lookups in both engines'
|
||||
`initSchema()` and `_upsertChunksOnce()` methods; each remains lazy inside a
|
||||
local `try/catch` because the gateway has a large provider/config closure and,
|
||||
more importantly, eager evaluation would occur before the catch and could
|
||||
turn a recoverable default/config-row fallback into a module-load failure.
|
||||
Every exception carries `engine-dynamic-import-ok` on the import line.
|
||||
`scripts/check-engine-dynamic-import.sh` enforces the rule. For history, use
|
||||
`git log -G'await[[:space:]]+import\\('`, not `git log -S`: a dynamic-to-static
|
||||
rewrite can preserve the searched token while changing its context.
|
||||
```
|
||||
|
||||
Do not add release tags, Windows-crash certainty, or historical branch names.
|
||||
|
||||
- [ ] **Step 2: Update the PGLite current-state entry in `KEY_FILES.md`**
|
||||
|
||||
Append this current-state sentence to the existing `src/core/pglite-engine.ts` entry, preserving the entry as one bullet:
|
||||
|
||||
```md
|
||||
Engine-path helper dependencies (`retry`, ontology, recency decay) bind statically; the only lazy imports are `ai/gateway.ts` in `initSchema` and `_upsertChunksOnce`, line-marked because their local catches preserve compiled-default and stored-config fallbacks that eager module evaluation would bypass.
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Update the Postgres current-state entry in `KEY_FILES.md`**
|
||||
|
||||
Append this sentence to the existing `src/core/postgres-engine.ts` entry:
|
||||
|
||||
```md
|
||||
Retry classifiers, ontology/recency helpers, and disconnect/pool-recovery audit writers bind statically; only the two `ai/gateway.ts` fallback lookups stay lazy and line-marked, in parity with PGLite.
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Update the migration current-state entry in `KEY_FILES.md`**
|
||||
|
||||
Append this sentence to the canonical `src/core/migrate.ts` entry (the broad runner entry, not the older v95-specific index note):
|
||||
|
||||
```md
|
||||
`retry-matcher.ts` and `timeline-dedup-repair.ts` are static dependencies because `runMigrations()` executes from live engine initialization; the engine dynamic-import guard scans this file with both engine implementations.
|
||||
```
|
||||
|
||||
Keep all three entries current-state only: no `v0.42.x`, branch, commit, “previously,” or “was/now” narration.
|
||||
|
||||
- [ ] **Step 5: Regenerate the llms bundles**
|
||||
|
||||
```bash
|
||||
bun run build:llms > .context/engine-dynamic-import-build-llms.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`; `llms.txt` and/or `llms-full.txt` update according to their configured linked/inlined status. Byte-identical output for a linked source is acceptable; the freshness test is authoritative.
|
||||
|
||||
- [ ] **Step 6: Run documentation freshness checks**
|
||||
|
||||
```bash
|
||||
bun test test/build-llms.test.ts > .context/engine-dynamic-import-llms-test.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`.
|
||||
|
||||
```bash
|
||||
bun run check:doc-history > .context/engine-dynamic-import-doc-history.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`; no release-history marker is introduced into current-state reference docs.
|
||||
|
||||
- [ ] **Step 7: Confirm prohibited release files remain untouched**
|
||||
|
||||
```bash
|
||||
git diff --name-only d7f52d8c..HEAD -- VERSION CHANGELOG.md TODOS.md
|
||||
```
|
||||
|
||||
Expected: no output.
|
||||
|
||||
- [ ] **Step 8: Commit documentation and generated bundles locally**
|
||||
|
||||
```bash
|
||||
git add CLAUDE.md docs/architecture/KEY_FILES.md llms.txt llms-full.txt
|
||||
```
|
||||
|
||||
```bash
|
||||
git commit -m "docs(engine): record static import invariant"
|
||||
```
|
||||
|
||||
Expected: one local documentation commit. If one generated bundle is byte-identical, Git simply omits it.
|
||||
|
||||
---
|
||||
|
||||
### Task 4: Verify and review the complete local reconciliation
|
||||
|
||||
**Files:**
|
||||
- Verify all files changed since `d7f52d8c`.
|
||||
- Do not create or modify release/publication metadata.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: Tasks 1–3.
|
||||
- Produces: full local verification evidence and an implementation diff ready for user review, not publication.
|
||||
|
||||
- [ ] **Step 1: Run the regression test and direct guard again**
|
||||
|
||||
```bash
|
||||
bun test test/scripts/check-engine-dynamic-import.test.ts > .context/engine-dynamic-import-final-test.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`; the full guard regression suite passes.
|
||||
|
||||
```bash
|
||||
bash scripts/check-engine-dynamic-import.sh > .context/engine-dynamic-import-final-guard.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`; three files scanned.
|
||||
|
||||
- [ ] **Step 2: Run TypeScript checking**
|
||||
|
||||
```bash
|
||||
bun run typecheck > .context/engine-dynamic-import-typecheck.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`. Report exact diagnostics if the branch or current Windows environment has a pre-existing failure.
|
||||
|
||||
- [ ] **Step 3: Run the authoritative verify dispatcher**
|
||||
|
||||
```bash
|
||||
bun run verify > .context/engine-dynamic-import-verify.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`, including `check:engine-dynamic-import`. On Windows, classify any per-check timeout from the complete log instead of treating the aggregate result as a source regression without evidence.
|
||||
|
||||
- [ ] **Step 4: Re-run focused tests as an ownership check**
|
||||
|
||||
```bash
|
||||
bun test test/chronicle-ontology.test.ts test/chronicle-ontology-ops.test.ts test/recency-decay.test.ts test/core/retry.test.ts test/retry-matcher.test.ts test/audit/pool-recovery-audit.test.ts test/migrate-retry.test.ts test/timeline-dedup-repair.test.ts > .context/engine-dynamic-import-final-focused.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`; record any infrastructure abort separately and rerun only the named file before classifying it.
|
||||
|
||||
- [ ] **Step 5: Run the llms freshness test after all documentation settles**
|
||||
|
||||
```bash
|
||||
bun test test/build-llms.test.ts > .context/engine-dynamic-import-final-llms.txt 2>&1; rc=$?; printf 'EXIT=%s\n' "$rc"; exit "$rc"
|
||||
```
|
||||
|
||||
Expected: exit `0`.
|
||||
|
||||
- [ ] **Step 6: Run whitespace and scope checks**
|
||||
|
||||
```bash
|
||||
git diff --check d7f52d8c..HEAD
|
||||
```
|
||||
|
||||
Expected: exit `0`, no output.
|
||||
|
||||
```bash
|
||||
git diff --name-only d7f52d8c..HEAD
|
||||
```
|
||||
|
||||
Expected files only:
|
||||
|
||||
```text
|
||||
CLAUDE.md
|
||||
docs/architecture/KEY_FILES.md
|
||||
docs/superpowers/plans/2026-07-28-engine-dynamic-import-reconciliation.md
|
||||
llms-full.txt
|
||||
llms.txt
|
||||
package.json
|
||||
scripts/check-engine-dynamic-import.sh
|
||||
scripts/check-engine-dynamic-import.ts
|
||||
scripts/run-verify-parallel.sh
|
||||
src/core/migrate.ts
|
||||
src/core/pglite-engine.ts
|
||||
src/core/postgres-engine.ts
|
||||
test/scripts/check-engine-dynamic-import.test.ts
|
||||
```
|
||||
|
||||
Either generated llms file may be absent if regeneration proves it byte-identical. `VERSION`, `CHANGELOG.md`, and `TODOS.md` must be absent.
|
||||
|
||||
- [ ] **Step 7: Review the exact implementation diff**
|
||||
|
||||
```bash
|
||||
git diff --stat d7f52d8c..HEAD && git diff d7f52d8c..HEAD -- src/core/pglite-engine.ts src/core/postgres-engine.ts src/core/migrate.ts scripts/check-engine-dynamic-import.sh test/scripts/check-engine-dynamic-import.test.ts package.json scripts/run-verify-parallel.sh CLAUDE.md docs/architecture/KEY_FILES.md
|
||||
```
|
||||
|
||||
Expected review findings:
|
||||
|
||||
- Exactly 13 safe `await import(...)` statements are removed.
|
||||
- Exactly four `ai/gateway.ts` imports remain, all marked on the same line.
|
||||
- All four gateway imports remain inside their original local `try/catch` fallback boundaries.
|
||||
- No accessor logic, fallback ordering, SQL, public signature, or engine parity behavior changes.
|
||||
- The parser-backed guard reports all violations plus read/parse failures, preserves CRLF line accounting, ignores comments/literals/type-only syntax, detects every runtime `import()` call expression, and accepts opt-outs only from real comment trivia on the same physical line.
|
||||
- The package script invokes the shell guard through Bash; `check:all` invokes that shell guard directly, and the parallel verify dispatcher invokes the package check.
|
||||
- Documentation is current-state and makes no deterministic Windows-crash claim.
|
||||
|
||||
**Observed Windows verification classification:** The authoritative aggregate completed with 25 of 33 checks passing. Individual reruns showed `check:test-names` and `typecheck` green; privacy/isolation exceeded Windows timing budgets; WASM failed in unrelated temporary-symlink setup; eval-glossary was CRLF/LF drift; resolver/brain-first findings predated and did not intersect this branch. The focused aggregate produced 103 pass / 5 fail: three setup-hook timeouts reproduced at the untouched base, and the known `migrate-retry` polling failure reproduced there. Its additional race-status assertion did not reproduce at base, so it remains an unresolved timing-sensitive limitation in untouched code—not evidence of an in-scope defect and not claimed as conclusively pre-existing.
|
||||
|
||||
- [ ] **Step 8: Commit the approved plan document locally**
|
||||
|
||||
The plan is an approved, tracked execution artifact and must not be left as an uncommitted file after implementation:
|
||||
|
||||
```bash
|
||||
git add docs/superpowers/plans/2026-07-28-engine-dynamic-import-reconciliation.md
|
||||
```
|
||||
|
||||
```bash
|
||||
git commit -m "docs: plan engine dynamic-import reconciliation"
|
||||
```
|
||||
|
||||
Expected: one local plan commit; no release metadata staged.
|
||||
|
||||
- [ ] **Step 9: Inspect final status without publishing**
|
||||
|
||||
```bash
|
||||
git status --short --branch
|
||||
```
|
||||
|
||||
Expected: branch `claude/kind-meitner-330c90` with a clean working tree. No push, PR, upstream comment, or other external side effect.
|
||||
|
||||
- [ ] **Step 10: Capture the completed milestone to memory**
|
||||
|
||||
Before writing, search MemPalace wing `gbrain` for this exact reconciliation to avoid duplication. Add a verbatim drawer recording exact base/head commits, the 13 hoists, four gateway opt-outs and rationale, guard/test/docs files, every verification command with exit code, and any environment-owned failures. Add a GBrain project timeline entry only if there is an existing relevant gbrain project page; do not create duplicate release metadata.
|
||||
|
||||
- [ ] **Step 11: Report the local result and ask separately before publication**
|
||||
|
||||
Report:
|
||||
|
||||
- exact local commits;
|
||||
- changed files;
|
||||
- test/check exit codes;
|
||||
- any blocked or pre-existing failures;
|
||||
- confirmation that release files were untouched;
|
||||
- confirmation that nothing was pushed or published.
|
||||
|
||||
Do not run any publication command. Wait for explicit user approval before any push, PR, or upstream interaction.
|
||||
@@ -1,142 +0,0 @@
|
||||
# Engine dynamic-import reconciliation design
|
||||
|
||||
**Date:** 2026-07-28
|
||||
|
||||
## Goal
|
||||
|
||||
Reconcile the overlapping engine dynamic-import changes from:
|
||||
|
||||
- `claude/hungry-edison-8bb1cd` at release commits `48ada48f` and `248bfe55`
|
||||
- `claude/elegant-gates-e5275e` at `ef4cf7a8`
|
||||
|
||||
onto a fresh branch from current `origin/master`, without merging or cherry-picking either lineage wholesale and without adding a release/version bump.
|
||||
|
||||
## Established state
|
||||
|
||||
At investigation time:
|
||||
|
||||
- `origin/master` was `6136e139972a5449630b4f47f5ed7b4cbe5b811b`, version `0.42.67.0`.
|
||||
- Upstream PR #3511 was still open, so trunk did not contain its two `chronicle/ontology.ts` hoists.
|
||||
- Neither source branch was an ancestor of trunk.
|
||||
- Trunk contained 17 dynamic imports in the three engine-path files:
|
||||
- 13 safe-hoist candidates: two ontology imports, nine engine helper/audit imports, and two migration imports.
|
||||
- Four `ai/gateway.ts` imports, all inside `try/catch` fallback paths.
|
||||
- `git log -G` showed the separate ontology, helper, migration, and gateway histories. `git log -S` is not suitable for this dynamic-to-static replacement because the relevant token can remain present while its context changes.
|
||||
- The guard from `ef4cf7a8` passed against that commit but failed against trunk. It also knew about only two gateway opt-outs because two `_upsertChunksOnce` gateway lookups landed later in trunk.
|
||||
|
||||
## Selected approach
|
||||
|
||||
Reconstruct the intended current state directly on fresh `origin/master`.
|
||||
|
||||
Do not merge or cherry-pick either old lineage. Selectively reproduce the desired source changes, adapt the guard to the current four gateway call sites, and write current-state documentation. This avoids importing stale release metadata, stale TODO claims, and unrelated lineage changes.
|
||||
|
||||
## Source changes
|
||||
|
||||
### Safe static imports
|
||||
|
||||
Hoist all 13 safe candidates:
|
||||
|
||||
- `src/core/pglite-engine.ts`
|
||||
- `valueHash`, `normalizeDimension`, `isNovelDimension` from `chronicle/ontology.ts`
|
||||
- `isRetryableConnError` through the existing `retry.ts` import
|
||||
- `resolveRecencyDecayMap`, `DEFAULT_FALLBACK` from `search/recency-decay.ts`
|
||||
- `src/core/postgres-engine.ts`
|
||||
- the same ontology, retry, and recency helpers
|
||||
- `isConnectionEndedError` from `retry-matcher.ts`
|
||||
- `logDbDisconnect` from `audit/db-disconnect-audit.ts`
|
||||
- `logPoolRecovery` from `audit/pool-recovery-audit.ts`
|
||||
- `src/core/migrate.ts`
|
||||
- `isStatementTimeoutError`, `isRetryableConnError` from `retry-matcher.ts`
|
||||
- `repairTimelineDedupIndex` from `timeline-dedup-repair.ts`
|
||||
|
||||
The implementation must keep the two engines in parity where the behavior is shared. Comments should describe current invariants, not repeat an unproven causal claim that these hoists fix the Windows test-runner crash.
|
||||
|
||||
### Deliberately lazy gateway imports
|
||||
|
||||
Keep all four `await import('./ai/gateway.ts')` call sites lazy:
|
||||
|
||||
- PGLite `initSchema`
|
||||
- PGLite `_upsertChunksOnce`
|
||||
- Postgres `initSchema`
|
||||
- Postgres `_upsertChunksOnce`
|
||||
|
||||
Each line receives the explicit `engine-dynamic-import-ok` marker and a concise nearby rationale.
|
||||
|
||||
The rationale has two parts:
|
||||
|
||||
1. The gateway's static closure includes the AI SDK, provider packages, and validation/config machinery, so eager loading would tax engine startup paths that do not otherwise need it.
|
||||
2. More importantly, each lookup is inside a `try/catch` that preserves a soft fallback (compiled defaults or the brain's stored embedding-model config). Hoisting the module would evaluate it before that catch can run and could convert a recoverable configuration/import failure into a module-load-time hard failure.
|
||||
|
||||
The guard must not allow unmarked gateway imports or a broad file-level exemption.
|
||||
|
||||
## Guard and wiring
|
||||
|
||||
Add `scripts/check-engine-dynamic-import.sh`, adapted from `ef4cf7a8`, with these properties:
|
||||
|
||||
- Default scan set:
|
||||
- `src/core/pglite-engine.ts`
|
||||
- `src/core/postgres-engine.ts`
|
||||
- `src/core/migrate.ts`
|
||||
- Normalize trailing CR before matching so CRLF checkouts cannot bypass the check.
|
||||
- Ignore comment-only lines.
|
||||
- Ignore only lines carrying `engine-dynamic-import-ok`.
|
||||
- Report every unmarked `await import(` with file and line.
|
||||
- Explain that contributors should prefer a static import and must justify a real opt-out.
|
||||
- Avoid asserting that every dynamic import deterministically crashes Windows; the measured evidence supports treating the pattern as an engine-path hardening invariant, while box-level commit exhaustion remained a confound in prior runs.
|
||||
|
||||
Wire it into:
|
||||
|
||||
- `package.json` as `check:engine-dynamic-import`
|
||||
- `package.json` `check:all`
|
||||
- `scripts/run-verify-parallel.sh`
|
||||
|
||||
Follow trunk's current rule that package scripts invoke repository shell scripts through `bash`.
|
||||
|
||||
## Regression coverage
|
||||
|
||||
Add an automated test for the guard. It must cover:
|
||||
|
||||
- A real dynamic import produces exit 1 and is reported.
|
||||
- A line carrying `engine-dynamic-import-ok` is allowed.
|
||||
- Line comments and block-comment lines do not produce findings.
|
||||
- The same violation is caught with CRLF input.
|
||||
- The default repository scan passes after the source reconciliation.
|
||||
|
||||
Use a temporary fixture rather than mutating tracked source files. Keep assertions path-portable.
|
||||
|
||||
The pre-fix red demonstration is the exact guard from `ef4cf7a8` run against current trunk: it exits 1 and reports the existing unmarked imports. The post-fix guard and test must pass.
|
||||
|
||||
## Documentation policy
|
||||
|
||||
Preserve current behavior, not either old release narrative:
|
||||
|
||||
- Do not modify `VERSION` or add a release `CHANGELOG.md` entry.
|
||||
- Do not copy old version headings or completed release TODO blocks.
|
||||
- Do not retain the old TODO claiming that extracting gateway accessors is necessarily the fix; the lazy imports are deliberately protected by their local soft-failure boundaries.
|
||||
- Add the cross-cutting no-unmarked-dynamic-import invariant to `CLAUDE.md`.
|
||||
- Update the current-state entries for `src/core/pglite-engine.ts`, `src/core/postgres-engine.ts`, and `src/core/migrate.ts` in `docs/architecture/KEY_FILES.md` where needed.
|
||||
- Regenerate `llms.txt` and `llms-full.txt` after the documentation edits.
|
||||
- Add a TODO only if implementation uncovers a real unresolved action.
|
||||
|
||||
Public documentation must use generic language and must not overstate the historical Windows crash causality.
|
||||
|
||||
## Verification
|
||||
|
||||
Capture full output to files before inspecting summaries. Run, at minimum:
|
||||
|
||||
1. The guard regression test.
|
||||
2. `bash scripts/check-engine-dynamic-import.sh`.
|
||||
3. Focused tests that exercise the touched engine, migration, retry, audit, and recency modules.
|
||||
4. `bun run typecheck`.
|
||||
5. `bun run verify`.
|
||||
6. `bun run build:llms` followed by `bun test test/build-llms.test.ts`.
|
||||
7. `git diff --check` and a final clean-status/diff review.
|
||||
|
||||
If platform contention or existing Windows suite defects block a broad test, report the exact command, exit code, and ownership classification rather than declaring success from a partial run.
|
||||
|
||||
## Git and publication boundary
|
||||
|
||||
- Work on `claude/kind-meitner-330c90`, reset locally to the exact investigated `origin/master` base.
|
||||
- Preserve the previous worktree tip under `claude/kind-meitner-330c90-pre-reconcile`.
|
||||
- Keep implementation and verification commits local.
|
||||
- Do not push, create a PR, comment upstream, or otherwise publish without explicit user approval after the local result is complete.
|
||||
@@ -13,7 +13,7 @@ Step-by-step walkthroughs that take you from zero to a working outcome. Concrete
|
||||
|
||||
These are the next tutorials on the roadmap. Open an issue if one of them is the one you need most; that's how we'll prioritize.
|
||||
|
||||
- **Set up GBrain for VC dealflow** — the operator's recipe. People pages for founders, companies with typed Facts fence carrying ARR / team-size / runway across dates, meetings auto-ingested, deal pages linking everything. Shows `gbrain whoknows`, `gbrain find_trajectory`, and `gbrain founder scorecard` on real workflows.
|
||||
- **Set up GBrain for VC dealflow** — the operator's recipe. People pages for founders, companies with typed Facts fence carrying ARR / team-size / runway across dates, meetings auto-ingested, deal pages linking everything. Shows `gbrain whoknows`, `gbrain find-trajectory`, and `gbrain founder scorecard` on real workflows.
|
||||
|
||||
- **Migrate your existing vault into GBrain** — for Notion / Obsidian / Roam users with a vault that doesn't match GBrain's default layout. Walks through `gbrain schema detect` → `suggest` → `review-candidates` so the brain learns your shape instead of forcing you to learn its.
|
||||
|
||||
|
||||
@@ -554,7 +554,7 @@ What to do next:
|
||||
|
||||
- **Wire ingestion** from external systems (Granola, Linear, Slack) using the [ingestion source contract](../skillpack-anatomy.md). Most companies want their meetings auto-ingested so the brain stays current without anyone typing notes.
|
||||
- **Set up team-specific dashboards** through the admin UI. Each team lead can have their own view of brain health and activity.
|
||||
- **Explore the rest of the brain layer.** `gbrain whoknows` (find the expert on a topic), `gbrain find_trajectory` (how a metric changed over time), `gbrain founder scorecard` (especially useful for VC and ops teams), the contradiction-detection cycle that surfaces conflicts between different people's notes.
|
||||
- **Explore the rest of the brain layer.** `gbrain whoknows` (find the expert on a topic), `gbrain find-trajectory` (how a metric changed over time), `gbrain founder scorecard` (especially useful for VC and ops teams), the contradiction-detection cycle that surfaces conflicts between different people's notes.
|
||||
|
||||
If you're building in this space (which YC has flagged as the [company-brain category in its Request for Startups](https://www.ycombinator.com/rfs#company-brain)), you might as well build on this. Everything described above is open source, MIT licensed, and what I run in production behind my own AI agents.
|
||||
|
||||
|
||||
@@ -115,21 +115,21 @@ You can use the same keys across multiple agents.
|
||||
|
||||
## Step 6: Install GBrain
|
||||
|
||||
Once OpenClaw is running:
|
||||
Once OpenClaw is running, installation is two commands — one in the brain repo, one in the agent workspace:
|
||||
|
||||
```bash
|
||||
gbrain install
|
||||
# In the BRAIN repo (the git repo that holds your markdown pages):
|
||||
gbrain init --supabase
|
||||
|
||||
# In the AGENT WORKSPACE repo (where OpenClaw runs):
|
||||
gbrain skillpack scaffold --all
|
||||
```
|
||||
|
||||
This installs:
|
||||
`gbrain init --supabase` walks a short wizard that asks for your Supabase connection string and creates the schema. You'll get that connection string in Step 7 — read 7a and 7b first so you paste the right one (the transaction pooler, not the direct connection). If you'd rather try things locally before paying for a database, `gbrain init --pglite` gives you a zero-config embedded engine instead; you can migrate to Supabase later with `gbrain migrate --to supabase`.
|
||||
|
||||
- About 60 skills
|
||||
- About 9 skill packs
|
||||
- Default brain structure
|
||||
- MCP server configuration
|
||||
- Supabase connection (for embeddings and search)
|
||||
`gbrain skillpack scaffold --all` copies the ~43 bundled skills into your agent workspace as first-class files you can edit freely. (The old managed-install model was retired in v0.36.0.0; see `docs/INSTALL.md` if you're upgrading from an older release.)
|
||||
|
||||
GBrain populates the brain repo with its default directory structure, skill files, and configuration. From this point, the agent has working memory and access to every skill.
|
||||
From this point, the agent has working memory and access to every skill.
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -32,7 +32,7 @@ gbrain schema sync --apply
|
||||
The sync backfills `page.type = 'meeting'` on all 4000 pages in 1000-row batches. Now:
|
||||
|
||||
- `gbrain whoknows "Q3 roadmap discussion"` routes through the meeting type, ranking by `expert_routing` signal (attendees, recency, salience) instead of raw text.
|
||||
- `gbrain extract-facts` runs on every meeting page automatically (because `extractable: true`), pulling typed facts like `attended_by=alice-example`, `date=2026-05-23`.
|
||||
- The `extract_facts` cycle runs on every meeting page automatically (because `extractable: true`), pulling typed facts like `attended_by=alice-example`, `date=2026-05-23`.
|
||||
- The downstream `think` skill can now answer "what did we decide about pricing in the last three roadmap meetings" by querying the meeting graph instead of grep'ing 4000 files.
|
||||
|
||||
One command. 4000 pages went from invisible to queryable. The content didn't change. The structure did.
|
||||
@@ -62,7 +62,7 @@ gbrain schema add-link-type led-by --page-type deal --target-type inves
|
||||
gbrain schema sync --apply
|
||||
```
|
||||
|
||||
Now `gbrain whoknows "Series A SaaS"` routes through `investor` and `portco` types specifically, not the noisy general type set. `gbrain graph-query alice-example --type intro-from --depth 2` walks two hops of intros to surface "Alice introduced you to Bob who introduced you to Charlie." `gbrain extract-facts` starts producing typed claims from the fence in your deal pages: `(deals/acme-seed, raise=2000000, valuation=15000000, lead=widget-vc, closed_at=2026-05-23)`.
|
||||
Now `gbrain whoknows "Series A SaaS"` routes through `investor` and `portco` types specifically, not the noisy general type set. `gbrain graph-query alice-example --type intro-from --depth 2` walks two hops of intros to surface "Alice introduced you to Bob who introduced you to Charlie." The `extract_facts` cycle starts producing typed claims from the fence in your deal pages: `(deals/acme-seed, raise=2000000, valuation=15000000, lead=widget-vc, closed_at=2026-05-23)`.
|
||||
|
||||
The CRM you've been promising yourself you'll set up next quarter? You just shipped it in 4 commands. It's downstream of your notes, not parallel to them.
|
||||
|
||||
@@ -143,7 +143,7 @@ Re-run the same `whoknows` query. Top-3 should shift, because the new type is no
|
||||
|
||||
Three things gbrain does that generic note systems can't:
|
||||
|
||||
**1. The brain knows the difference between a person and an idea.** Page-type matters at query time. `gbrain whoknows` only considers `expert_routing: true` types. `gbrain extract-facts` only runs on `extractable: true` types. `gbrain graph-query` walks declared link verbs. None of that works on a flat tag system because tags don't have semantics — they're labels. Types are first-class citizens with rules attached.
|
||||
**1. The brain knows the difference between a person and an idea.** Page-type matters at query time. `gbrain whoknows` only considers `expert_routing: true` types. The `extract_facts` cycle only runs on `extractable: true` types. `gbrain graph-query` walks declared link verbs. None of that works on a flat tag system because tags don't have semantics — they're labels. Types are first-class citizens with rules attached.
|
||||
|
||||
**2. Untyped content is invisible content.** If your meetings are typed as `note`, expert routing skips them, facts extraction ignores them, link inference doesn't fire. They exist on disk and they're indexed for text search, but the structural surfaces (whoknows, find_experts, recall, think) treat them as second-class. Adding a type isn't cosmetic; it's structural promotion.
|
||||
|
||||
|
||||
+4
-17
@@ -216,19 +216,6 @@ Per-file detail is in `docs/architecture/KEY_FILES.md`.
|
||||
text, the cast parses it). Guarded by `scripts/check-jsonb-pattern.sh` (template grep) +
|
||||
`scripts/check-jsonb-params.mjs` (positional AST scanner); the real backstop is the DATABASE_URL-gated
|
||||
e2e parity tests, since PGLite can't surface the bug. Full rule in `docs/ENGINES.md`.
|
||||
- **Engine-live paths avoid runtime dynamic `import()` for helper dependencies.** In
|
||||
`src/core/pglite-engine.ts`, `src/core/postgres-engine.ts`, and
|
||||
`src/core/migrate.ts`, dependencies previously reached through runtime dynamic
|
||||
imports use static top-level imports. The only current dynamic-`import()` exceptions
|
||||
are the four `ai/gateway.ts` lookups in both engines'
|
||||
`initSchema()` and `_upsertChunksOnce()` methods; each remains lazy inside a
|
||||
local `try/catch` because the gateway has a large provider/config closure and,
|
||||
more importantly, eager evaluation would occur before the catch and could
|
||||
turn a recoverable default/config-row fallback into a module-load failure.
|
||||
Every exception carries `engine-dynamic-import-ok` on the import line.
|
||||
`scripts/check-engine-dynamic-import.sh` enforces the rule. For history, use
|
||||
`git log -G'await[[:space:]]+import\\('`, not `git log -S`: a dynamic-to-static
|
||||
rewrite can preserve the searched token while changing its context.
|
||||
- **Engine parity.** `src/core/postgres-engine.ts` and `src/core/pglite-engine.ts` move in
|
||||
lockstep — a new method/SQL shape lands in BOTH, pinned by `test/e2e/engine-parity.test.ts`.
|
||||
Forward-referenced columns/indexes go in the bootstrap probe set (guarded by
|
||||
@@ -2329,7 +2316,7 @@ gbrain schema sync --apply
|
||||
The sync backfills `page.type = 'meeting'` on all 4000 pages in 1000-row batches. Now:
|
||||
|
||||
- `gbrain whoknows "Q3 roadmap discussion"` routes through the meeting type, ranking by `expert_routing` signal (attendees, recency, salience) instead of raw text.
|
||||
- `gbrain extract-facts` runs on every meeting page automatically (because `extractable: true`), pulling typed facts like `attended_by=alice-example`, `date=2026-05-23`.
|
||||
- The `extract_facts` cycle runs on every meeting page automatically (because `extractable: true`), pulling typed facts like `attended_by=alice-example`, `date=2026-05-23`.
|
||||
- The downstream `think` skill can now answer "what did we decide about pricing in the last three roadmap meetings" by querying the meeting graph instead of grep'ing 4000 files.
|
||||
|
||||
One command. 4000 pages went from invisible to queryable. The content didn't change. The structure did.
|
||||
@@ -2359,7 +2346,7 @@ gbrain schema add-link-type led-by --page-type deal --target-type inves
|
||||
gbrain schema sync --apply
|
||||
```
|
||||
|
||||
Now `gbrain whoknows "Series A SaaS"` routes through `investor` and `portco` types specifically, not the noisy general type set. `gbrain graph-query alice-example --type intro-from --depth 2` walks two hops of intros to surface "Alice introduced you to Bob who introduced you to Charlie." `gbrain extract-facts` starts producing typed claims from the fence in your deal pages: `(deals/acme-seed, raise=2000000, valuation=15000000, lead=widget-vc, closed_at=2026-05-23)`.
|
||||
Now `gbrain whoknows "Series A SaaS"` routes through `investor` and `portco` types specifically, not the noisy general type set. `gbrain graph-query alice-example --type intro-from --depth 2` walks two hops of intros to surface "Alice introduced you to Bob who introduced you to Charlie." The `extract_facts` cycle starts producing typed claims from the fence in your deal pages: `(deals/acme-seed, raise=2000000, valuation=15000000, lead=widget-vc, closed_at=2026-05-23)`.
|
||||
|
||||
The CRM you've been promising yourself you'll set up next quarter? You just shipped it in 4 commands. It's downstream of your notes, not parallel to them.
|
||||
|
||||
@@ -2440,7 +2427,7 @@ Re-run the same `whoknows` query. Top-3 should shift, because the new type is no
|
||||
|
||||
Three things gbrain does that generic note systems can't:
|
||||
|
||||
**1. The brain knows the difference between a person and an idea.** Page-type matters at query time. `gbrain whoknows` only considers `expert_routing: true` types. `gbrain extract-facts` only runs on `extractable: true` types. `gbrain graph-query` walks declared link verbs. None of that works on a flat tag system because tags don't have semantics — they're labels. Types are first-class citizens with rules attached.
|
||||
**1. The brain knows the difference between a person and an idea.** Page-type matters at query time. `gbrain whoknows` only considers `expert_routing: true` types. The `extract_facts` cycle only runs on `extractable: true` types. `gbrain graph-query` walks declared link verbs. None of that works on a flat tag system because tags don't have semantics — they're labels. Types are first-class citizens with rules attached.
|
||||
|
||||
**2. Untyped content is invisible content.** If your meetings are typed as `note`, expert routing skips them, facts extraction ignores them, link inference doesn't fire. They exist on disk and they're indexed for text search, but the structural surfaces (whoknows, find_experts, recall, think) treat them as second-class. Adding a type isn't cosmetic; it's structural promotion.
|
||||
|
||||
@@ -3910,7 +3897,7 @@ All 30 GBrain operations are available remotely, including `sync_brain` and
|
||||
directory where `gbrain serve` was launched. Symlinks, `..` traversal, and absolute
|
||||
paths outside cwd are rejected. Page slugs and filenames are allowlist-validated
|
||||
(alphanumeric + hyphens; no control chars, RTL overrides, or backslashes). Local
|
||||
CLI callers (`gbrain file upload ...`) keep unrestricted filesystem access since
|
||||
CLI callers (`gbrain files upload ...`) keep unrestricted filesystem access since
|
||||
the user owns the machine.
|
||||
|
||||
## Deployment Options
|
||||
|
||||
+1
-2
@@ -48,8 +48,7 @@
|
||||
"check:system-of-record": "bash scripts/check-system-of-record.sh",
|
||||
"check:admin-scope-drift": "bash scripts/check-admin-scope-drift.sh",
|
||||
"check:cli-exec": "bash scripts/check-cli-executable.sh",
|
||||
"check:engine-dynamic-import": "bash scripts/check-engine-dynamic-import.sh",
|
||||
"check:all": "bash scripts/check-privacy.sh && bash scripts/check-proposal-pii.sh && bash scripts/check-test-real-names.sh && bash scripts/check-jsonb-pattern.sh && bash scripts/check-source-id-projection.sh && bash scripts/check-source-config-leak.sh && bash scripts/check-progress-to-stdout.sh && bash scripts/check-no-tracked-symlinks.sh && bash scripts/check-no-legacy-getconnection.sh && bash scripts/check-test-isolation.sh && bash scripts/check-trailing-newline.sh && bash scripts/check-wasm-embedded.sh && bash scripts/check-exports-count.sh && bash scripts/check-admin-build.sh && bash scripts/check-admin-scope-drift.sh && bash scripts/check-cli-executable.sh && bash scripts/check-skill-brain-first.sh && bash scripts/check-operations-filter-bypass.sh && bash scripts/check-gateway-routed-no-direct-anthropic.sh && bash scripts/check-worker-pool-atomicity.sh && bash scripts/check-key-files-current-state.sh && bash scripts/check-no-double-retry.sh && bash scripts/check-batch-audit-site.sh && bash scripts/check-engine-dynamic-import.sh",
|
||||
"check:all": "bash scripts/check-privacy.sh && bash scripts/check-proposal-pii.sh && bash scripts/check-test-real-names.sh && bash scripts/check-jsonb-pattern.sh && bash scripts/check-source-id-projection.sh && bash scripts/check-source-config-leak.sh && bash scripts/check-progress-to-stdout.sh && bash scripts/check-no-tracked-symlinks.sh && bash scripts/check-no-legacy-getconnection.sh && bash scripts/check-test-isolation.sh && bash scripts/check-trailing-newline.sh && bash scripts/check-wasm-embedded.sh && bash scripts/check-exports-count.sh && bash scripts/check-admin-build.sh && bash scripts/check-admin-scope-drift.sh && bash scripts/check-cli-executable.sh && bash scripts/check-skill-brain-first.sh && bash scripts/check-operations-filter-bypass.sh && bash scripts/check-gateway-routed-no-direct-anthropic.sh && bash scripts/check-worker-pool-atomicity.sh && bash scripts/check-key-files-current-state.sh && bash scripts/check-no-double-retry.sh && bash scripts/check-batch-audit-site.sh",
|
||||
"check:gateway-routed": "bash scripts/check-gateway-routed-no-direct-anthropic.sh",
|
||||
"check:worker-pool-atomicity": "bash scripts/check-worker-pool-atomicity.sh",
|
||||
"check:doc-history": "bash scripts/check-key-files-current-state.sh",
|
||||
|
||||
@@ -1,41 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# CI guard: every `bun test` invocation in workflows and runner scripts must
|
||||
# pass an explicit --timeout.
|
||||
#
|
||||
# Why: bun ignores bunfig.toml's `timeout` key (verified on 1.3.14), so a bare
|
||||
# `bun test` gets the 5000ms default for BOTH tests and beforeAll/beforeEach/
|
||||
# afterAll/afterEach hooks. Hooks do NOT inherit a test's third-arg timeout —
|
||||
# a file whose tests all declare `}, 30_000)` still has a 5s hook budget, and
|
||||
# slow setup (Postgres connect + migrations, PGLite cold start) flakes on
|
||||
# loaded CI runners with the signature `(unnamed) [5001ms] ... hook timed out`
|
||||
# (the #3545 jsonb-parity failure). The CLI --timeout flag is the one measured
|
||||
# mechanism that raises the hook budget uniformly; per-hook second-arg
|
||||
# timeouts work too but don't scale to ~400 slow hooks.
|
||||
#
|
||||
# Usage: scripts/check-bun-test-timeout.sh
|
||||
# Exit: 0 when clean, 1 when a bare `bun test` invocation is found.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(git rev-parse --show-toplevel 2>/dev/null || pwd)"
|
||||
cd "$ROOT"
|
||||
|
||||
# Match executable `bun test` invocations. Exclude comment lines (#, //, *)
|
||||
# and lines that already carry --timeout anywhere.
|
||||
# Scope: workflows + runner scripts (the surfaces CI executes). package.json
|
||||
# script bodies route through scripts/ already; editing it is out of scope here.
|
||||
violations="$(grep -rnE '\bbun test\b' .github/workflows scripts 2>/dev/null \
|
||||
| grep -v -- '--timeout' \
|
||||
| grep -vE ':[[:space:]]*(#|//|\*)' \
|
||||
| grep -v 'check-bun-test-timeout' \
|
||||
|| true)"
|
||||
|
||||
if [ -n "$violations" ]; then
|
||||
echo "FAIL: bare 'bun test' without --timeout (5s default kills slow setup hooks):" >&2
|
||||
echo "$violations" >&2
|
||||
echo "" >&2
|
||||
echo "Add --timeout=60000 (see scripts/run-unit-shard.sh for the convention)." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "OK: every bun test invocation passes an explicit --timeout."
|
||||
@@ -1,31 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Engine-live paths use static imports by default. A line-level
|
||||
# `engine-dynamic-import-ok` marker is required for a justified lazy import.
|
||||
#
|
||||
# Historical Windows runs associated imports on these paths with abrupt Bun
|
||||
# test-process exits, but system-wide commit exhaustion remained a confound.
|
||||
# This guard therefore enforces a reviewed engine-path hardening invariant; it
|
||||
# does not claim every dynamic import deterministically crashes Windows.
|
||||
#
|
||||
# Usage:
|
||||
# bash scripts/check-engine-dynamic-import.sh
|
||||
# bash scripts/check-engine-dynamic-import.sh FILE [FILE...]
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" || exit 1
|
||||
|
||||
if [ "$#" -gt 0 ]; then
|
||||
FILES=("$@")
|
||||
else
|
||||
ROOT="$(git -C "$SCRIPT_DIR/.." rev-parse --show-toplevel 2>/dev/null || true)"
|
||||
[ -n "$ROOT" ] || ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
cd "$ROOT" || exit 1
|
||||
FILES=(
|
||||
src/core/pglite-engine.ts
|
||||
src/core/postgres-engine.ts
|
||||
src/core/migrate.ts
|
||||
)
|
||||
fi
|
||||
|
||||
exec bun "$SCRIPT_DIR/check-engine-dynamic-import.ts" "${FILES[@]}"
|
||||
@@ -1,80 +0,0 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
import { readFile } from 'node:fs/promises';
|
||||
import ts from 'typescript';
|
||||
|
||||
const MARKER = 'engine-dynamic-import-ok';
|
||||
const MARKER_TOKEN_CHAR = /[\p{ID_Continue}$-]/u;
|
||||
const files = process.argv.slice(2);
|
||||
const violations: string[] = [];
|
||||
const readErrors: string[] = [];
|
||||
|
||||
for (const file of files) {
|
||||
let sourceText: string;
|
||||
try {
|
||||
sourceText = await readFile(file, 'utf8');
|
||||
} catch (error) {
|
||||
const detail = error instanceof Error ? error.message : String(error);
|
||||
readErrors.push(`ERROR: cannot read input file ${file}: ${detail}`);
|
||||
continue;
|
||||
}
|
||||
|
||||
const sourceFile = ts.createSourceFile(
|
||||
file,
|
||||
sourceText,
|
||||
ts.ScriptTarget.Latest,
|
||||
true,
|
||||
ts.ScriptKind.TS,
|
||||
);
|
||||
const lines = sourceText.split(/\r?\n/);
|
||||
const markerLines = new Set<number>();
|
||||
|
||||
if (sourceFile.parseDiagnostics.length > 0) {
|
||||
const diagnostics = sourceFile.parseDiagnostics
|
||||
.map((diagnostic) => ts.flattenDiagnosticMessageText(diagnostic.messageText, ' '))
|
||||
.join('; ');
|
||||
readErrors.push(`ERROR: cannot parse input file ${file}: ${diagnostics}`);
|
||||
}
|
||||
|
||||
for (let markerPos = sourceText.indexOf(MARKER); markerPos >= 0; markerPos = sourceText.indexOf(MARKER, markerPos + MARKER.length)) {
|
||||
const before = Array.from(sourceText.slice(0, markerPos)).at(-1);
|
||||
const after = Array.from(sourceText.slice(markerPos + MARKER.length))[0];
|
||||
const standaloneMarker = (!before || !MARKER_TOKEN_CHAR.test(before))
|
||||
&& (!after || !MARKER_TOKEN_CHAR.test(after));
|
||||
const token = ts.getTokenAtPosition(sourceFile, markerPos);
|
||||
const insideToken = token.getStart(sourceFile) <= markerPos && markerPos < token.end;
|
||||
if (standaloneMarker && !insideToken) {
|
||||
markerLines.add(sourceFile.getLineAndCharacterOfPosition(markerPos).line);
|
||||
}
|
||||
}
|
||||
|
||||
function visit(node: ts.Node): void {
|
||||
if (ts.isCallExpression(node) && node.expression.kind === ts.SyntaxKind.ImportKeyword) {
|
||||
const { line } = sourceFile.getLineAndCharacterOfPosition(node.expression.getStart(sourceFile));
|
||||
const sourceLine = lines[line] ?? '';
|
||||
if (!markerLines.has(line)) {
|
||||
violations.push(` ${file}:${line + 1}:${sourceLine}`);
|
||||
}
|
||||
}
|
||||
ts.forEachChild(node, visit);
|
||||
}
|
||||
|
||||
visit(sourceFile);
|
||||
}
|
||||
|
||||
for (const error of readErrors) console.error(error);
|
||||
|
||||
if (violations.length > 0) {
|
||||
console.error('ERROR: unreviewed dynamic import on an engine-live path:');
|
||||
console.error();
|
||||
console.error(violations.join('\n'));
|
||||
console.error();
|
||||
console.error('Prefer a static top-level import. If lazy loading is load-bearing,');
|
||||
console.error("append 'engine-dynamic-import-ok' to that exact line and document");
|
||||
console.error('the startup or soft-failure boundary that requires it.');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (readErrors.length > 0) process.exit(1);
|
||||
|
||||
console.log(`check-engine-dynamic-import: ok (${files.length} file(s) scanned)`);
|
||||
@@ -1,114 +0,0 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Import an envelope-v0 file (a JSON serialization of AI chat history; format
|
||||
* spec: github.com/memvelope/memvelope) into a brain repo as one Markdown page
|
||||
* per conversation, which `gbrain sync` ingests.
|
||||
*
|
||||
* Usage:
|
||||
* node scripts/envelope-to-gbrain.mjs <envelope.mve.json> [outDir]
|
||||
*
|
||||
* Zero dependencies. Deterministic. No network. It does NOT call gbrain — it
|
||||
* only writes Markdown files.
|
||||
*
|
||||
* Output layout:
|
||||
* - One page per conversation, filename = date + conversation id (shared
|
||||
* titles cannot collide; the id is the natural key). A duplicate id
|
||||
* overwrites its own filename and warns on stderr; stdout reports DISTINCT
|
||||
* files written, not write calls.
|
||||
* - Frontmatter: `type: conversation` (keeps pages eligible for
|
||||
* conversation-facts extraction and chronicle behavior after sync), the
|
||||
* source provider, the conversation id, and `origin: memvelope/envelope-v0`.
|
||||
* - Page `date` is the first 10 chars of the conversation's ISO-8601
|
||||
* `created_at`. Body keeps message-id citations beside each speaker turn.
|
||||
*
|
||||
* Memory: the whole envelope is held in memory (no streaming); envelopes are
|
||||
* far smaller than the vendor exports they serialize.
|
||||
*
|
||||
* Verify:
|
||||
* node scripts/envelope-to-gbrain.mjs test/fixtures/memvelope/sample.mve.json /tmp/out
|
||||
* -> expect "wrote 1 markdown page(s)"
|
||||
* bun test test/envelope-to-gbrain.test.ts
|
||||
*
|
||||
* STATUS: live-verified against gbrain v0.42.56.0 on 2026-07-03: the sample
|
||||
* fixture -> 1 page; a real 662MB Claude export -> 353 conversations = 353
|
||||
* distinct pages (no collisions), searchable after sync with provenance and
|
||||
* message-id citations intact.
|
||||
*/
|
||||
|
||||
import { readFileSync, writeFileSync, mkdirSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
|
||||
const [, , envelopePath, outDir = './brain/conversations'] = process.argv;
|
||||
if (!envelopePath) {
|
||||
console.error('usage: node envelope-to-gbrain.mjs <envelope.mve.json> [outDir]');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const env = JSON.parse(readFileSync(envelopePath, 'utf8'));
|
||||
if (env.memvelope !== 'envelope-v0') {
|
||||
console.error(`not an envelope-v0 file (memvelope field = ${JSON.stringify(env.memvelope)})`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const slug = (s, fallback) =>
|
||||
(String(s || '').toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '') || fallback).slice(0, 60);
|
||||
|
||||
mkdirSync(outDir, { recursive: true });
|
||||
const filesWritten = new Set();
|
||||
let collisions = 0;
|
||||
const conversations = env.conversations || [];
|
||||
for (const [i, c] of conversations.entries()) {
|
||||
const date = (c.created_at || '').slice(0, 10);
|
||||
// Name the file by the conversation's own id — the natural unique key — so two
|
||||
// conversations that share a date and title can never silently overwrite each
|
||||
// other. The date only leads as a human/chronological sort prefix; the id
|
||||
// carries uniqueness. Positional fallback keeps names unique and deterministic
|
||||
// when an envelope omits an id.
|
||||
// One predicate for "this conversation carries its own id", shared by the
|
||||
// filename and the frontmatter below. Keeping it in a single place is what
|
||||
// stops the two from disagreeing about whether an id exists.
|
||||
const hasId = typeof c.id === 'string' && c.id.trim() !== '';
|
||||
const convId = hasId ? c.id.trim() : `conv-${i + 1}`;
|
||||
const name = `${date || '0000-00-00'}-${slug(convId, `conv-${i + 1}`)}.md`;
|
||||
// gbrain reads YAML frontmatter + markdown body; keep provenance in frontmatter.
|
||||
// Emit `type: conversation` so gbrain stores these as conversation pages rather
|
||||
// than defaulting to the generic `concept`. gbrain is open-typed — it takes an
|
||||
// explicit frontmatter `type` verbatim — and its conversation-aware features
|
||||
// (conversation-facts extraction, the conversation_format_coverage check,
|
||||
// chronicle eligibility) key off `type == 'conversation'`.
|
||||
const front = [
|
||||
'---',
|
||||
'type: conversation',
|
||||
`title: ${JSON.stringify(c.title || 'Untitled conversation')}`,
|
||||
`date: ${date || 'null'}`,
|
||||
// Every interpolated value is quoted. An envelope is a third-party file, so
|
||||
// a provider string carrying a newline would otherwise close this scalar and
|
||||
// inject arbitrary frontmatter keys into the page gbrain ingests.
|
||||
`source: ${JSON.stringify(env.meta?.source_provider || 'unknown')}`,
|
||||
// Omit the key entirely when the envelope carries no id, rather than
|
||||
// emitting the literal `undefined` or a synthesized `conv-N` — the positional
|
||||
// fallback names the file, but it is not a memvelope conversation id and
|
||||
// must not be recorded as one.
|
||||
...(hasId ? [`memvelope_conversation_id: ${JSON.stringify(convId)}`] : []),
|
||||
'origin: memvelope/envelope-v0',
|
||||
'---',
|
||||
'',
|
||||
].join('\n');
|
||||
const body = (c.messages || [])
|
||||
.map((m) => `**${m.role === 'user' ? 'Me' : 'Assistant'}** (${m.ts || 'no timestamp'} · ${m.id}):\n\n${m.text}`)
|
||||
.join('\n\n---\n\n');
|
||||
// Never lose a page silently: if two conversations still map to the same
|
||||
// filename (e.g. an envelope carrying duplicate ids), warn loudly instead of
|
||||
// overwriting in silence, and report the count of DISTINCT files written — not
|
||||
// the number of write calls, which is what hid the old title-collision bug.
|
||||
if (filesWritten.has(name)) {
|
||||
collisions += 1;
|
||||
console.warn(`warning: filename collision on "${name}" — conversation id ${JSON.stringify(c.id)} is not unique; overwriting the earlier page.`);
|
||||
}
|
||||
writeFileSync(join(outDir, name), front + `# ${c.title || 'Conversation'}\n\n` + body + '\n');
|
||||
filesWritten.add(name);
|
||||
}
|
||||
console.log(`wrote ${filesWritten.size} markdown page(s) to ${outDir} — point gbrain's sync at this directory.`);
|
||||
if (collisions) {
|
||||
console.warn(`warning: ${collisions} filename collision(s) — ${collisions} page(s) overwritten. Deduplicate conversation ids in the envelope to avoid data loss.`);
|
||||
}
|
||||
+2
-3
@@ -162,9 +162,8 @@ for f in "${files[@]}"; do
|
||||
if [ -n "${DATABASE_URL:-}" ]; then
|
||||
psql "$DATABASE_URL" -At -c "SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE pid != pg_backend_pid() AND datname = current_database()" >/dev/null 2>&1 || true
|
||||
fi
|
||||
# Hard outer timeout (180s per file). bun's --timeout covers tests AND
|
||||
# hooks (measured on 1.3.14), but it's timer-based: a PGLite WASM call
|
||||
# that blocks the event loop synchronously never lets the timer fire and
|
||||
# Hard outer timeout (180s per file). bun's --timeout is per-test; if a
|
||||
# PGLite WASM call hangs in beforeAll/afterAll, --timeout never fires and
|
||||
# the file wedges indefinitely. gtimeout/timeout SIGKILLs the file so the
|
||||
# suite advances. gtimeout (macOS via coreutils) preferred; timeout (Linux)
|
||||
# fallback; bare bun (no outer cap) if neither is installed.
|
||||
|
||||
@@ -64,7 +64,6 @@ CHECKS=(
|
||||
"check:source-scope-onboard"
|
||||
"check:no-double-retry"
|
||||
"check:batch-audit-site"
|
||||
"check:engine-dynamic-import"
|
||||
"check:worker-lock-renewal-shape"
|
||||
"typecheck"
|
||||
)
|
||||
|
||||
@@ -248,7 +248,7 @@ before submission.
|
||||
After the brain page is written, render to PDF using `skills/brain-pdf`:
|
||||
|
||||
```bash
|
||||
gbrain put_page # already done by the CLI; nothing to add here
|
||||
gbrain put # already done by the CLI; nothing to add here
|
||||
# Then invoke brain-pdf:
|
||||
# (see skills/brain-pdf/SKILL.md for the make-pdf invocation)
|
||||
```
|
||||
|
||||
@@ -73,13 +73,13 @@ stock worker auto-loads on startup) registers handlers before `start()`.
|
||||
Users who set `minion_mode: off` in `~/.gbrain/preferences.json` keep
|
||||
using `agentTurn`. Respect that. No auto-rewrite.
|
||||
|
||||
## Forward note (v0.12.0)
|
||||
## Forward note
|
||||
|
||||
GBrain v0.12.0 ships `gbrain cron`: a scheduler loop inside
|
||||
`gbrain jobs work` that owns cron expressions natively — no more
|
||||
handing off to host schedulers. Until v0.12.0 lands, the host
|
||||
scheduler keeps firing on schedule; v0.11.1 only replaces the execution
|
||||
layer (what the cron trigger *does*), not the scheduling layer.
|
||||
A native scheduler loop inside `gbrain jobs work` (owning cron
|
||||
expressions directly, with no host-scheduler hand-off) has been on the
|
||||
roadmap since v0.11.1 but has not shipped. The host scheduler keeps
|
||||
firing on schedule; this convention only replaces the execution layer
|
||||
(what the cron trigger *does*), not the scheduling layer.
|
||||
|
||||
## Related
|
||||
|
||||
|
||||
@@ -54,8 +54,8 @@ Ask the user what they want to track. Either:
|
||||
- Define a custom recipe with: source queries, classification rules, extraction schema,
|
||||
tracker page path, tracker format
|
||||
|
||||
Recipes are YAML files at `~/.gbrain/recipes/{name}.yaml`. Use `gbrain research init`
|
||||
to scaffold a new one.
|
||||
Recipes are YAML files at `~/.gbrain/recipes/{name}.yaml`. Scaffold a new one by
|
||||
copying a built-in recipe file and editing its fields.
|
||||
|
||||
### Phase 2: Search Sources
|
||||
|
||||
|
||||
@@ -201,7 +201,7 @@ Use the brain page template. MUST include:
|
||||
|
||||
### 4b. Entity pages (people, companies)
|
||||
For each entity mentioned:
|
||||
- Check if a brain page exists (`gbrain search "<name>"` or `gbrain get_page people/<slug>`).
|
||||
- Check if a brain page exists (`gbrain search "<name>"` or `gbrain get people/<slug>`).
|
||||
- If exists: update State, append Timeline entry citing this research.
|
||||
- If not: create with enrichment.
|
||||
|
||||
|
||||
@@ -112,7 +112,7 @@ gbrain query "<topic keywords>"
|
||||
# -d '{"model": "sonar-pro", "messages": [{"role":"user","content":"..."}]}'
|
||||
|
||||
# 4. Write the structured research page via put_page:
|
||||
gbrain put_page research/<slug> # via the put_page operation
|
||||
gbrain put research/<slug> # via the put_page operation
|
||||
|
||||
# 5. Cross-link entities mentioned (people, companies) per Iron Law.
|
||||
```
|
||||
|
||||
@@ -11,7 +11,7 @@ tools:
|
||||
- gbrain schema active
|
||||
- gbrain schema use
|
||||
- gbrain schema stats
|
||||
- gbrain pages restore
|
||||
- gbrain restore
|
||||
- mcp:run_onboard
|
||||
triggers:
|
||||
- "unify my types"
|
||||
@@ -143,7 +143,7 @@ WHERE source_id = 'default' AND frontmatter->>'legacy_type' IS NOT NULL;
|
||||
Page-to-alias and page-to-link source pages soft-delete with 72h TTL. Restore within that window:
|
||||
|
||||
```bash
|
||||
gbrain pages restore <slug>
|
||||
gbrain restore <slug>
|
||||
```
|
||||
|
||||
Revert the active pack flip:
|
||||
@@ -197,7 +197,7 @@ Outputs:
|
||||
- Active pack flipped to `gbrain-base-v2` atomically at end of successful run.
|
||||
|
||||
Side effects:
|
||||
- Source pages soft-deleted with 72h restore TTL (`gbrain pages restore <slug>`).
|
||||
- Source pages soft-deleted with 72h restore TTL (`gbrain restore <slug>`).
|
||||
- One-time cache invalidation on KNOBS_HASH_VERSION bump (5→6); self-healing in `cache.ttl_seconds`.
|
||||
- Query-time `--type X` alias-expands via `expandTypeFilter` (D14 back-compat).
|
||||
|
||||
@@ -212,7 +212,7 @@ DON'T:
|
||||
- Submit `unify-types` directly via the MCP `submit_job` op without `--allow-protected`. PROTECTED handlers require trusted local callers; remote MCP rejection is the intentional trust boundary.
|
||||
- Edit `mapping_rules` in `gbrain-base-v2.yaml` to skip clusters you don't trust. Fork the pack instead (`gbrain schema fork`) so the source-of-truth migration stays consistent across brains.
|
||||
- Run `unify-types` from inside an autopilot tick. The check is `manual_only` per D17 — autopilot deliberately never auto-fires it because pack upgrades are one-time consenting taxonomy decisions.
|
||||
- Hard-delete soft-deleted source pages before the 72h restore window. Use `gbrain pages restore <slug>` first if rollback is needed.
|
||||
- Hard-delete soft-deleted source pages before the 72h restore window. Use `gbrain restore <slug>` first if rollback is needed.
|
||||
- Assume `frontmatter.legacy_type` survives every roundtrip. The marker is canonical for the immediate post-migration window; downstream re-imports may overwrite it.
|
||||
|
||||
## Output Format
|
||||
|
||||
@@ -43,8 +43,9 @@ The Analysis section can interpret; the transcript section is sacred.
|
||||
|
||||
The user sends an audio or voice message via any channel (Telegram, voice
|
||||
memo upload, openclaw audio attachment). The host agent typically provides
|
||||
the transcript text. If not, transcribe via `gbrain transcription` (Groq
|
||||
Whisper by default; OpenAI fallback for audio > 25MB segmented via ffmpeg).
|
||||
the transcript text. If not, transcribe it with your host's transcription
|
||||
tool (Groq Whisper is fast and cheap; OpenAI Whisper works too — segment
|
||||
audio > 25MB via ffmpeg first).
|
||||
|
||||
## The pipeline
|
||||
|
||||
@@ -52,8 +53,9 @@ Whisper by default; OpenAI fallback for audio > 25MB segmented via ffmpeg).
|
||||
1. STORE → Upload original audio to gbrain storage backend
|
||||
(S3 / Supabase Storage / local — pluggable per
|
||||
src/core/storage.ts).
|
||||
2. TRANSCRIBE → Use the agent-provided transcript verbatim, OR call
|
||||
gbrain transcription if no transcript was supplied.
|
||||
2. TRANSCRIBE → Use the agent-provided transcript verbatim, OR
|
||||
transcribe the audio yourself (see "When to invoke")
|
||||
if no transcript was supplied.
|
||||
3. ROUTE → Apply the decision tree (below) to find the right
|
||||
destination directory.
|
||||
4. WRITE → Create / update the destination brain page; preserve the
|
||||
|
||||
+20
-1
@@ -55,12 +55,17 @@ export function bigintToStringReplacer(_key: string, value: unknown): unknown {
|
||||
}
|
||||
|
||||
// CLI-only commands that bypass the operation layer
|
||||
export const CLI_ONLY = new Set(['init', 'reinit-pglite', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'sync', 'extract', 'extract-conversation-facts', 'enrich', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'maintain', 'sources', 'mounts', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test', 'providers', 'storage', 'repos', 'code-def', 'code-refs', 'reindex', 'reindex-code', 'reindex-frontmatter', 'code-callers', 'code-callees', 'reconcile-links', 'frontmatter', 'auth', 'friction', 'claw-test', 'book-mirror', 'takes', 'think', 'salience', 'anomalies', 'calibration', 'transcripts', 'models', 'remote', 'recall', 'forget', 'edges-backfill', 'cache', 'ze-switch', 'retrieval-upgrade', 'founder', 'brainstorm', 'lsd', 'schema', 'capture', 'onboard', 'conversation-parser', 'status', 'connect', 'skillopt', 'quarantine', 'self-upgrade', 'advisor', 'watch', 'reindex-search-vector', 'backfill']);
|
||||
export const CLI_ONLY = new Set(['init', 'reinit-pglite', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'sync', 'extract', 'extract-conversation-facts', 'enrich', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'maintain', 'sources', 'mounts', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test', 'providers', 'storage', 'repos', 'code-def', 'code-refs', 'reindex', 'reindex-code', 'reindex-frontmatter', 'code-callers', 'code-callees', 'reconcile-links', 'frontmatter', 'auth', 'friction', 'claw-test', 'book-mirror', 'takes', 'think', 'salience', 'anomalies', 'calibration', 'transcripts', 'models', 'remote', 'recall', 'forget', 'edges-backfill', 'cache', 'ze-switch', 'retrieval-upgrade', 'founder', 'brainstorm', 'lsd', 'schema', 'capture', 'onboard', 'conversation-parser', 'status', 'connect', 'skillopt', 'quarantine', 'self-upgrade', 'advisor', 'watch', 'reindex-search-vector', 'pages', 'bench', 'backfill']);
|
||||
// CLI-only commands whose handlers print their own --help text. These are
|
||||
// excluded from the generic short-circuit so detailed per-command and
|
||||
// per-subcommand usage stays reachable.
|
||||
const CLI_ONLY_SELF_HELP = new Set([
|
||||
'upgrade', 'post-upgrade', 'check-update',
|
||||
// #3502 sweep: pages + bench print their own usage (pages.ts printHelp,
|
||||
// bench-publish.ts printHelp). Both were documented but undispatchable —
|
||||
// `pages` had a live handleCliOnly case but was missing from CLI_ONLY
|
||||
// (the #2035 calibration bug class); `bench` was never wired at all.
|
||||
'pages', 'bench',
|
||||
'embed', 'config',
|
||||
'skillpack', 'skillpack-check',
|
||||
'integrations', 'friction',
|
||||
@@ -1265,6 +1270,20 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
await runInit(args);
|
||||
return;
|
||||
}
|
||||
if (command === 'bench') {
|
||||
// #3502 sweep: `gbrain bench publish` was documented (docs/eval-bench.md,
|
||||
// KEY_FILES.md, and eval-gate's own --help text) but never dispatched —
|
||||
// the promised-but-unwired class retrieval-upgrade (#3390) fixed before.
|
||||
// Pure file-in/file-out (NDJSON → baseline); no DB, no engine.
|
||||
if (args[0] === 'publish') {
|
||||
const { runBenchPublish } = await import('./commands/bench-publish.ts');
|
||||
await runBenchPublish(args.slice(1));
|
||||
return;
|
||||
}
|
||||
console.error('Usage: gbrain bench publish --from <captured.ndjson> --to <X.baseline.ndjson> [flags]');
|
||||
console.error('Run `gbrain bench publish --help` for the full flag list.');
|
||||
process.exit(args[0] === '--help' || args[0] === '-h' ? 0 : 2);
|
||||
}
|
||||
// v0.37 fix wave (deferred TODO, shipped): one-command wipe-and-reinit.
|
||||
// Spawns its own engine internally so no pre-bound engine needed.
|
||||
if (command === 'reinit-pglite') {
|
||||
|
||||
+2
-12
@@ -4349,18 +4349,8 @@ export async function checkCycleFreshness(
|
||||
: `'${source.id}'`;
|
||||
const raw = source.config?.last_full_cycle_at;
|
||||
if (typeof raw !== 'string') {
|
||||
// #2540: WARN, not FAIL. This check iterates EVERY local_path source,
|
||||
// so on a multi-source install where only some vaults are cycled
|
||||
// (e.g. one nightly `gbrain dream --dir <vault>`), a never-cycled
|
||||
// sibling source turned doctor permanently red — which erodes the
|
||||
// check's signal until real staleness hides inside the noise (the
|
||||
// reporter's install masked genuinely stale sources for weeks this
|
||||
// way). "Never cycled" also fires on a source added minutes ago.
|
||||
// A source that HAS cycled and then went stale still escalates
|
||||
// through the warn/fail age thresholds below — that is the
|
||||
// regression signal this check exists for.
|
||||
issues.push(`Source ${display} has never completed a full cycle`);
|
||||
hasWarnings = true;
|
||||
hasFailures = true;
|
||||
continue;
|
||||
}
|
||||
const last = new Date(raw).getTime();
|
||||
@@ -4396,7 +4386,7 @@ export async function checkCycleFreshness(
|
||||
return {
|
||||
name: 'cycle_freshness',
|
||||
status: 'warn',
|
||||
message: `${issues.join('; ')}. Run \`gbrain dream --source <id>\` to cycle a source, or start \`gbrain autopilot\`.`,
|
||||
message: `${issues.join('; ')}.`,
|
||||
};
|
||||
}
|
||||
return {
|
||||
|
||||
+2
-10
@@ -16,7 +16,7 @@ interface FileRecord {
|
||||
filename: string;
|
||||
storage_path: string;
|
||||
mime_type: string | null;
|
||||
size_bytes: number | bigint | string | null;
|
||||
size_bytes: number;
|
||||
content_hash: string;
|
||||
metadata: Record<string, unknown>;
|
||||
created_at: string;
|
||||
@@ -42,14 +42,6 @@ function fileHash(filePath: string): string {
|
||||
return createHash('sha256').update(content).digest('hex');
|
||||
}
|
||||
|
||||
export function formatFileSizeKb(rawSizeBytes: number | bigint | string | null): string {
|
||||
if (rawSizeBytes == null) return '?';
|
||||
const sizeBytes = Number(rawSizeBytes);
|
||||
return Number.isFinite(sizeBytes) && sizeBytes >= 0
|
||||
? `${Math.round(sizeBytes / 1024)}KB`
|
||||
: '?';
|
||||
}
|
||||
|
||||
export async function runFiles(engine: BrainEngine, args: string[]) {
|
||||
const subcommand = args[0];
|
||||
|
||||
@@ -124,7 +116,7 @@ async function listFiles(engine: BrainEngine, slug?: string) {
|
||||
|
||||
console.log(`${rows.length} file(s):`);
|
||||
for (const row of rows) {
|
||||
const size = formatFileSizeKb(row.size_bytes as FileRecord['size_bytes']);
|
||||
const size = row.size_bytes ? `${Math.round(Number(row.size_bytes) / 1024)}KB` : '?';
|
||||
console.log(` ${row.page_slug || '(unlinked)'} / ${row.filename} [${size}, ${row.mime_type || '?'}]`);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
|
||||
import express from 'express';
|
||||
import type { Request, Response, NextFunction } from 'express';
|
||||
import type { Server as HttpServer } from 'http';
|
||||
import cookieParser from 'cookie-parser';
|
||||
import cors from 'cors';
|
||||
import rateLimit from 'express-rate-limit';
|
||||
@@ -47,7 +46,6 @@ import {
|
||||
type IngestionEvent,
|
||||
} from '../core/ingestion/types.ts';
|
||||
import { resolveOwnerHolder } from '../core/owner-holder.ts';
|
||||
import { registerCleanup } from '../core/process-cleanup.ts';
|
||||
|
||||
/**
|
||||
* /health endpoint timeout. 3s rather than 5s: Fly.io's default
|
||||
@@ -57,71 +55,6 @@ import { registerCleanup } from '../core/process-cleanup.ts';
|
||||
*/
|
||||
export const HEALTH_TIMEOUT_MS = 3000;
|
||||
|
||||
/** Exported so tests can type their structural fakes exactly (#3599). */
|
||||
export type HttpServerLifecycle = Pick<HttpServer, 'listening' | 'once' | 'off' | 'close'>;
|
||||
/** Exported so tests can type their structural fakes exactly (#3599). */
|
||||
export type SignalSource = Pick<NodeJS.Process, 'once' | 'off'>;
|
||||
type CleanupRegistrar = typeof registerCleanup;
|
||||
|
||||
/**
|
||||
* Keep the HTTP server strongly referenced and make the daemon lifetime
|
||||
* explicit instead of relying on runtime-specific event-loop behavior for an
|
||||
* unobserved `app.listen()` return value. The shared abnormal-termination
|
||||
* cleanup pass closes it before process exit.
|
||||
*/
|
||||
export function waitForHttpServerLifecycle(
|
||||
server: HttpServerLifecycle,
|
||||
options: {
|
||||
signals?: SignalSource;
|
||||
register?: CleanupRegistrar;
|
||||
} = {},
|
||||
): Promise<void> {
|
||||
const signals = options.signals ?? process;
|
||||
const register = options.register ?? registerCleanup;
|
||||
|
||||
return new Promise<void>((resolve, reject) => {
|
||||
let settled = false;
|
||||
let closePromise: Promise<void> | null = null;
|
||||
|
||||
const closeServer = (): Promise<void> => {
|
||||
if (closePromise) return closePromise;
|
||||
closePromise = new Promise<void>((closeResolve, closeReject) => {
|
||||
if (!server.listening) {
|
||||
closeResolve();
|
||||
return;
|
||||
}
|
||||
server.close((error?: Error) => {
|
||||
if (error) closeReject(error);
|
||||
else closeResolve();
|
||||
});
|
||||
});
|
||||
return closePromise;
|
||||
};
|
||||
|
||||
const deregister = register('http-server', closeServer);
|
||||
|
||||
const finish = (error?: Error) => {
|
||||
if (settled) return;
|
||||
settled = true;
|
||||
server.off('close', onClose);
|
||||
server.off('error', onError);
|
||||
signals.off('SIGINT', onSigint);
|
||||
deregister();
|
||||
if (error) reject(error);
|
||||
else resolve();
|
||||
};
|
||||
const onClose = () => finish();
|
||||
const onError = (error: Error) => finish(error);
|
||||
const onSigint = () => {
|
||||
void closeServer().catch(onError);
|
||||
};
|
||||
|
||||
server.once('close', onClose);
|
||||
server.once('error', onError);
|
||||
signals.once('SIGINT', onSigint);
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* v0.36.1.x #1024: bootstrap token resolution.
|
||||
*
|
||||
@@ -202,25 +135,6 @@ export type ProbeHealthResult =
|
||||
| { ok: true; status: 200; body: { status: 'ok'; version: string; engine: string; [k: string]: unknown } }
|
||||
| { ok: false; status: 503; body: { error: 'service_unavailable'; error_description: string } };
|
||||
|
||||
/** Exported so tests can type their structural fakes exactly (#3598). */
|
||||
export type AdminSseResponse = Pick<Response, 'setHeader' | 'flushHeaders' | 'write'>;
|
||||
|
||||
/**
|
||||
* Complete the admin EventSource handshake immediately.
|
||||
*
|
||||
* `flushHeaders()` alone can leave reverse proxies and browsers waiting for
|
||||
* the first response body bytes. An SSE comment is protocol-valid, ignored by
|
||||
* EventSource consumers, and makes the stream observable end-to-end without
|
||||
* fabricating an application event.
|
||||
*/
|
||||
export function openAdminSseStream(res: AdminSseResponse): void {
|
||||
res.setHeader('Content-Type', 'text/event-stream');
|
||||
res.setHeader('Cache-Control', 'no-cache');
|
||||
res.setHeader('Connection', 'keep-alive');
|
||||
res.flushHeaders();
|
||||
res.write(': connected\n\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure async health probe. Races `engine.getStats()` against a timeout,
|
||||
* returns a tagged result. No Express coupling — easy to unit-test with a
|
||||
@@ -1718,7 +1632,10 @@ export async function runServeHttp(engine: BrainEngine, options: ServeHttpOption
|
||||
// SSE live activity feed
|
||||
// ---------------------------------------------------------------------------
|
||||
app.get('/admin/events', requireAdmin, (req: Request, res: Response) => {
|
||||
openAdminSseStream(res);
|
||||
res.setHeader('Content-Type', 'text/event-stream');
|
||||
res.setHeader('Cache-Control', 'no-cache');
|
||||
res.setHeader('Connection', 'keep-alive');
|
||||
res.flushHeaders();
|
||||
|
||||
sseClients.add(res);
|
||||
req.on('close', () => sseClients.delete(res));
|
||||
@@ -2493,7 +2410,7 @@ export async function runServeHttp(engine: BrainEngine, options: ServeHttpOption
|
||||
// ---------------------------------------------------------------------------
|
||||
const clientCount = await sql`SELECT count(*)::int as count FROM oauth_clients`;
|
||||
|
||||
const httpServer = app.listen(port, bind, () => {
|
||||
app.listen(port, bind, () => {
|
||||
console.error(`
|
||||
╔══════════════════════════════════════════════════════╗
|
||||
║ GBrain MCP Server v${VERSION.padEnd(37)}║
|
||||
@@ -2518,6 +2435,4 @@ ${bootstrapFromEnv
|
||||
: `║ Admin Token (paste into /admin login): ║\n║ ${bootstrapToken.substring(0, 50)} ║\n║ ${bootstrapToken.substring(50).padEnd(50)} ║\n╚══════════════════════════════════════════════════════╝`}
|
||||
`);
|
||||
});
|
||||
|
||||
await waitForHttpServerLifecycle(httpServer);
|
||||
}
|
||||
|
||||
+4
-81
@@ -2874,17 +2874,10 @@ async function performSyncInner(engine: BrainEngine, opts: SyncOpts): Promise<Sy
|
||||
: await resolveSlugByPathOrSourcePath(engine, from, undefined);
|
||||
// The new path doesn't yet have a row, so resolve from path only.
|
||||
const newSlug = resolveSlugForPath(to);
|
||||
// #3056: the cheap rename is OBSERVED, not assumed. A zero-row UPDATE
|
||||
// doesn't throw, and a thrown collision used to be swallowed by an
|
||||
// empty catch — both fell through to importFile, which created/updated
|
||||
// the row at the new path while the old row stayed behind live. Both
|
||||
// shapes now fall through to the reconcile below.
|
||||
let renameApplied = false;
|
||||
try {
|
||||
renameApplied = (await engine.updateSlug(oldSlug, newSlug, renameOpts)) > 0;
|
||||
await engine.updateSlug(oldSlug, newSlug, renameOpts);
|
||||
} catch {
|
||||
// Destination slug occupied or invalid — treat as add; the reconcile
|
||||
// below removes the stale old row once the destination materialized.
|
||||
// Slug doesn't exist or collision, treat as add
|
||||
}
|
||||
// Reimport at new path (picks up content changes). Wrapped to match the
|
||||
// deletes/adds loops: a malformed renamed file is recorded to failedFiles
|
||||
@@ -2897,11 +2890,9 @@ async function performSyncInner(engine: BrainEngine, opts: SyncOpts): Promise<Sy
|
||||
// NAV-1 TOCTOU: refuse a destination that realpath-resolves outside the
|
||||
// repo (committed symlink pointing out).
|
||||
const filePath = join(gitContextRoot, to);
|
||||
let importResult: Awaited<ReturnType<typeof importFile>> | undefined;
|
||||
if (existsSync(filePath) && isPathSafe(filePath, gitContextRoot)) {
|
||||
try {
|
||||
const result = await importFile(engine, filePath, to, { noEmbed, sourceId: opts.sourceId, activePack: syncActivePack });
|
||||
importResult = result;
|
||||
if (result.status === 'imported') chunksCreated += result.chunks;
|
||||
else if (result.status === 'skipped' && (result as { error?: string }).error) {
|
||||
failedFiles.push({ path: to, error: String((result as { error?: string }).error) });
|
||||
@@ -2910,68 +2901,9 @@ async function performSyncInner(engine: BrainEngine, opts: SyncOpts): Promise<Sy
|
||||
failedFiles.push({ path: to, error: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
}
|
||||
// #3056 reconcile: the rename fell back to add semantics, so the row
|
||||
// that still represents the OLD path is the stale half of the rename
|
||||
// (git reported the old path gone; a plain delete of that path would
|
||||
// remove this row). Two safety rails, both from the #3252 review:
|
||||
//
|
||||
// 1. Delete only after the destination demonstrably materialized —
|
||||
// `imported`, or an errorless `skipped` AT the new slug. Identity
|
||||
// dedup can skip against the OLD row (result.slug === oldSlug),
|
||||
// in which case nothing landed at newSlug and deleting the old
|
||||
// row would destroy the only copy.
|
||||
// 2. Locate the stale row POSITIVELY by `source_path = from`, never
|
||||
// by the oldSlug guess — after a collision, a path-derived
|
||||
// fallback slug could name an unrelated (e.g. manually curated)
|
||||
// row. No source_path match → nothing is deleted (this also means
|
||||
// code-strategy imports, which don't populate source_path, fall
|
||||
// back safely to leaving the old row rather than guessing).
|
||||
//
|
||||
// A failed delete records a `<rename:…>` SENTINEL (not an ordinary
|
||||
// path failure): the gate hard-blocks the bookmark, and — unlike a
|
||||
// plain path row — the auto-skip valve can never chronic-skip it after
|
||||
// N attempts, which would advance the bookmark and make a transient
|
||||
// delete outage a permanent duplicate. The sentinel clears through the
|
||||
// ordinary success path once the rename converges on a later run.
|
||||
let reconcileFailed = false;
|
||||
if (!renameApplied && importResult !== undefined) {
|
||||
const destMaterialized = importResult.status === 'imported' ||
|
||||
(importResult.status === 'skipped' && !importResult.error && importResult.slug === newSlug);
|
||||
if (destMaterialized) {
|
||||
try {
|
||||
const staleMap = await engine.resolveSlugsByPaths([from], { sourceId: opts.sourceId ?? DEFAULT_SOURCE_ID });
|
||||
const staleSlug = staleMap.get(from);
|
||||
if (staleSlug !== undefined && staleSlug !== newSlug) {
|
||||
await engine.deletePage(staleSlug, renameOpts);
|
||||
deletedSlugs.add(staleSlug); // never hand a deleted slug to auto-embed
|
||||
serr(` [sync] rename reconciled: removed stale row ${staleSlug} (${from} -> ${to} fell back to add).`);
|
||||
} else if (staleSlug === undefined) {
|
||||
serr(` [sync] rename fallback: no row has source_path ${from}; stale row (if any) left in place.`);
|
||||
}
|
||||
} catch (e: unknown) {
|
||||
reconcileFailed = true;
|
||||
failedFiles.push({
|
||||
path: `<rename:${to}>`,
|
||||
error: `rename reconcile failed (stale row for ${from} not removed): ` +
|
||||
`${e instanceof Error ? e.message : String(e)}`,
|
||||
});
|
||||
}
|
||||
} else {
|
||||
serr(
|
||||
` [sync] rename fallback: ${from} -> ${to} did not materialize at ${newSlug} ` +
|
||||
`(import ${importResult.status}); old row left in place.`,
|
||||
);
|
||||
}
|
||||
}
|
||||
// Converged (cheap rename, clean reconcile, or nothing to reconcile):
|
||||
// clear any `<rename:…>` sentinel a previous failing run recorded.
|
||||
if (!reconcileFailed) succeededPaths.push(`<rename:${to}>`);
|
||||
pagesAffected.push(newSlug);
|
||||
deletedSlugs.delete(newSlug); // #1284: rename landed on a previously-deleted slug → embeddable again
|
||||
// A failed reconcile must NOT checkpoint: banking `to` would make the
|
||||
// resume filter skip this rename on the retry run, turning a transient
|
||||
// delete failure into a permanent duplicate — the exact bug being fixed.
|
||||
if (!reconcileFailed) await markCompleted(to);
|
||||
await markCompleted(to);
|
||||
progress.tick(1, newSlug);
|
||||
}
|
||||
progress.finish();
|
||||
@@ -3430,10 +3362,7 @@ async function performSyncInner(engine: BrainEngine, opts: SyncOpts): Promise<Sy
|
||||
|
||||
if (!gate.advanced) {
|
||||
const codeBreakdown = formatCodeBreakdown(failedFiles);
|
||||
// Two sentinel classes block here: `<head>` (pin ancestry broken) and
|
||||
// `<rename:…>` (#3056 — a rename-reconcile delete failed and advancing
|
||||
// would permanently bank the duplicate). Pick the message by which fired.
|
||||
if (gate.sentinelBlocked && failedFiles.some(f => f.path === '<head>')) {
|
||||
if (gate.sentinelBlocked) {
|
||||
serr(
|
||||
`\nSync blocked: repository history changed during sync (force-push / reset).\n` +
|
||||
`${codeBreakdown}\n\n` +
|
||||
@@ -3441,12 +3370,6 @@ async function performSyncInner(engine: BrainEngine, opts: SyncOpts): Promise<Sy
|
||||
`a commit that doesn't match the indexed tree. Re-run sync to re-pin against ` +
|
||||
`current HEAD.`,
|
||||
);
|
||||
} else if (gate.sentinelBlocked) {
|
||||
serr(
|
||||
`\nSync blocked: a rename left a stale duplicate that could not be removed:\n` +
|
||||
`${codeBreakdown}\n\n` +
|
||||
`The next 'gbrain sync' retries the reconcile from the same diff.`,
|
||||
);
|
||||
} else {
|
||||
const fileFailCount = failedFiles.filter(f => isSkippablePath(f.path)).length;
|
||||
serr(
|
||||
|
||||
+1
-10
@@ -895,17 +895,8 @@ export async function resolveSourceForDir(
|
||||
// (the cycleSourceId precedence) or 'default'.
|
||||
if (brainDir === null) return undefined;
|
||||
try {
|
||||
// #2540: exclude archived rows (dream's --source guard refuses to stamp
|
||||
// them, so an archived alias winning here means the stamp silently never
|
||||
// lands and doctor's cycle_freshness stays red on a healthy install) and
|
||||
// order deterministically so a duplicate registration of the same path
|
||||
// can't shadow the active source on whichever row the engine scans first.
|
||||
// Ordering matches listAllSources/sources-ops for operator-output parity.
|
||||
const rows = await engine.executeRaw<{ id: string }>(
|
||||
`SELECT id FROM sources
|
||||
WHERE local_path = $1 AND archived = false
|
||||
ORDER BY (id = 'default') DESC, id
|
||||
LIMIT 1`,
|
||||
`SELECT id FROM sources WHERE local_path = $1 LIMIT 1`,
|
||||
[brainDir],
|
||||
);
|
||||
if (rows[0]) return rows[0].id;
|
||||
|
||||
+1
-6
@@ -1951,13 +1951,8 @@ export interface BrainEngine {
|
||||
* preserved via stable page_id). `opts.sourceId` scopes the UPDATE — without
|
||||
* it, the bare `WHERE slug = old` matches every row across every source and
|
||||
* would either rename them all OR violate the (source_id, slug) UNIQUE.
|
||||
*
|
||||
* Returns the number of rows moved. 0 means the old slug had no row in the
|
||||
* scoped source — an UPDATE that matches nothing does NOT throw, so callers
|
||||
* that need to know whether the rename actually happened (the sync rename
|
||||
* path, #3056) must check the return value rather than rely on the catch.
|
||||
*/
|
||||
updateSlug(oldSlug: string, newSlug: string, opts?: { sourceId?: string }): Promise<number>;
|
||||
updateSlug(oldSlug: string, newSlug: string, opts?: { sourceId?: string }): Promise<void>;
|
||||
rewriteLinks(oldSlug: string, newSlug: string): Promise<void>;
|
||||
|
||||
/**
|
||||
|
||||
+2
-7
@@ -2,13 +2,6 @@ import type { BrainEngine } from './engine.ts';
|
||||
import { slugifyPath } from './sync.ts';
|
||||
import { getFtsLanguage } from './fts-language.ts';
|
||||
import { hnswMaxDimsForType } from './vector-index.ts';
|
||||
// runMigrations executes while an initialized engine is live. Keep its helper
|
||||
// modules in the static graph rather than importing them from async handlers.
|
||||
import {
|
||||
isStatementTimeoutError,
|
||||
isRetryableConnError,
|
||||
} from './retry-matcher.ts';
|
||||
import { repairTimelineDedupIndex } from './timeline-dedup-repair.ts';
|
||||
|
||||
/**
|
||||
* Schema migrations — run automatically on initSchema().
|
||||
@@ -5808,6 +5801,7 @@ async function runMigrationSQLWithRetry(
|
||||
m: Migration,
|
||||
sql: string,
|
||||
): Promise<void> {
|
||||
const { isStatementTimeoutError, isRetryableConnError } = await import('./retry-matcher.ts');
|
||||
// GBRAIN_MIGRATE_BACKOFF_MS lets tests skip the 5s/15s/45s backoff. In
|
||||
// production the env var is unset and the default cadence applies.
|
||||
const fastBackoff = process.env.GBRAIN_MIGRATE_BACKOFF_MS;
|
||||
@@ -6077,6 +6071,7 @@ export async function runMigrations(engine: BrainEngine): Promise<{ applied: num
|
||||
// reach the loop below). Best-effort + idempotent: a no-op on a healthy
|
||||
// index; `doctor` surfaces it independently if this ever fails.
|
||||
try {
|
||||
const { repairTimelineDedupIndex } = await import('./timeline-dedup-repair.ts');
|
||||
const r = await repairTimelineDedupIndex(engine);
|
||||
if (r.repaired) {
|
||||
console.error(
|
||||
|
||||
+11
-39
@@ -17,26 +17,7 @@ import type {
|
||||
SourceRow,
|
||||
} from './engine.ts';
|
||||
import { MAX_SEARCH_LIMIT, clampSearchLimit } from './engine.ts';
|
||||
// Engine-path imports stay static unless a call site carries an explicit
|
||||
// engine-dynamic-import-ok justification. The gateway is the only current
|
||||
// exception because its local try/catch preserves a soft fallback.
|
||||
import {
|
||||
withRetry,
|
||||
BULK_RETRY_OPTS,
|
||||
resolveBulkRetryOpts,
|
||||
computeNextDelay,
|
||||
isRetryableConnError,
|
||||
type BatchAuditSite,
|
||||
} from './retry.ts';
|
||||
import {
|
||||
valueHash,
|
||||
normalizeDimension,
|
||||
isNovelDimension,
|
||||
} from './chronicle/ontology.ts';
|
||||
import {
|
||||
resolveRecencyDecayMap,
|
||||
DEFAULT_FALLBACK,
|
||||
} from './search/recency-decay.ts';
|
||||
import { withRetry, BULK_RETRY_OPTS, resolveBulkRetryOpts, computeNextDelay, type BatchAuditSite } from './retry.ts';
|
||||
import { logBatchRetry as auditLogBatchRetry, logBatchExhausted as auditLogBatchExhausted } from './audit/batch-retry-audit.ts';
|
||||
import { runMigrations } from './migrate.ts';
|
||||
import { PGLITE_SCHEMA_SQL, getPGLiteSchema } from './pglite-schema.ts';
|
||||
@@ -438,9 +419,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
let dims: number = DEFAULT_EMBEDDING_DIMENSIONS;
|
||||
let model: string = DEFAULT_EMBEDDING_MODEL;
|
||||
try {
|
||||
// Keep the gateway lazy: its static closure is large, and evaluation inside
|
||||
// this try/catch preserves the unconfigured-gateway default fallback.
|
||||
const gw = await import('./ai/gateway.ts'); // engine-dynamic-import-ok
|
||||
const gw = await import('./ai/gateway.ts');
|
||||
// Both accessors THROW when the gateway is unconfigured (they never
|
||||
// return falsy), so the catch below is the only fallback path (#3461).
|
||||
dims = gw.getEmbeddingDimensions();
|
||||
@@ -2285,6 +2264,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
});
|
||||
} catch (err) {
|
||||
if (err instanceof Error && err.name === 'RetryAbortError') throw err;
|
||||
const { isRetryableConnError } = await import('./retry.ts');
|
||||
if (isRetryableConnError(err)) {
|
||||
auditLogBatchExhausted(auditSite, batchSize, opts.maxRetries + 1, err);
|
||||
}
|
||||
@@ -2350,9 +2330,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
// rationale — pglite mirrors it for parity.
|
||||
let resolvedModel: string | null = null;
|
||||
try {
|
||||
// Keep the gateway lazy so module-load failure remains inside this soft
|
||||
// fallback boundary; eager evaluation would bypass the config-row fallback.
|
||||
const gw = await import('./ai/gateway.ts'); // engine-dynamic-import-ok
|
||||
const gw = await import('./ai/gateway.ts');
|
||||
resolvedModel = gw.getEmbeddingModel();
|
||||
} catch {
|
||||
try {
|
||||
@@ -3864,6 +3842,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
|
||||
async mergeOntologyFact(obs: OntologyObservationInput): Promise<OntologyMergeResult> {
|
||||
const sourceId = obs.sourceId ?? 'default';
|
||||
const { valueHash, normalizeDimension, isNovelDimension } = await import('./chronicle/ontology.ts');
|
||||
const dimension = normalizeDimension(obs.dimension);
|
||||
const vh = valueHash(obs.value);
|
||||
const conf = obs.confidence ?? 0.7;
|
||||
@@ -5353,16 +5332,12 @@ export class PGLiteEngine implements BrainEngine {
|
||||
// pages_with_timeline) and v0.10.3 graph layer (link_coverage, timeline_coverage,
|
||||
// most_connected). Both coexist: master's brain_score is the composite
|
||||
// dashboard, v0.10.3 metrics give entity-page-level granularity.
|
||||
// #1305: every page-scoped count here excludes soft-deleted rows — same
|
||||
// posture as getStats — so brain_score moves when the user deletes pages.
|
||||
// Chunk/link counts stay raw (storage until the purge phase), matching
|
||||
// getStats, and destructive-removal counts elsewhere deliberately stay raw.
|
||||
const { rows: [h] } = await this.db.query(`
|
||||
WITH entity_pages AS (
|
||||
SELECT id, slug FROM pages WHERE type IN ('entity', 'person', 'company') AND deleted_at IS NULL
|
||||
SELECT id, slug FROM pages WHERE type IN ('entity', 'person', 'company')
|
||||
)
|
||||
SELECT
|
||||
(SELECT count(*) FROM pages WHERE deleted_at IS NULL) as page_count,
|
||||
(SELECT count(*) FROM pages) as page_count,
|
||||
(SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float /
|
||||
GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage,
|
||||
0 as stale_pages,
|
||||
@@ -5387,7 +5362,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
SELECT p.slug,
|
||||
(SELECT count(*) FROM links l WHERE l.from_page_id = p.id OR l.to_page_id = p.id)::int as link_count
|
||||
FROM pages p
|
||||
WHERE p.type IN ('entity', 'person', 'company') AND p.deleted_at IS NULL
|
||||
WHERE p.type IN ('entity', 'person', 'company')
|
||||
ORDER BY link_count DESC
|
||||
LIMIT 5
|
||||
`);
|
||||
@@ -5406,7 +5381,6 @@ export class PGLiteEngine implements BrainEngine {
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)) as islanded,
|
||||
EXISTS (SELECT 1 FROM timeline_entries te WHERE te.page_id = p.id) as has_timeline
|
||||
FROM pages p
|
||||
WHERE p.deleted_at IS NULL
|
||||
`);
|
||||
|
||||
const r = h as Record<string, unknown>;
|
||||
@@ -5501,18 +5475,15 @@ export class PGLiteEngine implements BrainEngine {
|
||||
}
|
||||
|
||||
// Sync
|
||||
async updateSlug(oldSlug: string, newSlug: string, opts?: { sourceId?: string }): Promise<number> {
|
||||
async updateSlug(oldSlug: string, newSlug: string, opts?: { sourceId?: string }): Promise<void> {
|
||||
newSlug = validateSlug(newSlug);
|
||||
const sourceId = opts?.sourceId ?? 'default';
|
||||
// Source-qualify so a rename in source A doesn't sweep up same-slug rows
|
||||
// in sources B/C/D (mirrors postgres-engine.ts).
|
||||
const result = await this.db.query(
|
||||
await this.db.query(
|
||||
`UPDATE pages SET slug = $1, updated_at = now() WHERE slug = $2 AND source_id = $3`,
|
||||
[newSlug, oldSlug, sourceId]
|
||||
);
|
||||
// #3056: rows moved — a zero-row UPDATE does not throw, so the count is
|
||||
// the only way callers can see the no-op.
|
||||
return result.affectedRows ?? 0;
|
||||
}
|
||||
|
||||
async rewriteLinks(_oldSlug: string, _newSlug: string): Promise<void> {
|
||||
@@ -6026,6 +5997,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
const recencyBias = opts.recency_bias ?? 'flat';
|
||||
let recencySql: string;
|
||||
if (recencyBias === 'on') {
|
||||
const { resolveRecencyDecayMap, DEFAULT_FALLBACK } = await import('./search/recency-decay.ts');
|
||||
recencySql = buildRecencyComponentSql({
|
||||
slugColumn: 'p.slug',
|
||||
dateExpr: 'COALESCE(p.effective_date, p.updated_at)',
|
||||
|
||||
+17
-44
@@ -13,29 +13,7 @@ import type {
|
||||
NewFact, FactListOpts, FactsHealth,
|
||||
SourceRow,
|
||||
} from './engine.ts';
|
||||
// Engine-path imports stay static unless a call site carries an explicit
|
||||
// engine-dynamic-import-ok justification. The gateway is the only current
|
||||
// exception because its local try/catch preserves a soft fallback.
|
||||
import {
|
||||
withRetry,
|
||||
BULK_RETRY_OPTS,
|
||||
resolveBulkRetryOpts,
|
||||
computeNextDelay,
|
||||
isRetryableConnError,
|
||||
type BatchAuditSite,
|
||||
} from './retry.ts';
|
||||
import { isConnectionEndedError } from './retry-matcher.ts';
|
||||
import {
|
||||
valueHash,
|
||||
normalizeDimension,
|
||||
isNovelDimension,
|
||||
} from './chronicle/ontology.ts';
|
||||
import {
|
||||
resolveRecencyDecayMap,
|
||||
DEFAULT_FALLBACK,
|
||||
} from './search/recency-decay.ts';
|
||||
import { logDbDisconnect } from './audit/db-disconnect-audit.ts';
|
||||
import { logPoolRecovery } from './audit/pool-recovery-audit.ts';
|
||||
import { withRetry, BULK_RETRY_OPTS, resolveBulkRetryOpts, computeNextDelay, type BatchAuditSite } from './retry.ts';
|
||||
import { logBatchRetry as auditLogBatchRetry, logBatchExhausted as auditLogBatchExhausted } from './audit/batch-retry-audit.ts';
|
||||
import type {
|
||||
DomainBankSampleOpts, CorpusSampleOpts, DomainBankRow,
|
||||
@@ -353,6 +331,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
// even a no-op disconnect (engine that was never connected) is
|
||||
// recorded — that case may itself be a caller-side bug worth seeing.
|
||||
try {
|
||||
const { logDbDisconnect } = await import('./audit/db-disconnect-audit.ts');
|
||||
logDbDisconnect('postgres', this._connectionStyle ?? 'unknown');
|
||||
} catch { /* best-effort; never block disconnect on audit failure */ }
|
||||
// v0.30.1: tear down the direct pool first if the manager owns one.
|
||||
@@ -402,9 +381,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
let dims: number = DEFAULT_EMBEDDING_DIMENSIONS;
|
||||
let model: string = DEFAULT_EMBEDDING_MODEL;
|
||||
try {
|
||||
// Keep the gateway lazy: its static closure is large, and evaluation inside
|
||||
// this try/catch preserves the unconfigured-gateway default fallback.
|
||||
const gw = await import('./ai/gateway.ts'); // engine-dynamic-import-ok
|
||||
const gw = await import('./ai/gateway.ts');
|
||||
// Both accessors THROW when the gateway is unconfigured (they never
|
||||
// return falsy), so the catch below is the only fallback path (#3461).
|
||||
dims = gw.getEmbeddingDimensions();
|
||||
@@ -2404,8 +2381,8 @@ export class PostgresEngine implements BrainEngine {
|
||||
if (err instanceof Error && err.name === 'RetryAbortError') throw err;
|
||||
// Best-effort exhausted-retry log. If the error wasn't retryable in
|
||||
// the first place, isRetryableConnError(err) is false and we skip.
|
||||
// retry.ts is already in this module's static graph through withRetry, so
|
||||
// classifying the exhausted error does not need a second runtime import.
|
||||
// Lazy-import to avoid a circular dep concern.
|
||||
const { isRetryableConnError } = await import('./retry.ts');
|
||||
if (isRetryableConnError(err)) {
|
||||
auditLogBatchExhausted(auditSite, batchSize, opts.maxRetries + 1, err);
|
||||
}
|
||||
@@ -2474,9 +2451,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
// is the LAST resort (fresh brain whose config row doesn't exist yet).
|
||||
let resolvedModel: string | null = null;
|
||||
try {
|
||||
// Keep the gateway lazy so module-load failure remains inside this soft
|
||||
// fallback boundary; eager evaluation would bypass the config-row fallback.
|
||||
const gw = await import('./ai/gateway.ts'); // engine-dynamic-import-ok
|
||||
const gw = await import('./ai/gateway.ts');
|
||||
resolvedModel = gw.getEmbeddingModel();
|
||||
} catch {
|
||||
try {
|
||||
@@ -4008,6 +3983,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
async mergeOntologyFact(obs: OntologyObservationInput): Promise<OntologyMergeResult> {
|
||||
const sql = this.sql;
|
||||
const sourceId = obs.sourceId ?? 'default';
|
||||
const { valueHash, normalizeDimension, isNovelDimension } = await import('./chronicle/ontology.ts');
|
||||
const dimension = normalizeDimension(obs.dimension);
|
||||
const vh = valueHash(obs.value);
|
||||
const conf = obs.confidence ?? 0.7;
|
||||
@@ -5456,16 +5432,12 @@ export class PostgresEngine implements BrainEngine {
|
||||
// no outbound links). The raw islanded list is filtered through the same
|
||||
// policy as `gbrain orphans` so convention pages do not count against
|
||||
// dashboard health.
|
||||
// #1305: every page-scoped count here excludes soft-deleted rows — same
|
||||
// posture as getStats — so brain_score moves when the user deletes pages.
|
||||
// Chunk/link counts stay raw (storage until the purge phase), matching
|
||||
// getStats, and destructive-removal counts elsewhere deliberately stay raw.
|
||||
const [h] = await sql`
|
||||
WITH entity_pages AS (
|
||||
SELECT id, slug FROM pages WHERE type IN ('entity', 'person', 'company') AND deleted_at IS NULL
|
||||
SELECT id, slug FROM pages WHERE type IN ('entity', 'person', 'company')
|
||||
)
|
||||
SELECT
|
||||
(SELECT count(*) FROM pages WHERE deleted_at IS NULL) as page_count,
|
||||
(SELECT count(*) FROM pages) as page_count,
|
||||
(SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float /
|
||||
GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage,
|
||||
0 as stale_pages,
|
||||
@@ -5487,7 +5459,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
SELECT p.slug,
|
||||
(SELECT count(*) FROM links l WHERE l.from_page_id = p.id OR l.to_page_id = p.id)::int as link_count
|
||||
FROM pages p
|
||||
WHERE p.type IN ('entity', 'person', 'company') AND p.deleted_at IS NULL
|
||||
WHERE p.type IN ('entity', 'person', 'company')
|
||||
ORDER BY link_count DESC
|
||||
LIMIT 5
|
||||
`;
|
||||
@@ -5506,7 +5478,6 @@ export class PostgresEngine implements BrainEngine {
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)) as islanded,
|
||||
EXISTS (SELECT 1 FROM timeline_entries te WHERE te.page_id = p.id) as has_timeline
|
||||
FROM pages p
|
||||
WHERE p.deleted_at IS NULL
|
||||
`;
|
||||
|
||||
const pageCount = Number(h.page_count);
|
||||
@@ -5598,17 +5569,14 @@ export class PostgresEngine implements BrainEngine {
|
||||
}
|
||||
|
||||
// Sync
|
||||
async updateSlug(oldSlug: string, newSlug: string, opts?: { sourceId?: string }): Promise<number> {
|
||||
async updateSlug(oldSlug: string, newSlug: string, opts?: { sourceId?: string }): Promise<void> {
|
||||
newSlug = validateSlug(newSlug);
|
||||
const sql = this.sql;
|
||||
const sourceId = opts?.sourceId ?? 'default';
|
||||
// Source-qualify so a rename in source A doesn't sweep up same-slug rows
|
||||
// in sources B/C/D (which would either rename them all OR fail the
|
||||
// (source_id, slug) UNIQUE if the new slug already exists in another source).
|
||||
const result = await sql`UPDATE pages SET slug = ${newSlug}, updated_at = now() WHERE slug = ${oldSlug} AND source_id = ${sourceId}`;
|
||||
// #3056: rows moved — a zero-row UPDATE does not throw, so the count is
|
||||
// the only way callers can see the no-op.
|
||||
return result.count ?? 0;
|
||||
await sql`UPDATE pages SET slug = ${newSlug}, updated_at = now() WHERE slug = ${oldSlug} AND source_id = ${sourceId}`;
|
||||
}
|
||||
|
||||
async rewriteLinks(_oldSlug: string, _newSlug: string): Promise<void> {
|
||||
@@ -5847,10 +5815,12 @@ export class PostgresEngine implements BrainEngine {
|
||||
let isReap = false;
|
||||
if (ctx?.error !== undefined) {
|
||||
try {
|
||||
const { isConnectionEndedError } = await import('./retry-matcher.ts');
|
||||
isReap = isConnectionEndedError(ctx.error);
|
||||
} catch { /* classification is best-effort */ }
|
||||
}
|
||||
try {
|
||||
const { logPoolRecovery } = await import('./audit/pool-recovery-audit.ts');
|
||||
logPoolRecovery(isReap ? 'reap_detected' : 'reconnect_other', ctx?.error);
|
||||
} catch { /* audit is best-effort */ }
|
||||
|
||||
@@ -5874,6 +5844,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
// New pool is live — discard the old one best-effort.
|
||||
if (oldSql) { try { await oldSql.end({ timeout: 5 }); } catch { /* swallow */ } }
|
||||
try {
|
||||
const { logPoolRecovery } = await import('./audit/pool-recovery-audit.ts');
|
||||
logPoolRecovery('reconnect_succeeded');
|
||||
} catch { /* best-effort */ }
|
||||
} catch (err) {
|
||||
@@ -5885,6 +5856,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
this._sql = oldSql;
|
||||
this.connectionManager = oldManager;
|
||||
try {
|
||||
const { logPoolRecovery } = await import('./audit/pool-recovery-audit.ts');
|
||||
logPoolRecovery('reconnect_failed', err);
|
||||
} catch { /* best-effort */ }
|
||||
throw err; // let batchRetry's backoff handle the retry
|
||||
@@ -6321,6 +6293,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
const recencyBias = opts.recency_bias ?? 'flat';
|
||||
let recencySql: string;
|
||||
if (recencyBias === 'on') {
|
||||
const { resolveRecencyDecayMap, DEFAULT_FALLBACK } = await import('./search/recency-decay.ts');
|
||||
recencySql = buildRecencyComponentSql({
|
||||
slugColumn: 'p.slug',
|
||||
dateExpr: 'COALESCE(p.effective_date, p.updated_at)',
|
||||
|
||||
@@ -48,32 +48,6 @@ import {
|
||||
|
||||
export const RRF_K = 60;
|
||||
const COMPILED_TRUTH_BOOST = 2.0;
|
||||
|
||||
/**
|
||||
* Which detail levels get the compiled_truth boost (#3430).
|
||||
*
|
||||
* ONLY `low`. The documented contract (`src/core/operations.ts`) is
|
||||
* "low (compiled truth only), medium (default, all with dedup), high (all
|
||||
* chunks)" — so `low` is the level that privileges compiled truth, and both
|
||||
* `medium` and `high` are supposed to see everything on equal footing.
|
||||
*
|
||||
* This was previously spelled `detail !== 'high'`, i.e. written as though
|
||||
* `high` were the special case. Because COMPILED_TRUTH_BOOST is applied AFTER
|
||||
* RRF normalization, and RRF's whole range over a 100-deep pool is 1/60 → 1/160,
|
||||
* a 2.0x multiplier is not a tilt — break-even is `2/(60+r) >= 1/60`, so any
|
||||
* boosted chunk inside the first 60 ranks outranks an unboosted rank-1 chunk.
|
||||
* At the default detail that made search categorically compiled-truth-only:
|
||||
* a page whose answer lived in a `fenced_code` chunk returned the prose chunk,
|
||||
* and the code chunk fell out of the window entirely.
|
||||
*
|
||||
* Extracted as a named predicate rather than left inline at three call sites so
|
||||
* the detail→boost mapping is directly testable. An inline expression can only
|
||||
* be covered through a full `hybridSearch` round trip, which is why the
|
||||
* original inversion went unnoticed.
|
||||
*/
|
||||
export function shouldBoostCompiledTruth(detail: string | null | undefined): boolean {
|
||||
return detail === 'low';
|
||||
}
|
||||
const pendingCacheWrites = new Set<Promise<unknown>>();
|
||||
|
||||
/**
|
||||
@@ -1195,7 +1169,7 @@ export async function hybridSearch(
|
||||
const noEmbedLists = [{ list: keywordResults, k: fk }];
|
||||
if (titleResults.length > 0) noEmbedLists.push({ list: titleResults, k: fk });
|
||||
if (relationalList.length > 0) noEmbedLists.push({ list: relationalList, k: fk });
|
||||
noEmbedResults = rrfFusionWeighted(noEmbedLists, shouldBoostCompiledTruth(detailResolved));
|
||||
noEmbedResults = rrfFusionWeighted(noEmbedLists, detailResolved !== 'high');
|
||||
}
|
||||
if (noEmbedResults.length > 0) {
|
||||
await runPostFusionStages(engine, noEmbedResults, postFusionOpts);
|
||||
@@ -1439,7 +1413,7 @@ export async function hybridSearch(
|
||||
const fallbackLists = [{ list: keywordResults, k: fk }];
|
||||
if (titleResults.length > 0) fallbackLists.push({ list: titleResults, k: fk });
|
||||
if (relationalList.length > 0) fallbackLists.push({ list: relationalList, k: fk });
|
||||
fallbackResults = rrfFusionWeighted(fallbackLists, shouldBoostCompiledTruth(detail));
|
||||
fallbackResults = rrfFusionWeighted(fallbackLists, detail !== 'high');
|
||||
}
|
||||
if (fallbackResults.length > 0) {
|
||||
await runPostFusionStages(engine, fallbackResults, postFusionOpts);
|
||||
@@ -1526,7 +1500,7 @@ export async function hybridSearch(
|
||||
// arms BEFORE fusion so the compiled-truth authority boost skips them.
|
||||
await stampUnverifiedExtractions(engine, allLists.flatMap((l) => l.list));
|
||||
|
||||
let fused = rrfFusionWeighted(allLists, shouldBoostCompiledTruth(detail));
|
||||
let fused = rrfFusionWeighted(allLists, detail !== 'high');
|
||||
|
||||
// Cosine re-scoring before dedup so semantically better chunks survive.
|
||||
// v0.36 (D9): hydrate from the active embedding column so rescore happens
|
||||
|
||||
@@ -766,7 +766,7 @@ export function attributeKnob<K extends keyof ModeBundle>(
|
||||
// written between the #3391 stale-fix (which changes which chunks count as
|
||||
// current) and the operator's migration run. Same one-time global cold-miss
|
||||
// pattern as the bumps above.
|
||||
export const KNOBS_HASH_VERSION = 14;
|
||||
export const KNOBS_HASH_VERSION = 13;
|
||||
|
||||
/**
|
||||
* v0.36 (D8 / CDX-2) — second-arg context for the cache key. The
|
||||
|
||||
@@ -1,37 +0,0 @@
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import { openAdminSseStream, type AdminSseResponse } from '../src/commands/serve-http.ts';
|
||||
|
||||
describe('admin SSE handshake', () => {
|
||||
test('flushes a protocol-valid comment immediately after the headers', () => {
|
||||
const calls: string[] = [];
|
||||
const headers = new Map<string, string>();
|
||||
|
||||
openAdminSseStream({
|
||||
setHeader(name: string, value: string | number | readonly string[]) {
|
||||
headers.set(name, String(value));
|
||||
calls.push(`header:${name}`);
|
||||
return this;
|
||||
},
|
||||
flushHeaders() {
|
||||
calls.push('flush');
|
||||
},
|
||||
write(chunk: unknown) {
|
||||
calls.push(`write:${String(chunk)}`);
|
||||
return true;
|
||||
},
|
||||
} as unknown as AdminSseResponse);
|
||||
|
||||
expect(headers).toEqual(new Map([
|
||||
['Content-Type', 'text/event-stream'],
|
||||
['Cache-Control', 'no-cache'],
|
||||
['Connection', 'keep-alive'],
|
||||
]));
|
||||
expect(calls).toEqual([
|
||||
'header:Content-Type',
|
||||
'header:Cache-Control',
|
||||
'header:Connection',
|
||||
'flush',
|
||||
'write:: connected\n\n',
|
||||
]);
|
||||
});
|
||||
});
|
||||
@@ -136,7 +136,7 @@ describe('D2 — knobsHash differs across cross-modal knob values', () => {
|
||||
return resolveSearchMode({ mode: 'balanced' });
|
||||
}
|
||||
|
||||
test('KNOBS_HASH_VERSION is 14 (cross-modal still appended; 13→14 compiled_truth boost scope #3430)', () => {
|
||||
test('KNOBS_HASH_VERSION is 13 (cross-modal still appended; 12→13 embedding-provider migration #3390)', () => {
|
||||
// v0.35 ladder: 1→2 reranker, 2→3 floor_ratio. v0.36 piggybacks on v=3
|
||||
// with 7 cross-modal knobs + column/provider context. v0.40.4 (salem) +
|
||||
// v0.39 T21 (master) bump to v=4 for graph_signals + schema-pack fields.
|
||||
@@ -146,8 +146,7 @@ describe('D2 — knobsHash differs across cross-modal knob values', () => {
|
||||
// v0.43: 9→10 relational recall arm. #1400: 10→11 query-side input_type
|
||||
// finally reaches asymmetric providers — pre-fix rows were keyed on
|
||||
// document-side query vectors. #2825: 11→12 hard-exclude fold (hx=).
|
||||
// #3430: 13→14 compiled_truth boost no longer applies at detail=medium.
|
||||
expect(KNOBS_HASH_VERSION).toBe(14);
|
||||
expect(KNOBS_HASH_VERSION).toBe(13);
|
||||
});
|
||||
|
||||
test('flipping unified_multimodal changes the hash', () => {
|
||||
|
||||
@@ -38,7 +38,7 @@ import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
import { resetPgliteState } from './helpers/reset-pglite.ts';
|
||||
import { withEnv, emptyHome } from './helpers/with-env.ts';
|
||||
import { runCycle, ALL_PHASES } from '../src/core/cycle.ts';
|
||||
import { mkdtempSync, writeFileSync, rmSync } from 'fs';
|
||||
import { mkdtempSync, writeFileSync } from 'fs';
|
||||
import { execSync } from 'child_process';
|
||||
import { tmpdir } from 'os';
|
||||
import { join } from 'path';
|
||||
@@ -139,25 +139,19 @@ describe('#2540 (i) — pack omitting optional phases, all enabled phases comple
|
||||
|
||||
describe('#2540 (ii) — an enabled phase that never completes still prevents the stamp', () => {
|
||||
test('every selected phase failing reports status=failed and does NOT stamp last_full_cycle_at', async () => {
|
||||
await withEnv({ GBRAIN_HOME: gbrainHome }, async () => {
|
||||
await withEnv({ GBRAIN_HOME: gbrainHome, OPENAI_API_KEY: undefined, ANTHROPIC_API_KEY: undefined }, async () => {
|
||||
await seedSource('always-fails');
|
||||
expect(await readLastFullCycleAt('always-fails')).toBeNull();
|
||||
|
||||
// Deterministic, environment-independent failure: run the sync phase
|
||||
// against a brain directory that no longer exists. The previous shape
|
||||
// ('embed' with OPENAI_API_KEY/ANTHROPIC_API_KEY unset) was
|
||||
// environment-sensitive — on a machine where any OTHER embedding
|
||||
// provider resolves (Voyage, ZeroEntropy, a local endpoint, …), embed
|
||||
// with zero stale chunks succeeds and the cycle reports 'clean',
|
||||
// flipping this test's expectation. A vanished checkout fails the
|
||||
// sync phase on every machine. This is NOT the fix under test; it's
|
||||
// the pre-existing "an enabled phase genuinely never completes" case
|
||||
// the issue says must keep failing doctor's check.
|
||||
rmSync(brainDir, { recursive: true, force: true });
|
||||
// embed is a real, always-enabled phase (no pack gate, no config
|
||||
// .enabled toggle). With no embedding provider key configured it
|
||||
// deterministically fails — this is NOT the fix under test, it's
|
||||
// the pre-existing "an enabled phase genuinely never completes"
|
||||
// case the issue says must keep failing doctor's check.
|
||||
const report = await runCycle(engine, {
|
||||
brainDir,
|
||||
sourceId: 'always-fails',
|
||||
phases: ['sync'],
|
||||
phases: ['embed'],
|
||||
});
|
||||
|
||||
expect(report.status).toBe('failed');
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
/**
|
||||
* #3502: docs must not reference nonexistent gbrain commands.
|
||||
*
|
||||
* `docs/tutorials/personal-brain.md` shipped a `gbrain install` step for two
|
||||
* months after the command it replaced was retired — every reader hit
|
||||
* "Unknown command: install". This guard scans README.md, docs/, and skills/
|
||||
* for `gbrain <verb>` invocations in code (fenced blocks + inline code spans)
|
||||
* and checks each verb against the live CLI surface: CLI_ONLY, operation
|
||||
* cliHints names (non-hidden), and aliases.
|
||||
*
|
||||
* Deliberately excluded (historical or speculative by design, per CLAUDE.md's
|
||||
* "historical docs are never rewritten" rule):
|
||||
* - docs/GBRAIN_V0.md — the v0 spec; documents v0's CLI
|
||||
* - docs/designs/, docs/plans/ — future/speculative design docs
|
||||
* - docs/migrations/, skills/migrations/ — per-release migration notes,
|
||||
* written against that release's CLI
|
||||
* - docs/UPGRADING_DOWNSTREAM_AGENTS.md — per-release upgrade chronicle
|
||||
*
|
||||
* Heuristics keep prose out: only fenced code + inline spans are scanned,
|
||||
* comment lines and diagram lines are skipped, and the verb must sit in
|
||||
* command position (start of command text, or after a shell operator).
|
||||
*/
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import { readdirSync, readFileSync, statSync } from 'fs';
|
||||
import { dirname, join, relative } from 'path';
|
||||
import { CLI_ONLY, cliAliases } from '../src/cli.ts';
|
||||
import { operations } from '../src/core/operations.ts';
|
||||
|
||||
const ROOT = dirname(import.meta.dir);
|
||||
|
||||
const EXCLUDED = [
|
||||
'docs/GBRAIN_V0.md',
|
||||
'docs/UPGRADING_DOWNSTREAM_AGENTS.md',
|
||||
'docs/designs/',
|
||||
'docs/plans/',
|
||||
'docs/migrations/',
|
||||
'skills/migrations/',
|
||||
];
|
||||
|
||||
/** Known-intentional references to commands that deliberately don't exist. */
|
||||
const ALLOWLIST: Record<string, string[]> = {
|
||||
// The doc explains that gbrain does NOT ship this command, on purpose.
|
||||
'docs/guides/rls-and-you.md': ['rls-exempt'],
|
||||
};
|
||||
|
||||
function validCommands(): Set<string> {
|
||||
const valid = new Set<string>(CLI_ONLY);
|
||||
for (const op of operations) {
|
||||
const name = op.cliHints?.name;
|
||||
if (name && !op.cliHints?.hidden) valid.add(name);
|
||||
}
|
||||
for (const alias of cliAliases.keys()) valid.add(alias);
|
||||
return valid;
|
||||
}
|
||||
|
||||
function* mdFiles(dir: string): Generator<string> {
|
||||
for (const entry of readdirSync(dir)) {
|
||||
const p = join(dir, entry);
|
||||
if (statSync(p).isDirectory()) yield* mdFiles(p);
|
||||
else if (p.endsWith('.md')) yield p;
|
||||
}
|
||||
}
|
||||
|
||||
interface CodeLine { code: string; line: number }
|
||||
|
||||
/** Fenced-block lines + inline code spans that START with `gbrain `. */
|
||||
function codeRegions(text: string): CodeLine[] {
|
||||
const out: CodeLine[] = [];
|
||||
const lines = text.split('\n');
|
||||
let inFence = false;
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const l = lines[i];
|
||||
if (/^\s*(```|~~~)/.test(l)) { inFence = !inFence; continue; }
|
||||
if (inFence) {
|
||||
const t = l.trim();
|
||||
if (/^(#|\/\/|--|\*)/.test(t)) continue; // comment lines
|
||||
if (/[│┌┐└┘├┤─═╔╗╚╝]/.test(l)) continue; // ASCII-art diagrams
|
||||
out.push({ code: l, line: i + 1 });
|
||||
continue;
|
||||
}
|
||||
for (const m of l.matchAll(/`(gbrain [^`]+)`/g)) out.push({ code: m[1], line: i + 1 });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** True when `gbrain` sits at command position (not mid-prose). */
|
||||
function commandPosition(prefix: string): boolean {
|
||||
const p = prefix.trimEnd();
|
||||
return p === '' || /[|;&`(={[]$/.test(p) || /\$$/.test(p);
|
||||
}
|
||||
|
||||
function scan(): string[] {
|
||||
const valid = validCommands();
|
||||
const violations: string[] = [];
|
||||
const files = [
|
||||
join(ROOT, 'README.md'),
|
||||
...mdFiles(join(ROOT, 'docs')),
|
||||
...mdFiles(join(ROOT, 'skills')),
|
||||
];
|
||||
for (const file of files) {
|
||||
const rel = relative(ROOT, file);
|
||||
if (EXCLUDED.some((e) => rel === e || rel.startsWith(e))) continue;
|
||||
const text = readFileSync(file, 'utf-8');
|
||||
for (const { code, line } of codeRegions(text)) {
|
||||
for (const m of code.matchAll(/\bgbrain\s+([A-Za-z][\w-]*)/g)) {
|
||||
const verb = m[1];
|
||||
if (!/^[a-z][a-z0-9_-]{2,}$/.test(verb)) continue; // flags, <slots>, v0.x
|
||||
if (!commandPosition(code.slice(0, m.index))) continue;
|
||||
if (valid.has(verb)) continue;
|
||||
if (ALLOWLIST[rel]?.includes(verb)) continue;
|
||||
violations.push(`${rel}:${line}: \`gbrain ${verb}\` is not a real command — ${code.trim().slice(0, 90)}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
return violations;
|
||||
}
|
||||
|
||||
describe('#3502 — docs reference only real gbrain commands', () => {
|
||||
test('every `gbrain <verb>` in README/docs/skills resolves to a live command', () => {
|
||||
const violations = scan();
|
||||
expect(violations).toEqual([]);
|
||||
});
|
||||
|
||||
test('the sanity anchors: install is dead, init/put/skillpack are live', () => {
|
||||
const valid = validCommands();
|
||||
expect(valid.has('install')).toBe(false); // retired v0.36.0.0 — the #3502 bug
|
||||
expect(valid.has('init')).toBe(true);
|
||||
expect(valid.has('put')).toBe(true);
|
||||
expect(valid.has('skillpack')).toBe(true);
|
||||
});
|
||||
|
||||
test('pages + bench are dispatchable (documented surfaces; #2035 bug class)', () => {
|
||||
// `pages` had a live handleCliOnly case but was dropped from CLI_ONLY;
|
||||
// `bench` (bench-publish.ts) was documented but never wired at all.
|
||||
expect(CLI_ONLY.has('pages')).toBe(true);
|
||||
expect(CLI_ONLY.has('bench')).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -79,38 +79,12 @@ describe('doctor checkCycleFreshness', () => {
|
||||
expect(result.message).toMatch(/gbrain dream --source/);
|
||||
});
|
||||
|
||||
test('source with NO last_full_cycle_at (never cycled) returns warn, not fail (#2540)', async () => {
|
||||
// #2540: never-cycled used to FAIL, which turned doctor permanently red
|
||||
// on any install that doesn't cycle every local_path source (e.g. one
|
||||
// nightly `dream --dir <vault>` plus other federated sources) — and on
|
||||
// any source added minutes ago. It surfaces as a warning; only a source
|
||||
// that HAS cycled and then went stale escalates to fail.
|
||||
test('source with NO last_full_cycle_at (never cycled) returns fail', async () => {
|
||||
await engine.executeRaw(`UPDATE sources SET local_path = NULL WHERE id = 'default'`);
|
||||
await seed('virgin');
|
||||
const result = await checkCycleFreshness(engine, { nowMs: NOW });
|
||||
expect(result.status).toBe('warn');
|
||||
expect(result.message).toMatch(/never completed a full cycle/);
|
||||
expect(result.message).toMatch(/gbrain dream --source/);
|
||||
});
|
||||
|
||||
test('reporter case (#2540): one cycled vault + never-cycled siblings is warn, not permanent fail', async () => {
|
||||
await engine.executeRaw(`UPDATE sources SET local_path = NULL WHERE id = 'default'`);
|
||||
await seed('nightly-vault', agoH(2)); // the one vault dreamt via --dir
|
||||
await seed('federated-a'); // never cycled
|
||||
await seed('federated-b'); // never cycled
|
||||
const result = await checkCycleFreshness(engine, { nowMs: NOW });
|
||||
expect(result.status).toBe('warn');
|
||||
expect(result.message).toMatch(/federated-a/);
|
||||
expect(result.message).toMatch(/federated-b/);
|
||||
expect(result.message).not.toMatch(/nightly-vault/);
|
||||
});
|
||||
|
||||
test('a previously-cycled source gone stale still fails even next to never-cycled sources', async () => {
|
||||
await engine.executeRaw(`UPDATE sources SET local_path = NULL WHERE id = 'default'`);
|
||||
await seed('stale', agoH(72)); // real regression signal
|
||||
await seed('virgin'); // never cycled — warn-only
|
||||
const result = await checkCycleFreshness(engine, { nowMs: NOW });
|
||||
expect(result.status).toBe('fail');
|
||||
expect(result.message).toMatch(/never completed a full cycle/);
|
||||
});
|
||||
|
||||
test('mixed sources: highest severity wins (fail > warn > ok)', async () => {
|
||||
|
||||
@@ -96,28 +96,6 @@ describe('gbrain dream --dir <path> freshness stamp (#1869)', () => {
|
||||
expect(await readLastFullCycleAt('mothballed')).toBeNull();
|
||||
});
|
||||
}, 60_000);
|
||||
|
||||
test('an ARCHIVED alias of the same path does not shadow the active source (#2540)', async () => {
|
||||
await withEnv({ GBRAIN_HOME: gbrainHome }, async () => {
|
||||
// Ordinary shape: a source was archived and re-added under a new id
|
||||
// pointing at the same checkout. Seed the archived twin FIRST so a
|
||||
// filterless `LIMIT 1` scan finds it first.
|
||||
await seedSource('retired-twin', true);
|
||||
await seedSource('active-twin', false);
|
||||
|
||||
const report = await runDream(engine, ['--dir', brainDir, '--phase', 'lint', '--json']);
|
||||
expect(report).toBeTruthy();
|
||||
if (report) expect(['ok', 'clean']).toContain(report.status);
|
||||
|
||||
// Pre-fix, resolveSourceForDir's exact match had no `archived = false`
|
||||
// filter and no ORDER BY, so the archived twin won the lookup; dream's
|
||||
// archived guard then (correctly) refused to stamp it — and the ACTIVE
|
||||
// source silently never got its stamp, leaving doctor's cycle_freshness
|
||||
// permanently stale on a healthy install.
|
||||
expect(await readLastFullCycleAt('active-twin')).not.toBeNull();
|
||||
expect(await readLastFullCycleAt('retired-twin')).toBeNull();
|
||||
});
|
||||
}, 60_000);
|
||||
});
|
||||
|
||||
/**
|
||||
|
||||
@@ -26,12 +26,8 @@ if (skip) {
|
||||
}
|
||||
|
||||
describeE2E('E2E: JSONB roundtrip — v0.12.1 reliability wave', () => {
|
||||
// 60s hook budget: setupDB runs connect + the full migration chain, which
|
||||
// exceeds bun's default 5s hook timeout on loaded CI runners. Hooks do NOT
|
||||
// inherit a test's third-arg timeout (verified on bun 1.3.14) — they need
|
||||
// their own second-arg budget. Same pattern as op-checkpoint-jsonb-parity.
|
||||
beforeAll(async () => { await setupDB(); }, 60_000);
|
||||
afterAll(async () => { await teardownDB(); }, 60_000);
|
||||
beforeAll(async () => { await setupDB(); });
|
||||
afterAll(async () => { await teardownDB(); });
|
||||
|
||||
test('putPage writes frontmatter as object, not double-encoded string', async () => {
|
||||
const engine = getEngine();
|
||||
|
||||
@@ -1,226 +0,0 @@
|
||||
/**
|
||||
* Pins the Memvelope envelope importer contract: deterministic markdown output,
|
||||
* provenance frontmatter, citation-bearing bodies, and loud collision handling.
|
||||
*/
|
||||
import { afterAll, describe, expect, test } from 'bun:test';
|
||||
import { mkdtempSync, rmSync, readdirSync, readFileSync, writeFileSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
// The same parser gbrain uses to ingest frontmatter (src/core/markdown.ts), so
|
||||
// the injection test asserts against the real consumer rather than a substring.
|
||||
import { safeLoad as yamlSafeLoad } from 'js-yaml';
|
||||
|
||||
const SCRIPT_PATH = join(import.meta.dir, '..', 'scripts', 'envelope-to-gbrain.mjs');
|
||||
const FIXTURE_PATH = join(import.meta.dir, 'fixtures', 'memvelope', 'sample.mve.json');
|
||||
const TEMP_DIRS: string[] = [];
|
||||
|
||||
afterAll(() => {
|
||||
for (const dir of TEMP_DIRS) {
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
function tempDir(): string {
|
||||
const dir = mkdtempSync(join(tmpdir(), 'envelope-to-gbrain-'));
|
||||
TEMP_DIRS.push(dir);
|
||||
return dir;
|
||||
}
|
||||
|
||||
async function runImporter(envelopePath: string, outDir = tempDir()) {
|
||||
// The script is plain Node-compatible ESM; Bun can execute it directly in CI
|
||||
// without requiring a separate node toolchain.
|
||||
const proc = Bun.spawn([process.execPath, SCRIPT_PATH, envelopePath, outDir], {
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
});
|
||||
await proc.exited;
|
||||
const stdout = await new Response(proc.stdout).text();
|
||||
const stderr = await new Response(proc.stderr).text();
|
||||
return { exitCode: proc.exitCode, stdout, stderr, outDir };
|
||||
}
|
||||
|
||||
function markdownFiles(dir: string): string[] {
|
||||
return readdirSync(dir).filter((name) => name.endsWith('.md')).sort();
|
||||
}
|
||||
|
||||
function readOnlyMarkdown(dir: string): string {
|
||||
const files = markdownFiles(dir);
|
||||
expect(files).toHaveLength(1);
|
||||
return readFileSync(join(dir, files[0]), 'utf8');
|
||||
}
|
||||
|
||||
describe('envelope-to-gbrain importer', () => {
|
||||
test('sample envelope writes exactly one markdown page and reports count', async () => {
|
||||
const result = await runImporter(FIXTURE_PATH);
|
||||
|
||||
expect(result.exitCode).toBe(0);
|
||||
expect(markdownFiles(result.outDir)).toHaveLength(1);
|
||||
expect(result.stdout).toContain('wrote 1 markdown page(s)');
|
||||
});
|
||||
|
||||
test('filename is keyed by conversation id with date prefix', async () => {
|
||||
const result = await runImporter(FIXTURE_PATH);
|
||||
|
||||
expect(result.exitCode).toBe(0);
|
||||
expect(markdownFiles(result.outDir)).toEqual(['2025-11-02-c-3f9a2b.md']);
|
||||
});
|
||||
|
||||
test('frontmatter carries conversation provenance fields', async () => {
|
||||
const result = await runImporter(FIXTURE_PATH);
|
||||
const page = readOnlyMarkdown(result.outDir);
|
||||
|
||||
expect(result.exitCode).toBe(0);
|
||||
expect(page).toContain('type: conversation');
|
||||
expect(page).toContain('title: "Onboarding Checklist Draft"');
|
||||
expect(page).toContain('date: 2025-11-02');
|
||||
expect(page).toContain('source: "chatgpt"');
|
||||
expect(page).toContain('memvelope_conversation_id: "c-3f9a2b"');
|
||||
expect(page).toContain('origin: memvelope/envelope-v0');
|
||||
});
|
||||
|
||||
test('body carries role labels and message-id citations', async () => {
|
||||
const result = await runImporter(FIXTURE_PATH);
|
||||
const page = readOnlyMarkdown(result.outDir);
|
||||
|
||||
expect(result.exitCode).toBe(0);
|
||||
expect(page).toContain('· m1');
|
||||
expect(page).toContain('· m4');
|
||||
expect(page).toContain('**Me**');
|
||||
expect(page).toContain('**Assistant**');
|
||||
});
|
||||
|
||||
test('output is deterministic across repeated runs', async () => {
|
||||
const first = await runImporter(FIXTURE_PATH);
|
||||
const second = await runImporter(FIXTURE_PATH);
|
||||
|
||||
expect(first.exitCode).toBe(0);
|
||||
expect(second.exitCode).toBe(0);
|
||||
expect(readOnlyMarkdown(first.outDir)).toBe(readOnlyMarkdown(second.outDir));
|
||||
});
|
||||
|
||||
test('duplicate conversation ids warn and report distinct files written', async () => {
|
||||
const inputDir = tempDir();
|
||||
const envelopePath = join(inputDir, 'duplicate.mve.json');
|
||||
writeFileSync(envelopePath, JSON.stringify({
|
||||
memvelope: 'envelope-v0',
|
||||
meta: { source_provider: 'chatgpt' },
|
||||
conversations: [
|
||||
{
|
||||
id: 'c-repeat',
|
||||
title: 'First repeated id',
|
||||
created_at: '2025-11-02T14:22:51.000Z',
|
||||
messages: [{ id: 'm1', role: 'user', ts: '2025-11-02T14:22:51.000Z', text: 'alice-example noted the first checklist draft.' }],
|
||||
},
|
||||
{
|
||||
id: 'c-repeat',
|
||||
title: 'Second repeated id',
|
||||
created_at: '2025-11-02T15:22:51.000Z',
|
||||
messages: [{ id: 'm2', role: 'assistant', ts: '2025-11-02T15:22:51.000Z', text: 'Assistant noted the repeated id collision.' }],
|
||||
},
|
||||
],
|
||||
}));
|
||||
|
||||
const result = await runImporter(envelopePath);
|
||||
|
||||
expect(result.exitCode).toBe(0);
|
||||
expect(result.stderr).toContain('warning: filename collision on "2025-11-02-c-repeat.md"');
|
||||
expect(result.stdout).toContain('wrote 1 markdown page(s)');
|
||||
expect(markdownFiles(result.outDir)).toHaveLength(1);
|
||||
});
|
||||
|
||||
test('missing or foreign format is rejected', async () => {
|
||||
const inputDir = tempDir();
|
||||
const envelopePath = join(inputDir, 'not-envelope.json');
|
||||
writeFileSync(envelopePath, JSON.stringify({ conversations: [] }));
|
||||
|
||||
const result = await runImporter(envelopePath);
|
||||
|
||||
expect(result.exitCode).toBe(1);
|
||||
expect(result.stderr).toContain('envelope-v0');
|
||||
});
|
||||
|
||||
test('missing conversation id uses positional fallback filename', async () => {
|
||||
const inputDir = tempDir();
|
||||
const envelopePath = join(inputDir, 'missing-id.mve.json');
|
||||
writeFileSync(envelopePath, JSON.stringify({
|
||||
memvelope: 'envelope-v0',
|
||||
meta: { source_provider: 'chatgpt' },
|
||||
conversations: [
|
||||
{
|
||||
title: 'Missing id example',
|
||||
created_at: '2025-11-02T14:22:51.000Z',
|
||||
messages: [{ id: 'm1', role: 'user', ts: '2025-11-02T14:22:51.000Z', text: 'alice-example asked for a fallback filename.' }],
|
||||
},
|
||||
],
|
||||
}));
|
||||
|
||||
const result = await runImporter(envelopePath);
|
||||
|
||||
expect(result.exitCode).toBe(0);
|
||||
expect(markdownFiles(result.outDir)).toEqual(['2025-11-02-conv-1.md']);
|
||||
});
|
||||
|
||||
test('missing conversation id omits the provenance key rather than emitting a value', async () => {
|
||||
const inputDir = tempDir();
|
||||
const envelopePath = join(inputDir, 'missing-id-frontmatter.mve.json');
|
||||
writeFileSync(envelopePath, JSON.stringify({
|
||||
memvelope: 'envelope-v0',
|
||||
meta: { source_provider: 'chatgpt' },
|
||||
conversations: [
|
||||
{
|
||||
title: 'Missing id example',
|
||||
created_at: '2025-11-02T14:22:51.000Z',
|
||||
messages: [{ id: 'm1', role: 'user', ts: '2025-11-02T14:22:51.000Z', text: 'alice-example asked about frontmatter.' }],
|
||||
},
|
||||
],
|
||||
}));
|
||||
|
||||
const result = await runImporter(envelopePath);
|
||||
const page = readOnlyMarkdown(result.outDir);
|
||||
|
||||
expect(result.exitCode).toBe(0);
|
||||
// Absent means absent: never the literal string `undefined`, and never the
|
||||
// positional filename fallback masquerading as a real conversation id.
|
||||
expect(page).not.toContain('memvelope_conversation_id');
|
||||
expect(page).not.toContain('undefined');
|
||||
expect(page).toContain('source: "chatgpt"');
|
||||
});
|
||||
|
||||
test('a provider string carrying a newline cannot inject frontmatter keys', async () => {
|
||||
const inputDir = tempDir();
|
||||
const envelopePath = join(inputDir, 'injecting-provider.mve.json');
|
||||
writeFileSync(envelopePath, JSON.stringify({
|
||||
memvelope: 'envelope-v0',
|
||||
meta: { source_provider: 'chatgpt\ntype: injected\nowner: attacker' },
|
||||
conversations: [
|
||||
{
|
||||
id: 'c-inject',
|
||||
title: 'Injection attempt',
|
||||
created_at: '2025-11-02T14:22:51.000Z',
|
||||
messages: [{ id: 'm1', role: 'user', ts: '2025-11-02T14:22:51.000Z', text: 'alice-example sent a hostile provider string.' }],
|
||||
},
|
||||
],
|
||||
}));
|
||||
|
||||
const result = await runImporter(envelopePath);
|
||||
const page = readOnlyMarkdown(result.outDir);
|
||||
const frontmatter = page.split('---')[1] ?? '';
|
||||
const parsed = yamlSafeLoad(frontmatter) as Record<string, unknown>;
|
||||
|
||||
expect(result.exitCode).toBe(0);
|
||||
// The newline is escaped inside a quoted scalar, so the hostile text stays
|
||||
// one value instead of becoming keys. Asserted structurally: a substring
|
||||
// check cannot tell a real key from the same characters inside a quoted
|
||||
// value, and would pass for the wrong reason.
|
||||
expect(Object.keys(parsed).sort()).toEqual([
|
||||
'date',
|
||||
'memvelope_conversation_id',
|
||||
'origin',
|
||||
'source',
|
||||
'title',
|
||||
'type',
|
||||
]);
|
||||
expect(parsed.type).toBe('conversation');
|
||||
expect(parsed.source).toBe('chatgpt\ntype: injected\nowner: attacker');
|
||||
});
|
||||
});
|
||||
+1
-20
@@ -4,7 +4,7 @@ import { join, basename } from 'path';
|
||||
import { createHash } from 'crypto';
|
||||
import { extname } from 'path';
|
||||
import { tmpdir } from 'os';
|
||||
import { collectFiles, formatFileSizeKb } from '../src/commands/files.ts';
|
||||
import { collectFiles } from '../src/commands/files.ts';
|
||||
import { operationsByName } from '../src/core/operations.ts';
|
||||
import * as db from '../src/core/db.ts';
|
||||
|
||||
@@ -51,25 +51,6 @@ afterAll(() => {
|
||||
rmSync(TMP, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe('formatFileSizeKb', () => {
|
||||
test('formats number, bigint, and string database values', () => {
|
||||
expect(formatFileSizeKb(35 * 1024)).toBe('35KB');
|
||||
expect(formatFileSizeKb(35n * 1024n)).toBe('35KB');
|
||||
expect(formatFileSizeKb('35840')).toBe('35KB');
|
||||
});
|
||||
|
||||
test('preserves zero-byte files instead of reporting an unknown size', () => {
|
||||
expect(formatFileSizeKb(0)).toBe('0KB');
|
||||
expect(formatFileSizeKb(0n)).toBe('0KB');
|
||||
});
|
||||
|
||||
test('reports missing or invalid sizes as unknown', () => {
|
||||
expect(formatFileSizeKb(null)).toBe('?');
|
||||
expect(formatFileSizeKb('not-a-number')).toBe('?');
|
||||
expect(formatFileSizeKb(-1)).toBe('?');
|
||||
});
|
||||
});
|
||||
|
||||
describe('getMimeType', () => {
|
||||
test('returns correct MIME for .jpg', () => {
|
||||
expect(getMimeType('photo.jpg')).toBe('image/jpeg');
|
||||
|
||||
-42
@@ -1,42 +0,0 @@
|
||||
{
|
||||
"memvelope": "envelope-v0",
|
||||
"meta": {
|
||||
"source_provider": "chatgpt",
|
||||
"conversation_count": 1,
|
||||
"message_count": 4
|
||||
},
|
||||
"conversations": [
|
||||
{
|
||||
"id": "c-3f9a2b",
|
||||
"title": "Onboarding Checklist Draft",
|
||||
"created_at": "2025-11-02T14:22:51.000Z",
|
||||
"updated_at": "2025-11-02T14:31:12.000Z",
|
||||
"messages": [
|
||||
{
|
||||
"id": "m1",
|
||||
"role": "user",
|
||||
"ts": "2025-11-02T14:22:51.000Z",
|
||||
"text": "alice-example is drafting acme-example's widget-co onboarding checklist and wants a concise first pass."
|
||||
},
|
||||
{
|
||||
"id": "m2",
|
||||
"role": "assistant",
|
||||
"ts": "2025-11-02T14:24:03.000Z",
|
||||
"text": "Start with account setup, workspace access, sample widget review, and a first-week check-in with the acme-example owner."
|
||||
},
|
||||
{
|
||||
"id": "m3",
|
||||
"role": "user",
|
||||
"ts": "2025-11-02T14:28:19.000Z",
|
||||
"text": "Add a note that bob-example should compare fund-a and fund-b reporting needs before the kickoff."
|
||||
},
|
||||
{
|
||||
"id": "m4",
|
||||
"role": "assistant",
|
||||
"ts": "2025-11-02T14:31:12.000Z",
|
||||
"text": "Include a pre-kickoff step for bob-example to list fund-a and fund-b reporting questions, then confirm owners with charlie-example."
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1,123 +0,0 @@
|
||||
/**
|
||||
* #1305 — getHealth() must exclude soft-deleted pages from every
|
||||
* page-scoped count, the same posture getStats() has had since v0.26.5.
|
||||
*
|
||||
* Pre-fix, getHealth counted raw `pages` rows: page_count and orphan_pages
|
||||
* included soft-deleted pages, the entity_pages CTE kept deleted entities in
|
||||
* the link/timeline coverage denominators and in most_connected, and
|
||||
* brain_score therefore never moved when a user soft-deleted pages.
|
||||
*
|
||||
* Boundary (deliberate): chunk- and link-scoped counts (embed_coverage,
|
||||
* missing_embeddings, link_count, dead_links) stay RAW — they occupy storage
|
||||
* until the autopilot purge phase, matching getStats. Destructive-removal
|
||||
* counts (purge paths, #2235) also deliberately count all rows and are
|
||||
* untouched here.
|
||||
*
|
||||
* Runs against PGLite — the fixed SQL shapes are identical in both engines.
|
||||
*/
|
||||
|
||||
import { describe, test, expect, beforeAll, afterAll, beforeEach } from 'bun:test';
|
||||
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
|
||||
let engine: PGLiteEngine;
|
||||
|
||||
beforeAll(async () => {
|
||||
engine = new PGLiteEngine();
|
||||
await engine.connect({});
|
||||
await engine.initSchema();
|
||||
}, 60_000);
|
||||
|
||||
afterAll(async () => {
|
||||
await engine.disconnect();
|
||||
});
|
||||
|
||||
beforeEach(async () => {
|
||||
for (const t of ['links', 'content_chunks', 'timeline_entries', 'tags', 'page_versions', 'pages']) {
|
||||
await (engine as any).db.exec(`DELETE FROM ${t}`);
|
||||
}
|
||||
});
|
||||
|
||||
async function seedNote(slug: string): Promise<void> {
|
||||
await engine.putPage(slug, { type: 'note', title: slug, compiled_truth: `content of ${slug}`, frontmatter: {} });
|
||||
}
|
||||
|
||||
async function pageId(slug: string): Promise<number> {
|
||||
return (await (engine as any).db.query(`SELECT id FROM pages WHERE slug=$1`, [slug])).rows[0].id;
|
||||
}
|
||||
|
||||
describe('#1305 — getHealth excludes soft-deleted pages', () => {
|
||||
test('page_count and orphan_pages match getStats after soft-delete (the issue repro)', async () => {
|
||||
for (let i = 0; i < 10; i++) await seedNote(`wiki/note-${i}`);
|
||||
for (let i = 0; i < 6; i++) await engine.softDeletePage(`wiki/note-${i}`);
|
||||
|
||||
const stats = await engine.getStats();
|
||||
const health = await engine.getHealth();
|
||||
expect(stats.page_count).toBe(4);
|
||||
// Pre-fix: 10 (raw rows). getHealth must agree with getStats.
|
||||
expect(health.page_count).toBe(4);
|
||||
// Pre-fix: 10 — deleted pages stayed in the islanded scan.
|
||||
expect(health.orphan_pages).toBe(4);
|
||||
});
|
||||
|
||||
test('brain_score moves when the user soft-deletes the islanded pages', async () => {
|
||||
// 2 connected pages + 8 islanded ones.
|
||||
await seedNote('wiki/hub');
|
||||
await seedNote('wiki/leaf');
|
||||
await (engine as any).db.query(
|
||||
`INSERT INTO links (from_page_id, to_page_id, link_type) VALUES ($1, $2, 'mentions')`,
|
||||
[await pageId('wiki/hub'), await pageId('wiki/leaf')],
|
||||
);
|
||||
for (let i = 0; i < 8; i++) await seedNote(`wiki/clutter-${i}`);
|
||||
|
||||
const before = await engine.getHealth();
|
||||
for (let i = 0; i < 8; i++) await engine.softDeletePage(`wiki/clutter-${i}`);
|
||||
const after = await engine.getHealth();
|
||||
|
||||
// Pre-fix both assertions fail: orphan_pages stayed 8 and brain_score
|
||||
// was byte-identical before/after the delete.
|
||||
expect(after.orphan_pages).toBe(0);
|
||||
expect(after.brain_score).toBeGreaterThan(before.brain_score);
|
||||
});
|
||||
|
||||
test('entity coverage denominators and most_connected exclude deleted entities', async () => {
|
||||
// Live entity: inbound link + timeline entry → full coverage.
|
||||
await engine.putPage('people/alice-example', { type: 'person', title: 'Alice', compiled_truth: 'a person', frontmatter: {} });
|
||||
await engine.putPage('people/bob-example', { type: 'person', title: 'Bob', compiled_truth: 'another person', frontmatter: {} });
|
||||
await seedNote('wiki/mentions-alice');
|
||||
const aliceId = await pageId('people/alice-example');
|
||||
await (engine as any).db.query(
|
||||
`INSERT INTO links (from_page_id, to_page_id, link_type) VALUES ($1, $2, 'mentions')`,
|
||||
[await pageId('wiki/mentions-alice'), aliceId],
|
||||
);
|
||||
await (engine as any).db.query(
|
||||
`INSERT INTO timeline_entries (page_id, date, summary) VALUES ($1, '2026-01-01', 'met alice')`,
|
||||
[aliceId],
|
||||
);
|
||||
|
||||
await engine.softDeletePage('people/bob-example');
|
||||
const h = await engine.getHealth();
|
||||
|
||||
// Pre-fix: bob stayed in the entity_pages CTE → coverage 0.5 each,
|
||||
// and bob appeared in most_connected.
|
||||
expect(h.link_coverage).toBe(1);
|
||||
expect(h.timeline_coverage).toBe(1);
|
||||
expect(h.most_connected.map((c) => c.slug)).not.toContain('people/bob-example');
|
||||
});
|
||||
|
||||
test('chunk storage counts stay raw (the deliberate boundary)', async () => {
|
||||
await seedNote('wiki/kept');
|
||||
await seedNote('wiki/gone');
|
||||
for (const slug of ['wiki/kept', 'wiki/gone']) {
|
||||
await (engine as any).db.query(
|
||||
`INSERT INTO content_chunks (page_id, chunk_index, chunk_text) VALUES ($1, 0, 'chunk')`,
|
||||
[await pageId(slug)],
|
||||
);
|
||||
}
|
||||
await engine.softDeletePage('wiki/gone');
|
||||
|
||||
const h = await engine.getHealth();
|
||||
// Soft-deleted pages' chunks still occupy storage until purge; the
|
||||
// missing_embeddings count keeps seeing them, same as getStats.
|
||||
expect(h.missing_embeddings).toBe(2);
|
||||
});
|
||||
});
|
||||
@@ -1,330 +0,0 @@
|
||||
import { afterEach, describe, expect, it, setDefaultTimeout } from 'bun:test';
|
||||
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { basename, join, resolve } from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
|
||||
const REPO_ROOT = resolve(import.meta.dir, '..', '..');
|
||||
const PACKAGE_JSON = resolve(REPO_ROOT, 'package.json');
|
||||
const GUARD = resolve(REPO_ROOT, 'scripts', 'check-engine-dynamic-import.sh');
|
||||
const VERIFY_DISPATCHER = resolve(REPO_ROOT, 'scripts', 'run-verify-parallel.sh');
|
||||
const BASH = process.platform === 'win32'
|
||||
? resolve(process.env.ProgramFiles ?? 'C:\\Program Files', 'Git', 'bin', 'bash.exe')
|
||||
: 'bash';
|
||||
const tempDirs: string[] = [];
|
||||
|
||||
setDefaultTimeout(30_000);
|
||||
|
||||
function fixture(name: string, content: string): string {
|
||||
const dir = mkdtempSync(join(tmpdir(), 'gbrain-engine-import-'));
|
||||
tempDirs.push(dir);
|
||||
const path = join(dir, name);
|
||||
writeFileSync(path, content, 'utf8');
|
||||
return path;
|
||||
}
|
||||
|
||||
function runGuard(files: string[] = [], cwd = REPO_ROOT) {
|
||||
const result = spawnSync(BASH, [GUARD, ...files], {
|
||||
cwd,
|
||||
encoding: 'utf8',
|
||||
timeout: 30_000,
|
||||
});
|
||||
return {
|
||||
code: result.status ?? -1,
|
||||
stdout: result.stdout ?? '',
|
||||
stderr: result.stderr ?? '',
|
||||
};
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
for (const dir of tempDirs.splice(0)) {
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
describe('check-engine-dynamic-import.sh', () => {
|
||||
it('exists', () => {
|
||||
expect(existsSync(GUARD)).toBe(true);
|
||||
});
|
||||
|
||||
it('rejects and reports an unmarked dynamic import', () => {
|
||||
const path = fixture('violator.ts', "async function load() {\n return await import('./helper.ts');\n}\n");
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:2:`);
|
||||
expect(result.stderr).toContain("await import('./helper.ts')");
|
||||
});
|
||||
|
||||
it('allows a same-line marker for multiple non-gateway imports and ignores comment-only matches', () => {
|
||||
const path = fixture(
|
||||
'allowed.ts',
|
||||
[
|
||||
"// await import('./comment.ts')",
|
||||
'/*',
|
||||
" * await import('./block-body.ts')",
|
||||
' */',
|
||||
"const first = import('./first.ts'); const second = import('./second.ts'); // engine-dynamic-import-ok",
|
||||
'',
|
||||
].join('\n'),
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(0);
|
||||
expect(result.stdout).toContain('check-engine-dynamic-import: ok (1 file(s) scanned)');
|
||||
});
|
||||
|
||||
it('rejects live code after a closed leading block comment', () => {
|
||||
const path = fixture(
|
||||
'leading-block-comment.ts',
|
||||
"/* load only when needed */ const helper = await import('./helper.ts');\n",
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:1:`);
|
||||
expect(result.stderr).toContain("await import('./helper.ts')");
|
||||
});
|
||||
|
||||
it('ignores dynamic-import text wholly inside a multiline block comment', () => {
|
||||
const path = fixture(
|
||||
'multiline-block-comment.ts',
|
||||
[
|
||||
'/*',
|
||||
" * await import('./comment-only.ts')",
|
||||
' */',
|
||||
'const value = 1;',
|
||||
'',
|
||||
].join('\n'),
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(0);
|
||||
expect(result.stdout).toContain('check-engine-dynamic-import: ok (1 file(s) scanned)');
|
||||
});
|
||||
|
||||
it('reports every violation across multiple files', () => {
|
||||
const first = fixture(
|
||||
'first-violator.ts',
|
||||
"const first = await import('./first.ts');\nconst second = await import('./second.ts');\n",
|
||||
);
|
||||
const second = fixture(
|
||||
'second-violator.ts',
|
||||
"/* explanation */ const third = await import('./third.ts');\n",
|
||||
);
|
||||
const result = runGuard([first, second]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(first)}:1:`);
|
||||
expect(result.stderr).toContain(`${basename(first)}:2:`);
|
||||
expect(result.stderr).toContain(`${basename(second)}:1:`);
|
||||
}, 30_000);
|
||||
|
||||
it('does not mistake comment delimiters inside literals for comments', () => {
|
||||
const path = fixture(
|
||||
'literal-delimiters.ts',
|
||||
[
|
||||
'const url = "https://example.test";',
|
||||
'const block = "/* not a comment";',
|
||||
'const template = `https://example.test`;',
|
||||
'const pattern = /\\/\\//;',
|
||||
"const helper = await import('./helper.ts');",
|
||||
'',
|
||||
].join('\n'),
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:5:`);
|
||||
}, 30_000);
|
||||
|
||||
it('detects bare and trivia-separated dynamic imports', () => {
|
||||
const path = fixture(
|
||||
'dynamic-import-syntax.ts',
|
||||
[
|
||||
"const first = import('./first.ts');",
|
||||
"const second = await import /* explanation */ ('./second.ts');",
|
||||
'',
|
||||
].join('\n'),
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:1:`);
|
||||
expect(result.stderr).toContain(`${basename(path)}:2:`);
|
||||
}, 30_000);
|
||||
|
||||
it('detects live code after a multiline block comment closes', () => {
|
||||
const path = fixture(
|
||||
'after-multiline-comment.ts',
|
||||
"/*\n * explanation\n */ const helper = await import('./helper.ts');\n",
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:3:`);
|
||||
});
|
||||
|
||||
it('requires the allow marker on the import line', () => {
|
||||
const path = fixture(
|
||||
'marker-line.ts',
|
||||
"// engine-dynamic-import-ok\nconst helper = await import('./helper.ts');\n",
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:2:`);
|
||||
});
|
||||
|
||||
it('does not accept marker text outside comment trivia or within longer comment tokens', () => {
|
||||
const path = fixture(
|
||||
'marker-text.ts',
|
||||
[
|
||||
"const first = import('./engine-dynamic-import-ok.ts');",
|
||||
"const marker = 'engine-dynamic-import-ok'; const second = import('./second.ts');",
|
||||
'const template = `prefix',
|
||||
'// engine-dynamic-import-ok ${import("./third.ts")}`;',
|
||||
"const fourth = import('./fourth.ts'); // no-engine-dynamic-import-ok: not approved",
|
||||
'',
|
||||
].join('\n'),
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:1:`);
|
||||
expect(result.stderr).toContain(`${basename(path)}:2:`);
|
||||
expect(result.stderr).toContain(`${basename(path)}:4:`);
|
||||
expect(result.stderr).toContain(`${basename(path)}:5:`);
|
||||
});
|
||||
|
||||
it('does not accept markers adjacent to Unicode identifier characters', () => {
|
||||
const path = fixture(
|
||||
'unicode-marker-text.ts',
|
||||
[
|
||||
"const first = import('./first.ts'); // noéengine-dynamic-import-ok: not approved",
|
||||
"const second = import('./second.ts'); // engine-dynamic-import-oké: not approved",
|
||||
"const third = import('./third.ts'); // éengine-dynamic-import-oké: not approved",
|
||||
"const fourth = import('./fourth.ts'); // nóengine-dynamic-import-ok: not approved",
|
||||
"const fifth = import('./fifth.ts'); // 𐐀engine-dynamic-import-ok: not approved",
|
||||
"const sixth = import('./sixth.ts'); // engine-dynamic-import-ok𐐀: not approved",
|
||||
'',
|
||||
].join('\n'),
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:1:`);
|
||||
expect(result.stderr).toContain(`${basename(path)}:2:`);
|
||||
expect(result.stderr).toContain(`${basename(path)}:3:`);
|
||||
expect(result.stderr).toContain(`${basename(path)}:4:`);
|
||||
expect(result.stderr).toContain(`${basename(path)}:5:`);
|
||||
expect(result.stderr).toContain(`${basename(path)}:6:`);
|
||||
});
|
||||
|
||||
it('allows a marker in real multiline comment trivia on the import line', () => {
|
||||
const path = fixture(
|
||||
'multiline-marker.ts',
|
||||
[
|
||||
'/* rationale',
|
||||
' * engine-dynamic-import-ok */ const helper = import("./helper.ts");',
|
||||
'',
|
||||
].join('\n'),
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(0);
|
||||
});
|
||||
|
||||
it('fails on TypeScript parse diagnostics', () => {
|
||||
const path = fixture('malformed.ts', 'const broken = ;\n');
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain('cannot parse input file');
|
||||
expect(result.stderr).toContain(basename(path));
|
||||
});
|
||||
|
||||
it('reports recovered-AST violations alongside parse diagnostics', () => {
|
||||
const path = fixture(
|
||||
'malformed-violator.ts',
|
||||
"const helper = import('./helper.ts');\nconst broken = ;\n",
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain('cannot parse input file');
|
||||
expect(result.stderr).toContain(`${basename(path)}:1:`);
|
||||
});
|
||||
|
||||
it('ignores type-position imports', () => {
|
||||
const path = fixture(
|
||||
'type-import.ts',
|
||||
"type Helper = import('./helper.ts').Helper;\n",
|
||||
);
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(0);
|
||||
});
|
||||
|
||||
it('reports readable-file violations alongside missing inputs', () => {
|
||||
const path = fixture('mixed-violator.ts', "const helper = import('./helper.ts');\n");
|
||||
const missing = join(tmpdir(), `gbrain-engine-import-missing-${process.pid}.ts`);
|
||||
const result = runGuard([path, missing]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:1:`);
|
||||
expect(result.stderr).toContain('cannot read input file');
|
||||
expect(result.stderr).toContain(basename(missing));
|
||||
});
|
||||
|
||||
it('fails when an explicit input file is missing', () => {
|
||||
const missing = join(tmpdir(), `gbrain-engine-import-missing-${process.pid}.ts`);
|
||||
const result = runGuard([missing]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain('cannot read input file');
|
||||
expect(result.stderr).toContain(basename(missing));
|
||||
});
|
||||
|
||||
it('resolves default inputs from the guard repository', () => {
|
||||
const foreign = mkdtempSync(join(tmpdir(), 'gbrain-engine-import-foreign-'));
|
||||
tempDirs.push(foreign);
|
||||
const foreignCore = join(foreign, 'src', 'core');
|
||||
mkdirSync(foreignCore, { recursive: true });
|
||||
expect(spawnSync('git', ['init', '-q'], { cwd: foreign }).status).toBe(0);
|
||||
writeFileSync(
|
||||
join(foreignCore, 'pglite-engine.ts'),
|
||||
"const foreign = import('./foreign.ts');\n",
|
||||
'utf8',
|
||||
);
|
||||
for (const name of ['postgres-engine.ts', 'migrate.ts']) {
|
||||
writeFileSync(join(foreignCore, name), '', 'utf8');
|
||||
}
|
||||
|
||||
const result = runGuard([], foreign);
|
||||
expect(result.code).toBe(0);
|
||||
expect(result.stdout).toContain('check-engine-dynamic-import: ok (3 file(s) scanned)');
|
||||
}, 30_000);
|
||||
|
||||
it('still catches a violation in CRLF input', () => {
|
||||
const path = fixture('crlf.ts', "async function load() {\r\n return await import('./helper.ts');\r\n}\r\n");
|
||||
const result = runGuard([path]);
|
||||
expect(result.code).toBe(1);
|
||||
expect(result.stderr).toContain(`${basename(path)}:2:`);
|
||||
});
|
||||
|
||||
it('passes on the reconciled repository sources', () => {
|
||||
const result = runGuard();
|
||||
expect(result.code).toBe(0);
|
||||
expect(result.stdout).toContain('check-engine-dynamic-import: ok (3 file(s) scanned)');
|
||||
}, 30_000);
|
||||
});
|
||||
|
||||
describe('engine dynamic-import guard wiring', () => {
|
||||
it('is invoked through bash by check:all', () => {
|
||||
const pkg = JSON.parse(readFileSync(PACKAGE_JSON, 'utf8')) as {
|
||||
scripts: Record<string, string>;
|
||||
};
|
||||
expect(pkg.scripts['check:engine-dynamic-import']).toBe(
|
||||
'bash scripts/check-engine-dynamic-import.sh',
|
||||
);
|
||||
expect(pkg.scripts['check:all']).toContain(
|
||||
'bash scripts/check-engine-dynamic-import.sh',
|
||||
);
|
||||
});
|
||||
|
||||
it('is listed by the authoritative verify dispatcher', () => {
|
||||
const result = spawnSync(BASH, [VERIFY_DISPATCHER, '--dry-list'], {
|
||||
cwd: REPO_ROOT,
|
||||
encoding: 'utf8',
|
||||
timeout: 30_000,
|
||||
});
|
||||
expect(result.status).toBe(0);
|
||||
expect(new Set((result.stdout ?? '').trim().split('\n'))).toContain(
|
||||
'check:engine-dynamic-import',
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -89,7 +89,7 @@ describe('alias_resolved boost stage', () => {
|
||||
});
|
||||
|
||||
describe('KNOBS_HASH_VERSION', () => {
|
||||
it('is 14 (13→14 compiled_truth boost no longer applies at detail=medium, so pre-fix rankings must be unreachable, #3430)', () => {
|
||||
expect(KNOBS_HASH_VERSION).toBe(14);
|
||||
it('is 13 (12→13 embedding-provider migration invalidates rows written against the prior embedding space, #3390)', () => {
|
||||
expect(KNOBS_HASH_VERSION).toBe(13);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,112 +0,0 @@
|
||||
/**
|
||||
* #3430: the compiled_truth boost must not apply at `detail=medium`.
|
||||
*
|
||||
* `COMPILED_TRUTH_BOOST = 2.0` is applied AFTER RRF score normalization. RRF's
|
||||
* entire dynamic range over a 100-deep pool is 1/60 → 1/160 (a factor of 2.67),
|
||||
* so a 2.0x multiplier consumes roughly three quarters of it. Break-even is
|
||||
* `2/(60+r) >= 1/60`, i.e. r <= 60 — so ANY boosted chunk in the first 60 ranks
|
||||
* outranks an unboosted rank-1 chunk. That is a categorical filter, not a tilt:
|
||||
* a page whose actual answer is in a `fenced_code` chunk returns the prose
|
||||
* chunk instead, and the code chunk leaves the result window entirely.
|
||||
*
|
||||
* The gate was written as `detail !== 'high'` — "high is special" — but the
|
||||
* documented contract in `src/core/operations.ts` is:
|
||||
*
|
||||
* low (compiled truth only), medium (default, all with dedup), high (all chunks)
|
||||
*
|
||||
* which makes LOW the special one. `low` already restricts to compiled_truth,
|
||||
* so a boost there is a no-op among equals; `medium` and `high` are both
|
||||
* supposed to see everything. Hence `detail === 'low'`.
|
||||
*
|
||||
* These tests pin the arithmetic, not the constant — they would still fail if
|
||||
* someone reintroduced a boost at medium with a different multiplier or behind
|
||||
* a score floor, which is why they assert final RANK rather than score.
|
||||
*/
|
||||
import { describe, test, expect } from 'bun:test';
|
||||
import { rrfFusion, RRF_K, shouldBoostCompiledTruth } from '../src/core/search/hybrid.ts';
|
||||
import { KNOBS_HASH_VERSION } from '../src/core/search/mode.ts';
|
||||
import type { SearchResult } from '../src/core/types.ts';
|
||||
|
||||
function chunk(slug: string, chunkSource: string): SearchResult {
|
||||
return { slug, chunk_source: chunkSource, chunk_text: 'x', title: slug, score: 0 } as unknown as SearchResult;
|
||||
}
|
||||
|
||||
/** One vector arm: the correct answer at rank 0, then `n` compiled_truth chunks. */
|
||||
function poolWithAnswerFirst(n: number): SearchResult[] {
|
||||
const list = [chunk('code/answer', 'fenced_code')];
|
||||
for (let i = 0; i < n; i++) list.push(chunk(`prose/p${i}`, 'compiled_truth'));
|
||||
return list;
|
||||
}
|
||||
|
||||
function rankOfAnswer(results: SearchResult[]): number {
|
||||
return results.findIndex((r) => r.slug === 'code/answer');
|
||||
}
|
||||
|
||||
describe('#3430: the detail→boost mapping itself', () => {
|
||||
// These are the assertions that actually FAIL on master. The rrfFusion tests
|
||||
// below pin the arithmetic but pass either way, because they pass the boost
|
||||
// flag explicitly — they cannot see how hybridSearch decides it. This is the
|
||||
// wiring.
|
||||
test('ONLY detail=low boosts compiled_truth', () => {
|
||||
expect(shouldBoostCompiledTruth('low')).toBe(true);
|
||||
expect(shouldBoostCompiledTruth('medium')).toBe(false);
|
||||
expect(shouldBoostCompiledTruth('high')).toBe(false);
|
||||
});
|
||||
|
||||
test('an absent detail does not boost — medium is the documented default', () => {
|
||||
// Callers that omit detail get medium semantics, so the unset case must
|
||||
// match medium, not low. A `!== 'high'` spelling gets this backwards.
|
||||
expect(shouldBoostCompiledTruth(undefined)).toBe(false);
|
||||
expect(shouldBoostCompiledTruth(null)).toBe(false);
|
||||
});
|
||||
|
||||
test('an unrecognized detail value does not boost', () => {
|
||||
// Fail-open toward showing everything rather than silently filtering.
|
||||
expect(shouldBoostCompiledTruth('')).toBe(false);
|
||||
expect(shouldBoostCompiledTruth('LOW')).toBe(false);
|
||||
expect(shouldBoostCompiledTruth('detailed')).toBe(false);
|
||||
});
|
||||
|
||||
test('the cache version was bumped so pre-fix rankings are unreachable', () => {
|
||||
// Results are cached AFTER fusion, so rows written under the old boost
|
||||
// semantics would otherwise be served under the new ones for the whole TTL.
|
||||
// 13 was the pre-fix value.
|
||||
expect(KNOBS_HASH_VERSION).toBeGreaterThanOrEqual(14);
|
||||
});
|
||||
});
|
||||
|
||||
describe('#3430: compiled_truth boost scope', () => {
|
||||
test('boost OFF (detail=medium/high) keeps the vector-ranked answer at rank 0', () => {
|
||||
// The regression this file exists for. Pre-fix, medium passed applyBoost=true
|
||||
// and the answer landed at rank n — outside a 20-result window for n >= 20.
|
||||
for (const n of [10, 20, 40, 80]) {
|
||||
const fused = rrfFusion([poolWithAnswerFirst(n)], RRF_K, false);
|
||||
expect(rankOfAnswer(fused), `n=${n}: answer must stay first without the boost`).toBe(0);
|
||||
}
|
||||
});
|
||||
|
||||
test('boost ON demonstrates the categorical displacement it causes', () => {
|
||||
// Documents WHY the boost cannot be on at medium. Not an endorsement of
|
||||
// these numbers — a characterization of the mechanism, so a future reader
|
||||
// sees the cost rather than re-deriving it.
|
||||
const observed = [10, 20, 40].map((n) => ({
|
||||
n,
|
||||
rank: rankOfAnswer(rrfFusion([poolWithAnswerFirst(n)], RRF_K, true)),
|
||||
}));
|
||||
// Displacement scales with pool composition: the answer is pushed back by
|
||||
// roughly one position per boosted chunk ahead of the break-even rank.
|
||||
for (const { n, rank } of observed) {
|
||||
expect(rank, `n=${n}: boosted chunks should displace the answer`).toBeGreaterThan(0);
|
||||
}
|
||||
// And past ~20 compiled_truth chunks it leaves a default-size window.
|
||||
expect(observed.find((o) => o.n === 20)!.rank).toBeGreaterThanOrEqual(20);
|
||||
});
|
||||
|
||||
test('with the boost off, compiled_truth still wins when the vector arm ranks it first', () => {
|
||||
// Guard against over-correcting: removing the boost must not penalize
|
||||
// compiled_truth, only stop privileging it.
|
||||
const list = [chunk('prose/answer', 'compiled_truth'), chunk('code/other', 'fenced_code')];
|
||||
const fused = rrfFusion([list], RRF_K, false);
|
||||
expect(fused[0].slug).toBe('prose/answer');
|
||||
});
|
||||
});
|
||||
@@ -413,10 +413,7 @@ describe('knobsHash determinism + cross-mode separation (CDX-4)', () => {
|
||||
// #3390/#3391: bumped 12→13 for the embedding-provider migration wave —
|
||||
// legacy callers hash prov=default before AND after a provider swap, so
|
||||
// pre-migration cache rows must become unreachable on upgrade.
|
||||
// v0.42.67.x bumped 13→14: the compiled_truth boost no longer applies at
|
||||
// detail=medium (#3430). Cached rows were ranked under the old semantics,
|
||||
// so they must become unreachable rather than be served under the new ones.
|
||||
expect(KNOBS_HASH_VERSION).toBe(14);
|
||||
expect(KNOBS_HASH_VERSION).toBe(13);
|
||||
});
|
||||
|
||||
test('T1 (codex): floor_ratio set vs unset produces DIFFERENT hashes (cache contamination prevention)', () => {
|
||||
@@ -581,8 +578,8 @@ describe('v0.40.4 — graph_signals knob', () => {
|
||||
});
|
||||
|
||||
describe('v0.42.3.0 — autocut knobs', () => {
|
||||
test('KNOBS_HASH_VERSION is 14 (13→14 compiled_truth boost scope fix, #3430)', () => {
|
||||
expect(KNOBS_HASH_VERSION).toBe(14);
|
||||
test('KNOBS_HASH_VERSION is 13 (12→13 embedding-migration wave, #3390/#3391)', () => {
|
||||
expect(KNOBS_HASH_VERSION).toBe(13);
|
||||
});
|
||||
|
||||
test('bundle defaults: conservative off, balanced/tokenmax on @0.20', () => {
|
||||
|
||||
@@ -64,10 +64,7 @@ describe('KNOBS_HASH_VERSION + version invariants', () => {
|
||||
// pre-fix document-side query vectors must not be served.
|
||||
// #2825: 11→12 to fold the resolved hard-exclude prefix list (hx=) —
|
||||
// cached rows leaked GBRAIN_SEARCH_EXCLUDE'd slugs across processes.
|
||||
// #3430: 13→14 — the compiled_truth boost no longer applies at
|
||||
// detail=medium. Results are cached after fusion, so rows ranked under
|
||||
// the old boost semantics must not be served under the new ones.
|
||||
expect(KNOBS_HASH_VERSION).toBe(14);
|
||||
expect(KNOBS_HASH_VERSION).toBe(13);
|
||||
});
|
||||
|
||||
test('hash is 16 hex chars regardless of reranker config', () => {
|
||||
|
||||
@@ -1,64 +0,0 @@
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import { EventEmitter } from 'events';
|
||||
import { waitForHttpServerLifecycle, type HttpServerLifecycle } from '../src/commands/serve-http.ts';
|
||||
|
||||
class FakeHttpServer extends EventEmitter {
|
||||
listening = true;
|
||||
closeCalls = 0;
|
||||
|
||||
close(callback?: (error?: Error) => void): this {
|
||||
this.closeCalls++;
|
||||
this.listening = false;
|
||||
queueMicrotask(() => {
|
||||
callback?.();
|
||||
this.emit('close');
|
||||
});
|
||||
return this;
|
||||
}
|
||||
}
|
||||
|
||||
describe('HTTP server lifecycle', () => {
|
||||
test('waits for shared cleanup to close the server', async () => {
|
||||
const server = new FakeHttpServer();
|
||||
const signals = new EventEmitter();
|
||||
let cleanup: (() => Promise<void>) | undefined;
|
||||
let deregistered = false;
|
||||
let resolved = false;
|
||||
|
||||
const lifecycle = waitForHttpServerLifecycle(server as unknown as HttpServerLifecycle, {
|
||||
signals: signals as unknown as NodeJS.Process,
|
||||
register(_name, fn) {
|
||||
cleanup = fn;
|
||||
return () => { deregistered = true; };
|
||||
},
|
||||
}).then(() => { resolved = true; });
|
||||
|
||||
await Promise.resolve();
|
||||
expect(resolved).toBe(false);
|
||||
expect(cleanup).toBeDefined();
|
||||
|
||||
await cleanup!();
|
||||
await lifecycle;
|
||||
|
||||
expect(server.closeCalls).toBe(1);
|
||||
expect(deregistered).toBe(true);
|
||||
expect(signals.listenerCount('SIGINT')).toBe(0);
|
||||
});
|
||||
|
||||
test('SIGINT closes the server through the same idempotent path', async () => {
|
||||
const server = new FakeHttpServer();
|
||||
const signals = new EventEmitter();
|
||||
|
||||
const lifecycle = waitForHttpServerLifecycle(server as unknown as HttpServerLifecycle, {
|
||||
signals: signals as unknown as NodeJS.Process,
|
||||
register() {
|
||||
return () => {};
|
||||
},
|
||||
});
|
||||
|
||||
signals.emit('SIGINT');
|
||||
await lifecycle;
|
||||
|
||||
expect(server.closeCalls).toBe(1);
|
||||
});
|
||||
});
|
||||
@@ -1,280 +0,0 @@
|
||||
/**
|
||||
* #3056 — sync rename path: a failed `updateSlug` must not leave a live
|
||||
* duplicate of the renamed page behind.
|
||||
*
|
||||
* Before the fix, the rename loop swallowed `updateSlug` failures with an
|
||||
* empty catch ("treat as add") and could not see a zero-row UPDATE at all
|
||||
* (updateSlug returned void). The run then fell through to importFile,
|
||||
* which created/updated the row at the new path — while the old row stayed
|
||||
* behind, live, with its slug occupied. Nothing was logged, no counter
|
||||
* moved, and the duplicate was permanent.
|
||||
*
|
||||
* The fix reconciles: when the cheap rename didn't move a row AND the
|
||||
* destination demonstrably materialized, the stale old row is located
|
||||
* positively by `source_path = from` and deleted. Two safety rails:
|
||||
*
|
||||
* - dedup-skip protection: identity dedup can skip the import against
|
||||
* the OLD row, in which case nothing landed at the destination and
|
||||
* deleting the old row would destroy the only copy — no reconcile.
|
||||
* - no slug-guess deletes: the stale row is found by source_path only;
|
||||
* an unrelated row that happens to sit at the guessed slug survives.
|
||||
*
|
||||
* A failed reconcile delete lands in failedFiles so the existing failure
|
||||
* gate blocks the bookmark and the next run retries the same rename diff.
|
||||
*/
|
||||
|
||||
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, test } from 'bun:test';
|
||||
import { execSync } from 'node:child_process';
|
||||
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
import { resetPgliteState } from './helpers/reset-pglite.ts';
|
||||
|
||||
let engine: PGLiteEngine;
|
||||
const repos: string[] = [];
|
||||
// Serial-file requirement: blocked runs write real rows to the sync-failure
|
||||
// ledger under the gbrain home — isolate it per test so the operator's
|
||||
// actual ledger is never touched (GBRAIN_HOME is the isolation lever;
|
||||
// process.env.HOME does not redirect Bun's os.homedir()).
|
||||
let tmpHome: string;
|
||||
const originalGbrainHome = process.env.GBRAIN_HOME;
|
||||
|
||||
beforeAll(async () => {
|
||||
engine = new PGLiteEngine();
|
||||
await engine.connect({});
|
||||
await engine.initSchema();
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
await engine.disconnect();
|
||||
});
|
||||
|
||||
beforeEach(async () => {
|
||||
tmpHome = mkdtempSync(join(tmpdir(), 'gbrain-3056-home-'));
|
||||
process.env.GBRAIN_HOME = tmpHome;
|
||||
await resetPgliteState(engine);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
if (originalGbrainHome !== undefined) process.env.GBRAIN_HOME = originalGbrainHome;
|
||||
else delete process.env.GBRAIN_HOME;
|
||||
try { rmSync(tmpHome, { recursive: true, force: true }); } catch { /* ignore */ }
|
||||
while (repos.length) {
|
||||
const d = repos.pop();
|
||||
if (d) rmSync(d, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
function personMd(title: string, body: string): string {
|
||||
return ['---', 'type: person', `title: ${title}`, '---', '', body].join('\n');
|
||||
}
|
||||
|
||||
/** Create a temp git repo seeded with the given files + an initial commit. */
|
||||
function mkRepo(files: Record<string, string>): string {
|
||||
const dir = mkdtempSync(join(tmpdir(), 'gbrain-3056-'));
|
||||
repos.push(dir);
|
||||
execSync('git init', { cwd: dir, stdio: 'pipe' });
|
||||
execSync('git config user.email "test@test.com"', { cwd: dir, stdio: 'pipe' });
|
||||
execSync('git config user.name "Test"', { cwd: dir, stdio: 'pipe' });
|
||||
for (const [rel, content] of Object.entries(files)) {
|
||||
mkdirSync(join(dir, rel, '..'), { recursive: true });
|
||||
writeFileSync(join(dir, rel), content);
|
||||
}
|
||||
execSync('git add -A && git commit -m "initial"', { cwd: dir, stdio: 'pipe' });
|
||||
return dir;
|
||||
}
|
||||
|
||||
const SYNC_OPTS = { noPull: true, noEmbed: true, noExtract: true, sourceId: 'default' } as const;
|
||||
|
||||
async function countPages(): Promise<number> {
|
||||
const rows = await engine.executeRaw<{ n: number | string }>(
|
||||
`SELECT count(*)::int AS n FROM pages WHERE source_id = 'default'`,
|
||||
);
|
||||
return Number(rows[0]?.n ?? 0);
|
||||
}
|
||||
|
||||
describe('updateSlug engine contract (#3056)', () => {
|
||||
test('returns 1 when the old slug row is moved', async () => {
|
||||
await engine.putPage('people/old', {
|
||||
type: 'person', title: 'Old', compiled_truth: 'body',
|
||||
}, { sourceId: 'default' });
|
||||
const moved = await engine.updateSlug('people/old', 'people/new', { sourceId: 'default' });
|
||||
expect(moved).toBe(1);
|
||||
expect(await engine.getPage('people/new')).not.toBeNull();
|
||||
});
|
||||
|
||||
test('returns 0 when the old slug has no row (the silent no-op case)', async () => {
|
||||
const moved = await engine.updateSlug('people/ghost', 'people/new', { sourceId: 'default' });
|
||||
expect(moved).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('#3056: rename fallback reconciles the stale old row', () => {
|
||||
test('collision: destination slug occupied → stale old row deleted after import lands', async () => {
|
||||
const { performSync } = await import('../src/commands/sync.ts');
|
||||
const repo = mkRepo({ 'people/carol.md': personMd('Carol', 'Carol is a person.') });
|
||||
await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
expect(await engine.getPage('people/carol')).not.toBeNull();
|
||||
|
||||
// A pre-existing row already occupies the rename destination, so
|
||||
// updateSlug throws (source_id, slug) UNIQUE and the loop falls back.
|
||||
await engine.putPage('people/dana', {
|
||||
type: 'person', title: 'Dana (stale)', compiled_truth: 'occupies the destination slug',
|
||||
}, { sourceId: 'default' });
|
||||
|
||||
execSync('git mv people/carol.md people/dana.md', { cwd: repo, stdio: 'pipe' });
|
||||
execSync('git commit -m "rename carol to dana"', { cwd: repo, stdio: 'pipe' });
|
||||
|
||||
const result = await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
expect(result.status).toBe('synced');
|
||||
|
||||
// The destination carries the renamed file's content...
|
||||
const dana = await engine.getPage('people/dana');
|
||||
expect(dana).not.toBeNull();
|
||||
expect(dana!.compiled_truth).toContain('Carol is a person.');
|
||||
|
||||
// ...and the stale old row is gone — no live duplicate.
|
||||
expect(await engine.getPage('people/carol')).toBeNull();
|
||||
expect(await countPages()).toBe(1);
|
||||
});
|
||||
|
||||
test('dedup-skip against the old row must NOT reconcile: the only copy survives', async () => {
|
||||
const { performSync } = await import('../src/commands/sync.ts');
|
||||
// frontmatter.id gives identity dedup a handle: the import at the new
|
||||
// path can skip as "identical to <old row>" — in which case NOTHING
|
||||
// landed at the destination and deleting the old row would destroy the
|
||||
// only copy of the content.
|
||||
const md = ['---', 'type: person', 'title: Carol', 'id: ext-3056', '---', '', 'Carol is a person.'].join('\n');
|
||||
const repo = mkRepo({ 'people/carol.md': md });
|
||||
await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
expect(await engine.getPage('people/carol')).not.toBeNull();
|
||||
|
||||
// Destination occupied → updateSlug throws → fallback path.
|
||||
await engine.putPage('people/dana', {
|
||||
type: 'person', title: 'Dana (stale)', compiled_truth: 'occupies the destination slug',
|
||||
}, { sourceId: 'default' });
|
||||
|
||||
execSync('git mv people/carol.md people/dana.md', { cwd: repo, stdio: 'pipe' });
|
||||
execSync('git commit -m "rename carol to dana"', { cwd: repo, stdio: 'pipe' });
|
||||
|
||||
await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
|
||||
// The import skipped against the OLD row (identity dedup), so the
|
||||
// destination never materialized with the renamed content — the
|
||||
// reconcile must not have deleted the old row, which still holds the
|
||||
// only copy.
|
||||
const carol = await engine.getPage('people/carol');
|
||||
expect(carol).not.toBeNull();
|
||||
expect(carol!.compiled_truth).toContain('Carol is a person.');
|
||||
});
|
||||
|
||||
test('reconcile never deletes by slug guess: unrelated manual row survives', async () => {
|
||||
const { performSync } = await import('../src/commands/sync.ts');
|
||||
const repo = mkRepo({ 'people/carol.md': personMd('Carol', 'Carol is a person.') });
|
||||
await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
|
||||
// The file's real row drifts to a divergent slug with no source_path
|
||||
// (unlocatable), and an UNRELATED manually-curated page happens to sit
|
||||
// at the path-derived slug a naive reconcile would guess.
|
||||
await engine.executeRaw(
|
||||
`UPDATE pages SET slug = 'people/carol-divergent', source_path = NULL
|
||||
WHERE source_id = 'default' AND slug = 'people/carol'`,
|
||||
);
|
||||
await engine.putPage('people/carol', {
|
||||
type: 'person', title: 'Manual Carol', compiled_truth: 'hand-authored, not from the file',
|
||||
}, { sourceId: 'default' });
|
||||
// Destination occupied → updateSlug throws UNIQUE → fallback path.
|
||||
await engine.putPage('people/dana', {
|
||||
type: 'person', title: 'Dana (stale)', compiled_truth: 'occupies the destination slug',
|
||||
}, { sourceId: 'default' });
|
||||
|
||||
execSync('git mv people/carol.md people/dana.md', { cwd: repo, stdio: 'pipe' });
|
||||
execSync('git commit -m "rename carol to dana"', { cwd: repo, stdio: 'pipe' });
|
||||
|
||||
const result = await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
expect(result.status).toBe('synced');
|
||||
|
||||
// The destination materialized with the file's content...
|
||||
const dana = await engine.getPage('people/dana');
|
||||
expect(dana).not.toBeNull();
|
||||
expect(dana!.compiled_truth).toContain('Carol is a person.');
|
||||
// ...but no row had source_path = from, so the reconcile deleted
|
||||
// NOTHING: the unrelated manual row at the guessed slug survives.
|
||||
const manual = await engine.getPage('people/carol');
|
||||
expect(manual).not.toBeNull();
|
||||
expect(manual!.compiled_truth).toContain('hand-authored');
|
||||
});
|
||||
|
||||
test('happy path: clean git mv rename keeps page_id and touches nothing else', async () => {
|
||||
const { performSync } = await import('../src/commands/sync.ts');
|
||||
const repo = mkRepo({ 'people/carol.md': personMd('Carol', 'Carol is a person.') });
|
||||
await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
const before = await engine.getPage('people/carol');
|
||||
expect(before).not.toBeNull();
|
||||
|
||||
execSync('git mv people/carol.md people/dana.md', { cwd: repo, stdio: 'pipe' });
|
||||
execSync('git commit -m "rename carol to dana"', { cwd: repo, stdio: 'pipe' });
|
||||
|
||||
const result = await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
expect(result.status).toBe('synced');
|
||||
|
||||
const after = await engine.getPage('people/dana');
|
||||
expect(after).not.toBeNull();
|
||||
expect(after!.id).toBe(before!.id); // cheap-path rename preserved the row
|
||||
expect(await engine.getPage('people/carol')).toBeNull();
|
||||
expect(await countPages()).toBe(1);
|
||||
});
|
||||
|
||||
test('reconcile failure blocks the bookmark and the next run retries to convergence', async () => {
|
||||
const { performSync } = await import('../src/commands/sync.ts');
|
||||
const repo = mkRepo({ 'people/carol.md': personMd('Carol', 'Carol is a person.') });
|
||||
await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
await engine.putPage('people/dana', {
|
||||
type: 'person', title: 'Dana (stale)', compiled_truth: 'occupies the destination slug',
|
||||
}, { sourceId: 'default' });
|
||||
|
||||
execSync('git mv people/carol.md people/dana.md', { cwd: repo, stdio: 'pipe' });
|
||||
execSync('git commit -m "rename carol to dana"', { cwd: repo, stdio: 'pipe' });
|
||||
|
||||
// Inject a transient failure into the reconcile delete.
|
||||
const origDelete = engine.deletePage.bind(engine);
|
||||
engine.deletePage = async () => { throw new Error('injected transient delete failure'); };
|
||||
let blocked;
|
||||
try {
|
||||
blocked = await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
} finally {
|
||||
engine.deletePage = origDelete;
|
||||
}
|
||||
|
||||
// The failed reconcile is not checkpointed past: the run blocks and the
|
||||
// stale duplicate is still visible. The failure is recorded as a
|
||||
// `<rename:…>` SENTINEL, which the auto-skip valve can never
|
||||
// chronic-skip — an outage lasting longer than the threshold must not
|
||||
// quietly bank the duplicate.
|
||||
expect(blocked.status).toBe('blocked_by_failures');
|
||||
expect(blocked.failedFiles).toBe(1);
|
||||
expect(await engine.getPage('people/carol')).not.toBeNull();
|
||||
const { loadSyncFailures } = await import('../src/core/sync-failure-ledger.ts');
|
||||
const openSentinels = loadSyncFailures().filter(
|
||||
f => f.path === '<rename:people/dana.md>' && f.state === 'open',
|
||||
);
|
||||
expect(openSentinels).toHaveLength(1);
|
||||
|
||||
// Next run (failure gone) retries the same rename diff and converges.
|
||||
const result = await performSync(engine, { repoPath: repo, ...SYNC_OPTS });
|
||||
expect(result.status).toBe('synced');
|
||||
expect(await engine.getPage('people/carol')).toBeNull();
|
||||
const dana = await engine.getPage('people/dana');
|
||||
expect(dana).not.toBeNull();
|
||||
expect(dana!.compiled_truth).toContain('Carol is a person.');
|
||||
expect(await countPages()).toBe(1);
|
||||
|
||||
// The convergence also clears the sentinel row — doctor must not keep
|
||||
// warning about a rename that has since reconciled.
|
||||
const remaining = loadSyncFailures().filter(
|
||||
f => f.path === '<rename:people/dana.md>' && f.state === 'open',
|
||||
);
|
||||
expect(remaining).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user